diff --git a/.github/workflows/release-container.yml b/.github/workflows/release-container.yml index 40d44e778..2c6674bbd 100644 --- a/.github/workflows/release-container.yml +++ b/.github/workflows/release-container.yml @@ -143,7 +143,9 @@ jobs: SOURCE_REVISION=${{ needs.verify.outputs.revision }} PREVIEW_VERSION=${{ needs.verify.outputs.version }} RUNTIME_SHA256=${{ needs.verify.outputs.runtime_sha256 }} - tags: docker.io/crowdb/crowdb-iceberg:${{ needs.verify.outputs.image_tag }} + tags: | + docker.io/crowdb/crowdb-iceberg:${{ needs.verify.outputs.image_tag }} + docker.io/crowdb/crowdb-iceberg:latest - uses: sigstore/cosign-installer@v4.1.2 - name: Sign published digest env: diff --git a/CHANGELOG.md b/CHANGELOG.md index 78fabdb2b..281d2658e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,8 +3,8 @@ # Changelog -CROWDB is preparing its first development release, `0.1.0-dev`. Publication -is pending; this is not a production release or a compatibility promise. +CROWDB is preparing `0.2.0`. This is a development version, not a +production release or a compatibility promise. CROWDB does not yet maintain compatibility for persisted data, WAL, metadata, or other on-disk formats. A newer checkout may be unable to read data created by @@ -21,7 +21,14 @@ policy. ## [Unreleased] -### 0.1.0-dev preparation +### 0.2.0 preparation + +- S3 and Iceberg own separate chunk types, storage policies, and write pools + behind one access-server process. +- Explicit single-node test and three-node production protection profiles, + with strip-level mirror and EC I/O and repairable degraded EC placement. + +## [0.1.0] - Single-node Linux amd64 container with native Iceberg REST catalog and FileIO, backed by CROWDB metadata, chunk storage and disk services. @@ -31,8 +38,8 @@ policy. - Host builds and runtime-only container packaging, with a manual Docker Hub publication workflow for version and commit tags, signatures, SBOM and provenance. -The intended image is `crowdb/crowdb-iceberg:v0.1.0-dev`; no published digest is -recorded yet. The GUI is not ready for this container. Multi-node deployment, +The preview image is `crowdb/crowdb-iceberg:0.1.0`; no published digest is +recorded here. Multi-node deployment, production hardening and data-format upgrades are outside this release. See the container deployment files for supported startup, persistence, diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 403f07876..9d40476d7 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -7,7 +7,7 @@ Thank you for contributing to CROWDB. ## Development status -CROWDB is under active development at version `0.1.0-dev`. It has not reached +CROWDB is under active development at version `0.2.0`. It has not reached alpha, is not recommended for production, and must be tested with disposable data. Compatibility is not yet maintained for persisted data, WAL, metadata, or other on-disk formats. A change may deliberately replace an unreleased format diff --git a/Cargo.lock b/Cargo.lock index d30b3a9a7..07d0719ad 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -510,15 +510,6 @@ version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "217698eaf96b4a3f0bc4f3662aaa55bdf913cd54d7204591faa790070c6d0853" -[[package]] -name = "crc32c" -version = "0.6.8" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a47af21622d091a8f0fb295b88bc886ac74efcc613efc19f5d0b21de5c89e47" -dependencies = [ - "rustc_version", -] - [[package]] name = "crc32fast" version = "1.5.0" @@ -618,7 +609,7 @@ checksum = "d0a5c400df2834b80a4c3327b3aad3a4c4cd4de0629063962b03235697506a28" [[package]] name = "crowdb-access-iceberg" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "async-trait", @@ -632,6 +623,7 @@ dependencies = [ "crowdb-chunk-client", "crowdb-chunk-kv-client", "crowdb-common", + "crowdb-kv-client", "crowdb-protocol", "data-encoding", "flatbuffers", @@ -653,7 +645,7 @@ dependencies = [ [[package]] name = "crowdb-access-multipart" -version = "0.1.0" +version = "0.2.0" dependencies = [ "crowdb-protocol", "md-5", @@ -663,7 +655,7 @@ dependencies = [ [[package]] name = "crowdb-access-s3" -version = "0.1.0" +version = "0.2.0" dependencies = [ "aes-gcm", "arc-swap", @@ -675,6 +667,8 @@ dependencies = [ "crowdb-access-multipart", "crowdb-chunk-client", "crowdb-chunk-kv-client", + "crowdb-common", + "crowdb-kv-client", "crowdb-protocol", "flatbuffers", "hmac", @@ -696,7 +690,7 @@ dependencies = [ [[package]] name = "crowdb-access-server" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "async-trait", @@ -710,6 +704,7 @@ dependencies = [ "crowdb-chunk-kv-client", "crowdb-chunkdb-client", "crowdb-common", + "crowdb-console-shared", "crowdb-diskdb-client", "crowdb-diskio-client", "crowdb-kv-client", @@ -722,6 +717,7 @@ dependencies = [ "hyper", "hyper-util", "md-5", + "openssl", "percent-encoding", "quick-xml", "reqwest", @@ -737,7 +733,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-client" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "async-trait", @@ -761,7 +757,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-kv" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "async-trait", @@ -780,7 +776,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-kv-client" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "async-trait", @@ -799,7 +795,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-kv-server" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "async-trait", @@ -825,7 +821,7 @@ dependencies = [ [[package]] name = "crowdb-chunk-stream" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "async-trait", @@ -847,7 +843,7 @@ dependencies = [ [[package]] name = "crowdb-chunkdb" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "axum", @@ -883,7 +879,7 @@ dependencies = [ [[package]] name = "crowdb-chunkdb-client" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "bytes", @@ -898,7 +894,7 @@ dependencies = [ [[package]] name = "crowdb-cli" -version = "0.1.0" +version = "0.2.0" dependencies = [ "axum", "chrono", @@ -929,7 +925,7 @@ dependencies = [ [[package]] name = "crowdb-common" -version = "0.1.0" +version = "0.2.0" dependencies = [ "crowdb-test-harness", "flate2", @@ -948,10 +944,11 @@ dependencies = [ [[package]] name = "crowdb-console-shared" -version = "0.1.0" +version = "0.2.0" dependencies = [ "async-trait", "axum", + "crowdb-chunkdb-client", "crowdb-kv-client", "crowdb-protocol", "crowdb-rpc-ffi", @@ -972,7 +969,7 @@ dependencies = [ [[package]] name = "crowdb-diskdb" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "axum", @@ -1008,7 +1005,7 @@ dependencies = [ [[package]] name = "crowdb-diskdb-client" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "bytes", @@ -1027,7 +1024,7 @@ dependencies = [ [[package]] name = "crowdb-diskio-client" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "bytes", @@ -1047,7 +1044,7 @@ dependencies = [ [[package]] name = "crowdb-kv" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "bytes", @@ -1080,7 +1077,7 @@ dependencies = [ [[package]] name = "crowdb-kv-client" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "axum", @@ -1104,7 +1101,7 @@ dependencies = [ [[package]] name = "crowdb-kv-server" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "axum", @@ -1129,7 +1126,7 @@ dependencies = [ [[package]] name = "crowdb-monitor" -version = "0.1.0" +version = "0.2.0" dependencies = [ "clap", "crowdb-diskio-client", @@ -1152,12 +1149,12 @@ dependencies = [ [[package]] name = "crowdb-protocol" -version = "0.1.0" +version = "0.2.0" dependencies = [ "bincode", "bytes", - "crc32c", "crc32fast", + "crowdb-common", "flatbuffers", "fs2", "getrandom 0.2.17", @@ -1171,7 +1168,7 @@ dependencies = [ [[package]] name = "crowdb-rpc-ffi" -version = "0.1.0" +version = "0.2.0" dependencies = [ "arc-swap", "bytes", @@ -1186,7 +1183,7 @@ dependencies = [ [[package]] name = "crowdb-test-harness" -version = "0.1.0" +version = "0.2.0" dependencies = [ "crowdb-chunkdb-client", "crowdb-diskdb-client", @@ -1203,7 +1200,7 @@ dependencies = [ [[package]] name = "crowdb-tree-ffi" -version = "0.1.0" +version = "0.2.0" dependencies = [ "bytes", "cc", @@ -1215,7 +1212,7 @@ dependencies = [ [[package]] name = "crowdb-web" -version = "0.1.0" +version = "0.2.0" dependencies = [ "async-trait", "axum", @@ -3112,6 +3109,7 @@ dependencies = [ "bytes", "encoding_rs", "futures-core", + "futures-util", "h2", "http", "http-body", @@ -3136,12 +3134,14 @@ dependencies = [ "tokio", "tokio-native-tls", "tokio-rustls", + "tokio-util", "tower", "tower-http", "tower-service", "url", "wasm-bindgen", "wasm-bindgen-futures", + "wasm-streams", "web-sys", "webpki-roots", ] @@ -4386,6 +4386,19 @@ dependencies = [ "wasmparser", ] +[[package]] +name = "wasm-streams" +version = "0.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "15053d8d85c7eccdbefef60f06769760a563c7f0a9d6902a13d35c7800b0ad65" +dependencies = [ + "futures-util", + "js-sys", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", +] + [[package]] name = "wasmparser" version = "0.244.0" diff --git a/Cargo.toml b/Cargo.toml index 2f43a6620..c65e21d85 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -40,7 +40,7 @@ exclude = ["third-party/hyper"] # `unsafe_code = "deny"`. [workspace.package] -version = "0.1.0" +version = "0.2.0" edition = "2021" rust-version = "1.75" license = "Apache-2.0" diff --git a/SECURITY.md b/SECURITY.md index 0fe289ccc..cda6bdc90 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -14,7 +14,7 @@ If you discover a security vulnerability in CROWDB, please report it responsibly ## Scope -CROWDB `0.1.0-dev` is a development version for evaluation with disposable data. +CROWDB `0.2.0` is a development version for evaluation with disposable data. There is no production support commitment, supported stable release series, or guaranteed response time. Security reports are reviewed by the maintainers. diff --git a/VERSION b/VERSION index 6e8bf73aa..0ea3a944b 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -0.1.0 +0.2.0 diff --git a/app/crowdb-access-server/Cargo.toml b/app/crowdb-access-server/Cargo.toml index ccdbe5109..555f3f011 100644 --- a/app/crowdb-access-server/Cargo.toml +++ b/app/crowdb-access-server/Cargo.toml @@ -12,7 +12,7 @@ workspace = true [features] default = ["s3"] -test-util = [] +test-util = ["crowdb-chunk-client/test-util"] s3-e2e = ["s3"] iceberg-e2e = [] s3 = [ @@ -29,6 +29,7 @@ crowdb-chunk-kv-client = { path = "../../lib/crowdb-chunk-kv-client" } crowdb-common = { path = "../../lib/crowdb-common/rust" } crowdb-kv-client = { path = "../../lib/crowdb-kv-client" } crowdb-protocol = { path = "../../lib/crowdb-protocol" } +crowdb-rpc-ffi = { path = "../../lib/crowdb-rpc/ffi" } futures = { version = "0.3", optional = true } http-body-util = "0.1" hyper = { workspace = true, features = ["http1", "server"] } @@ -37,6 +38,7 @@ percent-encoding = "2" quick-xml = "0.38" base64 = "0.22" md-5 = "0.10" +openssl = "0.10" serde_json = "1" serde = { version = "1", features = ["derive"] } sha2 = "0.10" @@ -49,16 +51,17 @@ thiserror = { workspace = true } [dev-dependencies] crowdb-access-server = { path = ".", default-features = false, features = ["test-util"] } +futures = "0.3" +crowdb-console-shared = { path = "../../lib/crowdb-console-shared" } hmac = "0.12" arc-swap = "1.9" async-trait = "0.1" crowdb-chunkdb-client = { path = "../../lib/crowdb-chunkdb-client" } crowdb-diskdb-client = { path = "../../lib/crowdb-diskdb-client" } crowdb-diskio-client = { path = "../../lib/crowdb-diskio-client", features = ["test-util"] } -crowdb-rpc-ffi = { path = "../../lib/crowdb-rpc/ffi" } crowdb-test-harness = { path = "../../lib/crowdb-test-harness", features = ["chunk-kv", "chunkdb", "diskdb", "diskio"] } serde_json = "1" -reqwest = { version = "0.12", default-features = false, features = ["rustls-tls"] } +reqwest = { version = "0.12", default-features = false, features = ["rustls-tls", "stream"] } tokio = { workspace = true, features = ["io-util", "macros", "net", "rt-multi-thread", "sync", "test-util"] } [[test]] @@ -67,10 +70,6 @@ path = "tests/s3_full_stack_test.rs" harness = false required-features = ["s3-e2e"] -[[bin]] -name = "crowdb-iceberg" -path = "src/iceberg_main.rs" - [[test]] name = "iceberg_full_stack_test" path = "tests/iceberg_full_stack_test.rs" diff --git a/app/crowdb-access-server/conf/crowdb_access_server_config.toml b/app/crowdb-access-server/conf/crowdb_access_server_config.toml index c9204d095..4b7ff00e9 100644 --- a/app/crowdb-access-server/conf/crowdb_access_server_config.toml +++ b/app/crowdb-access-server/conf/crowdb_access_server_config.toml @@ -1,6 +1,10 @@ # Canonical configuration for the S3 and Iceberg access processes. # Credentials and bearer tokens are supplied through environment variables. +[deployment] +mode = "production" +max_node_failures = 1 + [common] management_seeds = ["http://127.0.0.1:10000"] diskio_connections_per_endpoint = 2 @@ -23,6 +27,7 @@ queue_capacity = 1024 min_pipelines = 1 max_pipelines = 32 max_batch_bytes = 1048576 +chunk_capacity_bytes = 1073741824 [s3] listen = "127.0.0.1:8081" @@ -42,6 +47,7 @@ max_chunk_size = 1073741824 [iceberg] listen = "127.0.0.1:8181" native_budget_bytes = 268435456 +max_chunk_size = 1073741824 [iceberg.gc] enabled = false diff --git a/app/crowdb-access-server/src/config.rs b/app/crowdb-access-server/src/config.rs index b607fea83..97f388be2 100644 --- a/app/crowdb-access-server/src/config.rs +++ b/app/crowdb-access-server/src/config.rs @@ -12,6 +12,7 @@ use serde::{Deserialize, Serialize}; #[derive(Clone, Debug, Default, Deserialize, Serialize)] #[serde(default)] pub struct AccessConfig { + pub deployment: DeploymentConfig, pub common: CommonConfig, pub read: ReadConfig, pub small_write: SmallWriteConfig, @@ -19,6 +20,30 @@ pub struct AccessConfig { pub iceberg: IcebergConfig, } +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq, Deserialize, Serialize)] +#[serde(rename_all = "snake_case")] +pub enum DeploymentMode { + #[default] + Production, + TestSingleNode, +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +#[serde(default)] +pub struct DeploymentConfig { + pub mode: DeploymentMode, + pub max_node_failures: u32, +} + +impl Default for DeploymentConfig { + fn default() -> Self { + Self { + mode: DeploymentMode::Production, + max_node_failures: 1, + } + } +} + #[derive(Clone, Debug, Deserialize, Serialize)] #[serde(default)] pub struct CommonConfig { @@ -77,6 +102,7 @@ pub struct SmallWriteConfig { /// Data bytes per strip block used for the small-object routing boundary. pub disk_block_bytes: usize, pub conversion_enabled: bool, + pub mirror_copies: Option, pub ec_data: usize, pub ec_code: usize, pub memory_budget_bytes: usize, @@ -84,6 +110,7 @@ pub struct SmallWriteConfig { pub min_pipelines: usize, pub max_pipelines: usize, pub max_batch_bytes: usize, + pub chunk_capacity_bytes: u64, } impl Default for SmallWriteConfig { @@ -93,6 +120,7 @@ impl Default for SmallWriteConfig { threshold_ratio: 0.9, disk_block_bytes: 1024 * 1024, conversion_enabled: policy.conversion_enabled, + mirror_copies: None, ec_data: policy.conversion_data_num, ec_code: policy.conversion_code_num, memory_budget_bytes: policy.memory_budget, @@ -100,6 +128,7 @@ impl Default for SmallWriteConfig { min_pipelines: policy.min_pipelines, max_pipelines: policy.max_pipelines, max_batch_bytes: policy.max_batch_bytes, + chunk_capacity_bytes: policy.chunk_capacity, } } } @@ -120,6 +149,9 @@ impl SmallWriteConfig { pub fn policy(&self) -> SmallWritePolicy { SmallWritePolicy { conversion_enabled: self.conversion_enabled, + mirror_copies: self + .mirror_copies + .unwrap_or(SmallWritePolicy::default().mirror_copies), conversion_data_num: self.ec_data, conversion_code_num: self.ec_code, memory_budget: self.memory_budget_bytes, @@ -127,6 +159,7 @@ impl SmallWriteConfig { min_pipelines: self.min_pipelines, max_pipelines: self.max_pipelines, max_batch_bytes: self.max_batch_bytes, + chunk_capacity: self.chunk_capacity_bytes, ..SmallWritePolicy::default() } } @@ -135,6 +168,8 @@ impl SmallWriteConfig { #[derive(Clone, Debug, Default, Deserialize, Serialize)] #[serde(default)] pub struct S3Config { + /// Overrides the legacy shared small-write policy for S3 only. + pub small_write: Option, pub listen: Option, pub tenant: Option, pub region: Option, @@ -148,16 +183,47 @@ pub struct S3Config { pub ec_data: Option, pub ec_code: Option, pub max_chunk_size: Option, + pub large_memory_budget_bytes: Option, + pub large_prefetch_strips_per_chunk: Option, + pub large_prefetch_max_strips_per_batch: Option, + pub large_parallel_strip_writes: Option, + pub large_held_buffers: Option, + pub large_chunk_preparation_depth: Option, + pub large_mirror_copies: Option, } #[derive(Clone, Debug, Default, Deserialize, Serialize)] #[serde(default)] pub struct IcebergConfig { + /// Overrides the legacy shared small-write policy for Iceberg only. + pub small_write: Option, pub listen: Option, pub native_budget_bytes: Option, + pub ec_data: Option, + pub ec_code: Option, + pub max_chunk_size: Option, + pub large_memory_budget_bytes: Option, + pub large_prefetch_strips_per_chunk: Option, + pub large_prefetch_max_strips_per_batch: Option, + pub large_parallel_strip_writes: Option, + pub large_held_buffers: Option, + pub large_chunk_preparation_depth: Option, + pub large_mirror_copies: Option, pub gc: IcebergGcConfig, } +impl AccessConfig { + #[must_use] + pub fn s3_small_write(&self) -> &SmallWriteConfig { + self.s3.small_write.as_ref().unwrap_or(&self.small_write) + } + + #[must_use] + pub fn iceberg_small_write(&self) -> &SmallWriteConfig { + self.iceberg.small_write.as_ref().unwrap_or(&self.small_write) + } +} + #[derive(Clone, Debug, Default, Deserialize, Serialize)] #[serde(default)] pub struct IcebergGcConfig { @@ -198,22 +264,7 @@ impl BaseConfig for AccessConfig { { return Err("read.recovery_memory_bytes must be between 1 MiB and 4 GiB".into()); } - self.small_write - .policy() - .validate() - .map_err(|error| format!("invalid small_write config: {error}"))?; - if !self.small_write.threshold_ratio.is_finite() - || self.small_write.threshold_ratio <= 0.0 - || self.small_write.threshold_ratio > 1.0 - || self.small_write.disk_block_bytes < 128 * 1024 - || self.small_write.disk_block_bytes > 1024 * 1024 - || !self.small_write.disk_block_bytes.is_power_of_two() - || self.small_write.ec_data == 0 - || self.small_write.ec_data > 32 - || self.small_write.threshold_exclusive() > self.small_write.policy().object_limit - { - return Err("small_write strip capacity or threshold is invalid".into()); - } + self.validate_small_writes()?; if self.s3.ec_data == Some(0) || self.s3.ec_code == Some(0) { return Err("S3 EC data and code counts must be nonzero".into()); } @@ -223,6 +274,56 @@ impl BaseConfig for AccessConfig { if self.iceberg.native_budget_bytes == Some(0) { return Err("Iceberg native budget must be nonzero".into()); } + if self.iceberg.ec_data == Some(0) + || self.iceberg.ec_code == Some(0) + || self.iceberg.max_chunk_size == Some(0) + || self.s3.large_memory_budget_bytes == Some(0) + || self.s3.large_prefetch_strips_per_chunk == Some(0) + || self.s3.large_prefetch_max_strips_per_batch == Some(0) + || self.s3.large_parallel_strip_writes == Some(0) + || self.s3.large_held_buffers == Some(0) + || self.s3.large_chunk_preparation_depth == Some(0) + || self.iceberg.large_memory_budget_bytes == Some(0) + || self.iceberg.large_prefetch_strips_per_chunk == Some(0) + || self.iceberg.large_prefetch_max_strips_per_batch == Some(0) + || self.iceberg.large_parallel_strip_writes == Some(0) + || self.iceberg.large_held_buffers == Some(0) + || self.iceberg.large_chunk_preparation_depth == Some(0) + || self.s3.large_mirror_copies == Some(0) + || self.iceberg.large_mirror_copies == Some(0) + { + return Err("protocol large-write settings must be nonzero".into()); + } + match self.deployment.mode { + DeploymentMode::Production => { + if self.deployment.max_node_failures != 1 { + return Err("production requires max_node_failures = 1".into()); + } + if self.s3_small_write().policy().mirror_copies < 2 + || self.iceberg_small_write().policy().mirror_copies < 2 + || self.s3.large_mirror_copies == Some(1) + || self.iceberg.large_mirror_copies == Some(1) + { + return Err("production access writes require protected strips".into()); + } + } + DeploymentMode::TestSingleNode => { + if self.deployment.max_node_failures != 0 { + return Err("test_single_node requires max_node_failures = 0".into()); + } + for config in [self.s3_small_write(), self.iceberg_small_write()] { + if config.conversion_enabled + || config.policy().mirror_copies != 1 + || config.disk_block_bytes != 1024 * 1024 + { + return Err("test_single_node requires one-copy 1 MiB mirror writes".into()); + } + } + if self.s3.large_mirror_copies != Some(1) || self.iceberg.large_mirror_copies != Some(1) { + return Err("test_single_node requires one-copy large mirror strips".into()); + } + } + } if self.s3.small_object_limit == Some(0) || self.s3.list_scan_items == Some(0) || self.s3.list_scan_bytes == Some(0) @@ -242,6 +343,34 @@ impl BaseConfig for AccessConfig { } } +impl AccessConfig { + fn validate_small_writes(&self) -> Result<(), String> { + for config in [ + &self.small_write, + self.s3_small_write(), + self.iceberg_small_write(), + ] { + config + .policy() + .validate() + .map_err(|error| format!("invalid small_write config: {error}"))?; + if !config.threshold_ratio.is_finite() + || config.threshold_ratio <= 0.0 + || config.threshold_ratio > 1.0 + || config.disk_block_bytes < 128 * 1024 + || config.disk_block_bytes > 1024 * 1024 + || !config.disk_block_bytes.is_power_of_two() + || config.ec_data == 0 + || config.ec_data > 32 + || config.threshold_exclusive() > config.policy().object_limit + { + return Err("small_write strip capacity or threshold is invalid".into()); + } + } + Ok(()) + } +} + /// Remove a single `--config ` pair and load the named TOML file. /// /// # Errors diff --git a/app/crowdb-access-server/src/iceberg.rs b/app/crowdb-access-server/src/iceberg.rs index f6b55c9da..cdf009c16 100644 --- a/app/crowdb-access-server/src/iceberg.rs +++ b/app/crowdb-access-server/src/iceberg.rs @@ -34,15 +34,14 @@ pub use file_auth::authenticate_file_request; pub use file_body::{FileBodyError, FileReadBody, FileResponseBudget}; pub use file_complete::FileCompleteBody; pub use file_encoding::{FileEncodingError, FileUploadBody}; +pub use file_http::UploadFlowSnapshot; pub use file_request::{FileRequest, FileRequestError, MultipartRequest}; pub use file_response::{FileResponseError, FileS3ErrorCode, MultipartResponses}; pub use file_selection::{CompletePart, CompleteRequestError, CompleteResolveError, CompleteSelection}; pub use file_upload::{FileUploadBudget, FileUploadConstraints, FileUploadError}; pub use http::{serve, IcebergHttpService}; pub use metrics::{IcebergMetricsSnapshot, MetricCounts, ICEBERG_OUTCOME_NAMES, ICEBERG_ROUTE_NAMES}; -pub use runtime::{run, IcebergRuntimeConfig}; +pub use runtime::{run, run_with_shutdown, IcebergRuntimeConfig}; #[cfg(feature = "test-util")] pub use connection::active_io_for_tests; -#[cfg(feature = "test-util")] -pub use gc_runtime::budget::{BudgetedGcBlocks, BudgetedGcStore, GcIoBudget}; diff --git a/app/crowdb-access-server/src/iceberg/file_body.rs b/app/crowdb-access-server/src/iceberg/file_body.rs index fb98fe282..913cdcaa7 100644 --- a/app/crowdb-access-server/src/iceberg/file_body.rs +++ b/app/crowdb-access-server/src/iceberg/file_body.rs @@ -6,8 +6,9 @@ use std::sync::{ }; use std::task::{Context, Poll}; -use crowdb_access_iceberg::file::{ByteRange, FileBlockStore, FileIoError, FileReader, FileRecord}; -use crowdb_chunk_client::{ChunkReadStream, ReadError}; +use crowdb_access_iceberg::file::{ + ByteRange, FileBlockStore, FileIoError, FileLocationStream, FileReader, FileRecord, +}; use hyper::body::{Body, Bytes, Frame, SizeHint}; #[derive(Debug, thiserror::Error)] @@ -66,13 +67,7 @@ impl FileResponseBudget { start: 0, end: record.length, }); - Some( - store - .stream_client() - .ok_or(FileIoError::Bounds)? - .read_range_stream(&locations, interval.start, interval.end) - .map_err(FileIoError::from)?, - ) + store.stream_locations(&locations, interval.start, interval.end)? } else { None }; @@ -100,12 +95,13 @@ impl Drop for Permit { } type ReadFuture = Pin>, FileIoError>)> + Send>>; -type StreamFuture = Pin>)> + Send>>; +type StreamFuture = + Pin>)> + Send>>; pub struct FileReadBody { reader: Option, pending: Option, - stream: Option, + stream: Option, stream_pending: Option, remaining: u64, permit: Option, @@ -165,7 +161,7 @@ impl Body for FileReadBody { } Some(Err(error)) => { body.finish(); - Poll::Ready(Some(Err(FileIoError::Read(error)))) + Poll::Ready(Some(Err(error))) } _ => { body.finish(); diff --git a/app/crowdb-access-server/src/iceberg/file_encoding.rs b/app/crowdb-access-server/src/iceberg/file_encoding.rs index 2a61b9692..190f499b2 100644 --- a/app/crowdb-access-server/src/iceberg/file_encoding.rs +++ b/app/crowdb-access-server/src/iceberg/file_encoding.rs @@ -123,9 +123,15 @@ impl FileUploadBody { self.chunks.is_none() && self.length.is_some() } - #[must_use] - pub fn md5(&self) -> [u8; 16] { - self.content_md5.digest() + /// Native uploads validate MD5 after their independent digest pipe finishes. + pub fn defer_md5(&mut self) { + self.content_md5.defer(); + } + + /// # Errors + /// Rejects a declared Content-MD5 that differs from the completed digest pipe. + pub fn verify_deferred_md5(&self, digest: [u8; 16]) -> Result<(), FileEncodingError> { + self.content_md5.verify_deferred(digest) } pub(super) const fn failure(&self) -> Option { diff --git a/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs b/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs index 767696863..3b3046d9c 100644 --- a/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs +++ b/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs @@ -7,6 +7,7 @@ use super::FileEncodingError; pub(super) struct ContentMd5 { expected: Option<[u8; 16]>, digest: Md5, + deferred: bool, } impl ContentMd5 { @@ -23,6 +24,7 @@ impl ContentMd5 { Ok(Self { expected, digest: Md5::new(), + deferred: false, }) } @@ -35,10 +37,26 @@ impl ContentMd5 { } pub(super) fn update(&mut self, bytes: &[u8]) { - self.digest.update(bytes); + if !self.deferred { + self.digest.update(bytes); + } + } + + pub(super) fn defer(&mut self) { + self.deferred = true; + } + + pub(super) fn verify_deferred(&self, actual: [u8; 16]) -> Result<(), FileEncodingError> { + if !self.deferred || self.expected.is_some_and(|expected| expected != actual) { + return Err(FileEncodingError::Checksum); + } + Ok(()) } pub(super) fn verify(&self) -> Result<(), FileEncodingError> { + if self.deferred { + return Ok(()); + } if self.expected.is_some_and(|expected| self.digest() != expected) { return Err(FileEncodingError::Checksum); } diff --git a/app/crowdb-access-server/src/iceberg/file_http.rs b/app/crowdb-access-server/src/iceberg/file_http.rs index 02c4d608a..9f31fd358 100644 --- a/app/crowdb-access-server/src/iceberg/file_http.rs +++ b/app/crowdb-access-server/src/iceberg/file_http.rs @@ -9,10 +9,10 @@ use crowdb_access_iceberg::file::{ RangeError, }; use crowdb_access_iceberg::key::OperationId; +use crowdb_access_iceberg::storage::{default_large_write, own_large_write}; use crowdb_access_s3::auth::{RawAuthRequest, StreamingPayloadVerifier}; use crowdb_access_s3::native_buffer::{NativeBodyAllocator, NativeBodyReceiver}; -use crowdb_chunk_client::{ChunkClientConfig, LargeWritePolicy}; -use crowdb_common::ec::EcScheme; +use crowdb_chunk_client::LargeWritePolicy; use hyper::body::Incoming; use hyper::http::header::{ACCEPT_RANGES, CONTENT_LENGTH, CONTENT_RANGE, ETAG, RANGE}; use hyper::{Method, Request, Response, StatusCode}; @@ -26,9 +26,13 @@ use super::file_request::{FileRequest, FileRequestError}; use super::file_response::{FileS3ErrorCode, MultipartResponses}; use super::file_upload::FileUploadBudget; +mod metrics; mod multipart; mod stream; +use metrics::UploadFlowMetrics; +pub use metrics::UploadFlowSnapshot; + pub(super) struct FileHttp { repository: FileRepository, multipart: MultipartRepository, @@ -41,20 +45,23 @@ pub(super) struct FileHttp { small_threshold_exclusive: usize, large_write: LargeWritePolicy, native_allocator: Option>, + upload_metrics: Arc, region: String, limits: FileServiceLimits, } impl FileHttp { + pub(super) fn upload_metrics_snapshot(&self) -> UploadFlowSnapshot { + self.upload_metrics.snapshot() + } + pub(super) fn chunk_metrics( &self, ) -> Option<( crowdb_chunk_client::ReadFlowMetricsSnapshot, crowdb_chunk_client::SmallWriteMetricsSnapshot, )> { - self.blocks - .stream_client() - .map(|client| (client.read_flow_metrics(), client.small_write_metrics())) + self.blocks.chunk_metrics() } pub(super) fn new( @@ -82,11 +89,9 @@ impl FileHttp { responses: FileResponseBudget::new(64).map_err(|_| FileGrantError::Invalid)?, uploads: FileUploadBudget::new(64).map_err(|_| FileGrantError::Invalid)?, small_threshold_exclusive: crate::config::SmallWriteConfig::default().threshold_exclusive(), - large_write: LargeWritePolicy { - ec_scheme: EcScheme::new(8, 4), - client: Arc::new(ChunkClientConfig::default()), - }, + large_write: default_large_write(), native_allocator, + upload_metrics: Arc::new(UploadFlowMetrics::default()), region, limits: FileServiceLimits { max_request_bytes: 1024 * 1024 * 1024, @@ -105,13 +110,14 @@ impl FileHttp { Ok(()) } - pub(super) fn set_large_write(&mut self, policy: LargeWritePolicy) -> Result<(), FileGrantError> { + pub(super) fn set_large_write(&mut self, mut policy: LargeWritePolicy) -> Result<(), FileGrantError> { if policy.ec_scheme.data_num == 0 || policy.ec_scheme.code_num == 0 || policy.client.read_buffer_size == 0 { return Err(FileGrantError::Invalid); } + own_large_write(&mut policy); self.large_write = policy; Ok(()) } @@ -235,12 +241,12 @@ impl FileHttp { table: file_request.location.table(), file: crowdb_access_iceberg::key::FileId::random(), }; - if let Some(client) = self.blocks.stream_client() { - let sealed = stream::upload( - client, + if self.blocks.supports_stream_io() { + let published = stream::upload( + self.blocks.as_ref(), &self.uploads, admission, - &mut body, + body, owner, file_request.location.clone(), length, @@ -248,13 +254,16 @@ impl FileHttp { native_receiver, self.small_threshold_exclusive, &self.large_write, + &self.upload_metrics, + stream::Publication::Direct { + repository: &self.repository, + context, + }, ) .await?; - let published = self - .repository - .publish(context, &sealed) - .await - .map_err(catalog_error)?; + let stream::UploadedObject::Direct(published) = published else { + return Err(FileS3ErrorCode::InternalError); + }; let mut response = Response::new(IcebergBody::new(Vec::new())); set_header(&mut response, ETAG, &etag(&published))?; return Ok(response); diff --git a/app/crowdb-access-server/src/iceberg/file_http/metrics.rs b/app/crowdb-access-server/src/iceberg/file_http/metrics.rs new file mode 100644 index 000000000..bc51f9ec0 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_http/metrics.rs @@ -0,0 +1,327 @@ +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, serde::Serialize)] +pub struct UploadFlowSnapshot { + pub attempts: u64, + pub completed: u64, + pub failed: u64, + pub cancelled: u64, + pub logical_bytes: u64, + pub body_frames: u64, + pub body_poll_ns: u64, + pub body_waits: u64, + pub body_wait_ns: u64, + pub frames_prepared: u64, + pub frame_prepare_ns: u64, + pub digest_enqueues: u64, + pub digest_enqueue_ns: u64, + pub digest_process_ns: u64, + pub write_flow_pauses: u64, + pub write_flow_pause_ns: u64, + pub queued_buffers_peak: u64, + pub writer_feeds: u64, + pub writer_feed_ns: u64, + pub strip_prepare_waits: u64, + pub strip_prepare_wait_ns: u64, + pub strip_write_successes: u64, + pub strip_write_success_ns: u64, + pub strip_write_success_max_ns: u64, + pub mirror_uncommitted_peak: u64, + pub mirror_active_write_peak: u64, + pub writer_capacity_waits: u64, + pub writer_capacity_wait_ns: u64, + pub writer_finish_ns: u64, + pub digest_finish_ns: u64, + pub publication_attempts: u64, + pub publication_ns: u64, + pub multipart_completions: u64, + pub multipart_complete_ns: u64, + pub transfer_ns: u64, +} + +#[derive(Default)] +pub(super) struct UploadFlowMetrics { + attempts: AtomicU64, + completed: AtomicU64, + failed: AtomicU64, + cancelled: AtomicU64, + logical_bytes: AtomicU64, + body_frames: AtomicU64, + body_poll_ns: AtomicU64, + body_waits: AtomicU64, + body_wait_ns: AtomicU64, + frames_prepared: AtomicU64, + frame_prepare_ns: AtomicU64, + digest_enqueues: AtomicU64, + digest_enqueue_ns: AtomicU64, + digest_process_ns: AtomicU64, + write_flow_pauses: AtomicU64, + write_flow_pause_ns: AtomicU64, + queued_buffers_peak: AtomicU64, + writer_feeds: AtomicU64, + writer_feed_ns: AtomicU64, + strip_prepare_waits: AtomicU64, + strip_prepare_wait_ns: AtomicU64, + strip_write_successes: AtomicU64, + strip_write_success_ns: AtomicU64, + strip_write_success_max_ns: AtomicU64, + mirror_uncommitted_peak: AtomicU64, + mirror_active_write_peak: AtomicU64, + writer_capacity_waits: AtomicU64, + writer_capacity_wait_ns: AtomicU64, + writer_finish_ns: AtomicU64, + digest_finish_ns: AtomicU64, + publication_attempts: AtomicU64, + publication_ns: AtomicU64, + multipart_completions: AtomicU64, + multipart_complete_ns: AtomicU64, + transfer_ns: AtomicU64, +} + +impl UploadFlowMetrics { + pub(super) fn start(self: &Arc) -> UploadObservation { + UploadObservation { + metrics: Arc::clone(self), + started: Instant::now(), + outcome: None, + sample: UploadFlowSnapshot::default(), + } + } + + pub(super) fn publication(&self, elapsed: Duration) { + self.publication_attempts.fetch_add(1, Ordering::Relaxed); + self.publication_ns.fetch_add(nanos(elapsed), Ordering::Relaxed); + } + + pub(super) fn multipart_complete(&self, elapsed: Duration) { + self.multipart_completions.fetch_add(1, Ordering::Relaxed); + self.multipart_complete_ns + .fetch_add(nanos(elapsed), Ordering::Relaxed); + } + + pub(super) fn snapshot(&self) -> UploadFlowSnapshot { + UploadFlowSnapshot { + attempts: self.attempts.load(Ordering::Relaxed), + completed: self.completed.load(Ordering::Relaxed), + failed: self.failed.load(Ordering::Relaxed), + cancelled: self.cancelled.load(Ordering::Relaxed), + logical_bytes: self.logical_bytes.load(Ordering::Relaxed), + body_frames: self.body_frames.load(Ordering::Relaxed), + body_poll_ns: self.body_poll_ns.load(Ordering::Relaxed), + body_waits: self.body_waits.load(Ordering::Relaxed), + body_wait_ns: self.body_wait_ns.load(Ordering::Relaxed), + frames_prepared: self.frames_prepared.load(Ordering::Relaxed), + frame_prepare_ns: self.frame_prepare_ns.load(Ordering::Relaxed), + digest_enqueues: self.digest_enqueues.load(Ordering::Relaxed), + digest_enqueue_ns: self.digest_enqueue_ns.load(Ordering::Relaxed), + digest_process_ns: self.digest_process_ns.load(Ordering::Relaxed), + write_flow_pauses: self.write_flow_pauses.load(Ordering::Relaxed), + write_flow_pause_ns: self.write_flow_pause_ns.load(Ordering::Relaxed), + queued_buffers_peak: self.queued_buffers_peak.load(Ordering::Relaxed), + writer_feeds: self.writer_feeds.load(Ordering::Relaxed), + writer_feed_ns: self.writer_feed_ns.load(Ordering::Relaxed), + strip_prepare_waits: self.strip_prepare_waits.load(Ordering::Relaxed), + strip_prepare_wait_ns: self.strip_prepare_wait_ns.load(Ordering::Relaxed), + strip_write_successes: self.strip_write_successes.load(Ordering::Relaxed), + strip_write_success_ns: self.strip_write_success_ns.load(Ordering::Relaxed), + strip_write_success_max_ns: self.strip_write_success_max_ns.load(Ordering::Relaxed), + mirror_uncommitted_peak: self.mirror_uncommitted_peak.load(Ordering::Relaxed), + mirror_active_write_peak: self.mirror_active_write_peak.load(Ordering::Relaxed), + writer_capacity_waits: self.writer_capacity_waits.load(Ordering::Relaxed), + writer_capacity_wait_ns: self.writer_capacity_wait_ns.load(Ordering::Relaxed), + writer_finish_ns: self.writer_finish_ns.load(Ordering::Relaxed), + digest_finish_ns: self.digest_finish_ns.load(Ordering::Relaxed), + publication_attempts: self.publication_attempts.load(Ordering::Relaxed), + publication_ns: self.publication_ns.load(Ordering::Relaxed), + multipart_completions: self.multipart_completions.load(Ordering::Relaxed), + multipart_complete_ns: self.multipart_complete_ns.load(Ordering::Relaxed), + transfer_ns: self.transfer_ns.load(Ordering::Relaxed), + } + } +} + +pub(super) struct UploadObservation { + metrics: Arc, + started: Instant, + outcome: Option, + sample: UploadFlowSnapshot, +} + +impl UploadObservation { + pub(super) fn body_poll(&mut self, elapsed: Duration, has_frame: bool) { + self.sample.body_poll_ns += nanos(elapsed); + self.sample.body_frames += u64::from(has_frame); + } + + pub(super) fn body_wait(&mut self, elapsed: Duration) { + self.sample.body_waits += 1; + self.sample.body_wait_ns += nanos(elapsed); + } + + pub(super) fn payload(&mut self, bytes: usize) { + self.sample.logical_bytes += bytes as u64; + } + + pub(super) fn frame_prepare(&mut self, frames: usize, elapsed: Duration) { + self.sample.frames_prepared += u64::try_from(frames).unwrap_or(u64::MAX); + self.sample.frame_prepare_ns += nanos(elapsed); + } + + pub(super) fn digest_enqueue(&mut self, elapsed: Duration) { + self.sample.digest_enqueues += 1; + self.sample.digest_enqueue_ns += nanos(elapsed); + } + + pub(super) fn digest_process(&mut self, elapsed: Duration) { + self.sample.digest_process_ns += nanos(elapsed); + } + + pub(super) fn write_flow_pause(&mut self, elapsed: Duration) { + self.sample.write_flow_pauses += 1; + self.sample.write_flow_pause_ns += nanos(elapsed); + } + + pub(super) fn queued_buffers_peak(&mut self, peak: u64) { + self.sample.queued_buffers_peak = self.sample.queued_buffers_peak.max(peak); + } + + pub(super) fn writer_feeds(&mut self, count: u64, elapsed: Duration) { + self.sample.writer_feeds += count; + self.sample.writer_feed_ns += nanos(elapsed); + } + + pub(super) fn chunk_write_timing(&mut self, timing: crowdb_chunk_client::ChunkWriteTiming) { + self.sample.strip_prepare_waits += timing.strip_prepare_waits; + self.sample.strip_prepare_wait_ns += nanos(timing.strip_prepare_wait_time); + self.sample.strip_write_successes += timing.strip_write_successes; + self.sample.strip_write_success_ns += nanos(timing.strip_write_success_time); + self.sample.strip_write_success_max_ns = self + .sample + .strip_write_success_max_ns + .max(nanos(timing.strip_write_success_max)); + self.sample.mirror_uncommitted_peak = self + .sample + .mirror_uncommitted_peak + .max(timing.mirror_uncommitted_peak); + self.sample.mirror_active_write_peak = self + .sample + .mirror_active_write_peak + .max(timing.mirror_active_write_peak); + } + + pub(super) fn writer_capacity_waits(&mut self, count: u64, elapsed: Duration) { + self.sample.writer_capacity_waits += count; + self.sample.writer_capacity_wait_ns += nanos(elapsed); + } + + pub(super) fn writer_finish(&mut self, elapsed: Duration) { + self.sample.writer_finish_ns += nanos(elapsed); + } + + pub(super) fn digest_finish(&mut self, elapsed: Duration) { + self.sample.digest_finish_ns += nanos(elapsed); + } + + pub(super) fn complete(&mut self, success: bool) { + self.outcome = Some(success); + } +} + +impl Drop for UploadObservation { + fn drop(&mut self) { + self.metrics.attempts.fetch_add(1, Ordering::Relaxed); + match self.outcome { + Some(true) => &self.metrics.completed, + Some(false) => &self.metrics.failed, + None => &self.metrics.cancelled, + } + .fetch_add(1, Ordering::Relaxed); + self.metrics + .logical_bytes + .fetch_add(self.sample.logical_bytes, Ordering::Relaxed); + self.metrics + .body_frames + .fetch_add(self.sample.body_frames, Ordering::Relaxed); + self.metrics + .body_poll_ns + .fetch_add(self.sample.body_poll_ns, Ordering::Relaxed); + self.metrics + .body_waits + .fetch_add(self.sample.body_waits, Ordering::Relaxed); + self.metrics + .body_wait_ns + .fetch_add(self.sample.body_wait_ns, Ordering::Relaxed); + self.metrics + .frames_prepared + .fetch_add(self.sample.frames_prepared, Ordering::Relaxed); + self.metrics + .frame_prepare_ns + .fetch_add(self.sample.frame_prepare_ns, Ordering::Relaxed); + self.metrics + .digest_enqueues + .fetch_add(self.sample.digest_enqueues, Ordering::Relaxed); + self.metrics + .digest_enqueue_ns + .fetch_add(self.sample.digest_enqueue_ns, Ordering::Relaxed); + self.metrics + .digest_process_ns + .fetch_add(self.sample.digest_process_ns, Ordering::Relaxed); + self.metrics + .write_flow_pauses + .fetch_add(self.sample.write_flow_pauses, Ordering::Relaxed); + self.metrics + .write_flow_pause_ns + .fetch_add(self.sample.write_flow_pause_ns, Ordering::Relaxed); + self.metrics + .queued_buffers_peak + .fetch_max(self.sample.queued_buffers_peak, Ordering::Relaxed); + self.metrics + .writer_feeds + .fetch_add(self.sample.writer_feeds, Ordering::Relaxed); + self.metrics + .writer_feed_ns + .fetch_add(self.sample.writer_feed_ns, Ordering::Relaxed); + self.metrics + .strip_prepare_waits + .fetch_add(self.sample.strip_prepare_waits, Ordering::Relaxed); + self.metrics + .strip_prepare_wait_ns + .fetch_add(self.sample.strip_prepare_wait_ns, Ordering::Relaxed); + self.metrics + .strip_write_successes + .fetch_add(self.sample.strip_write_successes, Ordering::Relaxed); + self.metrics + .strip_write_success_ns + .fetch_add(self.sample.strip_write_success_ns, Ordering::Relaxed); + self.metrics + .strip_write_success_max_ns + .fetch_max(self.sample.strip_write_success_max_ns, Ordering::Relaxed); + self.metrics + .mirror_uncommitted_peak + .fetch_max(self.sample.mirror_uncommitted_peak, Ordering::Relaxed); + self.metrics + .mirror_active_write_peak + .fetch_max(self.sample.mirror_active_write_peak, Ordering::Relaxed); + self.metrics + .writer_capacity_waits + .fetch_add(self.sample.writer_capacity_waits, Ordering::Relaxed); + self.metrics + .writer_capacity_wait_ns + .fetch_add(self.sample.writer_capacity_wait_ns, Ordering::Relaxed); + self.metrics + .writer_finish_ns + .fetch_add(self.sample.writer_finish_ns, Ordering::Relaxed); + self.metrics + .digest_finish_ns + .fetch_add(self.sample.digest_finish_ns, Ordering::Relaxed); + self.metrics + .transfer_ns + .fetch_add(nanos(self.started.elapsed()), Ordering::Relaxed); + } +} + +fn nanos(duration: Duration) -> u64 { + duration.as_nanos().try_into().unwrap_or(u64::MAX) +} diff --git a/app/crowdb-access-server/src/iceberg/file_http/multipart.rs b/app/crowdb-access-server/src/iceberg/file_http/multipart.rs index e9a9de6a6..e89f5fa92 100644 --- a/app/crowdb-access-server/src/iceberg/file_http/multipart.rs +++ b/app/crowdb-access-server/src/iceberg/file_http/multipart.rs @@ -1,9 +1,10 @@ use std::sync::Arc; +use std::time::Instant; use crowdb_access_iceberg::catalog::{CatalogContext, CatalogError}; use crowdb_access_iceberg::file::{ FileIdentity, FileOperation, FileSealError, FileSealer, MultipartAdmissionLimits, MultipartPart, - MultipartPhase, MultipartSession, MultipartStreamPart, MultipartWorkError, + MultipartPhase, MultipartSession, MultipartWorkError, }; use crowdb_access_iceberg::key::{FileId, OperationId}; use crowdb_access_s3::auth::StreamingPayloadVerifier; @@ -130,26 +131,45 @@ impl FileHttp { pending: None, credit: None, }; - let policy = self - .admission - .initialize( - session.context, - MultipartAdmissionLimits { - max_sessions: 1024, - max_reserved_bytes: 64 * 1024 * 1024 * 1024 * 1024, - }, - ) - .await - .map_err(catalog_error)?; - admission - .check_create(&session, &policy) - .map_err(admission_error)?; - if !self - .admission - .reserve(&policy, &session, now_ms) - .await - .map_err(catalog_error)? - { + let mut reserved = false; + for attempt in 0..256_u64 { + let policy = match self + .admission + .initialize( + session.context, + MultipartAdmissionLimits { + max_sessions: 1024, + max_reserved_bytes: 64 * 1024 * 1024 * 1024 * 1024, + }, + ) + .await + { + Ok(policy) => policy, + Err(CatalogError::Busy) => { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1).min(16))).await; + continue; + } + Err(error) => return Err(catalog_error(error)), + }; + if policy.pending.is_some() { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1).min(16))).await; + continue; + } + admission + .check_create(&session, &policy) + .map_err(admission_error)?; + match self.admission.reserve(&policy, &session, now_ms).await { + Ok(true) => { + reserved = true; + break; + } + Ok(false) | Err(CatalogError::Busy) => { + tokio::time::sleep(std::time::Duration::from_millis((attempt + 1).min(16))).await; + } + Err(error) => return Err(catalog_error(error)), + } + } + if !reserved { return Err(FileS3ErrorCode::SlowDown); } let durable = self @@ -190,12 +210,12 @@ impl FileHttp { table: session.owner.table, file: FileId::random(), }; - let (tree, stream) = if let Some(client) = self.blocks.stream_client() { - let record = super::stream::upload( - client, + if self.blocks.supports_stream_io() { + let published = super::stream::upload( + self.blocks.as_ref(), &self.uploads, admission, - &mut body, + body, owner, session.location.clone(), length, @@ -203,69 +223,64 @@ impl FileHttp { native_receiver, self.small_threshold_exclusive, &self.large_write, + &self.upload_metrics, + super::stream::Publication::Part { + repository: &self.multipart, + session: &session, + number: part_number, + now_ms, + }, ) .await?; - ( - None, - Some(MultipartStreamPart { - length: record.length, - content: record.content, - }), + let super::stream::UploadedObject::Part(part) = published else { + return Err(FileS3ErrorCode::InternalError); + }; + return MultipartResponses::upload_part(&part) + .map(|response| response.map(IcebergBody::new)) + .map_err(|_| FileS3ErrorCode::InternalError); + } + let tree = admission + .receive( + &self.uploads, + &mut body, + self.blocks.clone(), + owner, + length, + digest, ) - } else { - let tree = admission - .receive( - &self.uploads, - &mut body, - self.blocks.clone(), - owner, - length, - digest, - ) - .await - .map_err(|error| { - body.failure() - .map_or_else(|| admission_error(error), encoding_error) - })?; - (Some(tree), None) - }; + .await + .map_err(|error| { + body.failure() + .map_or_else(|| admission_error(error), encoding_error) + })?; let mut part = MultipartPart { upload: session.upload, number: part_number, revision: 1, modified_ms: now_ms, owner, - tree, - stream, + tree: Some(tree), + stream: None, }; - if part.stream.is_some() { - part = self - .multipart - .put_stream_part(&session, &part, now_ms) - .await - .map_err(catalog_error)? - .ok_or(FileS3ErrorCode::SlowDown)?; - } else { - let before = self - .multipart - .part_for_upload(&session, part_number) - .await - .map_err(catalog_error)?; - part.revision = before.map_or(1, |part| part.revision.checked_add(1).unwrap_or(0)); - let pending = self - .multipart - .reserve_part_state(&session, &part, now_ms) - .await - .map_err(catalog_error)? - .ok_or(FileS3ErrorCode::SlowDown)?; - if !self - .multipart - .settle_part(&pending) - .await - .map_err(catalog_error)? - { - return Err(FileS3ErrorCode::SlowDown); - } + let before = self + .multipart + .part_for_upload(&session, part_number) + .await + .map_err(catalog_error)?; + part.revision = before.map_or(1, |part| part.revision.checked_add(1).unwrap_or(0)); + let pending = self + .multipart + .reserve_part_state(&session, &part, now_ms) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::SlowDown)?; + if !self + .multipart + .settle_part(&pending) + .await + .map_err(catalog_error)? + { + return Err(FileS3ErrorCode::SlowDown); } MultipartResponses::upload_part(&part) .map(|response| response.map(IcebergBody::new)) @@ -344,10 +359,10 @@ impl FileHttp { let resource = session.location.object_key(); let body = crate::iceberg::FileCompleteBody::new( async move { - service - .drive_complete(session, expected, now_ms, &url) - .await - .map(Response::into_body) + let started = Instant::now(); + let result = service.drive_complete(session, expected, now_ms, &url).await; + service.upload_metrics.multipart_complete(started.elapsed()); + result.map(Response::into_body) }, &resource, std::time::Duration::from_secs(10), diff --git a/app/crowdb-access-server/src/iceberg/file_http/stream.rs b/app/crowdb-access-server/src/iceberg/file_http/stream.rs index c509c61b5..fe266975b 100644 --- a/app/crowdb-access-server/src/iceberg/file_http/stream.rs +++ b/app/crowdb-access-server/src/iceberg/file_http/stream.rs @@ -1,26 +1,76 @@ use std::fmt::Write; +use std::future::Future; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; +use std::task::Poll; +use std::time::Instant; +use crowdb_access_iceberg::catalog::CatalogContext; use crowdb_access_iceberg::file::{ - ContentFormat, FileContent, FileIdentity, FileKind, FileLocation, FileRecord, + FileBlockStore, FileIdentity, FileLocation, FileRecord, FileRepository, MultipartPart, + MultipartRepository, MultipartSession, MultipartStreamPart, }; +use crowdb_access_iceberg::storage::IcebergFileWriter; use crowdb_access_s3::native_buffer::NativeBodyReceiver; -use crowdb_chunk_client::{ChunkIoClient, ChunkIoWriter, LargeWritePolicy}; -use crowdb_protocol::frame::MAX_FRAME_PAYLOAD_BYTES; +use crowdb_chunk_client::{FramedWriteBuffer, LargeWritePolicy}; +use crowdb_protocol::frame::{FrameMagic, MAX_FRAME_PAYLOAD_BYTES}; use http_body_util::BodyExt; -use hyper::body::Bytes; -use hyper::body::Incoming; -use sha2::{Digest, Sha256}; +use hyper::body::{Bytes, Incoming}; +use tokio::sync::mpsc; +use super::metrics::{UploadFlowMetrics, UploadObservation}; use super::{ - admission_error, multipart, FileS3ErrorCode, FileTransferAdmission, FileUploadBody, FileUploadBudget, + admission_error, catalog_error, multipart, FileS3ErrorCode, FileTransferAdmission, FileUploadBody, + FileUploadBudget, }; +use crate::upload_flow::digest_pipe::{DigestPipe, Digests}; +use crate::upload_flow::{drive_transfer, write_buffers, OfferStatus, UploadBuffer, WriteFlow}; -#[allow(clippy::too_many_arguments, clippy::too_many_lines)] +// Sixteen complete frames occupy one 1-MiB physical mirror strip. A full +// logical MiB would spill frame overhead into the next strip and serialize +// otherwise independent writes in the non-native receive path. +const TARGET_BUFFER_BYTES: usize = 16 * MAX_FRAME_PAYLOAD_BYTES; + +struct WriteObject<'a> { + body: FileUploadBody, + writer: IcebergFileWriter, + digest: DigestPipe, + held_buffers: usize, + admission: &'a FileTransferAdmission, + receiver: Option<&'a NativeBodyReceiver>, + owner: FileIdentity, + location: FileLocation, + declared_length: Option, + expected_sha256: Option<[u8; 32]>, + observation: UploadObservation, + publication: Publication<'a>, + metrics: &'a UploadFlowMetrics, +} + +pub(super) enum Publication<'a> { + Direct { + repository: &'a FileRepository, + context: CatalogContext, + }, + Part { + repository: &'a MultipartRepository, + session: &'a MultipartSession, + number: u16, + now_ms: u64, + }, +} + +pub(super) enum UploadedObject { + Direct(FileRecord), + Part(MultipartPart), +} + +#[allow(clippy::too_many_arguments)] pub(super) async fn upload( - client: &ChunkIoClient, + blocks: &dyn FileBlockStore, budget: &FileUploadBudget, admission: &FileTransferAdmission, - body: &mut FileUploadBody, + mut body: FileUploadBody, owner: FileIdentity, location: FileLocation, declared_length: Option, @@ -28,162 +78,336 @@ pub(super) async fn upload( native_receiver: Option<&NativeBodyReceiver>, small_threshold_exclusive: usize, large_write: &LargeWritePolicy, -) -> Result { + metrics: &Arc, + publication: Publication<'_>, +) -> Result { let _permit = budget.acquire().map_err(|_| FileS3ErrorCode::SlowDown)?; let small = declared_length .and_then(|length| usize::try_from(length).ok()) .filter(|length| *length < small_threshold_exclusive); - let handoff = native_receiver.filter(|_| small.is_none() && body.native_handoff_eligible()); - let mut writer: Box = if let Some(length) = small { - let key = location.to_string(); - if length <= MAX_FRAME_PAYLOAD_BYTES { - let mut small = client - .prepare_small_write_for_key(length, key.as_bytes()) - .await - .map_err(|_| FileS3ErrorCode::SlowDown)?; - small.require_durable_completion(); - Box::new(small) - } else { - Box::new( - client - .prepare_shared_object_write_for_key(length, key.as_bytes()) - .await - .map_err(|_| FileS3ErrorCode::SlowDown)?, - ) - } - } else { - let mut large = client.prepare_large_write(declared_length, large_write.clone()); - large - .wait_until_prepared() - .await - .map_err(|_| FileS3ErrorCode::SlowDown)?; - Box::new(large) - }; - if let Some(receiver) = handoff { + let receiver = native_receiver.filter(|_| small.is_none() && body.native_handoff_eligible()); + let writer = blocks + .prepare_upload_writer(&location.to_string(), small, declared_length, large_write) + .await + .map_err(|_| FileS3ErrorCode::SlowDown)? + .ok_or(FileS3ErrorCode::SlowDown)?; + if let Some(receiver) = receiver { receiver.enable_owner_handoff(); } - let mut length = 0u64; - let mut sha256 = expected_sha256.map(|_| Sha256::new()); - let target_buffer = usize::try_from(declared_length.unwrap_or(1024 * 1024)) - .unwrap_or(1024 * 1024) - .clamp(1, 1024 * 1024); - let mut pending = Vec::with_capacity(if handoff.is_some() { 0 } else { target_buffer }); - let transfer = async { - while let Some(frame) = body.frame().await { - let mut bytes = frame - .map_err(multipart::encoding_error)? - .into_data() - .map_err(|_| FileS3ErrorCode::InvalidRequest)?; - if bytes.is_empty() { - continue; - } - length = length - .checked_add(bytes.len() as u64) - .ok_or(FileS3ErrorCode::EntityTooLarge)?; - admission.check_bytes(length, length).map_err(admission_error)?; - if declared_length.is_some_and(|declared| length > declared) { - return Err(FileS3ErrorCode::InvalidRequest); + body.defer_md5(); + WriteObject { + body, + writer, + digest: DigestPipe::start(expected_sha256.is_some()), + held_buffers: large_write.client.large_held_buffers, + admission, + receiver, + owner, + location, + declared_length, + expected_sha256, + observation: metrics.start(), + publication, + metrics, + } + .run() + .await +} + +impl WriteObject<'_> { + async fn run(mut self) -> Result { + // One upload task polls both sides before yielding. + let (sender, receiver) = mpsc::channel(self.held_buffers); + let progress = AtomicU64::new(0); + let queued_peak = AtomicU64::new(0); + let flow = WriteFlow::new(sender, &progress, &queued_peak); + let target_buffer = usize::try_from(self.declared_length.unwrap_or(TARGET_BUFFER_BYTES as u64)) + .unwrap_or(TARGET_BUFFER_BYTES) + .clamp(1, TARGET_BUFFER_BYTES); + let receive = receive_body( + &mut self.body, + &mut self.digest, + self.admission, + self.receiver, + self.declared_length, + target_buffer, + flow, + &mut self.observation, + ); + let write = write_buffers(&mut self.writer, receiver, &progress, |_| { + FileS3ErrorCode::SlowDown + }); + let transfer = drive_transfer(receive, write, &progress).await; + self.observation + .queued_buffers_peak(queued_peak.load(Ordering::Relaxed)); + let (length, written) = match transfer { + Ok(result) => result, + Err(error) => { + let _ = self.digest.finish().await; + let _ = self.writer.on_error().await; + self.observation.complete(false); + return Err(error); } - if let Some(hash) = &mut sha256 { - hash.update(&bytes); + }; + self.observation.writer_feeds(written.feeds, written.feed_time); + self.observation + .writer_capacity_waits(written.capacity_waits, written.capacity_wait_time); + let digest_started = Instant::now(); + let digest = self + .digest + .finish() + .await + .map_err(|()| FileS3ErrorCode::InternalError); + self.observation.digest_finish(digest_started.elapsed()); + let etag = digest.and_then(|digest| self.validate_digest(&digest)); + let etag = match etag { + Ok(etag) => etag, + Err(error) => { + let _ = self.writer.on_error().await; + self.observation.complete(false); + return Err(error); } - while !writer.require_data() && !writer.input_complete() { - writer.wait_for_capacity().await; + }; + let started = Instant::now(); + let locations = self + .writer + .on_finish() + .await + .map_err(|_| FileS3ErrorCode::SlowDown); + self.observation.writer_finish(started.elapsed()); + if let Some(timing) = self.writer.write_timing() { + self.observation.chunk_write_timing(timing); + } + let locations = match locations { + Ok(locations) => locations, + Err(error) => { + let _ = self.writer.on_error().await; + self.observation.complete(false); + return Err(error); } - if let Some(receiver) = handoff { - if let Some(owner) = receiver.take_ready_owner() { - writer - .on_framed_data(Box::new(owner)) - .await - .map_err(|_| FileS3ErrorCode::SlowDown)?; - } - } else { - while !bytes.is_empty() { - let count = (target_buffer - pending.len()).min(bytes.len()); - pending.extend_from_slice(&bytes.split_to(count)); - if pending.len() == target_buffer { - writer - .on_data(Bytes::from(std::mem::replace( - &mut pending, - Vec::with_capacity(target_buffer), - ))) - .await - .map_err(|_| FileS3ErrorCode::SlowDown)?; - } - } + }; + let record = FileRecord::from_uploaded_locations( + self.owner.file, + self.location.clone(), + &locations, + length, + etag, + ) + .map_err(|_| FileS3ErrorCode::InternalError); + let record = match record { + Ok(record) => record, + Err(error) => { + self.observation.complete(false); + return Err(error); } + }; + let published = self.publish(record).await; + self.observation.complete(published.is_ok()); + published + } + + fn validate_digest(&mut self, digest: &Digests) -> Result { + self.observation.digest_process(digest.process_time); + self.body + .verify_deferred_md5(digest.md5) + .map_err(multipart::encoding_error)?; + if self + .expected_sha256 + .is_some_and(|expected| digest.sha256 != Some(expected)) + { + return Err(FileS3ErrorCode::InvalidRequest); + } + let mut etag = String::with_capacity(32); + for byte in digest.md5 { + write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); } - if let Some(receiver) = handoff { - if let Some(owner) = receiver - .finish_owner_when_ready() + Ok(etag) + } + + async fn publish(&mut self, record: FileRecord) -> Result { + let started = Instant::now(); + let published = match &self.publication { + Publication::Direct { repository, context } => repository + .publish(*context, &record) .await - .map_err(|_| FileS3ErrorCode::InvalidRequest)? - { - writer - .on_framed_data(Box::new(owner)) + .map(UploadedObject::Direct) + .map_err(catalog_error), + Publication::Part { + repository, + session, + number, + now_ms, + } => { + let part = MultipartPart { + upload: session.upload, + number: *number, + revision: 1, + modified_ms: *now_ms, + owner: self.owner, + tree: None, + stream: Some(MultipartStreamPart { + length: record.length, + content: record.content, + }), + }; + repository + .put_stream_part(session, &part, *now_ms) .await - .map_err(|_| FileS3ErrorCode::SlowDown)?; + .map_err(catalog_error) + .and_then(|part| part.map(UploadedObject::Part).ok_or(FileS3ErrorCode::SlowDown)) + } + }; + self.metrics.publication(started.elapsed()); + published + } +} + +#[allow(clippy::too_many_arguments, clippy::too_many_lines)] +async fn receive_body( + body: &mut FileUploadBody, + digest: &mut DigestPipe, + admission: &FileTransferAdmission, + native_receiver: Option<&NativeBodyReceiver>, + declared_length: Option, + target_buffer: usize, + flow: WriteFlow<'_>, + observation: &mut UploadObservation, +) -> Result { + let mut length = 0u64; + let mut pending = Vec::with_capacity(if native_receiver.is_some() { + 0 + } else { + target_buffer + }); + let mut digest_pending = Vec::with_capacity(16); + loop { + let started = Instant::now(); + let mut frame = std::pin::pin!(body.frame()); + let mut waited = None; + let next = std::future::poll_fn(|cx| match frame.as_mut().poll(cx) { + Poll::Pending => { + waited.get_or_insert_with(Instant::now); + Poll::Pending } + Poll::Ready(value) => Poll::Ready(value), + }) + .await; + if let Some(waited) = waited { + observation.body_wait(waited.elapsed()); } - if !pending.is_empty() { - writer - .on_data(Bytes::from(pending)) - .await - .map_err(|_| FileS3ErrorCode::SlowDown)?; + observation.body_poll(started.elapsed(), next.is_some()); + let Some(frame) = next else { break }; + let mut bytes = frame + .map_err(multipart::encoding_error)? + .into_data() + .map_err(|_| FileS3ErrorCode::InvalidRequest)?; + if bytes.is_empty() { + continue; } - if declared_length.is_some_and(|declared| declared != length) - || expected_sha256 - .zip(sha256) - .is_some_and(|(expected, hash)| <[u8; 32]>::from(hash.finalize()) != expected) - { + observation.payload(bytes.len()); + length = length + .checked_add(bytes.len() as u64) + .ok_or(FileS3ErrorCode::EntityTooLarge)?; + admission.check_bytes(length, length).map_err(admission_error)?; + if declared_length.is_some_and(|declared| length > declared) { return Err(FileS3ErrorCode::InvalidRequest); } - writer.on_finish().await.map_err(|_| FileS3ErrorCode::SlowDown) + if let Some(receiver) = native_receiver { + digest_pending.push(bytes); + if let Some(mut owner) = receiver.take_ready_owner() { + let started = Instant::now(); + owner + .prepare_frames(FrameMagic::RepoLargeV1, super::now_ms()?) + .map_err(|_| FileS3ErrorCode::InternalError)?; + observation.frame_prepare(owner.frame_count(), started.elapsed()); + handoff( + digest, + &flow, + UploadBuffer::Framed(Box::new(owner)), + std::mem::take(&mut digest_pending), + observation, + ) + .await?; + } + } else { + while !bytes.is_empty() { + let count = (target_buffer - pending.len()).min(bytes.len()); + let piece = bytes.split_to(count); + pending.extend_from_slice(&piece); + digest_pending.push(piece); + if pending.len() == target_buffer { + handoff( + digest, + &flow, + UploadBuffer::Data(Bytes::from(std::mem::replace( + &mut pending, + Vec::with_capacity(target_buffer), + ))), + std::mem::take(&mut digest_pending), + observation, + ) + .await?; + } + } + } } - .await; - let locations = match transfer { - Ok(locations) => locations, - Err(error) => { - let _ = writer.on_error().await; - return Err(error); + if let Some(receiver) = native_receiver { + if let Some(mut owner) = receiver + .finish_owner_when_ready() + .await + .map_err(|_| FileS3ErrorCode::InvalidRequest)? + { + let started = Instant::now(); + owner + .prepare_frames(FrameMagic::RepoLargeV1, super::now_ms()?) + .map_err(|_| FileS3ErrorCode::InternalError)?; + observation.frame_prepare(owner.frame_count(), started.elapsed()); + handoff( + digest, + &flow, + UploadBuffer::Framed(Box::new(owner)), + std::mem::take(&mut digest_pending), + observation, + ) + .await?; } - }; - let mut etag = String::with_capacity(32); - for byte in body.md5() { - write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); } - let content = - FileContent::from_locations(&locations, length, etag).map_err(|_| FileS3ErrorCode::InternalError)?; - let (kind, format) = format_for_location(&location); - let record = FileRecord { - file: owner.file, - location, - kind, - format, - length, - digest: [0; 32], - content, - hint: None, - }; - record.validate().map_err(|_| FileS3ErrorCode::InternalError)?; - Ok(record) + if !pending.is_empty() { + handoff( + digest, + &flow, + UploadBuffer::Data(Bytes::from(pending)), + std::mem::take(&mut digest_pending), + observation, + ) + .await?; + } + if declared_length.is_some_and(|declared| declared != length) { + return Err(FileS3ErrorCode::InvalidRequest); + } + Ok(length) } -fn format_for_location(location: &FileLocation) -> (FileKind, ContentFormat) { - let path = location.relative_key(); - let extension = std::path::Path::new(path).extension(); - let has_extension = |wanted: &str| extension.is_some_and(|value| value.eq_ignore_ascii_case(wanted)); - if has_extension("json") { - (FileKind::Metadata, ContentFormat::Json) - } else if has_extension("avro") { - (FileKind::Unbound, ContentFormat::Avro) - } else if has_extension("parquet") { - (FileKind::Unbound, ContentFormat::Parquet) - } else if has_extension("orc") { - (FileKind::Unbound, ContentFormat::Orc) - } else if has_extension("puffin") { - (FileKind::Unbound, ContentFormat::Puffin) - } else { - (FileKind::Unbound, ContentFormat::Opaque) +async fn handoff( + digest: &DigestPipe, + flow: &WriteFlow<'_>, + buffer: UploadBuffer, + payload: Vec, + observation: &mut UploadObservation, +) -> Result<(), FileS3ErrorCode> { + let offer_status = flow.offer(buffer).await.map_err(|()| FileS3ErrorCode::SlowDown)?; + let started = Instant::now(); + digest + .enqueue(payload) + .map_err(|()| FileS3ErrorCode::InternalError)?; + observation.digest_enqueue(started.elapsed()); + match offer_status { + OfferStatus::Continue => Ok(()), + OfferStatus::Pause => { + let started = Instant::now(); + flow.wait_ready().await.map_err(|()| FileS3ErrorCode::SlowDown)?; + observation.write_flow_pause(started.elapsed()); + Ok(()) + } } } diff --git a/app/crowdb-access-server/src/iceberg/gc_control.rs b/app/crowdb-access-server/src/iceberg/gc_control.rs index 3a784ebc5..bc22a3e82 100644 --- a/app/crowdb-access-server/src/iceberg/gc_control.rs +++ b/app/crowdb-access-server/src/iceberg/gc_control.rs @@ -75,7 +75,7 @@ pub(super) async fn manage( }; show(&task); } - _ => return Err("usage: crowdb-iceberg gc limits | start-table UUID TABLE_ID | start-retired UUID CATALOG_ID EPOCH | inspect|pause|resume|retry CATALOG_ID TASK_ID".into()), + _ => return Err("usage: crowdb-access-server iceberg gc limits | start-table UUID TABLE_ID | start-retired UUID CATALOG_ID EPOCH | inspect|pause|resume|retry CATALOG_ID TASK_ID".into()), } Ok(()) } diff --git a/app/crowdb-access-server/src/iceberg/gc_runtime.rs b/app/crowdb-access-server/src/iceberg/gc_runtime.rs index e84dc73bd..249df7e34 100644 --- a/app/crowdb-access-server/src/iceberg/gc_runtime.rs +++ b/app/crowdb-access-server/src/iceberg/gc_runtime.rs @@ -4,15 +4,16 @@ use crate::config::IcebergGcConfig; use crowdb_access_iceberg::{ catalog::{CatalogContext, CatalogRepository, RootState, RoutedCatalogStore}, file::FileBlockStore, - gc::{GcLimits, GcPhase, GcRepository, GcScan, GcStore, GcSystemScan, GcTaskKind, GcWorker}, + gc::{ + BudgetedGcBlocks, BudgetedGcStore, GcIoBudget, GcLimits, GcPhase, GcRepository, GcScan, GcStore, + GcSystemScan, GcTaskKind, GcWorker, + }, key::{CatalogId, CatalogScope, IcebergKey, SystemScope}, operation::{ManagementAction, ManagementPhase}, record::StorageRecord, }; use crowdb_chunk_client::ChunkIoClient; -pub(super) mod budget; - #[derive(Clone, Default)] struct ScanPosition { task: Vec, @@ -71,7 +72,7 @@ impl GcRuntimeConfig { if limits.minimum_retention_ms < GcLimits::default().minimum_retention_ms { return Err("GC retention must be at least seven days".into()); } - let interval_ms = setting_or("CROWDB_ICEBERG_GC_INTERVAL_MS", file.interval_ms, 1000_u64)?; + let interval_ms = setting_or("CROWDB_ICEBERG_GC_INTERVAL_MS", file.interval_ms, 60_000_u64)?; if !(100..=60_000).contains(&interval_ms) { return Err("GC interval must be between 100 and 60000 milliseconds".into()); } @@ -168,13 +169,18 @@ pub(super) async fn run( if !config.enabled { return std::future::pending().await; } - let budget = Arc::new(budget::GcIoBudget::new(&config)); - let metered_store = Arc::new(budget::BudgetedGcStore::new(store.clone(), budget.clone())); - let native = Arc::new(crowdb_access_iceberg::file::NativeFileBlocks::new( + let budget = Arc::new(GcIoBudget::new( + config.kv_bytes, + config.kv_requests, + config.chunk_bytes, + config.chunk_requests, + )); + let metered_store = Arc::new(BudgetedGcStore::new(store.clone(), budget.clone())); + let blocks: Arc = Arc::new(BudgetedGcBlocks::native( chunks, metered_store.clone(), + budget.clone(), )); - let blocks: Arc = Arc::new(budget::BudgetedGcBlocks::new(native, budget.clone())); let repository = GcRepository::new(metered_store.clone()); let worker = match GcWorker::new(repository, blocks, config.limits) { Ok(worker) => worker, @@ -243,9 +249,9 @@ pub(super) async fn run( } async fn scan_and_advance( - store: Arc, + store: Arc, worker: &GcWorker, - budget: &budget::GcIoBudget, + budget: &GcIoBudget, catalog: CatalogId, after: ScanPosition, active: Option, @@ -323,7 +329,7 @@ async fn scan_and_advance( } async fn scan_retired( - store: Arc, + store: Arc, worker: &GcWorker, after: Vec, limits: GcLimits, @@ -368,7 +374,7 @@ async fn scan_retired( } async fn scan_purge( - store: Arc, + store: Arc, context: CatalogContext, after: Vec, limits: GcLimits, diff --git a/app/crowdb-access-server/src/iceberg/http.rs b/app/crowdb-access-server/src/iceberg/http.rs index 24a191b2a..c019ca03d 100644 --- a/app/crowdb-access-server/src/iceberg/http.rs +++ b/app/crowdb-access-server/src/iceberg/http.rs @@ -63,6 +63,7 @@ impl IcebergHttpService { snapshot.chunk_read = Some(read); snapshot.chunk_small_write = Some(write); } + snapshot.upload_flow = self.files.as_ref().map(|files| files.upload_metrics_snapshot()); snapshot } diff --git a/app/crowdb-access-server/src/iceberg/metrics.rs b/app/crowdb-access-server/src/iceberg/metrics.rs index f538e846e..0cf30c448 100644 --- a/app/crowdb-access-server/src/iceberg/metrics.rs +++ b/app/crowdb-access-server/src/iceberg/metrics.rs @@ -8,6 +8,7 @@ use std::{ }; use super::routes::Route; +use crowdb_access_iceberg::catalog::{CatalogStoreOperationCounts, CatalogStoreOperationMeter}; const ROUTE_COUNT: usize = 9; const OUTCOME_COUNT: usize = 7; @@ -40,6 +41,7 @@ pub struct MetricCounts { pub response_bytes: u64, pub dispatch_latency_ns: u64, pub lifetime_ns: u64, + pub catalog: CatalogStoreOperationCounts, } #[derive(Clone, Debug, Eq, PartialEq, serde::Serialize)] @@ -51,6 +53,7 @@ pub struct IcebergMetricsSnapshot { pub selected_versions: [u64; 3], pub chunk_read: Option, pub chunk_small_write: Option, + pub upload_flow: Option, pub catalog: Option, } @@ -60,6 +63,10 @@ struct Counters { response_bytes: AtomicU64, dispatch_latency_ns: AtomicU64, lifetime_ns: AtomicU64, + catalog_get: AtomicU64, + catalog_compare_exchange: AtomicU64, + catalog_scan: AtomicU64, + catalog_conditional_delete: AtomicU64, } impl Counters { @@ -70,6 +77,10 @@ impl Counters { response_bytes: AtomicU64::new(0), dispatch_latency_ns: AtomicU64::new(0), lifetime_ns: AtomicU64::new(0), + catalog_get: AtomicU64::new(0), + catalog_compare_exchange: AtomicU64::new(0), + catalog_scan: AtomicU64::new(0), + catalog_conditional_delete: AtomicU64::new(0), } } @@ -80,6 +91,12 @@ impl Counters { response_bytes: self.response_bytes.load(Ordering::Relaxed), dispatch_latency_ns: self.dispatch_latency_ns.load(Ordering::Relaxed), lifetime_ns: self.lifetime_ns.load(Ordering::Relaxed), + catalog: CatalogStoreOperationCounts { + get: self.catalog_get.load(Ordering::Relaxed), + compare_exchange: self.catalog_compare_exchange.load(Ordering::Relaxed), + scan: self.catalog_scan.load(Ordering::Relaxed), + conditional_delete: self.catalog_conditional_delete.load(Ordering::Relaxed), + }, } } } @@ -110,6 +127,7 @@ impl IcebergMetrics { selected_versions: array::from_fn(|index| self.selected_versions[index].load(Ordering::Relaxed)), chunk_read: None, chunk_small_write: None, + upload_flow: None, catalog: None, } } @@ -125,6 +143,7 @@ pub(super) struct RequestObservation { status: AtomicU16, retry: AtomicU8, version: AtomicU8, + catalog: CatalogStoreOperationMeter, } impl RequestObservation { @@ -139,6 +158,7 @@ impl RequestObservation { status: AtomicU16::new(0), retry: AtomicU8::new(0), version: AtomicU8::new(0), + catalog: CatalogStoreOperationMeter::default(), }) } @@ -151,6 +171,10 @@ impl RequestObservation { pub(super) fn response_bytes(&self, length: usize) { self.response_bytes.fetch_add(length as u64, Ordering::Relaxed); } + + pub(super) fn catalog_meter(&self) -> &CatalogStoreOperationMeter { + &self.catalog + } } impl Drop for RequestObservation { @@ -179,6 +203,15 @@ impl Drop for RequestObservation { counters .lifetime_ns .fetch_add(elapsed_ns(self.started), Ordering::Relaxed); + let catalog = self.catalog.snapshot(); + counters.catalog_get.fetch_add(catalog.get, Ordering::Relaxed); + counters + .catalog_compare_exchange + .fetch_add(catalog.compare_exchange, Ordering::Relaxed); + counters.catalog_scan.fetch_add(catalog.scan, Ordering::Relaxed); + counters + .catalog_conditional_delete + .fetch_add(catalog.conditional_delete, Ordering::Relaxed); let retry = self.retry.load(Ordering::Relaxed); if retry != 0 { self.metrics.retry[usize::from(retry - 1)].fetch_add(1, Ordering::Relaxed); @@ -199,7 +232,8 @@ tokio::task_local! { } pub(super) async fn observe(span: Arc, future: F) -> F::Output { - REQUEST_OBSERVATION.scope(span, future).await + let meter = span.catalog_meter().clone(); + meter.observe(REQUEST_OBSERVATION.scope(span, future)).await } pub(super) fn record_request_bytes(length: usize) { diff --git a/app/crowdb-access-server/src/iceberg/runtime.rs b/app/crowdb-access-server/src/iceberg/runtime.rs index 4c143c2fa..df23488e8 100644 --- a/app/crowdb-access-server/src/iceberg/runtime.rs +++ b/app/crowdb-access-server/src/iceberg/runtime.rs @@ -2,22 +2,16 @@ use std::sync::Arc; use std::time::{Duration, SystemTime, UNIX_EPOCH}; use crowdb_access_iceberg::catalog::{ - Capabilities, CatalogError, CatalogLifecycle, CatalogRepository, ClearBounds, ManagementPrivilege, - RootState, RoutedCatalogStore, + Capabilities, CatalogError, CatalogLifecycle, CatalogRepository, ManagementPrivilege, RootState, + RoutedCatalogStore, }; use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; +use crowdb_access_iceberg::storage::{connect, foreground_blocks, IcebergLargeWriteSettings}; use crowdb_access_iceberg::wire::BearerAuthenticator; use crowdb_access_s3::native_buffer::NativeBodyAllocator; -use crowdb_chunk_client::{ - ChunkClientConfig, ChunkIoClient, ChunkIoClientConfig, ChunkReadPolicy, LargeWritePolicy, - SmallWritePolicy, -}; -use crowdb_chunk_kv_client::{ - ChunkKvClient, ChunkKvRpcTransport, ClientConfig, Group0ChunkKvRangeCatalogSource, -}; -use crowdb_common::ec::EcScheme; -use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; +use crowdb_chunk_client::ChunkIoClient; use tokio::net::TcpListener; +use tokio::sync::watch; use super::{serve, IcebergHttpService}; use crate::config::{load_args, AccessConfig}; @@ -82,8 +76,19 @@ impl IcebergRuntimeConfig { /// # Errors /// Returns configuration, authentication, storage, management or listener failures. -pub async fn run() -> Result<(), BoxError> { - let (access_config, arguments) = load_args(std::env::args().skip(1).collect())?; +pub async fn run(arguments: Vec) -> Result<(), BoxError> { + run_with_shutdown(arguments, None).await +} + +/// Runs Iceberg with an optional coordinated process shutdown signal. +/// +/// # Errors +/// Returns configuration, authentication, storage, management or listener failures. +pub async fn run_with_shutdown( + arguments: Vec, + shutdown: Option>, +) -> Result<(), BoxError> { + let (access_config, arguments) = load_args(arguments)?; let config = IcebergRuntimeConfig::from_config(&access_config)?; if arguments.len() > 7 { return Err("too many Iceberg command arguments".into()); @@ -91,20 +96,19 @@ pub async fn run() -> Result<(), BoxError> { let (repository, store, chunks) = connect( config.management_seeds.clone(), access_config.read.policy(), - access_config.small_write.policy(), + access_config.iceberg_small_write().policy(), access_config.common.diskio_connections_per_endpoint, access_config.common.diskio_rpc_workers, ) .await?; let result = if arguments.is_empty() || arguments == ["serve"] { Box::pin(start_listener( - &config.listen, + config, repository, store, - config.authentication, chunks.clone(), - config.management_seeds, access_config, + shutdown, )) .await } else if arguments.first().is_some_and(|argument| argument == "gc") { @@ -125,55 +129,19 @@ pub async fn run() -> Result<(), BoxError> { Ok(()) } -async fn connect( - seeds: Vec, - read_policy: ChunkReadPolicy, - small_write: SmallWritePolicy, - diskio_connections_per_endpoint: usize, - diskio_rpc_workers: u32, -) -> Result<(Arc, Arc, ChunkIoClient), BoxError> { - let control = Arc::new(CrowdbKvClient::new(KvConfig::new(seeds.clone()))); - let client_config = ClientConfig::default(); - let source = Arc::new(Group0ChunkKvRangeCatalogSource::from_shared(Arc::clone(&control))); - let transport = Arc::new(ChunkKvRpcTransport::new( - client_config.max_owner_connections, - 1, - 2, - )); - let client = Arc::new(ChunkKvClient::new(client_config, source, transport)?); - client.refresh_catalog().await?; - let chunks = ChunkIoClient::connect_with_kv_read_policy( - ChunkIoClientConfig { - management_seeds: seeds, - diskio_connections_per_endpoint, - diskio_rpc_workers, - small_write, - }, - control, - read_policy, - ) - .await?; - let store = Arc::new(RoutedCatalogStore::new(client)); - let repository = Arc::new(CatalogRepository::new( - store.clone(), - ClearBounds { - request_ms: 300_000, - delegated_access_ms: 900_000, - ..ClearBounds::default() - }, - )?); - Ok((repository, store, chunks)) -} - async fn start_listener( - address: &str, + runtime: IcebergRuntimeConfig, repository: Arc, store: Arc, - authentication: BearerAuthenticator, chunks: ChunkIoClient, - management_seeds: Vec, access_config: AccessConfig, + shutdown: Option>, ) -> Result<(), BoxError> { + let IcebergRuntimeConfig { + listen: address, + management_seeds, + authentication, + } = runtime; let gc_config = super::gc_runtime::GcRuntimeConfig::from_config(&access_config.iceberg.gc)?; for _ in 0..600 { match repository.recover(now_ms()?).await { @@ -190,24 +158,13 @@ async fn start_listener( if timeout.is_zero() || timeout > Duration::from_secs(300) { return Err("catalog request timeout is outside server bounds".into()); } - let blocks: Arc = Arc::new( - crowdb_access_iceberg::file::NativeFileBlocks::new(chunks.clone(), store.clone()), - ); + let blocks = foreground_blocks(chunks.clone(), store.clone()); let native_budget = access_config .iceberg .native_budget_bytes .unwrap_or(256 * 1024 * 1024); let native_allocator = Arc::new(NativeBodyAllocator::new(native_budget, 1024 * 1024)?); - let large_write = LargeWritePolicy { - ec_scheme: EcScheme::new( - access_config.small_write.ec_data, - access_config.small_write.ec_code, - ), - client: Arc::new(ChunkClientConfig { - read_buffer_size: access_config.small_write.disk_block_bytes, - ..ChunkClientConfig::default() - }), - }; + let large_write = iceberg_large_write(&access_config)?; let mut service = IcebergHttpService::new(repository.clone(), authentication, timeout) .with_namespaces(store.clone())? .with_fileio_native( @@ -216,7 +173,7 @@ async fn start_listener( "us-east-1".into(), Some(native_allocator), )? - .with_small_object_threshold(access_config.small_write.threshold_exclusive())? + .with_small_object_threshold(access_config.iceberg_small_write().threshold_exclusive())? .with_large_write_policy(large_write)?; if authority.admission_bounds.delegated_access_ms >= 900_000 { let endpoint = @@ -228,46 +185,103 @@ async fn start_listener( tracing::warn!("table routes disabled: persisted catalog delegation bound is below fifteen minutes"); } let service = Arc::new(service); - let listener = TcpListener::bind(address).await?; + let listener = TcpListener::bind(&address).await?; tracing::info!(%address, "Iceberg listener ready"); - let serving = serve(listener, service, async { - let _ = tokio::signal::ctrl_c().await; - }); + let serving = serve(listener, service, wait_for_shutdown(shutdown)); let multipart = Box::pin(super::file_recovery::run( repository.clone(), store.clone(), blocks.clone(), )); let tables = super::table_recovery::run(repository.clone(), store.clone(), blocks.clone()); - let (gc_store, gc_chunks) = if gc_config.enabled { - let (_, gc_store, gc_chunks) = connect( - management_seeds, - access_config.read.policy(), - access_config.small_write.policy(), - access_config.common.diskio_connections_per_endpoint, - access_config.common.diskio_rpc_workers, - ) - .await?; - (gc_store, Some(gc_chunks)) - } else { - (store.clone(), None) - }; + let (gc_store, gc_chunks) = + connect_gc_pool(&access_config, management_seeds, store.clone(), gc_config.enabled).await?; let gc_client = gc_chunks.clone().unwrap_or_else(|| chunks.clone()); + let gc_failure_client = gc_client.clone(); let gc = super::gc_runtime::run(repository.clone(), gc_store, gc_client, gc_config); - tokio::select! { - result = serving => result?, - () = super::recovery::run(repository, crowdb_access_iceberg::namespace::NamespaceRecovery::new(store)) => {} - () = multipart => {} - () = tables => {} - () = gc => {} - } - if let Some(gc_chunks) = gc_chunks { - gc_chunks.shutdown_small_writes().await?; - } + let result: Result<(), BoxError> = tokio::select! { + result = serving => result.map_err(Into::into), + () = chunks.wait_for_small_write_manager_failure() => { + Err("Iceberg small-write manager stopped unexpectedly".into()) + } + () = gc_failure_client.wait_for_small_write_manager_failure(), if gc_chunks.is_some() => { + Err("Iceberg GC small-write manager stopped unexpectedly".into()) + } + () = super::recovery::run(repository, crowdb_access_iceberg::namespace::NamespaceRecovery::new(store)) => { + Err("Iceberg namespace recovery stopped unexpectedly".into()) + } + () = multipart => Err("Iceberg multipart recovery stopped unexpectedly".into()), + () = tables => Err("Iceberg table recovery stopped unexpectedly".into()), + () = gc => Err("Iceberg GC stopped unexpectedly".into()), + }; + let gc_shutdown = if let Some(gc_chunks) = gc_chunks { + gc_chunks.shutdown_small_writes().await + } else { + Ok(()) + }; + result?; + gc_shutdown?; tracing::info!("Iceberg listener drained"); Ok(()) } +async fn connect_gc_pool( + access_config: &AccessConfig, + management_seeds: Vec, + store: Arc, + enabled: bool, +) -> Result<(Arc, Option), BoxError> { + if !enabled { + return Ok((store, None)); + } + let (_, gc_store, gc_chunks) = connect( + management_seeds, + access_config.read.policy(), + access_config.iceberg_small_write().policy(), + access_config.common.diskio_connections_per_endpoint, + access_config.common.diskio_rpc_workers, + ) + .await?; + Ok((gc_store, Some(gc_chunks))) +} + +fn iceberg_large_write( + access_config: &AccessConfig, +) -> Result { + let small = access_config.iceberg_small_write(); + Ok(IcebergLargeWriteSettings { + ec_data: access_config.iceberg.ec_data.unwrap_or(small.ec_data), + ec_code: access_config.iceberg.ec_code.unwrap_or(small.ec_code), + disk_block_bytes: small.disk_block_bytes, + mirror_copies: access_config.iceberg.large_mirror_copies, + max_chunk_size: access_config.iceberg.max_chunk_size, + memory_budget_bytes: access_config.iceberg.large_memory_budget_bytes, + prefetch_strips_per_chunk: access_config.iceberg.large_prefetch_strips_per_chunk, + prefetch_max_strips_per_batch: access_config.iceberg.large_prefetch_max_strips_per_batch, + parallel_strip_writes: access_config.iceberg.large_parallel_strip_writes, + held_buffers: access_config.iceberg.large_held_buffers, + chunk_preparation_depth: access_config.iceberg.large_chunk_preparation_depth, + } + .policy()?) +} + +async fn wait_for_shutdown(mut shutdown: Option>) { + if let Some(receiver) = shutdown.as_mut() { + tokio::select! { + _ = tokio::signal::ctrl_c() => {} + () = async { + loop { + if *receiver.borrow() || receiver.changed().await.is_err() { + break; + } + } + } => {} + } + } else { + let _ = tokio::signal::ctrl_c().await; + } +} + async fn manage( repository: &CatalogRepository, authentication: &BearerAuthenticator, @@ -310,7 +324,7 @@ async fn manage( Some("rename") if arguments.len() == 4 => ManagementAction::Rename, Some("clear") if arguments.len() == 5 => ManagementAction::Clear, Some("activate") if arguments.len() == 5 => ManagementAction::Activate, - _ => return Err("usage: crowdb-iceberg initialize UUIDv7 NAME | rename UUIDv7 NAME EPOCH | clear UUIDv7 NAME EPOCH CONFIRM_CATALOG_ID | activate UUIDv7 NAME EPOCH CAPABILITY_BITS_HEX | status | inspect | serve".into()), + _ => return Err("usage: crowdb-access-server iceberg initialize UUIDv7 NAME | rename UUIDv7 NAME EPOCH | clear UUIDv7 NAME EPOCH CONFIRM_CATALOG_ID | activate UUIDv7 NAME EPOCH CAPABILITY_BITS_HEX | status | inspect | serve".into()), }; let request = ManagementRequest { identity: RequestIdentity::parse(&arguments[1], now_ms()?)?, diff --git a/app/crowdb-access-server/src/iceberg/table_limits.rs b/app/crowdb-access-server/src/iceberg/table_limits.rs index b16a12722..32dfccdf3 100644 --- a/app/crowdb-access-server/src/iceberg/table_limits.rs +++ b/app/crowdb-access-server/src/iceberg/table_limits.rs @@ -47,7 +47,7 @@ pub(super) fn commits() -> CommitProofLimits { }; let parquet = ParquetMetadataLimits { footer_bytes: 1024 * 1024, - values: 100_000, + values: 500_000, depth: 32, schema_elements: 4096, row_groups: 10_000, diff --git a/app/crowdb-access-server/src/iceberg/table_write.rs b/app/crowdb-access-server/src/iceberg/table_write.rs index 3311a11dc..13f1dbbc2 100644 --- a/app/crowdb-access-server/src/iceberg/table_write.rs +++ b/app/crowdb-access-server/src/iceberg/table_write.rs @@ -154,7 +154,7 @@ impl TableWrites { } let target = request::parse(&uri); let result = match target { - Ok(target) => self.mutate(&record, capabilities, target, bytes, now).await, + Ok(target) => Box::pin(self.mutate(&record, capabilities, target, bytes, now)).await, Err(error) => Err(error), }; let (status, body) = self.outcome_response(result).await?; diff --git a/app/crowdb-access-server/src/iceberg_main.rs b/app/crowdb-access-server/src/iceberg_main.rs deleted file mode 100644 index 484dfeaef..000000000 --- a/app/crowdb-access-server/src/iceberg_main.rs +++ /dev/null @@ -1,5 +0,0 @@ -#[tokio::main] -async fn main() -> Result<(), Box> { - tracing_subscriber::fmt::init(); - crowdb_access_server::iceberg::run().await -} diff --git a/app/crowdb-access-server/src/lib.rs b/app/crowdb-access-server/src/lib.rs index 3029821ef..95a9a6243 100644 --- a/app/crowdb-access-server/src/lib.rs +++ b/app/crowdb-access-server/src/lib.rs @@ -7,6 +7,7 @@ pub mod config; mod http_receive; pub mod iceberg; mod multipart_complete; +mod upload_flow; #[cfg(feature = "s3")] pub mod credentials; diff --git a/app/crowdb-access-server/src/main.rs b/app/crowdb-access-server/src/main.rs index 7c092373e..c2947bc3a 100644 --- a/app/crowdb-access-server/src/main.rs +++ b/app/crowdb-access-server/src/main.rs @@ -18,6 +18,8 @@ use crowdb_access_s3::metrics::{DependencyHealth, S3Health, S3Metrics}; #[cfg(feature = "s3")] use crowdb_access_s3::native_buffer::NativeBodyAllocator; #[cfg(feature = "s3")] +use crowdb_access_s3::storage::{S3LargeWriteSettings, S3WriteSettings}; +#[cfg(feature = "s3")] use crowdb_access_server::config::{load_args, AccessConfig}; #[cfg(feature = "s3")] use crowdb_access_server::credentials::CredentialAuthority; @@ -26,20 +28,33 @@ use crowdb_access_server::s3::{serve, ProductionS3Operations, S3Dispatcher, S3Se #[cfg(feature = "s3")] use crowdb_access_server::storage::S3StorageClients; #[cfg(feature = "s3")] -use crowdb_chunk_client::SmallWritePolicy; -#[cfg(feature = "s3")] -#[cfg(feature = "s3")] -use crowdb_common::ec::EcScheme; +use crowdb_chunk_client::LargeWritePolicy; #[cfg(feature = "s3")] use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; #[cfg(feature = "s3")] use tokio::net::TcpListener; +#[cfg(feature = "s3")] +use tokio::sync::watch; #[tokio::main] async fn main() -> Result<(), Box> { - tracing_subscriber::fmt().with_writer(std::io::stderr).init(); + let mut args: Vec = std::env::args().skip(1).collect(); + if args.first().is_some_and(|arg| arg == "iceberg") { + args.remove(0); + init_access_logging()?; + return crowdb_access_server::iceberg::run(args) + .await + .map_err(|error| -> Box { error }); + } + let s3_only = args.first().is_some_and(|arg| arg == "s3"); + if s3_only { + args.remove(0); + } + init_access_logging()?; #[cfg(feature = "s3")] - let (access_config, remaining_args) = load_args(std::env::args().skip(1).collect())?; + let (access_config, remaining_args) = load_args(args.clone())?; + #[cfg(not(feature = "s3"))] + let _ = args; #[cfg(feature = "s3")] if matches!( remaining_args.first().map(String::as_str), @@ -52,12 +67,69 @@ async fn main() -> Result<(), Box> { return Err("unexpected S3 server arguments".into()); } #[cfg(feature = "s3")] - run_s3(&access_config).await?; + if !s3_only && access_config.s3.listen.is_none() && std::env::var_os("CROWDB_S3_LISTEN").is_none() { + return Err("S3 listen address is required when starting both access listeners".into()); + } + #[cfg(feature = "s3")] + if s3_only { + run_s3(&access_config, None).await?; + } else { + let (shutdown_tx, shutdown_rx) = watch::channel(false); + let s3 = run_s3(&access_config, Some(shutdown_rx.clone())); + let iceberg = crowdb_access_server::iceberg::run_with_shutdown(args, Some(shutdown_rx)); + tokio::pin!(s3, iceberg); + tokio::select! { + result = &mut s3 => { + let _ = shutdown_tx.send(true); + let other = iceberg.await; + result?; + other.map_err(|error| -> Box { error })?; + } + result = &mut iceberg => { + let _ = shutdown_tx.send(true); + let other = s3.await; + result.map_err(|error| -> Box { error })?; + other?; + } + } + } + #[cfg(not(feature = "s3"))] + crowdb_access_server::iceberg::run(args) + .await + .map_err(|error| -> Box { error })?; + Ok(()) +} + +fn init_access_logging() -> Result<(), std::io::Error> { + tracing_subscriber::fmt() + .with_writer(std::io::stderr) + .with_env_filter( + tracing_subscriber::EnvFilter::try_from_default_env() + .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("warn")), + ) + .init(); + let log_dir = std::env::var("CROWDB_ACCESS_LOG_DIR").unwrap_or_default(); + if !log_dir.is_empty() { + std::fs::create_dir_all(&log_dir)?; + } + crowdb_rpc_ffi::init_logging( + &log_dir, + if log_dir.is_empty() { "warn" } else { "info" }, + 30, + 5, + "crowdb-access-rpc", + ); + if !log_dir.is_empty() { + crowdb_rpc_ffi::add_log_stderr("warn"); + } Ok(()) } #[cfg(feature = "s3")] -async fn run_s3(access_config: &AccessConfig) -> Result<(), Box> { +async fn run_s3( + access_config: &AccessConfig, + shutdown: Option>, +) -> Result<(), Box> { if let Some(address) = access_config .s3 .listen @@ -77,12 +149,12 @@ async fn run_s3(access_config: &AccessConfig) -> Result<(), Box Result<(), Box Result<(), Box> = tokio::select! { + result = serve(listener, handler, wait_for_shutdown(shutdown)) => result.map_err(Into::into), + () = chunks.wait_for_small_write_manager_failure() => { + Err("S3 small-write manager stopped unexpectedly".into()) + } + }; health.stop(); expiry_task.abort(); let shutdown_result = chunks.shutdown_small_writes().await; @@ -153,6 +229,36 @@ async fn run_s3(access_config: &AccessConfig) -> Result<(), Box) { + if let Some(path) = std::env::var_os("CROWDB_TEST_STOP_S3_MANAGER_FILE") { + tokio::spawn(async move { + while !std::path::Path::new(&path).exists() { + tokio::time::sleep(Duration::from_millis(50)).await; + } + let _ = chunks.stop_small_write_manager_for_test().await; + }); + } +} + +#[cfg(feature = "s3")] +async fn wait_for_shutdown(mut shutdown: Option>) { + if let Some(receiver) = shutdown.as_mut() { + tokio::select! { + _ = tokio::signal::ctrl_c() => {} + () = async { + loop { + if *receiver.borrow() || receiver.changed().await.is_err() { + break; + } + } + } => {} + } + } else { + let _ = tokio::signal::ctrl_c().await; + } +} + #[cfg(feature = "s3")] fn start_multipart_expiry(operations: Arc) -> tokio::task::JoinHandle<()> { tokio::spawn(async move { @@ -222,10 +328,10 @@ fn s3_service_config( tenant: TenantId, continuation_key: Vec, small_write_limit: usize, - ec_scheme: EcScheme, + large_write: LargeWritePolicy, ) -> Result> { let mut config = S3ServiceConfig::basic(tenant, continuation_key, small_write_limit); - config.large_write.ec_scheme = ec_scheme; + config.large_write = large_write; if let Some(limit) = access.s3.list_scan_items { config.list_scan_items = limit; } @@ -239,52 +345,32 @@ fn s3_service_config( configured_usize(access.s3.small_object_limit, "CROWDB_S3_SMALL_OBJECT_LIMIT")? .unwrap_or(config.small_object_limit) .min(small_write_limit.saturating_sub(1)); - configure_large_write(&mut config, access)?; Ok(config) } #[cfg(feature = "s3")] -fn configure_large_write( - config: &mut S3ServiceConfig, - access: &AccessConfig, -) -> Result<(), Box> { - Arc::make_mut(&mut config.large_write.client).read_buffer_size = access.small_write.disk_block_bytes; - if let Some(max_chunk_size) = configured_u64(access.s3.max_chunk_size, "CROWDB_S3_MAX_CHUNK_SIZE")? { - if max_chunk_size == 0 { - return Err("CROWDB S3 max chunk size must be nonzero".into()); - } - Arc::make_mut(&mut config.large_write.client).max_chunk_size = max_chunk_size; - } - Ok(()) -} - -#[cfg(feature = "s3")] -fn s3_ec_scheme(access: &AccessConfig) -> Result> { - let ec_data = - configured_usize(access.s3.ec_data, "CROWDB_S3_EC_DATA")?.unwrap_or(access.small_write.ec_data); - let ec_code = - configured_usize(access.s3.ec_code, "CROWDB_S3_EC_CODE")?.unwrap_or(access.small_write.ec_code); - if ec_data == 0 || ec_data > 32 || ec_code == 0 { - return Err("CROWDB S3 EC data and code counts are invalid".into()); - } - Ok(EcScheme::new(ec_data, ec_code)) -} - -#[cfg(feature = "s3")] -fn s3_write_routing( +fn s3_write_policies( access: &AccessConfig, -) -> Result<(EcScheme, SmallWritePolicy, usize), Box> { - let ec_scheme = s3_ec_scheme(access)?; - let mut config = access.small_write.clone(); - config.ec_data = ec_scheme.data_num; - config.ec_code = ec_scheme.code_num; - let policy = config.policy(); - policy.validate()?; - let threshold = config.threshold_exclusive(); - if threshold > policy.object_limit { - return Err("S3 small-object threshold exceeds the shared writer limit".into()); +) -> Result> { + let small = access.s3_small_write(); + Ok(S3WriteSettings { + small: small.policy(), + threshold_ratio: small.threshold_ratio, + disk_block_bytes: small.disk_block_bytes, + ec_data: configured_usize(access.s3.ec_data, "CROWDB_S3_EC_DATA")?.unwrap_or(small.ec_data), + ec_code: configured_usize(access.s3.ec_code, "CROWDB_S3_EC_CODE")?.unwrap_or(small.ec_code), + large: S3LargeWriteSettings { + mirror_copies: access.s3.large_mirror_copies, + max_chunk_size: configured_u64(access.s3.max_chunk_size, "CROWDB_S3_MAX_CHUNK_SIZE")?, + memory_budget_bytes: access.s3.large_memory_budget_bytes, + prefetch_strips_per_chunk: access.s3.large_prefetch_strips_per_chunk, + prefetch_max_strips_per_batch: access.s3.large_prefetch_max_strips_per_batch, + parallel_strip_writes: access.s3.large_parallel_strip_writes, + held_buffers: access.s3.large_held_buffers, + chunk_preparation_depth: access.s3.large_chunk_preparation_depth, + }, } - Ok((ec_scheme, policy, threshold)) + .policies()?) } #[cfg(feature = "s3")] diff --git a/app/crowdb-access-server/src/s3/operations.rs b/app/crowdb-access-server/src/s3/operations.rs index 7a0a4752b..3c2d9f7a1 100644 --- a/app/crowdb-access-server/src/s3/operations.rs +++ b/app/crowdb-access-server/src/s3/operations.rs @@ -16,10 +16,7 @@ use crowdb_access_s3::object::{self, ListObjectsV2Request, ObjectMetadataError}; use crowdb_access_s3::publication::PublicationRequest; use crowdb_access_s3::retrieval::{self, ObjectHeaders, RetrievalError}; use crowdb_access_s3::route::{S3Operation, S3Route}; -use crowdb_access_s3::streaming::{ - publish_completed_locations, write_body_with_checksums_buffered, - write_native_body_with_checksums_metered, PutErrorCode, PutOutcome, -}; +use crowdb_access_s3::streaming::{publish_completed_locations, PutErrorCode, PutOutcome}; use crowdb_chunk_client::{ ChunkClientConfig, ChunkIoWriter, IoError, LargeWritePolicy, PreparedLargeWrite, SharedObjectWriter, }; @@ -40,6 +37,9 @@ use super::{error_response, full_body, install_body_receive_provider, BoxError, use crowdb_access_s3::wire; mod multipart; +mod upload; + +use upload::write_object_body; const DEFAULT_LIST_LIMIT: usize = 1_000; const DEFAULT_LIST_SCAN_BYTES: usize = 4 * 1024 * 1024; @@ -265,31 +265,17 @@ impl ProductionS3Operations { receiver.enable_owner_handoff(); } let mut body = request.into_body(); - let write_result = if let Some(receiver) = native_receiver.as_deref() { - write_native_body_with_checksums_metered( - &mut body, - &mut writer, - receiver, - content_md5.as_deref(), - payload_sha256.as_deref(), - self.metrics.as_deref(), - ) - .await - } else { - let receive_bytes = content_length - .and_then(|length| usize::try_from(length).ok()) - .unwrap_or(1024 * 1024) - .clamp(1, 1024 * 1024); - write_body_with_checksums_buffered( - &mut body, - &mut writer, - content_md5.as_deref(), - payload_sha256.as_deref(), - receive_bytes, - self.metrics.as_deref(), - ) - .await - }; + let write_result = write_object_body( + &mut body, + &mut writer, + native_receiver.as_deref(), + content_length, + content_md5.as_deref(), + payload_sha256.as_deref(), + self.config.large_write.client.large_held_buffers, + self.metrics.as_deref(), + ) + .await; let (etag, checksum) = match write_result { Ok(result) => result, Err(outcome) => { @@ -297,10 +283,10 @@ impl ProductionS3Operations { return Err(map_put_outcome(&outcome)); } }; - let locations = writer - .on_finish() - .await - .map_err(|_| S3ErrorCode::ServiceUnavailable)?; + let locations = writer.on_finish().await.map_err(|error| { + tracing::warn!(%error, "S3 PUT chunk finalization failed"); + S3ErrorCode::ServiceUnavailable + })?; let logical_length = locations.iter().map(|location| location.logical_length).sum(); if content_length.is_some_and(|expected| expected != logical_length) { return Err(S3ErrorCode::InvalidRequest); @@ -494,10 +480,10 @@ impl ProductionS3Operations { .storage .chunks .prepare_large_write(content_length, self.config.large_write.clone()); - prepared - .wait_until_prepared() - .await - .map_err(|_| S3ErrorCode::ServiceUnavailable)?; + prepared.wait_until_prepared().await.map_err(|error| { + tracing::warn!(%error, "S3 large write preparation failed"); + S3ErrorCode::ServiceUnavailable + })?; Ok(ObjectWriter::Large(Box::new(prepared))) } } @@ -661,6 +647,19 @@ fn map_object_error(error: &ObjectMetadataError) -> S3ErrorCode { } fn map_put_outcome(outcome: &PutOutcome) -> S3ErrorCode { + if matches!( + outcome, + PutOutcome::Timeout + | PutOutcome::Error { + code: PutErrorCode::ChunkWrite + | PutErrorCode::LocationEncoding + | PutErrorCode::MetadataEncoding + | PutErrorCode::KvRejected, + .. + } + ) { + tracing::warn!(?outcome, "S3 PUT failed"); + } match outcome { PutOutcome::Success => S3ErrorCode::InternalError, PutOutcome::Timeout => S3ErrorCode::ServiceUnavailable, diff --git a/app/crowdb-access-server/src/s3/operations/multipart.rs b/app/crowdb-access-server/src/s3/operations/multipart.rs index c96aa520a..ec651d0e8 100644 --- a/app/crowdb-access-server/src/s3/operations/multipart.rs +++ b/app/crowdb-access-server/src/s3/operations/multipart.rs @@ -11,9 +11,6 @@ use crowdb_access_s3::metadata::{ MultipartRepositoryError, MultipartSessionRecord, }; use crowdb_access_s3::route::{S3Operation, S3Route}; -use crowdb_access_s3::streaming::{ - write_body_with_checksums_buffered, write_native_body_with_checksums_metered, -}; use crowdb_access_s3::wire; use crowdb_chunk_client::ChunkIoWriter; use http_body_util::BodyExt as _; @@ -24,7 +21,8 @@ use hyper::{Request, Response, StatusCode}; use super::{ content_length, full_body, install_body_receive_provider, map_put_outcome, required_bucket, required_key, - response, strict_header, unix_millis, xml_response, ProductionS3Operations, Query, ResponseBody, + response, strict_header, unix_millis, write_object_body, xml_response, ProductionS3Operations, Query, + ResponseBody, }; use crate::multipart_complete::{CompleteRequestError, CompleteSelection}; @@ -175,30 +173,17 @@ impl ProductionS3Operations { receiver.enable_owner_handoff(); } let mut body = request.into_body(); - let written = if let Some(receiver) = native_receiver.as_deref() { - write_native_body_with_checksums_metered( - &mut body, - &mut writer, - receiver, - content_md5.as_deref(), - payload_sha256.as_deref(), - self.metrics.as_deref(), - ) - .await - } else { - let receive_bytes = usize::try_from(length) - .unwrap_or(1024 * 1024) - .clamp(1, 1024 * 1024); - write_body_with_checksums_buffered( - &mut body, - &mut writer, - content_md5.as_deref(), - payload_sha256.as_deref(), - receive_bytes, - self.metrics.as_deref(), - ) - .await - }; + let written = write_object_body( + &mut body, + &mut writer, + native_receiver.as_deref(), + Some(length), + content_md5.as_deref(), + payload_sha256.as_deref(), + self.config.large_write.client.large_held_buffers, + self.metrics.as_deref(), + ) + .await; let (etag, _) = match written { Ok(value) => value, Err(outcome) => { diff --git a/app/crowdb-access-server/src/s3/operations/upload.rs b/app/crowdb-access-server/src/s3/operations/upload.rs new file mode 100644 index 000000000..4e7c3a1fd --- /dev/null +++ b/app/crowdb-access-server/src/s3/operations/upload.rs @@ -0,0 +1,191 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! One S3 PUT or `UploadPart` body through the shared object write flow. + +use std::sync::atomic::AtomicU64; +use std::time::{SystemTime, UNIX_EPOCH}; + +use crowdb_access_s3::integrity::{validate_completed_digests, IntegrityError}; +use crowdb_access_s3::metrics::S3Metrics; +use crowdb_access_s3::native_buffer::NativeBodyReceiver; +use crowdb_access_s3::streaming::{PutErrorCode, PutOutcome}; +use crowdb_chunk_client::FramedWriteBuffer; +use crowdb_protocol::frame::FrameMagic; +use http_body_util::BodyExt; +use hyper::body::{Bytes, Incoming}; +use tokio::sync::mpsc; + +use crate::upload_flow::digest_pipe::DigestPipe; +use crate::upload_flow::{drive_transfer, write_buffers, OfferStatus, UploadBuffer, WriteFlow}; + +use super::ObjectWriter; + +const TARGET_BUFFER_BYTES: usize = 1024 * 1024; + +fn failed(code: PutErrorCode, message: &impl ToString) -> PutOutcome { + PutOutcome::Error { + code, + message: message.to_string(), + } +} + +#[allow(clippy::too_many_arguments)] +pub(super) async fn write_object_body( + body: &mut Incoming, + writer: &mut ObjectWriter, + native_receiver: Option<&NativeBodyReceiver>, + declared_length: Option, + expected_content_md5: Option<&str>, + expected_payload_sha256: Option<&str>, + held_buffers: usize, + metrics: Option<&S3Metrics>, +) -> Result<(String, Vec), PutOutcome> { + let (sender, receiver) = mpsc::channel(held_buffers); + let progress = AtomicU64::new(0); + let queued_peak = AtomicU64::new(0); + let flow = WriteFlow::new(sender, &progress, &queued_peak); + let mut digest = DigestPipe::start(expected_payload_sha256.is_some()); + let receive = receive_body(body, native_receiver, declared_length, flow, &digest, metrics); + let write = write_buffers(writer, receiver, &progress, |error| { + failed(PutErrorCode::ChunkWrite, &error) + }); + let (length, _) = drive_transfer(receive, write, &progress).await?; + if declared_length.is_some_and(|expected| expected != length) { + return Err(failed( + PutErrorCode::BodyRead, + &"body length differs from Content-Length", + )); + } + let digests = digest + .finish() + .await + .map_err(|()| failed(PutErrorCode::ChunkWrite, &"digest worker failed"))?; + validate_completed_digests( + digests.md5, + digests.sha256, + expected_content_md5, + expected_payload_sha256, + ) + .map_err(|error| match error { + IntegrityError::InvalidDigest => failed(PutErrorCode::InvalidDigest, &error), + IntegrityError::Mismatch => failed(PutErrorCode::BadDigest, &error), + IntegrityError::InvalidPayloadDigest => failed(PutErrorCode::InvalidPayloadDigest, &error), + IntegrityError::PayloadMismatch => failed(PutErrorCode::PayloadMismatch, &error), + }) +} + +async fn receive_body( + body: &mut Incoming, + native_receiver: Option<&NativeBodyReceiver>, + declared_length: Option, + flow: WriteFlow<'_>, + digest: &DigestPipe, + metrics: Option<&S3Metrics>, +) -> Result { + let target = usize::try_from(declared_length.unwrap_or(TARGET_BUFFER_BYTES as u64)) + .unwrap_or(TARGET_BUFFER_BYTES) + .clamp(1, TARGET_BUFFER_BYTES); + let mut length = 0u64; + let mut pending = Vec::with_capacity(if native_receiver.is_some() { 0 } else { target }); + let mut digest_pending = Vec::with_capacity(16); + loop { + let Some(frame) = body.frame().await else { break }; + let mut bytes = frame + .map_err(|error| failed(PutErrorCode::BodyRead, &error))? + .into_data() + .map_err(|_| failed(PutErrorCode::BodyRead, &"unexpected non-data body frame"))?; + if bytes.is_empty() { + continue; + } + length = length + .checked_add(bytes.len() as u64) + .ok_or_else(|| failed(PutErrorCode::BodyRead, &"body length overflow"))?; + if declared_length.is_some_and(|expected| length > expected) { + return Err(failed(PutErrorCode::BodyRead, &"body exceeds Content-Length")); + } + if let Some(metrics) = metrics { + metrics.record_checksum_bytes(bytes.len()); + } + if let Some(receiver) = native_receiver { + digest_pending.push(bytes); + if let Some(owner) = receiver.take_ready_owner() { + handoff_owner(owner, &flow, digest, std::mem::take(&mut digest_pending)).await?; + } + } else { + while !bytes.is_empty() { + let count = (target - pending.len()).min(bytes.len()); + let piece = bytes.split_to(count); + pending.extend_from_slice(&piece); + digest_pending.push(piece); + if pending.len() == target { + let owner = Bytes::from(std::mem::replace(&mut pending, Vec::with_capacity(target))); + handoff( + &flow, + digest, + UploadBuffer::Data(owner), + std::mem::take(&mut digest_pending), + ) + .await?; + } + } + } + } + if let Some(receiver) = native_receiver { + if let Some(owner) = receiver + .finish_owner_when_ready() + .await + .map_err(|error| failed(PutErrorCode::BodyRead, &error))? + { + handoff_owner(owner, &flow, digest, std::mem::take(&mut digest_pending)).await?; + } + } + if !pending.is_empty() { + handoff( + &flow, + digest, + UploadBuffer::Data(Bytes::from(pending)), + std::mem::take(&mut digest_pending), + ) + .await?; + } + Ok(length) +} + +async fn handoff_owner( + mut owner: crowdb_access_s3::native_buffer::NativeFramedOwner, + flow: &WriteFlow<'_>, + digest: &DigestPipe, + payload: Vec, +) -> Result<(), PutOutcome> { + let now_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |duration| { + u64::try_from(duration.as_millis()).unwrap_or(u64::MAX) + }); + owner + .prepare_frames(FrameMagic::RepoLargeV1, now_ms) + .map_err(|error| failed(PutErrorCode::ChunkWrite, &error))?; + handoff(flow, digest, UploadBuffer::Framed(Box::new(owner)), payload).await +} + +async fn handoff( + flow: &WriteFlow<'_>, + digest: &DigestPipe, + buffer: UploadBuffer, + payload: Vec, +) -> Result<(), PutOutcome> { + let offer = flow + .offer(buffer) + .await + .map_err(|()| failed(PutErrorCode::ChunkWrite, &"write flow closed"))?; + digest + .enqueue(payload) + .map_err(|()| failed(PutErrorCode::ChunkWrite, &"digest capacity exhausted"))?; + if matches!(offer, OfferStatus::Pause) { + flow.wait_ready() + .await + .map_err(|()| failed(PutErrorCode::ChunkWrite, &"write flow closed"))?; + } + Ok(()) +} diff --git a/app/crowdb-access-server/src/storage.rs b/app/crowdb-access-server/src/storage.rs index ff9d1eb45..3a7373a45 100644 --- a/app/crowdb-access-server/src/storage.rs +++ b/app/crowdb-access-server/src/storage.rs @@ -1,89 +1,6 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Production client wiring for stateless S3 handlers. +//! Compatibility re-export for application callers. -use std::sync::Arc; - -use crowdb_access_s3::metadata::ChunkKvMetadataStore; -use crowdb_chunk_client::{ChunkIoClient, ChunkIoClientConfig, ChunkReadPolicy, SmallWritePolicy}; -use crowdb_chunk_kv_client::{ - ChunkKvClient, ChunkKvRpcTransport, ClientConfig as ChunkKvConfig, Group0ChunkKvRangeCatalogSource, -}; -use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; - -#[derive(Clone)] -pub struct S3StorageClients { - pub control: Arc, - pub metadata: Arc, - pub chunks: Arc, -} - -#[derive(Debug, thiserror::Error)] -pub enum StorageConnectError { - #[error("Chunk-KV client configuration failed: {0}")] - ChunkKv(String), - #[error("chunk I/O discovery failed: {0}")] - ChunkIo(String), -} - -impl S3StorageClients { - /// Connects metadata and chunk clients through one discovery client. - /// - /// # Errors - /// - /// Returns before readiness on configuration or discovery failure. - pub async fn connect( - management_seeds: Vec, - diskio_connections_per_endpoint: usize, - diskio_rpc_workers: u32, - small_write: SmallWritePolicy, - ) -> Result { - Self::connect_with_read_policy( - management_seeds, - diskio_connections_per_endpoint, - diskio_rpc_workers, - small_write, - ChunkReadPolicy::default(), - ) - .await - } - - /// # Errors - /// Returns an error when the management or `DiskIO` connection cannot be established. - pub async fn connect_with_read_policy( - management_seeds: Vec, - diskio_connections_per_endpoint: usize, - diskio_rpc_workers: u32, - small_write: SmallWritePolicy, - read_policy: ChunkReadPolicy, - ) -> Result { - let kv = Arc::new(CrowdbKvClient::new(KvConfig::new(management_seeds.clone()))); - let config = ChunkKvConfig::default(); - let catalog = Arc::new(Group0ChunkKvRangeCatalogSource::from_shared(Arc::clone(&kv))); - let transport = Arc::new(ChunkKvRpcTransport::new(config.max_owner_connections, 1, 2)); - let metadata = ChunkKvClient::new(config, catalog, transport) - .map_err(|error| StorageConnectError::ChunkKv(error.to_string()))?; - metadata - .refresh_catalog() - .await - .map_err(|error| StorageConnectError::ChunkKv(error.to_string()))?; - let chunks = ChunkIoClient::connect_with_kv_read_policy( - ChunkIoClientConfig { - management_seeds, - diskio_connections_per_endpoint, - diskio_rpc_workers, - small_write, - }, - Arc::clone(&kv), - read_policy, - ) - .await - .map_err(|error| StorageConnectError::ChunkIo(error.to_string()))?; - Ok(Self { - control: kv, - metadata: Arc::new(ChunkKvMetadataStore::new(Arc::new(metadata))), - chunks: Arc::new(chunks), - }) - } -} +pub use crowdb_access_s3::storage::{S3StorageClients, StorageConnectError}; diff --git a/app/crowdb-access-server/src/upload_flow.rs b/app/crowdb-access-server/src/upload_flow.rs new file mode 100644 index 000000000..8b9b3de82 --- /dev/null +++ b/app/crowdb-access-server/src/upload_flow.rs @@ -0,0 +1,151 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Bounded object-body handoff shared by S3 and Iceberg uploads. + +use std::future::{poll_fn, Future}; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::task::Poll; +use std::time::{Duration, Instant}; + +use crowdb_chunk_client::{ChunkIoWriter, FramedWriteBuffer, IoError}; +use hyper::body::Bytes; +use tokio::sync::mpsc; + +pub(crate) mod digest_pipe; + +pub(crate) enum UploadBuffer { + Framed(Box), + Data(Bytes), +} + +pub(crate) enum OfferStatus { + Continue, + Pause, +} + +pub(crate) struct WriteFlow<'a> { + sender: mpsc::Sender, + progress: &'a AtomicU64, + queued_peak: &'a AtomicU64, +} + +impl<'a> WriteFlow<'a> { + pub(crate) fn new( + sender: mpsc::Sender, + progress: &'a AtomicU64, + queued_peak: &'a AtomicU64, + ) -> Self { + Self { + sender, + progress, + queued_peak, + } + } + + pub(crate) async fn offer(&self, buffer: UploadBuffer) -> Result { + self.sender.send(buffer).await.map_err(|_| ())?; + self.progress.fetch_add(1, Ordering::Relaxed); + self.queued_peak.fetch_max( + u64::try_from(self.sender.max_capacity() - self.sender.capacity()).unwrap_or(u64::MAX), + Ordering::Relaxed, + ); + Ok(if self.sender.capacity() == 0 { + OfferStatus::Pause + } else { + OfferStatus::Continue + }) + } + + pub(crate) async fn wait_ready(&self) -> Result<(), ()> { + let permit = self.sender.reserve().await.map_err(|_| ())?; + drop(permit); + Ok(()) + } +} + +#[derive(Default)] +pub(crate) struct WriteStats { + pub(crate) feeds: u64, + pub(crate) feed_time: Duration, + pub(crate) capacity_waits: u64, + pub(crate) capacity_wait_time: Duration, +} + +pub(crate) async fn drive_transfer( + receive: R, + write: W, + progress: &AtomicU64, +) -> Result<(u64, WriteStats), E> +where + R: Future>, + W: Future>, +{ + let mut receive = Some(Box::pin(receive)); + let mut write = Some(Box::pin(write)); + let mut length = None; + let mut written = None; + poll_fn(|cx| { + for _ in 0..32 { + let before = progress.load(Ordering::Relaxed); + if let Some(Poll::Ready(result)) = receive.as_mut().map(|future| future.as_mut().poll(cx)) { + match result { + Ok(value) => length = Some(value), + Err(error) => return Poll::Ready(Err(error)), + } + receive = None; + } + if let Some(Poll::Ready(result)) = write.as_mut().map(|future| future.as_mut().poll(cx)) { + match result { + Ok(value) => written = Some(value), + Err(error) => return Poll::Ready(Err(error)), + } + write = None; + } + if let Some(length) = length { + if let Some(written) = written.take() { + return Poll::Ready(Ok((length, written))); + } + } + if progress.load(Ordering::Relaxed) == before { + return Poll::Pending; + } + } + cx.waker().wake_by_ref(); + Poll::Pending + }) + .await +} + +pub(crate) async fn write_buffers( + writer: &mut W, + mut receiver: mpsc::Receiver, + progress: &AtomicU64, + map_error: fn(IoError) -> E, +) -> Result +where + W: ChunkIoWriter + ?Sized, +{ + let mut stats = WriteStats::default(); + loop { + while !writer.require_data() && !writer.input_complete() { + let started = Instant::now(); + writer.wait_for_capacity().await; + stats.capacity_waits += 1; + stats.capacity_wait_time += started.elapsed(); + } + let Some(buffer) = receiver.recv().await else { + break; + }; + progress.fetch_add(1, Ordering::Relaxed); + let started = Instant::now(); + match buffer { + UploadBuffer::Framed(owner) => writer.on_framed_data(owner).await, + UploadBuffer::Data(bytes) => writer.on_data(bytes).await, + } + .map_err(map_error)?; + stats.feed_time += started.elapsed(); + stats.feeds += 1; + } + Ok(stats) +} diff --git a/app/crowdb-access-server/src/upload_flow/digest_pipe.rs b/app/crowdb-access-server/src/upload_flow/digest_pipe.rs new file mode 100644 index 000000000..47f7bacec --- /dev/null +++ b/app/crowdb-access-server/src/upload_flow/digest_pipe.rs @@ -0,0 +1,83 @@ +use hyper::body::Bytes; +use openssl::hash::{Hasher, MessageDigest}; +use std::time::{Duration, Instant}; +use tokio::sync::mpsc; +use tokio::task::JoinHandle; + +/// One upload's checksum pipeline. Queueing never controls socket backpressure; +/// the write flow does. Bytes clones retain the received buffer until OpenSSL +/// has consumed it, while the writer may use the same buffer. +pub(crate) struct DigestPipe { + sender: Option>, + worker: Option>>, +} + +struct DigestBatch { + payload: Vec, +} + +pub(crate) struct Digests { + pub md5: [u8; 16], + pub sha256: Option<[u8; 32]>, + pub process_time: Duration, +} + +impl DigestPipe { + pub(crate) fn start(check_sha256: bool) -> Self { + let (sender, mut receiver) = mpsc::channel::(1024); + let worker = tokio::task::spawn_blocking(move || { + let mut md5 = Hasher::new(MessageDigest::md5()).map_err(|_| ())?; + let mut sha256 = check_sha256 + .then(|| Hasher::new(MessageDigest::sha256()).map_err(|_| ())) + .transpose()?; + let mut process_time = Duration::ZERO; + while let Some(batch) = receiver.blocking_recv() { + let started = Instant::now(); + for bytes in batch.payload { + md5.update(&bytes).map_err(|_| ())?; + if let Some(sha256) = &mut sha256 { + sha256.update(&bytes).map_err(|_| ())?; + } + } + process_time += started.elapsed(); + } + Ok(Digests { + process_time, + md5: md5 + .finish() + .map_err(|_| ())? + .as_ref() + .try_into() + .map_err(|_| ())?, + sha256: sha256 + .as_mut() + .map(|sha256| { + sha256 + .finish() + .map_err(|_| ())? + .as_ref() + .try_into() + .map_err(|_| ()) + }) + .transpose()?, + }) + }); + Self { + sender: Some(sender), + worker: Some(worker), + } + } + + pub(crate) fn enqueue(&self, payload: Vec) -> Result<(), ()> { + self.sender + .as_ref() + .ok_or(())? + .try_send(DigestBatch { payload }) + .map_err(|_| ()) + } + + pub(crate) async fn finish(&mut self) -> Result { + self.sender.take(); + self.worker.take().ok_or(())?.await.map_err(|_| ())? + } +} diff --git a/app/crowdb-access-server/tests/common/iceberg_commit_case.rs b/app/crowdb-access-server/tests/common/iceberg_commit_case.rs index c12455c74..c1883bd42 100644 --- a/app/crowdb-access-server/tests/common/iceberg_commit_case.rs +++ b/app/crowdb-access-server/tests/common/iceberg_commit_case.rs @@ -115,7 +115,10 @@ impl TestCommitCase { .await .unwrap(); assert_eq!(replay.status(), 200); - assert_eq!(replay.bytes().await.unwrap(), bytes); + assert_eq!( + durable_response(&replay.bytes().await.unwrap()), + durable_response(&bytes) + ); let changed = post(endpoint, &self.path, &self.identity, &format!("{} ", self.body)) .await .unwrap(); @@ -143,6 +146,26 @@ impl TestCommitCase { } } +fn durable_response(bytes: &[u8]) -> Value { + let mut response: Value = serde_json::from_slice(bytes).unwrap(); + if let Some(config) = response.get_mut("config").and_then(Value::as_object_mut) { + // File grants are issued per response; the catalog result is the replayed state. + let credentials = [ + "s3.access-key-id", + "s3.secret-access-key", + "s3.session-token", + "s3.session-token-expires-at-ms", + ]; + if credentials.iter().any(|key| config.contains_key(*key)) { + for key in credentials { + let value = config.remove(key).expect("incomplete S3 credentials"); + assert!(value.as_str().is_some_and(|text| !text.is_empty())); + } + } + } + response +} + fn properties() -> Value { let mut properties = serde_json::Map::new(); for field in 0..16 { diff --git a/app/crowdb-access-server/tests/common/iceberg_commit_child.rs b/app/crowdb-access-server/tests/common/iceberg_commit_child.rs index 1003d790d..6178cfac0 100644 --- a/app/crowdb-access-server/tests/common/iceberg_commit_child.rs +++ b/app/crowdb-access-server/tests/common/iceberg_commit_child.rs @@ -39,7 +39,11 @@ pub async fn run() { management_seeds: seeds, diskio_connections_per_endpoint: 2, diskio_rpc_workers: 2, - small_write: SmallWritePolicy::default(), + small_write: SmallWritePolicy { + mirror_copies: 1, + conversion_enabled: false, + ..SmallWritePolicy::default() + }, }, control, ) diff --git a/app/crowdb-access-server/tests/common/iceberg_file_worker.rs b/app/crowdb-access-server/tests/common/iceberg_file_worker.rs index 2f0b35194..00f5791c2 100644 --- a/app/crowdb-access-server/tests/common/iceberg_file_worker.rs +++ b/app/crowdb-access-server/tests/common/iceberg_file_worker.rs @@ -106,7 +106,8 @@ impl TestWorker { let address = reservation.local_addr().unwrap(); drop(reservation); Self( - Command::new(env!("CARGO_BIN_EXE_crowdb-iceberg")) + Command::new(env!("CARGO_BIN_EXE_crowdb-access-server")) + .arg("iceberg") .env("CROWDB_MANAGEMENT_SEEDS", seeds.join(",")) .env("CROWDB_ICEBERG_LISTEN", address.to_string()) .env("CROWDB_ICEBERG_READ_TOKEN", "r".repeat(32)) diff --git a/app/crowdb-access-server/tests/common/iceberg_process.rs b/app/crowdb-access-server/tests/common/iceberg_process.rs index 27a1d103c..0fae48c1f 100644 --- a/app/crowdb-access-server/tests/common/iceberg_process.rs +++ b/app/crowdb-access-server/tests/common/iceberg_process.rs @@ -82,8 +82,16 @@ impl TestIcebergProcess { } pub fn command(seeds: &[String]) -> Command { - let mut command = Command::new(env!("CARGO_BIN_EXE_crowdb-iceberg")); + let mut command = Command::new(env!("CARGO_BIN_EXE_crowdb-access-server")); command + .arg("iceberg") + .args([ + "--config", + concat!( + env!("CARGO_MANIFEST_DIR"), + "/tests/common/iceberg_single_node.toml" + ), + ]) .env("CROWDB_MANAGEMENT_SEEDS", seeds.join(",")) .env("CROWDB_ICEBERG_READ_TOKEN", "r".repeat(32)) .env("CROWDB_ICEBERG_WRITE_TOKEN", "w".repeat(32)) diff --git a/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock b/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock index 129f16a2e..81511f201 100644 --- a/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock +++ b/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.lock @@ -583,7 +583,7 @@ checksum = "a31eee39dddec8330830986fcd7625edb5a24ec90ea038215273bbc3adb08ac6" [[package]] name = "crowdb-iceberg-rust-client-fixture" -version = "0.1.0" +version = "0.2.0" dependencies = [ "iceberg", "iceberg-catalog-rest", diff --git a/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml b/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml index faac12267..d49585567 100644 --- a/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml +++ b/app/crowdb-access-server/tests/common/iceberg_rust/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "crowdb-iceberg-rust-client-fixture" -version = "0.1.0" +version = "0.2.0" edition = "2021" publish = false diff --git a/app/crowdb-access-server/tests/common/iceberg_signed_file.rs b/app/crowdb-access-server/tests/common/iceberg_signed_file.rs index a4bb32b5c..7c4ebf45a 100644 --- a/app/crowdb-access-server/tests/common/iceberg_signed_file.rs +++ b/app/crowdb-access-server/tests/common/iceberg_signed_file.rs @@ -2,10 +2,12 @@ use super::common::now_ms; use base64::engine::general_purpose::STANDARD; use base64::Engine; use hmac::{Hmac, Mac}; +use hyper::body::Bytes; use md5::Md5; use reqwest::{Client, Method, Response}; use sha2::{Digest, Sha256}; use std::fmt::Write; +use std::io; fn hex(bytes: &[u8]) -> String { let mut result = String::new(); @@ -27,6 +29,12 @@ pub struct TestFileClient { pub address: std::net::SocketAddr, } +struct SignedPayload { + body: reqwest::Body, + hash: String, + content_md5: Option, +} + impl TestFileClient { pub async fn send(&self, method: Method, path: &str, query: &str, body: &[u8], md5: bool) -> Response { self.send_range(method, path, query, body, md5, None).await @@ -56,15 +64,121 @@ impl TestFileClient { md5: bool, range: Option<&str>, ) -> reqwest::RequestBuilder { + let content_md5 = md5.then(|| STANDARD.encode(Md5::digest(body))); + let hash = if md5 { + "UNSIGNED-PAYLOAD".to_owned() + } else { + hex(&Sha256::digest(body)) + }; + self.signed_request( + method, + path, + query, + SignedPayload { + body: reqwest::Body::from(body.to_vec()), + hash, + content_md5, + }, + range, + ) + } + + #[allow(dead_code)] + pub async fn send_repeated( + &self, + method: Method, + path: &str, + block: Bytes, + repetitions: usize, + md5: [u8; 16], + ) -> Response { + self.send_repeated_with_query(method, path, "", block, repetitions, md5) + .await + } + + #[allow(dead_code)] + pub async fn send_repeated_with_query( + &self, + method: Method, + path: &str, + query: &str, + block: Bytes, + repetitions: usize, + md5: [u8; 16], + ) -> Response { + let length = block.len() * repetitions; + let frames = futures::stream::iter((0..repetitions).map(move |_| Ok::<_, io::Error>(block.clone()))); + self.signed_request( + method, + path, + query, + SignedPayload { + body: reqwest::Body::wrap_stream(frames), + hash: "UNSIGNED-PAYLOAD".to_owned(), + content_md5: Some(STANDARD.encode(md5)), + }, + None, + ) + .header("content-length", length) + .send() + .await + .unwrap() + } + + #[allow(dead_code)] + pub fn request_stream( + &self, + method: Method, + path: &str, + length: usize, + md5: [u8; 16], + receiver: tokio::sync::mpsc::Receiver>, + ) -> reqwest::RequestBuilder { + let frames = futures::stream::unfold(receiver, |mut receiver| async move { + receiver.recv().await.map(|frame| (frame, receiver)) + }); + self.signed_request( + method, + path, + "", + SignedPayload { + body: reqwest::Body::wrap_stream(frames), + hash: "UNSIGNED-PAYLOAD".to_owned(), + content_md5: Some(STANDARD.encode(md5)), + }, + None, + ) + .header("content-length", length) + } + + fn signed_request( + &self, + method: Method, + path: &str, + query: &str, + payload: SignedPayload, + range: Option<&str>, + ) -> reqwest::RequestBuilder { + let SignedPayload { + body, + hash, + content_md5, + } = payload; let now = chrono::DateTime::::from_timestamp_millis(i64::try_from(now_ms()).unwrap()).unwrap(); let date = now.format("%Y%m%dT%H%M%SZ").to_string(); let short = now.format("%Y%m%d").to_string(); - let hash = hex(&Sha256::digest(body)); let host = self.address.to_string(); - let names = "host;x-amz-content-sha256;x-amz-date;x-amz-security-token"; + let names = if content_md5.is_some() { + "content-md5;host;x-amz-content-sha256;x-amz-date;x-amz-security-token" + } else { + "host;x-amz-content-sha256;x-amz-date;x-amz-security-token" + }; + let md5_header = content_md5 + .as_ref() + .map_or_else(String::new, |value| format!("content-md5:{value}\n")); let canonical = format!( - "{}\n{path}\n{query}\nhost:{host}\nx-amz-content-sha256:{hash}\nx-amz-date:{date}\nx-amz-security-token:{}\n\n{names}\n{hash}", + "{}\n{path}\n{query}\n{md5_header}host:{host}\nx-amz-content-sha256:{hash}\nx-amz-date:{date}\nx-amz-security-token:{}\n\n{names}\n{hash}", method.as_str(), self.credentials.session_token() ); let date_key = mac( @@ -97,9 +211,9 @@ impl TestFileClient { .header("x-amz-date", date) .header("x-amz-security-token", self.credentials.session_token()) .header("authorization", authorization) - .body(body.to_vec()); - if md5 { - request = request.header("content-md5", STANDARD.encode(Md5::digest(body))); + .body(body); + if let Some(content_md5) = content_md5 { + request = request.header("content-md5", content_md5); } if let Some(range) = range { request = request.header("range", range); diff --git a/app/crowdb-access-server/tests/common/iceberg_single_node.toml b/app/crowdb-access-server/tests/common/iceberg_single_node.toml new file mode 100644 index 000000000..6321d79d8 --- /dev/null +++ b/app/crowdb-access-server/tests/common/iceberg_single_node.toml @@ -0,0 +1,13 @@ +[deployment] +mode = "test_single_node" +max_node_failures = 0 + +[small_write] +conversion_enabled = false +mirror_copies = 1 + +[s3] +large_mirror_copies = 1 + +[iceberg] +large_mirror_copies = 1 diff --git a/app/crowdb-access-server/tests/common/iceberg_stack.rs b/app/crowdb-access-server/tests/common/iceberg_stack.rs index 96b98e292..39afaa2d8 100644 --- a/app/crowdb-access-server/tests/common/iceberg_stack.rs +++ b/app/crowdb-access-server/tests/common/iceberg_stack.rs @@ -6,7 +6,8 @@ use crowdb_chunk_kv_client::{ ChunkKvClient, ChunkKvRpcTransport, ClientConfig, Group0ChunkKvRangeCatalogSource, }; use crowdb_diskio_client::{DiskId as DiskIoDiskId, TestWireDiskioClient}; -use crowdb_protocol::common::{DiskId, HwStatus, NodeValue, RackValue}; +use crowdb_kv_client::{ClientConfig as KvClientConfig, CrowdbKvClient, KVClusterMetaClient}; +use crowdb_protocol::common::{DiskId, HwStatus, NodeValue, RackValue, ReplicaValue}; use crowdb_protocol::diskdb::rpc::{DiskGroupValue, DiskType, DiskValue}; use crowdb_rpc_ffi::RpcServer; use crowdb_test_harness::chunk_kv::ChunkKvProcess; @@ -123,8 +124,8 @@ impl TestIcebergStack { cluster.runtime_mut(), &seeds, ChunkdbStartOptions { + test_single_node: true, placement_mode: ChunkdbPlacementMode::UnsafeColocated, - repair_allow_unsafe_placement: true, ..ChunkdbStartOptions::default() }, ); @@ -160,6 +161,28 @@ impl TestIcebergStack { async fn seed(cluster: &KvCluster) { let hardware = cluster.make_hardware_client(); + let kv = CrowdbKvClient::new(KvClientConfig::new(cluster.mgmt_endpoints.clone())); + kv.seed_leader(0, 0, cluster.group0_leader_endpoint.clone()); + let metadata = KVClusterMetaClient::new(kv); + metadata.add_store(0, &[0]).await.unwrap(); + for (group_id, endpoint) in [ + (0, &cluster.group0_leader_endpoint), + (1, &cluster.group1_leader_endpoint), + ] { + metadata.add_group(0, group_id).await.unwrap(); + metadata + .add_replica(&ReplicaValue { + store_id: 0, + group_id, + replica_id: 1, + node_id: 0, + role: String::new(), + voting: true, + endpoint: endpoint.clone(), + }) + .await + .unwrap(); + } hardware .add_rack( 1, diff --git a/app/crowdb-access-server/tests/config_test.rs b/app/crowdb-access-server/tests/config_test.rs index 028f46251..aebd4ff60 100644 --- a/app/crowdb-access-server/tests/config_test.rs +++ b/app/crowdb-access-server/tests/config_test.rs @@ -13,9 +13,10 @@ fn tracked_access_configs_load_and_set_bounded_read_resources() { load_from_file(&root.join("conf/crowdb_access_server_config.toml")).unwrap(); let container: AccessConfig = load_from_file(&root.join("../../container/single-node-container/templates/access.toml")).unwrap(); - for (config, expected_ec, expected_threshold) in - [(canonical, (8, 4), 7_549_748), (container, (2, 1), 1_887_437)] + for (config, expected_ec, expected_threshold, expected_budget) in + [(canonical, (8, 4), 7_549_748, 1), (container, (2, 1), 943_719, 0)] { + assert_eq!(config.deployment.max_node_failures, expected_budget); assert_eq!(config.read.stream_slots, 3); assert_eq!(config.read.stream_window_bytes, 1024 * 1024); assert_eq!(config.read.global_stream_bytes, 256 * 1024 * 1024); @@ -41,6 +42,16 @@ fn tracked_access_configs_load_and_set_bounded_read_resources() { } } +#[test] +fn deployment_rejects_mismatched_node_failure_budget() { + let mut config = AccessConfig::default(); + config.deployment.max_node_failures = 0; + assert_eq!( + config.validate(), + Err("production requires max_node_failures = 1".to_string()) + ); +} + #[test] fn config_argument_is_removed_from_service_commands() { let root = Path::new(env!("CARGO_MANIFEST_DIR")); @@ -93,3 +104,28 @@ fn unsupported_disk_block_size_is_rejected_before_routing() { config.small_write.disk_block_bytes = 768 * 1024; assert!(config.validate().is_err()); } + +#[test] +fn protocol_small_write_overrides_are_independent() { + let mut config = AccessConfig::default(); + config.s3.small_write = Some(SmallWriteConfig { + ec_data: 4, + ec_code: 2, + ..SmallWriteConfig::default() + }); + config.iceberg.small_write = Some(SmallWriteConfig { + ec_data: 2, + ec_code: 1, + ..SmallWriteConfig::default() + }); + assert!(config.validate().is_ok()); + assert_eq!(config.s3_small_write().ec_data, 4); + assert_eq!(config.iceberg_small_write().ec_data, 2); + config.s3.small_write.as_mut().unwrap().chunk_capacity_bytes = 32 * 1024 * 1024; + assert_eq!(config.s3_small_write().policy().chunk_capacity, 32 * 1024 * 1024); + assert_eq!( + config.iceberg_small_write().policy().chunk_capacity, + 1024 * 1024 * 1024 + ); + assert_eq!(config.small_write.ec_data, 8); +} diff --git a/app/crowdb-access-server/tests/iceberg_auth_test.rs b/app/crowdb-access-server/tests/iceberg_auth_test.rs index ec8abc8a9..c3af6f8b9 100644 --- a/app/crowdb-access-server/tests/iceberg_auth_test.rs +++ b/app/crowdb-access-server/tests/iceberg_auth_test.rs @@ -9,8 +9,9 @@ fn writer_configuration_fails_before_backend_connection() { Some("m".repeat(32)), Some("c".repeat(32)), ] { - let mut command = Command::new(env!("CARGO_BIN_EXE_crowdb-iceberg")); + let mut command = Command::new(env!("CARGO_BIN_EXE_crowdb-access-server")); command + .arg("iceberg") .env("CROWDB_MANAGEMENT_SEEDS", "127.0.0.1:1") .env("CROWDB_ICEBERG_READ_TOKEN", "r".repeat(32)) .env("CROWDB_ICEBERG_MANAGE_TOKEN", "m".repeat(32)) diff --git a/app/crowdb-access-server/tests/iceberg_file_http_test.rs b/app/crowdb-access-server/tests/iceberg_file_http_test.rs index e53b4d4c0..adb2a8965 100644 --- a/app/crowdb-access-server/tests/iceberg_file_http_test.rs +++ b/app/crowdb-access-server/tests/iceberg_file_http_test.rs @@ -22,9 +22,11 @@ use crowdb_access_iceberg::file::{ use crowdb_access_iceberg::key::OperationId; use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; use crowdb_access_iceberg::wire::BearerAuthenticator; -use crowdb_access_server::config::SmallWriteConfig; +use crowdb_access_server::config::AccessConfig; +use crowdb_common::config::load_from_file; use crowdb_protocol::chunkdb::rpc::{QueryChunkRequest, Strip}; use crowdb_test_harness::chunkdb::make_client as make_chunkdb_client; +use futures::StreamExt; use md5::{Digest, Md5}; use reqwest::{Client, Method}; use std::fmt::Write as _; @@ -172,6 +174,28 @@ fn path(table: TableLocation, key: &str) -> String { format!("/{}/{}", table.bucket(), table.file(key).unwrap().object_key()) } +#[tokio::test(flavor = "multi_thread", worker_threads = 8)] +async fn concurrent_multipart_creates_share_admission() { + let (_stack, _process, client, table) = setup().await; + let mut requests = tokio::task::JoinSet::new(); + for index in 0..24 { + let object = path(table, &format!("data/concurrent-{index}.parquet")); + let request = client.request(Method::POST, &object, "uploads=", b"", false, None); + requests.spawn(async move { request.send().await.unwrap() }); + } + while let Some(result) = requests.join_next().await { + let response = result.unwrap(); + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + } +} + +fn fixture_config() -> AccessConfig { + load_from_file( + &std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("tests/common/iceberg_single_node.toml"), + ) + .unwrap() +} + async fn catalog_counts(client: &TestFileClient) -> (u64, u64, u64, u64) { let response: serde_json::Value = client .client @@ -202,6 +226,24 @@ fn catalog_delta(before: (u64, u64, u64, u64), after: (u64, u64, u64, u64)) -> S ) } +async fn file_request_counts(client: &TestFileClient) -> (u64, u64) { + let response: serde_json::Value = client + .client + .get(format!("http://{}/_crowdb/metrics", client.address)) + .bearer_auth("m".repeat(32)) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + let catalog = &response["routes"][6][0]["catalog"]; + ( + catalog["get"].as_u64().unwrap(), + catalog["compare_exchange"].as_u64().unwrap(), + ) +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] #[ignore = "native null-DiskIO release performance fixture"] async fn native_file_5_mib_profile() { @@ -274,6 +316,7 @@ async fn native_file_5_mib_profile() { } #[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[allow(clippy::too_many_lines)] async fn signed_standard_put_get_and_multipart_publish_unbound_files() { let (stack, _process, client, table) = setup().await; let parquet = b"PAR1datafoot\x04\0\0\0PAR1"; @@ -310,10 +353,24 @@ async fn signed_standard_put_get_and_multipart_publish_unbound_files() { let medium = path(table, "data/medium.parquet"); let medium_bytes = (0..1_200_000) - .map(|index| (index % 251) as u8) + .map(|index| u8::try_from(index % 251).unwrap()) .collect::>(); + let before = file_request_counts(&client).await; + let started = Instant::now(); let response = client.send(Method::PUT, &medium, "", &medium_bytes, true).await; + let elapsed = started.elapsed(); assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + let after = file_request_counts(&client).await; + assert!( + elapsed < std::time::Duration::from_secs(2), + "1.2 MiB PUT took {elapsed:?}" + ); + assert!( + after.0 - before.0 <= 15 && after.1 - before.1 <= 2, + "1.2 MiB PUT used {} gets and {} CAS operations", + after.0 - before.0, + after.1 - before.1 + ); let stored = repository .load( client.credentials.grant().context, @@ -336,7 +393,7 @@ async fn signed_standard_put_get_and_multipart_publish_unbound_files() { assert_eq!(range.status(), 206); assert_eq!( range.bytes().await.unwrap().as_ref(), - &medium_bytes[1048550..1048601] + &medium_bytes[1_048_550..1_048_601] ); let metrics: serde_json::Value = Client::new() .get(format!("http://{}/_crowdb/metrics", client.address)) @@ -401,12 +458,11 @@ async fn signed_standard_put_get_and_multipart_publish_unbound_files() { let mut composite = Md5::new(); composite.update(Md5::digest(first)); composite.update(Md5::digest(second)); - let expected_etag = composite - .finalize() - .iter() - .map(|byte| format!("{byte:02x}")) - .collect::() - + "-2"; + let mut expected_etag = String::with_capacity(34); + for byte in composite.finalize() { + write!(&mut expected_etag, "{byte:02x}").unwrap(); + } + expected_etag.push_str("-2"); let published = repository .load( client.credentials.grant().context, @@ -449,6 +505,7 @@ async fn signed_standard_put_get_and_multipart_publish_unbound_files() { async fn ordinary_put_size_matrix_streams_and_reads_ranges() { let (stack, _process, client, table) = setup_with_bounds_and_file_limit( ClearBounds { + request_ms: 120_000, delegated_access_ms: 900_000, ..ClearBounds::default() }, @@ -460,7 +517,9 @@ async fn ordinary_put_size_matrix_streams_and_reads_ranges() { for size in [10 * 1024, 1024 * 1024, 12 * 1024 * 1024, 100 * 1024 * 1024] { let key = format!("data/size-{size}.parquet"); let object = path(table, &key); - let bytes = (0..size).map(|offset| (offset % 256) as u8).collect::>(); + let bytes = (0..size) + .map(|offset| u8::try_from(offset % 256).unwrap()) + .collect::>(); let expected = Md5::digest(&bytes); let started = Instant::now(); let put = client.send(Method::PUT, &object, "", &bytes, true).await; @@ -487,12 +546,14 @@ async fn ordinary_put_size_matrix_streams_and_reads_ranges() { .unwrap() .chunk .unwrap(); - let ec = chunk.strips.iter().find_map(|strip| match strip.strip.as_ref() { - Some(Strip::EcStrip(ec)) => Some(ec), - _ => None, - }); - let ec = ec.expect("large Iceberg write has an EC strip"); - assert_eq!((ec.data_num, ec.code_num), (8, 4)); + let expected_copies = fixture_config().iceberg.large_mirror_copies.unwrap(); + assert!(chunk.strips.iter().any(|strip| strip.sealed_length > 0)); + for strip in &chunk.strips { + let Some(Strip::MirrorStrip(mirror)) = strip.strip.as_ref() else { + panic!("single-node large Iceberg write has a mirror strip"); + }; + assert_eq!(mirror.segments.len(), usize::try_from(expected_copies).unwrap()); + } } let get_started = Instant::now(); let mut response = client.send(Method::GET, &object, "", b"", false).await; @@ -519,10 +580,249 @@ async fn ordinary_put_size_matrix_streams_and_reads_ranges() { } } +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn slow_socket_resumes_upload_after_writer_drains() { + const BLOCK_BYTES: usize = 1024 * 1024; + let (_stack, _process, client, table) = setup_with_bounds_and_file_limit( + ClearBounds { + request_ms: 120_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + 16 * 1024 * 1024, + 16 * 1024 * 1024, + ) + .await; + let block = hyper::body::Bytes::from(vec![41; BLOCK_BYTES]); + let mut md5 = Md5::new(); + md5.update(&block); + md5.update(&block); + let digest: [u8; 16] = md5.finalize().into(); + let object = path(table, "data/slow-socket.parquet"); + let (sender, receiver) = tokio::sync::mpsc::channel(1); + let request = client.request_stream(Method::PUT, &object, 2 * BLOCK_BYTES, digest, receiver); + let upload = tokio::spawn(async move { request.send().await.unwrap() }); + sender.send(Ok(block.clone())).await.unwrap(); + // Let the first strip finish so no disk task remains to wake the upload. + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + sender.send(Ok(block)).await.unwrap(); + drop(sender); + let response = tokio::time::timeout(std::time::Duration::from_secs(30), upload) + .await + .expect("socket readiness must resume the idle upload") + .unwrap(); + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + let get = client.send(Method::GET, &object, "", b"", false).await; + assert_eq!(get.status(), 200); + assert_eq!(get.bytes().await.unwrap().len(), 2 * BLOCK_BYTES); + let metrics: serde_json::Value = Client::new() + .get(format!("http://{}/_crowdb/metrics", client.address)) + .bearer_auth("m".repeat(32)) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + assert!(metrics["upload_flow"]["body_waits"].as_u64().unwrap() > 0); + assert!(metrics["upload_flow"]["body_wait_ns"].as_u64().unwrap() > 0); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "manual 100 MiB upload profile"] +async fn repeated_100_mib_put_streams_without_client_payload_copy() { + const BLOCK_BYTES: usize = 1024 * 1024; + const BLOCK_COUNT: usize = 100; + let (_stack, _process, client, table) = setup_with_bounds_and_file_limit( + ClearBounds { + request_ms: 120_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + 128 * 1024 * 1024, + 128 * 1024 * 1024, + ) + .await; + let block = hyper::body::Bytes::from(vec![37; BLOCK_BYTES]); + let mut md5 = Md5::new(); + for _ in 0..BLOCK_COUNT { + md5.update(&block); + } + let digest: [u8; 16] = md5.finalize().into(); + let object = path(table, "data/repeated-100-mib.parquet"); + let started = Instant::now(); + let response = client + .send_repeated(Method::PUT, &object, block.clone(), BLOCK_COUNT, digest) + .await; + let put_elapsed = started.elapsed(); + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + let metrics: serde_json::Value = Client::new() + .get(format!("http://{}/_crowdb/metrics", client.address)) + .bearer_auth("m".repeat(32)) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + let upload_flow = &metrics["upload_flow"]; + assert_eq!(upload_flow["attempts"], 1); + assert_eq!(upload_flow["completed"], 1); + assert_eq!(upload_flow["logical_bytes"], BLOCK_BYTES * BLOCK_COUNT); + assert!(upload_flow["strip_write_successes"].as_u64().unwrap() > 0); + assert!(upload_flow["strip_write_success_ns"].as_u64().unwrap() > 0); + let head = client.send(Method::HEAD, &object, "", b"", false).await; + assert_eq!(head.status(), 200); + assert_eq!( + head.headers()["content-length"], + (BLOCK_BYTES * BLOCK_COUNT).to_string() + ); + let last = format!( + "bytes={}-{}", + BLOCK_BYTES * BLOCK_COUNT - 128, + BLOCK_BYTES * BLOCK_COUNT - 1 + ); + let range = client + .send_range(Method::GET, &object, "", b"", false, Some(&last)) + .await; + assert_eq!(range.status(), 206); + assert_eq!(range.bytes().await.unwrap().as_ref(), &[37; 128]); + let invalid = path(table, "data/repeated-invalid-md5.parquet"); + let rejected = client + .send_repeated(Method::PUT, &invalid, block, 1, [0; 16]) + .await; + assert_eq!(rejected.status(), 400); + assert!(rejected.text().await.unwrap().contains("BadDigest")); + let absent = client.send(Method::HEAD, &invalid, "", b"", false).await; + assert_eq!(absent.status(), 404); + let sha_object = path(table, "data/signed-sha256.parquet"); + let sha = client + .send(Method::PUT, &sha_object, "", b"sha256 payload", false) + .await; + assert_eq!(sha.status(), 200, "{}", sha.text().await.unwrap()); + println!("iceberg repeated 100 MiB PUT={put_elapsed:?}"); + println!("iceberg direct upload flow metrics={upload_flow}"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "manual 100 MiB multipart upload profile"] +async fn repeated_100_mib_multipart_upload_profile() { + const BLOCK_BYTES: usize = 1024 * 1024; + const PART_COUNT: usize = 13; + let (_stack, _process, client, table) = setup_with_bounds_and_file_limit( + ClearBounds { + request_ms: 120_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + 128 * 1024 * 1024, + 128 * 1024 * 1024, + ) + .await; + let block = hyper::body::Bytes::from(vec![37; BLOCK_BYTES]); + let digests = [8, 4].map(|repetitions| { + let mut md5 = Md5::new(); + for _ in 0..repetitions { + md5.update(&block); + } + <[u8; 16]>::from(md5.finalize()) + }); + let object = path(table, "data/repeated-100-mib-multipart.parquet"); + let upload_started = Instant::now(); + let created = client.send(Method::POST, &object, "uploads=", b"", false).await; + assert_eq!(created.status(), 200, "{}", created.text().await.unwrap()); + let created_body = created.text().await.unwrap(); + let upload = created_body + .split_once("") + .unwrap() + .1 + .split_once("") + .unwrap() + .0; + let create_elapsed = upload_started.elapsed(); + let part_started = Instant::now(); + let parts = futures::stream::iter(1..=PART_COUNT) + .map(|number| { + let block = block.clone(); + let query = format!("partNumber={number}&uploadId={upload}"); + let object = object.clone(); + let repetitions = if number == PART_COUNT { 4 } else { 8 }; + let digest = if repetitions == 4 { digests[1] } else { digests[0] }; + let client = &client; + async move { + let started = Instant::now(); + let response = client + .send_repeated_with_query(Method::PUT, &object, &query, block, repetitions, digest) + .await; + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + let etag = response.headers()["etag"].to_str().unwrap().to_owned(); + (number, etag, started.elapsed()) + } + }) + .buffer_unordered(4) + .collect::>() + .await; + let parts_elapsed = part_started.elapsed(); + let mut parts = parts; + parts.sort_unstable_by_key(|part| part.0); + let mut manifest = String::from( + "", + ); + for (number, etag, _) in &parts { + write!( + manifest, + "{etag}{number}" + ) + .unwrap(); + } + manifest.push_str(""); + let complete_started = Instant::now(); + let completed = client + .send( + Method::POST, + &object, + &format!("uploadId={upload}"), + manifest.as_bytes(), + false, + ) + .await; + assert_eq!(completed.status(), 200, "{}", completed.text().await.unwrap()); + let completed_body = completed.text().await.unwrap(); + assert!( + completed_body.contains(""), + "{completed_body}" + ); + let complete_elapsed = complete_started.elapsed(); + let last = format!("bytes={}-{}", 100 * BLOCK_BYTES - 128, 100 * BLOCK_BYTES - 1); + let range = client + .send_range(Method::GET, &object, "", b"", false, Some(&last)) + .await; + assert_eq!(range.status(), 206); + assert_eq!(range.bytes().await.unwrap().as_ref(), &[37; 128]); + println!( + "iceberg 100 MiB multipart create={create_elapsed:?} parts={parts_elapsed:?} complete={complete_elapsed:?} slowest_part={:?}", + parts.iter().map(|part| part.2).max().unwrap() + ); + let metrics: serde_json::Value = Client::new() + .get(format!("http://{}/_crowdb/metrics", client.address)) + .bearer_auth("m".repeat(32)) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + println!("iceberg upload flow metrics={}", metrics["upload_flow"]); + assert_eq!(metrics["upload_flow"]["attempts"], PART_COUNT); + assert_eq!(metrics["upload_flow"]["completed"], PART_COUNT); + assert_eq!(metrics["upload_flow"]["logical_bytes"], 100 * BLOCK_BYTES); + assert_eq!(metrics["upload_flow"]["multipart_completions"], 1); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn small_routing_is_strict_at_the_strip_threshold() { let (_stack, _process, client, table) = setup().await; - let threshold = SmallWriteConfig::default().threshold_exclusive(); + let threshold = fixture_config().iceberg_small_write().threshold_exclusive(); let completed = async || { let metrics: serde_json::Value = Client::new() .get(format!("http://{}/_crowdb/metrics", client.address)) diff --git a/app/crowdb-access-server/tests/iceberg_full_stack_test.rs b/app/crowdb-access-server/tests/iceberg_full_stack_test.rs index dcb0f750c..788a46608 100644 --- a/app/crowdb-access-server/tests/iceberg_full_stack_test.rs +++ b/app/crowdb-access-server/tests/iceberg_full_stack_test.rs @@ -69,6 +69,104 @@ async fn execute(repository: &CatalogRepository, request: ManagementRequest) -> .unwrap() } +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn metadata_create_and_commit_stay_within_operation_budget() { + let stack = TestIcebergStack::start().await; + let repository = CatalogRepository::new( + stack.store().await, + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + ) + .unwrap(); + execute( + &repository, + request(ManagementAction::Initialize, "metadata-budget", None), + ) + .await; + common::activate(&repository).await; + let frontend = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + let origin = format!("http://{}", frontend.address); + let client = reqwest::Client::builder() + .timeout(Duration::from_secs(10)) + .build() + .unwrap(); + let namespace = client + .post(format!("{origin}/v1/namespaces")) + .bearer_auth("w".repeat(32)) + .json(&serde_json::json!({"namespace": ["budget"]})) + .send() + .await + .unwrap(); + assert_eq!(namespace.status(), 200, "{}", namespace.text().await.unwrap()); + + let before = metadata_counters(&client, &origin).await; + let started = std::time::Instant::now(); + let created = client + .post(format!("{origin}/v1/namespaces/budget/tables")) + .bearer_auth("w".repeat(32)) + .json( + &serde_json::json!({"name":"small", "schema":{"type":"struct", "schema-id":0, + "fields":[{"id":1,"name":"value","type":"long","required":false}]}}), + ) + .send() + .await + .unwrap(); + let create_duration = started.elapsed(); + assert_eq!(created.status(), 200, "{}", created.text().await.unwrap()); + let after_create = metadata_counters(&client, &origin).await; + + let started = std::time::Instant::now(); + let updated = client + .post(format!("{origin}/v1/namespaces/budget/tables/small")) + .bearer_auth("w".repeat(32)) + .json(&serde_json::json!({"requirements":[], "updates":[ + {"action":"set-properties", "updates":{"sample":"done"}}]})) + .send() + .await + .unwrap(); + let update_duration = started.elapsed(); + assert_eq!(updated.status(), 200, "{}", updated.text().await.unwrap()); + let after_update = metadata_counters(&client, &origin).await; + + let create_get = after_create.0 - before.0; + let create_cas = after_create.1 - before.1; + let update_get = after_update.0 - after_create.0; + let update_cas = after_update.1 - after_create.1; + eprintln!( + "metadata budget: create={create_duration:?} get={create_get} cas={create_cas}; \ + update={update_duration:?} get={update_get} cas={update_cas}" + ); + assert!(create_duration < Duration::from_secs(2)); + assert!(update_duration < Duration::from_secs(2)); + assert!( + create_get <= 160 && create_cas <= 30, + "table create performed too many storage operations" + ); + assert!( + update_get <= 160 && update_cas <= 20, + "table update performed too many storage operations" + ); +} + +async fn metadata_counters(client: &reqwest::Client, origin: &str) -> (u64, u64) { + let response = client + .get(format!("{origin}/_crowdb/metrics")) + .bearer_auth("m".repeat(32)) + .send() + .await + .unwrap(); + assert_eq!(response.status(), 200); + let value: serde_json::Value = response.json().await.unwrap(); + let catalog = &value["routes"][4][0]["catalog"]; + ( + catalog["get"].as_u64().unwrap(), + catalog["compare_exchange"].as_u64().unwrap(), + ) +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn catalog_recovery_survives_real_chunk_kv_restart() { let mut stack = TestIcebergStack::start().await; diff --git a/app/crowdb-access-server/tests/iceberg_gc_budget_test.rs b/app/crowdb-access-server/tests/iceberg_gc_budget_test.rs index 77220d091..379018af5 100644 --- a/app/crowdb-access-server/tests/iceberg_gc_budget_test.rs +++ b/app/crowdb-access-server/tests/iceberg_gc_budget_test.rs @@ -7,10 +7,9 @@ use async_trait::async_trait; use crowdb_access_iceberg::file::{ChunkRoot, FileBlockStore, FileIdentity, FileIoError}; use crowdb_access_iceberg::{ catalog::{CasOutcome, CatalogStore, StoreError, StoredValue}, - gc::{GcScan, GcStore, GcSystemScan}, + gc::{BudgetedGcBlocks, BudgetedGcStore, GcIoBudget, GcScan, GcStore, GcSystemScan}, record::MAX_RECORD_BYTES, }; -use crowdb_access_server::iceberg::{BudgetedGcBlocks, BudgetedGcStore, GcIoBudget}; use crowdb_chunk_client::ReclaimOutcome; use crowdb_chunk_kv_client::MultiScanPage; use crowdb_protocol::chunk_kv::ClientRequestId; @@ -92,7 +91,7 @@ impl FileBlockStore for TestBlocks { #[tokio::test] async fn chunk_io_budget_rejects_work_before_dispatch_and_resets_per_step() { - let budget = Arc::new(GcIoBudget::for_tests(1024, 8, 24, 2)); + let budget = Arc::new(GcIoBudget::new(1024, 8, 24, 2)); let inner = Arc::new(TestBlocks::default()); let blocks = BudgetedGcBlocks::new(inner.clone(), budget.clone()); let root = ChunkRoot { @@ -116,7 +115,7 @@ async fn chunk_io_budget_rejects_work_before_dispatch_and_resets_per_step() { #[tokio::test] async fn kv_budget_rejects_work_before_dispatch_and_resets_per_step() { - let budget = Arc::new(GcIoBudget::for_tests( + let budget = Arc::new(GcIoBudget::new( u64::try_from(MAX_RECORD_BYTES).unwrap() + 1, 1, 24, diff --git a/app/crowdb-access-server/tests/iceberg_gc_control_test.rs b/app/crowdb-access-server/tests/iceberg_gc_control_test.rs index e3c0204b1..12212cc10 100644 --- a/app/crowdb-access-server/tests/iceberg_gc_control_test.rs +++ b/app/crowdb-access-server/tests/iceberg_gc_control_test.rs @@ -16,7 +16,11 @@ use crowdb_access_iceberg::{ record::StorageRecord, table::{head_key, TableHead, TableLifecycle, TablePurgeTask}, }; +use crowdb_access_s3::storage::S3StorageClients; +use crowdb_chunk_client::{ChunkIoWriter, ChunkReadPolicy, SmallWritePolicy}; +use hyper::body::Bytes; use sha2::{Digest, Sha256}; +use std::sync::atomic::{AtomicUsize, Ordering}; use std::sync::Arc; fn command(stack: &common::TestIcebergStack, token: char, arguments: &[&str]) -> std::process::Output { @@ -410,6 +414,31 @@ assert not catalog.namespace_exists(namespace) } } +async fn write_s3_during_gc(seeds: Vec, progress: Arc) { + let s3 = S3StorageClients::connect_with_read_policy( + seeds, + 2, + 1, + SmallWritePolicy { + conversion_enabled: false, + mirror_copies: 1, + ..SmallWritePolicy::default() + }, + ChunkReadPolicy::default(), + ) + .await + .unwrap(); + for index in 0..64 { + let data = Bytes::from(vec![u8::try_from(index).unwrap(); 128]); + let mut writer = s3.chunks.prepare_small_write(data.len()).await.unwrap(); + writer.on_data(data).await.unwrap(); + writer.on_finish().await.unwrap(); + progress.fetch_add(1, Ordering::Release); + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + } + s3.chunks.shutdown_small_writes().await.unwrap(); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] #[ignore = "requires the pinned PyIceberg environment"] async fn official_sdk_foreground_progresses_under_gc_backlog() { @@ -463,6 +492,11 @@ with ThreadPoolExecutor(max_workers=4) as executor: .arg(format!("http://{}", server.address)) .spawn() .unwrap(); + let s3_progress = Arc::new(AtomicUsize::new(0)); + let s3_writes = tokio::spawn(write_s3_during_gc( + stack.cluster.mgmt_endpoints.clone(), + Arc::clone(&s3_progress), + )); let repository = GcRepository::new(store); let identity = OperationId::from_bytes(table.as_bytes()).unwrap(); let overlapped = tokio::time::timeout(std::time::Duration::from_secs(90), async { @@ -475,6 +509,7 @@ with ThreadPoolExecutor(max_workers=4) as executor: .await .unwrap() .is_some_and(|task| task.revision > 1) + && s3_progress.load(Ordering::Acquire) > 0 { break true; } @@ -494,6 +529,8 @@ with ThreadPoolExecutor(max_workers=4) as executor: .await .unwrap(); assert!(status.success(), "official SDK foreground operations failed"); + s3_writes.await.unwrap(); + assert_eq!(s3_progress.load(Ordering::Acquire), 64); assert!( overlapped, "GC did not advance while the SDK requests were active" diff --git a/app/crowdb-access-server/tests/protocol_http_policy_test.rs b/app/crowdb-access-server/tests/protocol_http_policy_test.rs new file mode 100644 index 000000000..6b7797420 --- /dev/null +++ b/app/crowdb-access-server/tests/protocol_http_policy_test.rs @@ -0,0 +1,477 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::net::{SocketAddr, TcpListener}; +use std::path::Path; +use std::process::{Child, Command, Stdio}; +use std::time::{Duration, SystemTime, UNIX_EPOCH}; + +use crowdb_access_iceberg::catalog::{Capabilities, ManagementPrivilege}; +use crowdb_access_iceberg::file::{FileGrant, FileGrantIssuer, FileOperation, FileOperations, TableLocation}; +use crowdb_access_iceberg::key::OperationId; +use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; +use crowdb_access_iceberg::storage; +use crowdb_access_iceberg::wire::BearerAuthenticator; +use crowdb_chunk_client::{ + ChunkIoClient, ChunkIoClientConfig, ChunkIoWriter, ChunkReadPolicy, IoError, SmallWritePolicy, +}; +use crowdb_chunkdb_client::ChunkdbRpcTransport; +use crowdb_console_shared::{ + config::{ConsoleConfig, ServiceType}, + lifecycle, + ops::s3, +}; +use crowdb_protocol::chunkdb::rpc::{ChunkType, ListChunksRequest, Strip}; +use crowdb_test_harness::test_dirs::TestDir; +use reqwest::{Client, Method}; + +mod common { + pub fn now_ms() -> u64 { + super::now_ms() + } +} + +#[path = "common/iceberg_signed_file.rs"] +mod signed; + +fn now_ms() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap() + .as_millis() + .try_into() + .unwrap() +} + +fn free_address() -> SocketAddr { + let reservation = TcpListener::bind("127.0.0.1:0").unwrap(); + reservation.local_addr().unwrap() +} + +struct RunningAccess(Child); + +impl Drop for RunningAccess { + fn drop(&mut self) { + let _ = self.0.kill(); + let _ = self.0.wait(); + } +} + +struct StopClusterOnDrop<'a>(&'a Path); + +impl Drop for StopClusterOnDrop<'_> { + fn drop(&mut self) { + let _ = s3::stop(self.0); + } +} + +async fn initialize_iceberg(seeds: Vec) { + let small = SmallWritePolicy { + conversion_enabled: false, + mirror_copies: 2, + ..SmallWritePolicy::default() + }; + let (repository, _, chunks) = storage::connect(seeds, ChunkReadPolicy::default(), small, 2, 2) + .await + .unwrap(); + for (action, epoch, capabilities) in [ + (ManagementAction::Initialize, 0, None), + ( + ManagementAction::Activate, + 1, + Some(Capabilities::from_bits(0x3fff).unwrap()), + ), + ] { + repository + .execute( + ManagementRequest { + identity: RequestIdentity { + operation: OperationId::random(), + issued_ms: now_ms(), + }, + principal: "manager".into(), + action, + expected_epoch: epoch, + display_name: "protected-http".into(), + confirmation: None, + capabilities, + }, + ManagementPrivilege::Manage, + now_ms(), + ) + .await + .unwrap(); + } + chunks.shutdown_small_writes().await.unwrap(); +} + +async fn start_access( + config: &Path, + seeds: &[String], + s3_addr: SocketAddr, + iceberg_addr: SocketAddr, +) -> RunningAccess { + start_access_with_fault(config, seeds, s3_addr, iceberg_addr, None).await +} + +async fn start_access_with_fault( + config: &Path, + seeds: &[String], + s3_addr: SocketAddr, + iceberg_addr: SocketAddr, + stop_s3_manager_file: Option<&Path>, +) -> RunningAccess { + let mut command = Command::new(env!("CARGO_BIN_EXE_crowdb-access-server")); + command + .args(["--config", config.to_str().unwrap()]) + .env("CROWDB_MANAGEMENT_SEEDS", seeds.join(",")) + .env("CROWDB_S3_LISTEN", s3_addr.to_string()) + .env("CROWDB_S3_TENANT", "local") + .env( + "CROWDB_S3_MASTER_KEY", + "1111111111111111111111111111111111111111111111111111111111111111", + ) + .env("CROWDB_S3_REGION", "us-east-1") + .env("CROWDB_S3_TRUSTED_NETWORK", "true") + .env("CROWDB_ICEBERG_LISTEN", iceberg_addr.to_string()) + .env("CROWDB_ICEBERG_READ_TOKEN", "r".repeat(32)) + .env("CROWDB_ICEBERG_WRITE_TOKEN", "w".repeat(32)) + .env("CROWDB_ICEBERG_MANAGE_TOKEN", "m".repeat(32)) + .env("CROWDB_ICEBERG_CLEAR_TOKEN", "c".repeat(32)) + .env("CROWDB_ICEBERG_GC_ENABLED", "0") + .stdout(Stdio::inherit()) + .stderr(Stdio::inherit()); + if let Some(path) = stop_s3_manager_file { + command.env("CROWDB_TEST_STOP_S3_MANAGER_FILE", path); + } + let child = command.spawn().unwrap(); + let mut process = RunningAccess(child); + let client = Client::new(); + tokio::time::timeout(Duration::from_secs(30), async { + loop { + if let Some(status) = process.0.try_wait().unwrap() { + panic!("combined access process exited before readiness: {status}"); + } + let s3_ready = client + .get(format!("http://{s3_addr}/_crowdb/health/ready")) + .send() + .await + .is_ok_and(|response| response.status().is_success()); + let iceberg_ready = client + .get(format!("http://{iceberg_addr}/v1/config")) + .bearer_auth("r".repeat(32)) + .send() + .await + .is_ok_and(|response| response.status().is_success()); + if s3_ready && iceberg_ready { + break; + } + tokio::time::sleep(Duration::from_millis(50)).await; + } + }) + .await + .unwrap(); + process +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "starts a complete simulated three-rack production storage stack"] +async fn combined_http_listeners_keep_protocol_chunk_policies_separate() { + let dir = TestDir::new("access-protected-http-policy").unwrap(); + s3::start_protected_test_cluster(dir.path()).await.unwrap(); + let _cleanup = StopClusterOnDrop(dir.path()); + let (cluster, _) = s3::load(dir.path()).unwrap(); + let seeds = cluster + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| server.url.clone()) + .collect::>(); + initialize_iceberg(seeds.clone()).await; + + let s3_addr = free_address(); + let iceberg_addr = loop { + let address = free_address(); + if address != s3_addr { + break address; + } + }; + let config = dir.path().join("combined-access.toml"); + std::fs::write( + &config, + "[s3]\nec_data = 2\nec_code = 1\nlarge_prefetch_strips_per_chunk = 2\nlarge_memory_budget_bytes = 67108864\n[s3.small_write]\nconversion_enabled = false\nmirror_copies = 2\n[iceberg]\nec_data = 4\nec_code = 2\nlarge_prefetch_strips_per_chunk = 3\nlarge_memory_budget_bytes = 100663296\n[iceberg.small_write]\nconversion_enabled = false\nmirror_copies = 2\n", + ) + .unwrap(); + let _access = start_access(&config, &seeds, s3_addr, iceberg_addr).await; + write_s3(s3_addr).await; + write_iceberg(iceberg_addr, &seeds).await; + assert_chunk_layouts(&cluster).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "starts a complete simulated three-rack production storage stack"] +async fn terminal_s3_storage_failure_stops_both_access_listeners() { + let dir = TestDir::new("access-storage-failure").unwrap(); + s3::start_protected_test_cluster(dir.path()).await.unwrap(); + let _cleanup = StopClusterOnDrop(dir.path()); + let (cluster, _) = s3::load(dir.path()).unwrap(); + let seeds = cluster + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| server.url.clone()) + .collect::>(); + initialize_iceberg(seeds.clone()).await; + let s3_addr = free_address(); + let iceberg_addr = loop { + let address = free_address(); + if address != s3_addr { + break address; + } + }; + let config = dir.path().join("combined-access.toml"); + std::fs::write( + &config, + "[s3.small_write]\nmirror_copies = 2\n[iceberg.small_write]\nmirror_copies = 2\n", + ) + .unwrap(); + let sentinel = dir.path().join("stop-s3-manager"); + let mut access = start_access_with_fault(&config, &seeds, s3_addr, iceberg_addr, Some(&sentinel)).await; + let s3_client = s3::S3HttpClient::new(format!("http://{s3_addr}")).unwrap(); + s3_client + .request(Method::PUT, Some("failure"), None, &[], None, None) + .await + .unwrap(); + s3_client + .request( + Method::PUT, + Some("failure"), + Some("small"), + &[], + Some(vec![0x42; 1024]), + None, + ) + .await + .unwrap(); + std::fs::write(&sentinel, b"stop").unwrap(); + let status = tokio::time::timeout(Duration::from_secs(15), async { + loop { + if let Some(status) = access.0.try_wait().unwrap() { + break status; + } + tokio::time::sleep(Duration::from_millis(50)).await; + } + }) + .await + .expect("combined access process must stop after storage manager failure"); + assert!(!status.success()); + assert!(tokio::net::TcpStream::connect(s3_addr).await.is_err()); + assert!(tokio::net::TcpStream::connect(iceberg_addr).await.is_err()); +} + +async fn write_s3(s3_addr: SocketAddr) { + let s3_client = s3::S3HttpClient::new(format!("http://{s3_addr}")).unwrap(); + s3_client + .request(Method::PUT, Some("policy"), None, &[], None, None) + .await + .unwrap(); + for (name, bytes) in [ + ("small-a", vec![0x31; 60 * 1024]), + ("small-b", vec![0x33; 20 * 1024]), + ("large", vec![0x32; 2 * 1024 * 1024]), + ] { + s3_client + .request( + Method::PUT, + Some("policy"), + Some(name), + &[], + Some(bytes.clone()), + None, + ) + .await + .unwrap(); + let (_, read) = s3_client + .request(Method::GET, Some("policy"), Some(name), &[], None, None) + .await + .unwrap(); + assert_eq!(read, bytes); + } +} + +async fn write_iceberg(iceberg_addr: SocketAddr, seeds: &[String]) { + let client = Client::new(); + let endpoint = format!("http://{iceberg_addr}"); + let namespace = client + .post(format!("{endpoint}/v1/namespaces")) + .bearer_auth("w".repeat(32)) + .json(&serde_json::json!({"namespace": ["analytics"]})) + .send() + .await + .unwrap(); + assert_eq!(namespace.status(), 200, "{}", namespace.text().await.unwrap()); + let draft = client + .post(format!("{endpoint}/v1/namespaces/analytics/tables")) + .bearer_auth("w".repeat(32)) + .json(&serde_json::json!({"name": "files", "stage-create": true, + "schema": {"type": "struct", "fields": []}})) + .send() + .await + .unwrap(); + assert_eq!(draft.status(), 200, "{}", draft.text().await.unwrap()); + let draft: serde_json::Value = draft.json().await.unwrap(); + let table: TableLocation = format!("{}/", draft["metadata"]["location"].as_str().unwrap()) + .parse() + .unwrap(); + let authenticator = + BearerAuthenticator::new(&"r".repeat(32), &"w".repeat(32), &"m".repeat(32), &"c".repeat(32)).unwrap(); + let issuer = FileGrantIssuer::new(authenticator.namespace_token_key(), 900_000).unwrap(); + let (repository, _, chunks) = storage::connect( + seeds.to_vec(), + ChunkReadPolicy::default(), + SmallWritePolicy { + mirror_copies: 2, + conversion_enabled: false, + ..SmallWritePolicy::default() + }, + 2, + 2, + ) + .await + .unwrap(); + let context = repository.status().await.unwrap().0.context; + let credentials = issuer + .issue(FileGrant { + context, + table: table.table, + principal: [7; 32], + nonce: OperationId::random(), + issued_ms: now_ms() - 1_000, + expires_ms: now_ms() + 600_000, + operations: FileOperations::new(&[FileOperation::Get, FileOperation::Put, FileOperation::Head]) + .unwrap(), + max_request_bytes: 8 * 1024 * 1024, + max_file_bytes: 8 * 1024 * 1024, + }) + .unwrap(); + let file_client = signed::TestFileClient { + client, + credentials, + address: iceberg_addr, + }; + for (name, bytes) in [ + ("small-a", vec![0x41; 800 * 1024]), + ("small-b", vec![0x43; 200 * 1024]), + ("large", vec![0x42; 4 * 1024 * 1024]), + ] { + let path = format!( + "/{}/{}", + table.bucket(), + table.file(&format!("data/{name}.bin")).unwrap().object_key() + ); + let put = file_client.send(Method::PUT, &path, "", &bytes, true).await; + assert_eq!(put.status(), 200, "{}", put.text().await.unwrap()); + let get = file_client.send(Method::GET, &path, "", b"", false).await; + assert_eq!(get.status(), 200); + assert_eq!(get.bytes().await.unwrap().as_ref(), bytes); + } + chunks.shutdown_small_writes().await.unwrap(); +} + +async fn assert_chunk_layouts(cluster: &ConsoleConfig) { + let transport = ChunkdbRpcTransport::new(); + let mut saw = [false; 4]; + for server in cluster + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Chunkdb) + { + let listed = transport + .send_list_chunks( + server.rpc_url.as_deref().unwrap(), + &ListChunksRequest { + max_keys: 1_024, + ..ListChunksRequest::default() + }, + ) + .await + .unwrap(); + for chunk in listed.chunks { + let is_s3 = chunk.chunk_type == ChunkType::S3 as i32; + let is_iceberg = chunk.chunk_type == ChunkType::IcebergTable as i32; + if !is_s3 && !is_iceberg { + continue; + } + assert_eq!( + chunk.id.unwrap().high >> 56, + u64::try_from(chunk.chunk_type).unwrap() + ); + for strip in chunk.strips { + match strip.strip.unwrap() { + Strip::MirrorStrip(mirror) => { + assert_eq!(mirror.segments.len(), 2); + saw[if is_s3 { 0 } else { 2 }] = true; + } + Strip::EcStrip(ec) => { + assert_eq!((ec.data_num, ec.code_num), if is_s3 { (2, 1) } else { (4, 2) }); + saw[if is_s3 { 1 } else { 3 }] = true; + } + } + } + } + } + assert!( + saw.into_iter().all(|seen| seen), + "both protocols must write small mirror and large EC strips" + ); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "stops all real DiskIO processes in a simulated three-rack production cluster"] +async fn protected_two_copy_write_stops_after_repair_and_chunk_rotation_fail() { + let dir = TestDir::new("access-protected-mirror-failure").unwrap(); + s3::start_protected_test_cluster(dir.path()).await.unwrap(); + let _cleanup = StopClusterOnDrop(dir.path()); + let (cluster, _) = s3::load(dir.path()).unwrap(); + let seeds = cluster + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| server.url.clone()) + .collect(); + let client = ChunkIoClient::connect(ChunkIoClientConfig { + management_seeds: seeds, + diskio_connections_per_endpoint: 2, + diskio_rpc_workers: 1, + small_write: SmallWritePolicy { + chunk_type: ChunkType::S3, + conversion_enabled: false, + mirror_copies: 2, + ..SmallWritePolicy::default() + }, + }) + .await + .unwrap(); + for server in cluster + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Diskio) + { + lifecycle::stop_pid_with_timeout(server.pid.unwrap(), Duration::from_secs(5)).unwrap(); + } + let result = tokio::time::timeout(Duration::from_secs(15), async { + let data = hyper::body::Bytes::from_static(b"failed-protected-write"); + let mut writer = client.prepare_small_write(data.len()).await?; + writer.on_data(data).await?; + writer.on_finish().await.map(|_| ()) + }) + .await + .expect("failed DiskIO write exceeded the 15-second fault budget"); + assert!(matches!(result, Err(IoError::WriteFailed(_))), "{result:?}"); + let metrics = client.small_write_metrics(); + assert_eq!(metrics.completed, 0); + assert_eq!(metrics.failed, 1); + assert_eq!(metrics.exhausted_repairs, 2); + assert_eq!(metrics.repairs_avoiding_rotation, 0); + let _ = client.shutdown_small_writes().await; +} diff --git a/app/crowdb-access-server/tests/protocol_production_policy_test.rs b/app/crowdb-access-server/tests/protocol_production_policy_test.rs new file mode 100644 index 000000000..30cbe5777 --- /dev/null +++ b/app/crowdb-access-server/tests/protocol_production_policy_test.rs @@ -0,0 +1,212 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::path::Path; +use std::sync::Arc; + +use crowdb_access_iceberg::storage::{self as iceberg_storage, IcebergLargeWriteSettings}; +use crowdb_access_s3::storage::{S3LargeWriteSettings, S3StorageClients, S3WritePolicies, S3WriteSettings}; +use crowdb_chunk_client::{ + ChunkIoClient, ChunkIoWriter, ChunkReadPolicy, LargeWritePolicy, SmallWritePolicy, +}; +use crowdb_chunkdb_client::{ChunkdbClient, ChunkdbRpcTransport}; +use crowdb_console_shared::{config::ServiceType, ops::s3}; +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, ServiceRegistryClient}; +use crowdb_protocol::chunkdb::rpc::{ChunkType, Location, QueryChunkRequest, Strip}; +use crowdb_test_harness::test_dirs::TestDir; +use hyper::body::Bytes; + +const MIB: usize = 1024 * 1024; + +struct StopClusterOnDrop<'a>(&'a Path); + +impl Drop for StopClusterOnDrop<'_> { + fn drop(&mut self) { + let _ = s3::stop(self.0); + } +} + +async fn write_small(client: &ChunkIoClient, payload: Bytes) -> Location { + let mut writer = client.prepare_small_write(payload.len()).await.unwrap(); + writer.on_data(payload).await.unwrap(); + writer.on_finish().await.unwrap().remove(0) +} + +fn production_policies() -> (S3WritePolicies, LargeWritePolicy, SmallWritePolicy) { + let s3_small = SmallWritePolicy { + conversion_enabled: false, + mirror_copies: 2, + memory_budget: 64 * MIB, + ..SmallWritePolicy::default() + }; + let s3 = S3WriteSettings { + small: s3_small, + threshold_ratio: 0.5, + disk_block_bytes: MIB, + ec_data: 2, + ec_code: 1, + large: S3LargeWriteSettings { + max_chunk_size: Some(8 * MIB as u64), + memory_budget_bytes: Some(64 * MIB), + prefetch_strips_per_chunk: Some(2), + ..S3LargeWriteSettings::default() + }, + } + .policies() + .unwrap(); + let iceberg_large = IcebergLargeWriteSettings { + ec_data: 4, + ec_code: 2, + disk_block_bytes: MIB, + mirror_copies: None, + max_chunk_size: Some(16 * MIB as u64), + memory_budget_bytes: Some(96 * MIB), + prefetch_strips_per_chunk: Some(3), + prefetch_max_strips_per_batch: None, + parallel_strip_writes: None, + held_buffers: None, + chunk_preparation_depth: Some(1), + } + .policy() + .unwrap(); + let iceberg_small = SmallWritePolicy { + conversion_enabled: false, + mirror_copies: 2, + memory_budget: 96 * MIB, + ..SmallWritePolicy::default() + }; + (s3, iceberg_large, iceberg_small) +} + +async fn assert_chunk_layouts( + seeds: Vec, + s3_small: Location, + iceberg_small: Location, + s3_large: Location, + iceberg_large: Location, +) { + let registry = ServiceRegistryClient::new(CrowdbKvClient::new(ClientConfig::new(seeds))); + let chunkdb = ChunkdbClient::new(registry, Arc::new(ChunkdbRpcTransport::new())); + for (location, expected_type, expected_ec) in [ + (s3_small, ChunkType::S3, None), + (iceberg_small, ChunkType::IcebergTable, None), + (s3_large, ChunkType::S3, Some((2, 1))), + (iceberg_large, ChunkType::IcebergTable, Some((4, 2))), + ] { + let chunk = chunkdb + .query_chunk(QueryChunkRequest { + chunk_id: location.chunk_id, + }) + .await + .unwrap() + .chunk + .unwrap(); + assert_eq!(chunk.chunk_type, expected_type as i32); + assert_eq!(chunk.id.unwrap().high >> 56, expected_type as u64); + let strip = &chunk.strips[0]; + match (strip.strip.as_ref().unwrap(), expected_ec) { + (Strip::MirrorStrip(mirror), None) => assert_eq!(mirror.segments.len(), 2), + (Strip::EcStrip(ec), Some((data, code))) => { + assert_eq!((ec.data_num, ec.code_num), (data, code)); + } + _ => panic!("protocol chunk used a different protection policy"), + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +#[ignore = "starts a complete simulated three-rack production storage stack"] +async fn s3_and_iceberg_keep_distinct_policies_on_protected_storage() { + let dir = TestDir::new("access-production-protocol-policy").unwrap(); + s3::start_protected_test_cluster(dir.path()).await.unwrap(); + let _cleanup = StopClusterOnDrop(dir.path()); + let (config, _) = s3::load(dir.path()).unwrap(); + let seeds = config + .servers + .iter() + .filter(|server| server.service_type == ServiceType::Kv) + .map(|server| server.url.clone()) + .collect::>(); + + let (s3_policies, iceberg_large, iceberg_small) = production_policies(); + assert_ne!(s3_policies.large.ec_scheme, iceberg_large.ec_scheme); + assert_ne!( + s3_policies.large.client.prefetch_strips_per_chunk, + iceberg_large.client.prefetch_strips_per_chunk + ); + assert_ne!( + s3_policies.large.client.memory_budget, + iceberg_large.client.memory_budget + ); + + let s3_storage = S3StorageClients::connect_with_read_policy( + seeds.clone(), + 2, + 1, + s3_policies.small, + ChunkReadPolicy::default(), + ) + .await + .unwrap(); + let (_, _, iceberg_chunks) = + iceberg_storage::connect(seeds.clone(), ChunkReadPolicy::default(), iceberg_small, 2, 1) + .await + .unwrap(); + let s3_chunks = Arc::clone(&s3_storage.chunks); + let (s3_small, iceberg_small) = tokio::join!( + write_small(&s3_chunks, Bytes::from_static(b"s3-small")), + write_small(&iceberg_chunks, Bytes::from_static(b"iceberg-small")) + ); + let s3_data = vec![0x31; 2 * MIB]; + let iceberg_data = vec![0x42; 4 * MIB]; + let (s3_large, iceberg_large_result) = tokio::join!( + s3_chunks + .prepare_large_write(Some(s3_data.len() as u64), s3_policies.large) + .write_stream(s3_data.as_slice()), + iceberg_chunks + .prepare_large_write(Some(iceberg_data.len() as u64), iceberg_large) + .write_stream(iceberg_data.as_slice()) + ); + let s3_large = s3_large.unwrap(); + let iceberg_large_result = iceberg_large_result.unwrap(); + assert_eq!( + s3_chunks + .read_object(std::slice::from_ref(&s3_small)) + .await + .unwrap() + .concat(), + b"s3-small" + ); + assert_eq!( + iceberg_chunks + .read_object(std::slice::from_ref(&iceberg_small)) + .await + .unwrap() + .concat(), + b"iceberg-small" + ); + assert_eq!( + s3_chunks.read_object(&s3_large.locations).await.unwrap().concat(), + s3_data + ); + assert_eq!( + iceberg_chunks + .read_object(&iceberg_large_result.locations) + .await + .unwrap() + .concat(), + iceberg_data + ); + + assert_chunk_layouts( + seeds, + s3_small, + iceberg_small, + s3_large.locations[0].clone(), + iceberg_large_result.locations[0].clone(), + ) + .await; + s3_chunks.shutdown_small_writes().await.unwrap(); + iceberg_chunks.shutdown_small_writes().await.unwrap(); + s3::delete(dir.path()).unwrap(); +} diff --git a/app/crowdb-access-server/tests/s3_full_stack_test.rs b/app/crowdb-access-server/tests/s3_full_stack_test.rs index 7c7f2176a..53da26bca 100644 --- a/app/crowdb-access-server/tests/s3_full_stack_test.rs +++ b/app/crowdb-access-server/tests/s3_full_stack_test.rs @@ -130,6 +130,15 @@ fn main() { async fn run_suite() { let mut stack = start_full_stack().await; + if let Ok(method) = std::env::var("CROWDB_S3_E2E_ONLY") { + assert!( + BOTO3_CASES.contains(&method.as_str()), + "unknown focused S3 case: {method}" + ); + stack.run_one_boto3_case(&method); + stack.rpc.stop(); + return; + } println!("\nrunning {TEST_COUNT} tests"); stack.run_boto3_cases(); stack.run_restart_cases().await; @@ -139,6 +148,20 @@ async fn run_suite() { } impl FullStackSetup { + fn run_one_boto3_case(&self, method: &str) { + let context = Boto3CaseContext { + listen: &self.listen, + second_listen: &self.second_listen, + access_key: &self.access_key, + secret_key: &self.secret_key, + access_server: self.access_server.as_ref().expect("primary access server"), + chunk_kv: &self.chunk_kv, + }; + let case = TestCase::start(&format!("boto3::{method}")); + run_boto3_case(method, &context); + case.pass(); + } + fn run_boto3_cases(&self) { let context = Boto3CaseContext { listen: &self.listen, @@ -531,6 +554,7 @@ fn start_access_server( let log_path = service_root.join("log").join("access-server.log"); let log = std::fs::File::create(&log_path).expect("create access-server log"); let child = Command::new(access_binary) + .arg("s3") .env("CROWDB_S3_LISTEN", &listen) .env("CROWDB_MANAGEMENT_SEEDS", seeds) .env("CROWDB_S3_TENANT", "boto3-e2e") diff --git a/app/crowdb-chunk-kv-server/src/config.rs b/app/crowdb-chunk-kv-server/src/config.rs index f197667bf..af8754e64 100644 --- a/app/crowdb-chunk-kv-server/src/config.rs +++ b/app/crowdb-chunk-kv-server/src/config.rs @@ -167,6 +167,8 @@ pub struct StorageConfig { pub metadata_store_id: u64, pub stream_writer_lease_ms: u64, pub stream_mirror_copies: u32, + pub tree_chunk_capacity_bytes: u64, + pub stream_chunk_capacity_bytes: u64, pub diskio_connections_per_endpoint: usize, pub diskio_rpc_workers: u32, } @@ -176,7 +178,9 @@ impl Default for StorageConfig { Self { metadata_store_id: 1, stream_writer_lease_ms: 30_000, - stream_mirror_copies: 3, + stream_mirror_copies: 2, + tree_chunk_capacity_bytes: 256 * 1024 * 1024, + stream_chunk_capacity_bytes: 256 * 1024 * 1024, diskio_connections_per_endpoint: 1, diskio_rpc_workers: 2, } @@ -187,11 +191,15 @@ impl StorageConfig { fn validate(&self) -> Result<(), ConfigError> { if self.stream_writer_lease_ms == 0 || self.stream_mirror_copies == 0 + || self.stream_mirror_copies > 5 + || !(1024 * 1024..=256 * 1024 * 1024).contains(&self.tree_chunk_capacity_bytes) + || !(1024 * 1024..=256 * 1024 * 1024).contains(&self.stream_chunk_capacity_bytes) || self.diskio_connections_per_endpoint == 0 || self.diskio_rpc_workers == 0 { return Err(ConfigError::Invalid( - "storage lease, connections, and workers must be nonzero".into(), + "storage lease, mirror copies, connections, workers, and tree chunk capacity must be valid" + .into(), )); } Ok(()) diff --git a/app/crowdb-chunk-kv-server/src/storage.rs b/app/crowdb-chunk-kv-server/src/storage.rs index 5813d0535..3faecebfa 100644 --- a/app/crowdb-chunk-kv-server/src/storage.rs +++ b/app/crowdb-chunk-kv-server/src/storage.rs @@ -47,6 +47,8 @@ pub struct ChunkKvStorage { chunk_io: ChunkIoClient, streams: Arc, tree_transport: Arc, + tree_mirror_copies: u32, + tree_chunk_capacity_bytes: u64, metadata_store_id: u64, } @@ -72,12 +74,14 @@ impl ChunkKvStorage { ) .await .map_err(|error| StorageRuntimeError::ChunkIo(error.to_string()))?; - Self::from_parts( + Self::from_parts_with_mirror_copies( kv, chunk_io, config.storage.metadata_store_id, config.storage.stream_writer_lease_ms, config.storage.stream_mirror_copies, + config.storage.tree_chunk_capacity_bytes, + config.storage.stream_chunk_capacity_bytes, ) .await } @@ -88,19 +92,34 @@ impl ChunkKvStorage { metadata_store_id: u64, writer_lease_ms: u64, stream_mirror_copies: u32, + tree_chunk_capacity_bytes: u64, + stream_chunk_capacity_bytes: u64, ) -> Result { + let stream_config = StreamConfig { + chunk_capacity_bytes: stream_chunk_capacity_bytes, + ..StreamConfig::default() + }; let streams = Arc::new( ProductionStreamRuntime::new_with_mirror_copies( Arc::clone(&kv), &chunk_io, writer_lease_ms, ChunkReadPolicy::default(), - StreamConfig::default(), + stream_config, stream_mirror_copies, ) .map_err(|error| StorageRuntimeError::Stream(error.to_string()))?, ); - Self::assemble(kv, chunk_io, streams, metadata_store_id, writer_lease_ms).await + Self::assemble( + kv, + chunk_io, + streams, + metadata_store_id, + writer_lease_ms, + stream_mirror_copies, + tree_chunk_capacity_bytes, + ) + .await } /// Assembles production adapters from already connected process clients. @@ -121,6 +140,8 @@ impl ChunkKvStorage { metadata_store_id, writer_lease_ms, stream_mirror_copies, + 256 * 1024 * 1024, + 256 * 1024 * 1024, ) .await } @@ -131,6 +152,8 @@ impl ChunkKvStorage { streams: Arc, metadata_store_id: u64, writer_lease_ms: u64, + mirror_copies: u32, + tree_chunk_capacity_bytes: u64, ) -> Result { let (chunkdb, disks) = chunk_io .native_storage_routes() @@ -151,6 +174,7 @@ impl ChunkKvStorage { writer_lease_ms, rpc_timeout_ms: writer_lease_ms, completion_capacity: 1_024, + mirror_copies, }) .map_err(|error| StorageRuntimeError::Tree(error.to_string()))?, ); @@ -159,6 +183,8 @@ impl ChunkKvStorage { chunk_io, streams, tree_transport, + tree_mirror_copies: mirror_copies, + tree_chunk_capacity_bytes, metadata_store_id, }) } @@ -187,9 +213,11 @@ impl ChunkKvStorage { /// Returns an invalid native page-store or transport error. pub fn open_tree_page_store( &self, - options: ChunkPageStoreOptions, + mut options: ChunkPageStoreOptions, catalog: Arc, ) -> Result, StorageRuntimeError> { + options.mirror_copies = self.tree_mirror_copies; + options.max_chunk_bytes = self.tree_chunk_capacity_bytes; PageStore::open_chunk(options, catalog, Some(&self.tree_transport)) .map(Arc::new) .map_err(|error| StorageRuntimeError::Tree(error.to_string())) @@ -298,6 +326,8 @@ impl ChunkKvStorage { iu_size: 0, max_concurrent_packs: 0, materialization_bytes_per_pass: 0, + mirror_copies: 0, + max_chunk_bytes: 0, }, binding.metadata_group_id, ) @@ -441,6 +471,8 @@ impl ChunkKvStorage { iu_size: 0, max_concurrent_packs: 0, materialization_bytes_per_pass: 0, + mirror_copies: 0, + max_chunk_bytes: 0, }, config.metadata_group_id, ) @@ -636,6 +668,8 @@ impl ChunkKvStorage { iu_size: 0, max_concurrent_packs: 0, materialization_bytes_per_pass: 0, + mirror_copies: 0, + max_chunk_bytes: 0, }, metadata_group_id, ) diff --git a/app/crowdb-chunk-kv-server/tests/config_test.rs b/app/crowdb-chunk-kv-server/tests/config_test.rs index e4e5c87bb..ccd016bea 100644 --- a/app/crowdb-chunk-kv-server/tests/config_test.rs +++ b/app/crowdb-chunk-kv-server/tests/config_test.rs @@ -22,6 +22,8 @@ fn defaults_close_the_documented_timing_contract() { assert_eq!(config.max_split_catchup_lag_records, 1_024); assert_eq!(config.storage.metadata_store_id, 1); assert_eq!(config.storage.stream_writer_lease_ms, 30_000); + assert_eq!(config.storage.tree_chunk_capacity_bytes, 256 * 1024 * 1024); + assert_eq!(config.storage.stream_chunk_capacity_bytes, 256 * 1024 * 1024); assert_eq!(config.storage.diskio_connections_per_endpoint, 1); assert_eq!(config.storage.diskio_rpc_workers, 2); assert_eq!(config.rpc_workers, 2); @@ -68,4 +70,10 @@ fn invalid_identity_address_and_capacity_fail_closed() { config.storage.metadata_store_id = 0; config.storage.stream_writer_lease_ms = 0; assert!(config.validate().is_err()); + config.storage.stream_writer_lease_ms = 30_000; + config.storage.tree_chunk_capacity_bytes = 257 * 1024 * 1024; + assert!(config.validate().is_err()); + config.storage.tree_chunk_capacity_bytes = 256 * 1024 * 1024; + config.storage.stream_chunk_capacity_bytes = 257 * 1024 * 1024; + assert!(config.validate().is_err()); } diff --git a/app/crowdb-chunkdb/conf/crowdb_chunkdb_config.toml b/app/crowdb-chunkdb/conf/crowdb_chunkdb_config.toml index 56c1d506d..dacc0e8ec 100644 --- a/app/crowdb-chunkdb/conf/crowdb_chunkdb_config.toml +++ b/app/crowdb-chunkdb/conf/crowdb_chunkdb_config.toml @@ -1,5 +1,14 @@ # Canonical crowdb-chunkdb startup configuration. # Omitted fields use the typed defaults documented in chunkdb_config.rs. +[deployment] +mode = "production" +max_node_failures = 1 + [server] rpc_workers = 2 + +[conversion_io] +normal_connections_per_endpoint = 1 +priority_connections_per_endpoint = 1 +rpc_workers = 2 diff --git a/app/crowdb-chunkdb/src/chunkdb_config.rs b/app/crowdb-chunkdb/src/chunkdb_config.rs index 5b1221726..5bc65ef7a 100644 --- a/app/crowdb-chunkdb/src/chunkdb_config.rs +++ b/app/crowdb-chunkdb/src/chunkdb_config.rs @@ -19,9 +19,37 @@ pub enum PlacementMode { UnsafeColocated, } +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum DeploymentMode { + #[default] + Production, + TestSingleNode, + /// Legacy colocated EC fixtures; rejected by release builds. + TestUnsafePlacement, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(default)] +pub struct DeploymentConfig { + pub mode: DeploymentMode, + pub max_node_failures: u32, +} + +impl Default for DeploymentConfig { + fn default() -> Self { + Self { + mode: DeploymentMode::Production, + max_node_failures: 1, + } + } +} + /// Top-level configuration for a chunkdb instance. #[derive(Debug, Clone, Default, Serialize, Deserialize)] pub struct ChunkdbConfig { + #[serde(default)] + pub deployment: DeploymentConfig, #[serde(default)] pub server: ServerConfig, #[serde(default)] @@ -42,6 +70,27 @@ pub struct ChunkdbConfig { pub placement_rebalance: PlacementRebalanceConfig, #[serde(default)] pub reservation: ReservationConfig, + #[serde(default)] + pub conversion_io: ConversionIoConfig, +} + +/// DiskIO transport used by background conversion and repair. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(default)] +pub struct ConversionIoConfig { + pub normal_connections_per_endpoint: usize, + pub priority_connections_per_endpoint: usize, + pub rpc_workers: u32, +} + +impl Default for ConversionIoConfig { + fn default() -> Self { + Self { + normal_connections_per_endpoint: 1, + priority_connections_per_endpoint: 1, + rpc_workers: 2, + } + } } /// Placement safety policy. @@ -60,6 +109,35 @@ pub struct PlacementConfig { impl BaseConfig for ChunkdbConfig { fn validate(&self) -> Result<(), String> { + match self.deployment.mode { + DeploymentMode::Production => { + if self.deployment.max_node_failures != 1 { + return Err("production requires max_node_failures = 1".into()); + } + if self.placement.mode != PlacementMode::Protected + || self.placement.allow_unsafe_ec + || self.placement.allow_degraded_failure_domains + { + return Err("production deployment requires protected placement".into()); + } + } + DeploymentMode::TestSingleNode => { + if self.deployment.max_node_failures != 0 { + return Err("test_single_node requires max_node_failures = 0".into()); + } + if self.placement.mode != PlacementMode::UnsafeColocated { + return Err("test_single_node requires explicit unsafe_colocated placement".into()); + } + if self.conversion.enabled { + return Err("test_single_node must disable mirror-to-EC conversion".into()); + } + } + DeploymentMode::TestUnsafePlacement => { + if !cfg!(debug_assertions) { + return Err("test_unsafe_placement is unavailable in release builds".into()); + } + } + } if self.server.rpc_workers == 0 { return Err("server.rpc_workers must be > 0".into()); } @@ -94,6 +172,12 @@ impl BaseConfig for ChunkdbConfig { self.placement_repair.validate()?; self.placement_rebalance.validate()?; self.reservation.validate()?; + if self.conversion_io.normal_connections_per_endpoint == 0 + || self.conversion_io.priority_connections_per_endpoint == 0 + || self.conversion_io.rpc_workers == 0 + { + return Err("conversion_io connections and RPC workers must be > 0".into()); + } Ok(()) } } diff --git a/app/crowdb-chunkdb/src/conversion.rs b/app/crowdb-chunkdb/src/conversion.rs index ea9b7b783..3179105db 100644 --- a/app/crowdb-chunkdb/src/conversion.rs +++ b/app/crowdb-chunkdb/src/conversion.rs @@ -72,6 +72,7 @@ pub enum ConversionError { /// Coordinates client-side no-reread conversion with durable task takeover. pub struct ConversionCoordinator { + enabled: bool, lifecycle: Arc, tasks: Arc, wake: Option>, @@ -451,6 +452,7 @@ impl ConversionCoordinator { #[must_use] pub fn new(lifecycle: Arc, tasks: Arc) -> Self { Self { + enabled: true, lifecycle, tasks, wake: None, @@ -469,6 +471,12 @@ impl ConversionCoordinator { self } + #[must_use] + pub fn with_enabled(mut self, enabled: bool) -> Self { + self.enabled = enabled; + self + } + #[must_use] pub fn with_policy( mut self, @@ -485,6 +493,9 @@ impl ConversionCoordinator { } pub async fn reconcile_reservations(&self, max_groups: u32, now_ms: u64) -> Result { + if !self.enabled { + return Ok(0); + } let cursor = self.reservation_scan_cursor.load_full(); let groups = self .lifecycle @@ -544,6 +555,11 @@ impl ConversionCoordinator { claim_lease_ms: u64, now_ms: u64, ) -> Result { + if !self.enabled { + return Err(ConversionError::Payload( + "conversion is disabled by deployment mode".into(), + )); + } let chunk = self.lifecycle.query_chunk(&chunk_id).await?; validate_source(&chunk, expected_modify_ts, start_index, &old_strips)?; let task_id = conversion_task_id(&old_strips, data_num, code_num)?; @@ -655,6 +671,11 @@ impl ConversionCoordinator { client_owner: u64, now_ms: u64, ) -> Result { + if !self.enabled { + return Err(ConversionError::Payload( + "conversion is disabled by deployment mode".into(), + )); + } let task = self .tasks .get(&chunk_id, TASK_KIND_MIRROR_TO_EC, &task_id) @@ -706,6 +727,11 @@ impl ConversionCoordinator { code_num: u32, now_ms: u64, ) -> Result { + if !self.enabled { + return Err(ConversionError::Payload( + "conversion is disabled by deployment mode".into(), + )); + } let chunk = self.lifecycle.query_chunk(&chunk_id).await?; self.admit_groups(&chunk, data_num, code_num, now_ms, false, 0) .await @@ -749,6 +775,11 @@ impl ConversionCoordinator { min_age_ms: u64, now_ms: u64, ) -> Result { + if !self.enabled { + return Err(ConversionError::Payload( + "conversion is disabled by deployment mode".into(), + )); + } let limit = if max_chunks == 0 { 256 } else { max_chunks }; let start_after = self.scan_cursor.load_full(); let chunks = self.lifecycle.list_chunks(start_after.as_deref(), limit).await?; diff --git a/app/crowdb-chunkdb/src/conversion/io.rs b/app/crowdb-chunkdb/src/conversion/io.rs index f9222e9ae..57dd6eb71 100644 --- a/app/crowdb-chunkdb/src/conversion/io.rs +++ b/app/crowdb-chunkdb/src/conversion/io.rs @@ -5,11 +5,14 @@ use std::sync::Arc; +use arc_swap::ArcSwapOption; use bytes::Bytes; use crowdb_diskio_client::{DiskId, DiskioClient, DiskioClientConfig, Durability, SegmentTarget}; use crowdb_kv_client::{HardwareClient, ServiceRegistryClient}; use crowdb_protocol::diskdb::rpc::Segment; +use crate::chunkdb_config::ConversionIoConfig; + #[derive(Debug, thiserror::Error)] pub enum ConversionIoError { #[error("conversion DiskIO topology error: {0}")] @@ -20,54 +23,76 @@ pub enum ConversionIoError { /// Conversion-specific policy adapter over the shared semantic client. pub struct ConversionDiskIo { - client: Option>, + client: ArcSwapOption, + config: ConversionIoConfig, } impl ConversionDiskIo { + pub fn deferred(config: ConversionIoConfig) -> Self { + Self { + client: ArcSwapOption::empty(), + config, + } + } + #[cfg(feature = "test-util")] #[must_use] pub fn empty_for_tests() -> Self { - Self { client: None } + Self::deferred(ConversionIoConfig::default()) } pub async fn connect( service: &ServiceRegistryClient, hardware: &HardwareClient, ) -> Result { + Self::connect_with_config(service, hardware, &ConversionIoConfig::default()).await + } + + pub async fn connect_with_config( + service: &ServiceRegistryClient, + hardware: &HardwareClient, + config: &ConversionIoConfig, + ) -> Result { + let io = Self::deferred(config.clone()); + io.refresh(service, hardware).await?; + Ok(io) + } + + pub async fn refresh( + &self, + service: &ServiceRegistryClient, + hardware: &HardwareClient, + ) -> Result<(), ConversionIoError> { + if let Some(client) = self.client.load_full() { + return client + .refresh() + .await + .map(|_| ()) + .map_err(|error| ConversionIoError::Topology(error.to_string())); + } let client = DiskioClient::connect_with_clients( service.clone(), hardware.clone(), DiskioClientConfig { - normal_connections_per_endpoint: 1, - priority_connections_per_endpoint: 1, + normal_connections_per_endpoint: self.config.normal_connections_per_endpoint, + priority_connections_per_endpoint: self.config.priority_connections_per_endpoint, + rpc_workers: self.config.rpc_workers, ..DiskioClientConfig::default() }, ) .await .map_err(|error| ConversionIoError::Topology(error.to_string()))?; - Ok(Self { - client: Some(Arc::new(client)), - }) - } - - pub async fn refresh( - &self, - _service: &ServiceRegistryClient, - _hardware: &HardwareClient, - ) -> Result<(), ConversionIoError> { - self.client()? - .refresh() - .await - .map(|_| ()) - .map_err(|error| ConversionIoError::Topology(error.to_string())) + self.client.store(Some(Arc::new(client))); + Ok(()) } pub async fn read_segment(&self, segment: &Segment, unit_bytes: u64) -> Result { let target = target(segment, unit_bytes)?; let length = u32::try_from(target.capacity()) .map_err(|_| ConversionIoError::Io("segment read size exceeds u32".into()))?; - self.client()? - .read(target, 0, length, self.client()?.normal_options().priority()) + let client = self.client()?; + client + .read(target, 0, length, client.normal_options().priority()) .await .map_err(|error| ConversionIoError::Io(error.to_string())) } @@ -82,15 +107,16 @@ impl ConversionDiskIo { length: u32, ) -> Result { #[cfg(feature = "test-util")] - if self.client.is_none() { + if self.client.load().is_none() { return Ok(Bytes::from(vec![ 0; usize::try_from(length).expect("u32 fits usize") ])); } let target = target(segment, unit_bytes)?; - self.client()? - .read(target, offset, length, self.client()?.normal_options().priority()) + let client = self.client()?; + client + .read(target, offset, length, client.normal_options().priority()) .await .map_err(|error| ConversionIoError::Io(error.to_string())) } @@ -102,13 +128,14 @@ impl ConversionDiskIo { data: Bytes, ) -> Result<(), ConversionIoError> { let target = target(segment, unit_bytes)?; - self.client()? + let client = self.client()?; + client .write( target, 0, data, Durability::Buffered, - self.client()?.normal_options().priority(), + client.normal_options().priority(), ) .await .map_err(|error| ConversionIoError::Io(error.to_string())) @@ -116,16 +143,17 @@ impl ConversionDiskIo { pub async fn fsync_segment(&self, segment: &Segment) -> Result<(), ConversionIoError> { let disk_id = disk_id(segment)?; - self.client()? - .fsync(disk_id, self.client()?.normal_options().priority()) + let client = self.client()?; + client + .fsync(disk_id, client.normal_options().priority()) .await .map_err(|error| ConversionIoError::Io(error.to_string())) } - fn client(&self) -> Result<&DiskioClient, ConversionIoError> { + fn client(&self) -> Result, ConversionIoError> { self.client - .as_deref() - .ok_or_else(|| ConversionIoError::Topology("test DiskIO client is not connected".into())) + .load_full() + .ok_or_else(|| ConversionIoError::Topology("background DiskIO client is not connected".into())) } } diff --git a/app/crowdb-chunkdb/src/lifecycle/handler.rs b/app/crowdb-chunkdb/src/lifecycle/handler.rs index d7e30515d..3802be73c 100644 --- a/app/crowdb-chunkdb/src/lifecycle/handler.rs +++ b/app/crowdb-chunkdb/src/lifecycle/handler.rs @@ -127,6 +127,7 @@ pub use reservation::{ /// Lifecycle handler — orchestrates allocate/append/seal/delete/query/list. pub struct LifecycleHandler { + deployment_mode: Option, store: Arc, allocator: Arc, topology: TopologyCache, @@ -187,6 +188,7 @@ impl LifecycleHandler { #[must_use] pub fn new(store: Arc, allocator: Arc, topology: TopologyCache) -> Self { Self { + deployment_mode: None, store, allocator, topology, @@ -236,6 +238,99 @@ impl LifecycleHandler { self } + #[must_use] + pub fn with_deployment_mode(mut self, mode: crate::chunkdb_config::DeploymentMode) -> Self { + self.deployment_mode = Some(mode); + self + } + + fn validate_strip_layout( + &self, + strip_type: ProtoStripType, + data_num: u32, + code_num: u32, + copy_count: u32, + capacity_kb: u32, + ) -> Result<(), LifecycleError> { + use crate::chunkdb_config::DeploymentMode; + if strip_type == ProtoStripType::Mirror && copy_count > 5 { + return Err(LifecycleError::InvalidRequest( + "mirror strips support at most five copies".into(), + )); + } + match self.deployment_mode { + Some(DeploymentMode::TestSingleNode) => { + if strip_type != ProtoStripType::Mirror + || copy_count != 1 + || data_num != 0 + || code_num != 0 + || capacity_kb != 1024 + { + return Err(LifecycleError::InvalidRequest( + "test_single_node requires 1 MiB mirror strips with one copy each".into(), + )); + } + } + Some(DeploymentMode::Production) if strip_type == ProtoStripType::Mirror && copy_count < 2 => { + return Err(LifecycleError::InvalidRequest( + "production mirror strips require at least two copies".into(), + )); + } + Some(DeploymentMode::Production) => {} + Some(DeploymentMode::TestUnsafePlacement) => {} + None => {} + } + Ok(()) + } + + fn validate_replacement_layouts(&self, strips: &[ChunkStrip]) -> Result<(), LifecycleError> { + for strip in strips { + match strip.strip.as_ref() { + Some(Strip::MirrorStrip(mirror)) => self.validate_strip_layout( + ProtoStripType::Mirror, + 0, + 0, + u32::try_from(mirror.segments.len()).unwrap_or(u32::MAX), + strip.capacity, + )?, + Some(Strip::EcStrip(ec)) => self.validate_strip_layout( + ProtoStripType::Ec, + ec.data_num, + ec.code_num, + 0, + strip.capacity, + )?, + None => { + return Err(LifecycleError::InvalidRequest( + "replacement strip has no body".into(), + )) + } + } + } + Ok(()) + } + + fn protected_degraded_layout( + &self, + strip_type: ProtoStripType, + snapshot: &crate::topology::TopologySnapshot, + ) -> Option { + use std::collections::HashSet; + + if self.deployment_mode != Some(crate::chunkdb_config::DeploymentMode::Production) { + return None; + } + if strip_type != ProtoStripType::Mirror { + return None; + } + let healthy_nodes: HashSet<_> = snapshot + .healthy_disk_groups() + .into_iter() + .map(|group| group.node_id) + .collect(); + (healthy_nodes.len() == 2).then_some(StripAllocType::Mirror { copy_count: 2 }) + } + /// Attach the persistent task store used for foreground degraded EC admission. #[must_use] pub fn with_placement_tasks(mut self, tasks: Arc) -> Self { @@ -347,6 +442,7 @@ impl LifecycleHandler { writer_lease_ms: u64, owner_key: Vec, ) -> Result { + self.validate_strip_layout(strip_type, data_num, code_num, copy_count, write_granularity_kb)?; if !chunk_owner_key_matches_type(chunk_type, &owner_key) { return Err(LifecycleError::InvalidRequest( "chunk owner key does not match chunk type".into(), @@ -356,6 +452,11 @@ impl LifecycleHandler { Some(id) => id, None => self.generate_owned_chunk_id(chunk_type)?, }; + if (id.high >> 56) != u64::from(chunk_type as u8) { + return Err(LifecycleError::InvalidRequest( + "chunk id prefix does not match chunk type".into(), + )); + } self.check_range(&id)?; let mut allocation_guard = AllocationMetricGuard::new(self.metrics.clone()); @@ -382,18 +483,20 @@ impl LifecycleHandler { let snap = self.topology.snapshot(); - let mirror_copies = if copy_count == 0 { 3 } else { copy_count as usize }; - let strip_alloc_type = match strip_type { - ProtoStripType::Mirror => StripAllocType::Mirror { - copy_count: mirror_copies, - }, - ProtoStripType::Ec => StripAllocType::Ec { - data_num: data_num as usize, - code_num: code_num as usize, - }, - }; + let mirror_copies = if copy_count == 0 { 2 } else { copy_count as usize }; + let strip_alloc_type = + self.protected_degraded_layout(strip_type, &snap) + .unwrap_or(match strip_type { + ProtoStripType::Mirror => StripAllocType::Mirror { + copy_count: mirror_copies, + }, + ProtoStripType::Ec => StripAllocType::Ec { + data_num: data_num as usize, + code_num: code_num as usize, + }, + }); - let constraints = self.placement_constraints(); + let constraints = self.allocation_constraints(strip_type, &snap); // Convert write_granularity (KB) to unit_count using the unit // size from the topology snapshot. Fall back to treating KB as // units if unit_size_bytes is unavailable (0). @@ -675,6 +778,8 @@ impl LifecycleHandler { copy_count: u32, unit_count: u32, ) -> Result { + let capacity_kb = unit_count.saturating_mul(self.topology.snapshot().unit_size_bytes() / 1024); + self.validate_strip_layout(strip_type, data_num, code_num, copy_count, capacity_kb)?; self.check_range(chunk_id)?; let mut guard = if let Some(locks) = &self.locks { @@ -705,18 +810,20 @@ impl LifecycleHandler { } let snap = self.topology.snapshot(); - let mirror_copies = if copy_count == 0 { 3 } else { copy_count as usize }; - let strip_alloc_type = match strip_type { - ProtoStripType::Mirror => StripAllocType::Mirror { - copy_count: mirror_copies, - }, - ProtoStripType::Ec => StripAllocType::Ec { - data_num: data_num as usize, - code_num: code_num as usize, - }, - }; + let mirror_copies = if copy_count == 0 { 2 } else { copy_count as usize }; + let strip_alloc_type = + self.protected_degraded_layout(strip_type, &snap) + .unwrap_or(match strip_type { + ProtoStripType::Mirror => StripAllocType::Mirror { + copy_count: mirror_copies, + }, + ProtoStripType::Ec => StripAllocType::Ec { + data_num: data_num as usize, + code_num: code_num as usize, + }, + }); - let constraints = self.placement_constraints(); + let constraints = self.allocation_constraints(strip_type, &snap); let start_seq = if chunk.next_strip_sequence == 0 { chunk .strips @@ -860,7 +967,7 @@ impl LifecycleHandler { self.store.delete_reservation_group(chunk_id, &group_id).await?; } } - seal_written_ec_strips(&mut chunk, seal_length, now_ms); + seal_written_strips(&mut chunk, seal_length, now_ms); close_acknowledged_strips(&mut chunk, now_ms); self.store.put_chunk(&chunk).await?; @@ -1120,6 +1227,7 @@ impl LifecycleHandler { "strip replacement ranges must be non-empty".into(), )); } + self.validate_replacement_layouts(replacement_strips)?; let mut guard = if let Some(locks) = &self.locks { Some( locks @@ -1461,6 +1569,11 @@ impl LifecycleHandler { data_num: u32, code_num: u32, ) -> Result { + if self.deployment_mode == Some(crate::chunkdb_config::DeploymentMode::TestSingleNode) { + return Err(LifecycleError::InvalidRequest( + "test_single_node disables mirror-to-EC conversion".into(), + )); + } self.check_range(chunk_id)?; let first = old_strips .first() @@ -1789,6 +1902,29 @@ impl LifecycleHandler { constraints } + fn allocation_constraints( + &self, + strip_type: ProtoStripType, + snapshot: &crate::topology::TopologySnapshot, + ) -> PlacementConstraints { + let constraints = self.placement_constraints(); + if self.deployment_mode != Some(crate::chunkdb_config::DeploymentMode::Production) + || strip_type != ProtoStripType::Ec + { + return constraints; + } + let healthy_nodes = snapshot + .healthy_disk_groups() + .into_iter() + .map(|group| group.node_id) + .collect::>(); + if healthy_nodes.len() == 2 { + constraints.allow_degraded_ec() + } else { + constraints + } + } + fn admit_placement_repairs(&self, chunk: &Chunk) { let Some(tasks) = self.placement_tasks.clone() else { return; @@ -2007,19 +2143,18 @@ fn close_acknowledged_strips(chunk: &mut Chunk, now_ms: u64) { chunk.closed_strip_sequence = last_closed; } -fn seal_written_ec_strips(chunk: &mut Chunk, seal_length: u32, now_ms: u64) { +fn seal_written_strips(chunk: &mut Chunk, seal_length: u32, now_ms: u64) { let mut remaining = seal_length; for strip in &mut chunk.strips { let written = remaining.min(strip.capacity); if written == 0 { break; } - let Some(crowdb_protocol::chunkdb::rpc::Strip::EcStrip(ec)) = strip.strip.as_mut() else { - continue; - }; strip.sealed_length = written; strip.sealed_ts_ms = now_ms; - ec.ec_state = crowdb_protocol::chunkdb::rpc::EcState::Parity as i32; + if let Some(crowdb_protocol::chunkdb::rpc::Strip::EcStrip(ec)) = strip.strip.as_mut() { + ec.ec_state = crowdb_protocol::chunkdb::rpc::EcState::Parity as i32; + } remaining -= written; } } diff --git a/app/crowdb-chunkdb/src/lifecycle/handler/reservation.rs b/app/crowdb-chunkdb/src/lifecycle/handler/reservation.rs index f1f10ff77..720aef943 100644 --- a/app/crowdb-chunkdb/src/lifecycle/handler/reservation.rs +++ b/app/crowdb-chunkdb/src/lifecycle/handler/reservation.rs @@ -372,6 +372,16 @@ impl LifecycleHandler { fence: ReservationFence, spec: ReserveGroupSpec, ) -> Result { + let capacity_kb = spec + .strip_size + .saturating_mul(self.topology.snapshot().unit_size_bytes() / 1024); + self.validate_strip_layout( + super::ProtoStripType::Mirror, + spec.conversion_data_num, + spec.conversion_code_num, + spec.copy_count, + capacity_kb, + )?; self.check_range(chunk_id)?; validate_reserve_spec(fence, spec)?; let mut guard = self.acquire_reservation_guard(chunk_id).await?; diff --git a/app/crowdb-chunkdb/src/main.rs b/app/crowdb-chunkdb/src/main.rs index cb0ad473d..2ae672256 100644 --- a/app/crowdb-chunkdb/src/main.rs +++ b/app/crowdb-chunkdb/src/main.rs @@ -10,7 +10,7 @@ use std::time::{Duration, SystemTime, UNIX_EPOCH}; use clap::Parser; use crowdb_chunkdb::ad_hoc::{AdHocRecoveryManager, AdHocRecoveryShared}; use crowdb_chunkdb::allocator::{ChunkAllocator, DiskdbClientPool}; -use crowdb_chunkdb::chunkdb_config::{ChunkdbConfig, PlacementMode}; +use crowdb_chunkdb::chunkdb_config::{ChunkdbConfig, DeploymentMode, PlacementMode}; use crowdb_chunkdb::conversion::io::ConversionDiskIo; use crowdb_chunkdb::conversion::{ConversionCoordinator, MirrorToEcTaskHandler}; use crowdb_chunkdb::finalize::FinalizeChunkTaskHandler; @@ -36,8 +36,8 @@ use crowdb_chunkdb::topology::{ }; use crowdb_common::metrics::{MetricsRegistry, MetricsRunner}; use crowdb_kv_client::{ - ClientConfig, CrowdbKvClient, DomainMonitorClient, HardwareClient, RangeBindingClient, - ServiceRegistryClient, WatchNotifyClient, + ClientConfig, CrowdbKvClient, DomainMonitorClient, HardwareClient, KVClusterMetaClient, + RangeBindingClient, ServiceRegistryClient, WatchNotifyClient, }; use tracing::{error, info, warn}; @@ -196,6 +196,10 @@ async fn main() { error!("initial topology refresh failed; refusing readiness"); return; }; + if let Err(error) = validate_voting_topology(&kv, config.deployment.mode).await { + error!(%error, "KV voting topology does not satisfy deployment mode; refusing readiness"); + return; + } let monitor_request = crowdb_protocol::chunk_kv::EnsureDomainMonitorRequest { descriptor: crowdb_protocol::chunk_kv::DomainMonitorDescriptor { domain: "chunkdb".into(), @@ -367,6 +371,7 @@ async fn main() { // Lifecycle handler. let handler = Arc::new( LifecycleHandler::new(Arc::clone(&store), allocator, cache) + .with_deployment_mode(config.deployment.mode) .with_placement_tasks(Arc::clone(&task_store)) .with_range_guard(Arc::clone(&range_guard)) .with_locks(Arc::clone(&lock_map)) @@ -409,6 +414,7 @@ async fn main() { let relocation = Arc::new(RelocationCoordinator::new(Arc::clone(&task_manager))); let conversion = Arc::new( ConversionCoordinator::new(Arc::clone(&handler), Arc::clone(&task_store)) + .with_enabled(config.deployment.mode != DeploymentMode::TestSingleNode) .with_wake(task_manager.wake_handle()) .with_policy( config.conversion.data_num, @@ -590,113 +596,108 @@ async fn main() { config.repair.memory_bytes, Arc::clone(&workflow_metrics.repair), )); - let (task_scanner_handle, conversion_route_refresh_handle, ad_hoc_manager) = - match ConversionDiskIo::connect( - &ServiceRegistryClient::from_shared(Arc::clone(&kv)), - &HardwareClient::from_shared(Arc::clone(&kv)), - ) - .await - { - Ok(io) => { - let io = Arc::new(io); - let conversion_task_handler = Arc::new(MirrorToEcTaskHandler::new( - Arc::clone(&handler), - Arc::clone(&task_store), - Arc::clone(&io), - Arc::clone(&workflow_metrics.conversion), - config.conversion.max_bandwidth_mbps, - config.conversion.max_concurrency, - )); - let repair_task_handler = Arc::new( - RepairStripTaskHandler::new( - Arc::clone(&handler), - Arc::clone(&task_manager), - Arc::clone(&io), - config.repair.memory_bytes, - config.repair.max_concurrency, - config.repair.allow_unsafe_placement, - Arc::clone(&workflow_metrics.repair), - ) - .with_ad_hoc(Arc::clone(&ad_hoc_shared)), - ); - let placement_repair_task_handler = Arc::new(PlacementRepairTaskHandler::new( - Arc::clone(&handler), - Arc::clone(&task_manager), - Arc::clone(&io), - Arc::clone(&workflow_metrics.placement), - )); - let task_handlers: Vec> = vec![ - Arc::new(FinalizeChunkTaskHandler::new( - Arc::clone(&handler), - Arc::clone(&io), - )), - conversion_task_handler, - repair_task_handler, - placement_repair_task_handler, - Arc::new(RelocateSegmentTaskHandler::new( - Arc::clone(&handler), - Arc::clone(&task_manager), - )), - ]; - let executor = Arc::new( - TaskExecutor::new( - Arc::clone(&task_manager), - config - .conversion - .max_concurrency - .saturating_add(config.repair.max_concurrency) - .saturating_add(config.placement_repair.max_concurrency), - task_handlers, - ) - .expect("unique conversion task handler"), - ); - let ad_hoc_manager = Arc::new(AdHocRecoveryManager::new( - Arc::clone(&ad_hoc_shared), - Arc::clone(&handler), - Arc::clone(&pool), - Arc::clone(&repair), - Arc::clone(&task_store), - Arc::clone(&task_manager), - Arc::clone(&executor), - )); - let scanner = TaskScanner::new( - Arc::clone(&task_store), - Arc::clone(&task_manager), - Arc::clone(&executor), - 256, - Duration::from_secs(1), - ); - let scanner_stop = stop_rx.clone(); - let scanner_handle = tokio::spawn(async move { scanner.run(scanner_stop).await }); - let service = ServiceRegistryClient::from_shared(Arc::clone(&kv)); - let hardware = HardwareClient::from_shared(Arc::clone(&kv)); - let mut refresh_stop = stop_rx.clone(); - let refresh_interval = Duration::from_secs(u64::from(config.topology.refresh_interval_secs)); - let refresh_handle = tokio::spawn(async move { - let mut ticker = tokio::time::interval(refresh_interval); - ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); - loop { - tokio::select! { - _ = ticker.tick() => { - if let Err(error) = io.refresh(&service, &hardware).await { - warn!(%error, "background conversion DiskIO route refresh failed"); - } - } - changed = refresh_stop.changed() => { - if changed.is_err() || *refresh_stop.borrow() { - return; - } - } + let io = Arc::new(ConversionDiskIo::deferred(config.conversion_io.clone())); + let service = ServiceRegistryClient::from_shared(Arc::clone(&kv)); + let hardware = HardwareClient::from_shared(Arc::clone(&kv)); + if let Err(error) = io.refresh(&service, &hardware).await { + warn!(%error, "background DiskIO discovery will retry while task execution remains enabled"); + } + let (task_scanner_handle, conversion_route_refresh_handle, ad_hoc_manager) = { + let conversion_task_handler = Arc::new(MirrorToEcTaskHandler::new( + Arc::clone(&handler), + Arc::clone(&task_store), + Arc::clone(&io), + Arc::clone(&workflow_metrics.conversion), + config.conversion.max_bandwidth_mbps, + config.conversion.max_concurrency, + )); + let repair_task_handler = Arc::new( + RepairStripTaskHandler::new( + Arc::clone(&handler), + Arc::clone(&task_manager), + Arc::clone(&io), + config.repair.memory_bytes, + config.repair.max_concurrency, + config.repair.allow_unsafe_placement, + Arc::clone(&workflow_metrics.repair), + ) + .with_ad_hoc(Arc::clone(&ad_hoc_shared)), + ); + let placement_repair_task_handler = Arc::new(PlacementRepairTaskHandler::new( + Arc::clone(&handler), + Arc::clone(&task_manager), + Arc::clone(&io), + Arc::clone(&workflow_metrics.placement), + )); + let mut task_handlers: Vec> = vec![ + Arc::new(FinalizeChunkTaskHandler::new( + Arc::clone(&handler), + Arc::clone(&io), + )), + repair_task_handler, + placement_repair_task_handler, + Arc::new(RelocateSegmentTaskHandler::new( + Arc::clone(&handler), + Arc::clone(&task_manager), + )), + ]; + if config.deployment.mode != DeploymentMode::TestSingleNode { + task_handlers.push(conversion_task_handler); + } + let executor = Arc::new( + TaskExecutor::new( + Arc::clone(&task_manager), + config + .conversion + .max_concurrency + .saturating_add(config.repair.max_concurrency) + .saturating_add(config.placement_repair.max_concurrency), + task_handlers, + ) + .expect("unique conversion task handler"), + ); + let ad_hoc_manager = Arc::new(AdHocRecoveryManager::new( + Arc::clone(&ad_hoc_shared), + Arc::clone(&handler), + Arc::clone(&pool), + Arc::clone(&repair), + Arc::clone(&task_store), + Arc::clone(&task_manager), + Arc::clone(&executor), + )); + let scanner = TaskScanner::new( + Arc::clone(&task_store), + Arc::clone(&task_manager), + Arc::clone(&executor), + 256, + Duration::from_secs(1), + ); + let scanner_stop = stop_rx.clone(); + let scanner_handle = tokio::spawn(async move { scanner.run(scanner_stop).await }); + let service = ServiceRegistryClient::from_shared(Arc::clone(&kv)); + let hardware = HardwareClient::from_shared(Arc::clone(&kv)); + let mut refresh_stop = stop_rx.clone(); + let refresh_interval = Duration::from_secs(u64::from(config.topology.refresh_interval_secs)); + let refresh_handle = tokio::spawn(async move { + let mut ticker = tokio::time::interval(refresh_interval); + ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + loop { + tokio::select! { + _ = ticker.tick() => { + if let Err(error) = io.refresh(&service, &hardware).await { + warn!(%error, "background conversion DiskIO route refresh failed"); } } - }); - (Some(scanner_handle), Some(refresh_handle), Some(ad_hoc_manager)) - } - Err(error) => { - warn!(%error, "background conversion DiskIO is unavailable; client fast path remains enabled"); - (None, None, None) + changed = refresh_stop.changed() => { + if changed.is_err() || *refresh_stop.borrow() { + return; + } + } + } } - }; + }); + (Some(scanner_handle), Some(refresh_handle), Some(ad_hoc_manager)) + }; let rpc_service = Arc::new( ChunkdbRpcService::new(Arc::clone(&handler), Arc::clone(&workflow_metrics), rpc_rt_handle) .with_conversion(Arc::clone(&conversion)) @@ -1048,6 +1049,53 @@ fn load_config(args: &Cli) -> ChunkdbConfig { config } +async fn validate_voting_topology(kv: &Arc, mode: DeploymentMode) -> Result<(), String> { + use std::collections::{BTreeMap, BTreeSet}; + + if mode == DeploymentMode::TestUnsafePlacement { + return Ok(()); + } + + let metadata = KVClusterMetaClient::from_shared(Arc::clone(kv)); + let groups = metadata + .list_all_groups() + .await + .map_err(|error| format!("cannot read KV groups: {error}"))?; + let replicas = metadata + .list_all_replicas() + .await + .map_err(|error| format!("cannot read KV replicas: {error}"))?; + let mut voters: BTreeMap<(u64, u64), BTreeSet> = BTreeMap::new(); + for replica in replicas.into_iter().filter(|replica| replica.voting) { + voters + .entry((replica.store_id, replica.group_id)) + .or_default() + .insert(replica.node_id); + } + if groups.is_empty() { + return Err("KV voting topology is empty".into()); + } + for group in groups { + let store_id = group.store_id; + let group_id = group.group_id; + let nodes = voters.remove(&(store_id, group_id)).unwrap_or_default(); + let valid = match mode { + DeploymentMode::Production => nodes.len() >= 3, + DeploymentMode::TestSingleNode => nodes.len() == 1, + DeploymentMode::TestUnsafePlacement => { + unreachable!("test fixtures bypass voting topology validation") + } + }; + if !valid { + return Err(format!( + "KV group {store_id}/{group_id} has {} voting nodes, incompatible with {mode:?}", + nodes.len() + )); + } + } + Ok(()) +} + /// Replace the port portion of a `host:port` address string. fn replace_port(addr: &str, port: u16) -> String { if let Some(idx) = addr.rfind(':') { diff --git a/app/crowdb-chunkdb/src/selector.rs b/app/crowdb-chunkdb/src/selector.rs index 50568f857..c864410ce 100644 --- a/app/crowdb-chunkdb/src/selector.rs +++ b/app/crowdb-chunkdb/src/selector.rs @@ -67,7 +67,8 @@ impl ChunkPlacementStrategy for ProtectedPlacementStrategy { } fn permits_degraded_disk(&self, constraints: &PlacementConstraints, ec: bool) -> bool { - constraints.allow_degraded_failure_domains && (!ec || constraints.allow_unsafe_ec) + constraints.allow_degraded_failure_domains + && (!ec || constraints.allow_unsafe_ec || constraints.allow_degraded_ec) } fn permits_unsafe_ec(&self, configured: bool) -> bool { @@ -193,6 +194,8 @@ pub struct PlacementConstraints { pub exclude_disk_groups: Vec, /// Permit EC placement that exceeds the safe per-node failure bound. pub allow_unsafe_ec: bool, + /// Permit EC placement across two survivors of a three-node production cluster. + pub allow_degraded_ec: bool, /// Permit a plan that cannot satisfy every requested failure domain. pub allow_degraded_failure_domains: bool, /// Ordering used to choose among otherwise eligible domains. @@ -231,6 +234,13 @@ impl PlacementConstraints { self } + #[must_use] + pub fn allow_degraded_ec(mut self) -> Self { + self.allow_degraded_ec = true; + self.allow_degraded_failure_domains = true; + self + } + #[must_use] pub fn allow_degraded_failure_domains(mut self) -> Self { self.allow_degraded_failure_domains = true; @@ -356,7 +366,11 @@ pub(super) fn finish_plan( ec_shape: bool, ) -> Result { let protection = assess_entries(&entries, loss_budget); - if ec_shape && protection.max_fragments_per_node > loss_budget && !constraints.allow_unsafe_ec { + if ec_shape + && protection.max_fragments_per_node > loss_budget + && !constraints.allow_unsafe_ec + && !constraints.allow_degraded_ec + { return Err(PlacementError::UnsafePlacementRequired); } // A single-copy mirror has no recoverable domain-loss budget. It still diff --git a/app/crowdb-chunkdb/src/selector/ec.rs b/app/crowdb-chunkdb/src/selector/ec.rs index ede657246..3876f27e1 100644 --- a/app/crowdb-chunkdb/src/selector/ec.rs +++ b/app/crowdb-chunkdb/src/selector/ec.rs @@ -70,10 +70,30 @@ impl EcPlacement { ); } - if !constraints.allow_unsafe_ec { + if !constraints.allow_unsafe_ec && !constraints.allow_degraded_ec { return Err(PlacementError::UnsafePlacementRequired); } + if constraints.allow_degraded_ec { + if node_count != 2 { + return Err(PlacementError::UnsafePlacementRequired); + } + let balanced_limit = total_blocks.div_ceil(2); + let entries = try_distribute(snap, &by_rack, total_blocks, balanced_limit, constraints).ok_or( + PlacementError::InsufficientNodes { + needed: total_blocks, + available: node_count, + }, + )?; + return finish_plan( + snap, + entries, + u32::try_from(code_num).unwrap_or(u32::MAX), + constraints, + true, + ); + } + // Fall back to unsafe mode: max `total_blocks` per node (i.e. // no practical limit — just spread as evenly as possible). warn!( diff --git a/app/crowdb-chunkdb/src/service/chunkdb_rpc_service/wire.rs b/app/crowdb-chunkdb/src/service/chunkdb_rpc_service/wire.rs index 01af6afdb..54316c99c 100644 --- a/app/crowdb-chunkdb/src/service/chunkdb_rpc_service/wire.rs +++ b/app/crowdb-chunkdb/src/service/chunkdb_rpc_service/wire.rs @@ -234,6 +234,8 @@ pub(super) fn proto_chunk_type(fb: FBChunkType) -> Option { FBChunkType::BtreePage => Some(ProtoChunkType::BtreePage), FBChunkType::PageIndex => Some(ProtoChunkType::PageIndex), FBChunkType::Stream => Some(ProtoChunkType::Stream), + FBChunkType::S3 => Some(ProtoChunkType::S3), + FBChunkType::IcebergTable => Some(ProtoChunkType::IcebergTable), _ => None, } } @@ -987,6 +989,8 @@ pub(super) fn fb_chunk_type(t: ProtoChunkType) -> FBChunkType { ProtoChunkType::BtreePage => FBChunkType::BtreePage, ProtoChunkType::PageIndex => FBChunkType::PageIndex, ProtoChunkType::Stream => FBChunkType::Stream, + ProtoChunkType::S3 => FBChunkType::S3, + ProtoChunkType::IcebergTable => FBChunkType::IcebergTable, } } diff --git a/app/crowdb-chunkdb/tests/common/cluster.rs b/app/crowdb-chunkdb/tests/common/cluster.rs index 76c80a2a2..dfb9f23b2 100644 --- a/app/crowdb-chunkdb/tests/common/cluster.rs +++ b/app/crowdb-chunkdb/tests/common/cluster.rs @@ -803,11 +803,11 @@ pub async fn wait_for_disks_ready( } /// Wait until the topology cache contains the seeded healthy disk-groups. -async fn wait_for_topology_ready(topology: &TopologyCache) { +async fn wait_for_topology_ready(topology: &TopologyCache, expected_disk_groups: usize) { let deadline = Instant::now() + Duration::from_secs(10); loop { let snap = topology.snapshot(); - if snap.healthy_disk_groups().len() >= seeded_dg_ids().len() { + if snap.healthy_disk_groups().len() >= expected_disk_groups { return; } assert!( @@ -841,6 +841,14 @@ impl ChunkdbHarness { } pub async fn start_with_layout_validity(cluster: &KvCluster, layout_validity: Duration) -> Self { + Self::start_with_disk_group_count(cluster, layout_validity, seeded_dg_ids().len()).await + } + + pub async fn start_with_disk_group_count( + cluster: &KvCluster, + layout_validity: Duration, + expected_disk_groups: usize, + ) -> Self { let kv = cluster.make_crowdb_client(); // Topology cache + refresh loop. @@ -852,7 +860,7 @@ impl ChunkdbHarness { run_refresh_loop(refresh_cache, hw, Duration::from_secs(5), stop_rx).await; }); - wait_for_topology_ready(&topology).await; + wait_for_topology_ready(&topology, expected_disk_groups).await; // Binding cache — all buckets to store 0, group 1. let bindings = BindingCache::new(); diff --git a/app/crowdb-chunkdb/tests/config_test.rs b/app/crowdb-chunkdb/tests/config_test.rs index 43b069d33..70ed09b69 100644 --- a/app/crowdb-chunkdb/tests/config_test.rs +++ b/app/crowdb-chunkdb/tests/config_test.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crowdb_chunkdb::chunkdb_config::{ChunkdbConfig, PlacementMode}; +use crowdb_chunkdb::chunkdb_config::{ChunkdbConfig, DeploymentMode, PlacementMode}; use crowdb_chunkdb::selector::FailureDomainPriority; use crowdb_common::config::BaseConfig; @@ -9,6 +9,7 @@ use crowdb_common::config::BaseConfig; fn rpc_workers_defaults_and_validates() { let config: ChunkdbConfig = toml::from_str("[server]\n").expect("partial config parses"); assert_eq!(config.server.rpc_workers, 2); + assert_eq!(config.conversion_io.rpc_workers, 2); config.validate().expect("default workers validate"); let mut invalid = config; @@ -26,6 +27,7 @@ fn tracked_config_file_loads_and_validates() { .join("crowdb_chunkdb_config.toml"); let config = crowdb_common::config::load_from_file::(&path).expect("load tracked config"); assert_eq!(config.server.rpc_workers, 2); + assert_eq!(config.conversion_io.normal_connections_per_endpoint, 1); assert_eq!( config.placement.failure_domain_priority, FailureDomainPriority::RackFirst @@ -33,6 +35,37 @@ fn tracked_config_file_loads_and_validates() { assert!(!config.placement.allow_degraded_failure_domains); } +#[test] +fn single_node_container_declares_test_only_deployment() { + let path = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../../container/single-node-container/templates/chunkdb.toml"); + let config = crowdb_common::config::load_from_file::(&path).unwrap(); + assert_eq!(config.deployment.mode, DeploymentMode::TestSingleNode); + assert_eq!(config.deployment.max_node_failures, 0); + assert_eq!(config.placement.mode, PlacementMode::UnsafeColocated); + assert_eq!(config.conversion_io.rpc_workers, 1); +} + +#[test] +fn conversion_io_transport_rejects_zero_resources() { + let mut config = ChunkdbConfig::default(); + config.conversion_io.priority_connections_per_endpoint = 0; + assert_eq!( + config.validate(), + Err("conversion_io connections and RPC workers must be > 0".to_string()) + ); +} + +#[test] +fn unsafe_fixture_mode_is_explicit_and_only_available_to_debug_builds() { + let config: ChunkdbConfig = toml::from_str( + "[deployment]\nmode = \"test_unsafe_placement\"\n[placement]\nmode = \"unsafe_colocated\"\nallow_unsafe_ec = true\n", + ) + .unwrap(); + assert_eq!(config.deployment.mode, DeploymentMode::TestUnsafePlacement); + assert_eq!(config.validate().is_ok(), cfg!(debug_assertions)); +} + #[test] fn placement_policy_parses_both_priorities() { let rack: ChunkdbConfig = toml::from_str( @@ -62,6 +95,32 @@ fn unsafe_colocated_placement_mode_is_explicit() { let colocated: ChunkdbConfig = toml::from_str("[placement]\nmode = \"unsafe_colocated\"\n").expect("mode parses"); assert_eq!(colocated.placement.mode, PlacementMode::UnsafeColocated); + assert!(colocated.validate().is_err()); + + let single: ChunkdbConfig = toml::from_str( + "[deployment]\nmode = \"test_single_node\"\nmax_node_failures = 0\n[placement]\nmode = \"unsafe_colocated\"\n", + ) + .expect("explicit test mode parses"); + assert_eq!(single.deployment.mode, DeploymentMode::TestSingleNode); + single.validate().expect("explicit test mode validates"); + + let mut bad_budget = single; + bad_budget.deployment.max_node_failures = 1; + assert_eq!( + bad_budget.validate(), + Err("test_single_node requires max_node_failures = 0".to_string()) + ); + + let mut bad_production_budget = ChunkdbConfig::default(); + bad_production_budget.deployment.max_node_failures = 0; + assert_eq!( + bad_production_budget.validate(), + Err("production requires max_node_failures = 1".to_string()) + ); + + let mut unsafe_production = protected; + unsafe_production.placement.allow_unsafe_ec = true; + assert!(unsafe_production.validate().is_err()); } #[test] diff --git a/app/crowdb-chunkdb/tests/full_stack_test.rs b/app/crowdb-chunkdb/tests/full_stack_test.rs index 48019423f..d986fd2d4 100644 --- a/app/crowdb-chunkdb/tests/full_stack_test.rs +++ b/app/crowdb-chunkdb/tests/full_stack_test.rs @@ -20,7 +20,7 @@ use common::cluster::{ DiskdbServer, KvCluster, DATA_GROUP_ID, STORE_ID, }; use crowdb_chunkdb::allocator::StripAllocType; -use crowdb_chunkdb::chunkdb_config::PlacementRebalanceConfig; +use crowdb_chunkdb::chunkdb_config::{DeploymentMode, PlacementRebalanceConfig}; use crowdb_chunkdb::conversion::io::ConversionDiskIo; use crowdb_chunkdb::conversion::{decode_payload, ConversionCoordinator, MirrorToEcTaskHandler}; use crowdb_chunkdb::finalize::FinalizeChunkTaskHandler; @@ -41,6 +41,7 @@ use crowdb_chunkdb::task::{ RelocateSegmentTaskHandler, SegmentOwnerResolver, TaskAdmission, TaskClaim, TaskExecutor, TaskHandler, TaskManager, TaskOutcome, TaskScanner, TaskStore, }; +use crowdb_chunkdb::topology::build_snapshot; use crowdb_chunkdb_client::ChunkdbRpcTransport; use crowdb_common::metrics::MetricsRegistry; use crowdb_protocol::chunk_task::{ @@ -53,7 +54,7 @@ use crowdb_protocol::chunkdb::rpc::{ QuerySegmentOwnerRequest, RelocateSegmentHandoffRequest, RelocationHandoffDisposition, SegmentOwnerDisposition, Strip, StripReservationAction, StripReservationState, StripType, }; -use crowdb_protocol::common::{ChunkId, DiskGroupUsageSummary}; +use crowdb_protocol::common::{ChunkId, DiskGroupUsageSummary, HwStatus}; use crowdb_protocol::diskdb::rpc::RelocationJournalPhase; use crowdb_protocol::{port::alloc as port_alloc, ServicePort}; use crowdb_test_harness::diskio::{DiskioGroup0Identity, DiskioProcess, DiskioStartOpts}; @@ -334,6 +335,13 @@ async fn diskio_routes_cover_every_group_in_the_two_rack_fixture() { ) .await; let _diskdb = DiskdbServer::start_with_disk_groups_and_zones(&cluster, &disk_groups, 32).await; + let service = cluster.make_service_registry_client(); + let hardware = cluster.make_hardware_client(); + let io = ConversionDiskIo::deferred(crowdb_chunkdb::chunkdb_config::ConversionIoConfig::default()); + assert!( + io.refresh(&service, &hardware).await.is_err(), + "DiskIO routes must be unavailable before the services start" + ); let diskio = start_diskio_groups( &cluster, &[ @@ -346,11 +354,9 @@ async fn diskio_routes_cover_every_group_in_the_two_rack_fixture() { ], 2_000, ); - let service = cluster.make_service_registry_client(); - let hardware = cluster.make_hardware_client(); let deadline = tokio::time::Instant::now() + Duration::from_secs(15); loop { - if ConversionDiskIo::connect(&service, &hardware).await.is_ok() { + if io.refresh(&service, &hardware).await.is_ok() { break; } assert!( @@ -837,6 +843,419 @@ fn task_value() -> ChunkTaskValue { struct CompleteTaskHandler; +#[tokio::test] +async fn explicit_single_node_mode_rejects_ec_and_extra_copies() { + if std::env::var("CROWDB_KV_SERVER_BIN").is_err() && common::cluster::crowdb_kv_server_bin().is_none() { + eprintln!("skipping: crowdb-kv-server binary is unavailable"); + return; + } + let cluster = KvCluster::start().await; + seed_hardware(&cluster.make_hardware_client()).await; + let _diskdb = DiskdbServer::start(&cluster).await; + let harness = ChunkdbHarness::start(&cluster).await; + let handler = LifecycleHandler::new( + Arc::clone(&harness.store), + Arc::clone(&harness.allocator), + harness.topology.clone(), + ) + .with_deployment_mode(DeploymentMode::TestSingleNode) + .with_locks(Arc::new(ChunkLockMap::new( + 10_000, + Arc::new(LifecycleMetrics::new()), + Duration::from_secs(60), + ))); + for (strip_type, copies, size_kb) in [ + (StripType::Ec, 0, 1024), + (StripType::Mirror, 2, 1024), + (StripType::Mirror, 1, 512), + ] { + assert!(matches!( + handler + .allocate_chunk(None, size_kb, 1, strip_type, 0, 0, copies, ChunkType::Repo, 0, 0) + .await, + Err(LifecycleError::InvalidRequest(_)) + )); + } + let chunk = handler + .allocate_chunk( + None, + 1024, + 1, + StripType::Mirror, + 0, + 0, + 1, + ChunkType::Repo, + 71, + 30_000, + ) + .await + .unwrap(); + assert_eq!(chunk.strips[0].capacity, 1024); + assert!(matches!(chunk.strips[0].strip, Some(Strip::MirrorStrip(_)))); + let id = chunk.id.unwrap(); + let fence = ReservationFence { + expected_modify_ts: chunk.modify_ts, + writer_epoch: 71, + lease_generation: 1, + lease_ms: 30_000, + }; + for copies in [0, 2] { + let group = crowdb_protocol::generate_chunk_id(ChunkType::Repo as u8).to_proto(); + assert!(matches!( + handler + .reserve_strip_group( + &id, + &group, + fence, + ReserveGroupSpec { + strip_size: 1, + strip_count: 1, + copy_count: copies, + conversion_data_num: 0, + conversion_code_num: 0, + }, + ) + .await, + Err(LifecycleError::InvalidRequest(_)) + )); + } + assert!(matches!( + handler.allocate_conversion_strip(&id, &chunk.strips, 1, 1).await, + Err(LifecycleError::InvalidRequest(_)) + )); + let mut invalid_replacement = chunk.strips[0].clone(); + let Some(Strip::MirrorStrip(mirror)) = invalid_replacement.strip.as_mut() else { + panic!("single-node strip must be a mirror"); + }; + mirror.segments.push(mirror.segments[0]); + let operation = crowdb_protocol::generate_chunk_id(ChunkType::Repo as u8).to_proto(); + assert!(matches!( + handler + .replace_chunk_strip_range( + &id, + chunk.modify_ts, + 0, + &chunk.strips, + &[invalid_replacement], + operation, + ) + .await, + Err(LifecycleError::InvalidRequest(_)) + )); + assert_single_node_reservation(&handler, &id, fence).await; +} + +async fn assert_single_node_reservation(handler: &LifecycleHandler, id: &ChunkId, fence: ReservationFence) { + let group = crowdb_protocol::generate_chunk_id(ChunkType::Repo as u8).to_proto(); + let reserved = handler + .reserve_strip_group( + id, + &group, + fence, + ReserveGroupSpec { + strip_size: 1, + strip_count: 1, + copy_count: 1, + conversion_data_num: 0, + conversion_code_num: 0, + }, + ) + .await + .expect("one-copy single-node reservation"); + let strip = reserved.group.unwrap().strips[0].clone(); + assert_eq!(strip.capacity, 1024); + let mut invalid = strip.clone(); + let Some(Strip::MirrorStrip(mirror)) = invalid.strip.as_mut() else { + panic!("reserved strip must be a mirror"); + }; + mirror.segments.push(mirror.segments[0]); + let operation = crowdb_protocol::generate_chunk_id(ChunkType::Repo as u8).to_proto(); + assert!(matches!( + handler + .replace_chunk_strip_range(id, reserved.chunk.modify_ts, 1, &[strip], &[invalid], operation,) + .await, + Err(LifecycleError::InvalidRequest(_)) + )); +} + +async fn assert_invalid_production_mirror_copies(handler: &LifecycleHandler) { + for copies in [0, 1] { + assert!(matches!( + handler + .allocate_chunk( + None, + 1024, + 1, + StripType::Mirror, + 0, + 0, + copies, + ChunkType::S3, + 0, + 0 + ) + .await, + Err(LifecycleError::InvalidRequest(_)) + )); + } +} + +#[tokio::test] +async fn production_ec_stays_degraded_ec_and_mirrors_use_two_copies_after_one_node_loss() { + if std::env::var("CROWDB_KV_SERVER_BIN").is_err() && common::cluster::crowdb_kv_server_bin().is_none() { + eprintln!("skipping: crowdb-kv-server binary is unavailable"); + return; + } + let cluster = KvCluster::start().await; + let hardware = cluster.make_hardware_client(); + let groups = seed_hardware_layout_with_zones( + &hardware, + &[(100, vec![10]), (101, vec![11]), (102, vec![12])], + 32, + ) + .await; + let _diskdb = DiskdbServer::start_with_disk_groups_and_zones(&cluster, &groups, 32).await; + let harness = ChunkdbHarness::start_with_disk_group_count(&cluster, Duration::from_secs(30), 3).await; + hardware + .set_node_status(102, 12, HwStatus::Offline) + .await + .unwrap(); + harness + .topology + .replace(build_snapshot(&hardware).await.expect("refreshed hardware")); + let handler = LifecycleHandler::new( + Arc::clone(&harness.store), + Arc::clone(&harness.allocator), + harness.topology.clone(), + ) + .with_deployment_mode(DeploymentMode::Production); + assert_invalid_production_mirror_copies(&handler).await; + let degraded = handler + .allocate_chunk(None, 1024, 1, StripType::Ec, 2, 1, 0, ChunkType::Repo, 0, 0) + .await + .unwrap(); + let Some(Strip::EcStrip(ec)) = °raded.strips[0].strip else { + panic!("degraded allocation must retain EC geometry"); + }; + assert_eq!(ec.segments.len(), 3); + assert!(degraded.strips[0].placement_repair_required); + let degraded_small = handler + .allocate_chunk(None, 1024, 1, StripType::Mirror, 0, 0, 3, ChunkType::S3, 0, 0) + .await + .unwrap(); + let Some(Strip::MirrorStrip(small_mirror)) = °raded_small.strips[0].strip else { + panic!("degraded small allocation must use protected mirrors"); + }; + assert_eq!(small_mirror.segments.len(), 2); + let appended = handler + .append_chunk( + °raded_small.id.unwrap(), + degraded_small.modify_ts, + 1, + StripType::Mirror, + 0, + 0, + 3, + 1, + ) + .await + .unwrap(); + let Some(Strip::MirrorStrip(appended_mirror)) = &appended.strips[0].strip else { + panic!("degraded append must use protected mirrors"); + }; + assert_eq!(appended_mirror.segments.len(), 2); + hardware.set_node_status(102, 12, HwStatus::Up).await.unwrap(); + harness + .topology + .replace(build_snapshot(&hardware).await.expect("recovered hardware")); + let healthy = handler + .allocate_chunk(None, 1024, 1, StripType::Ec, 2, 1, 0, ChunkType::Repo, 0, 0) + .await + .unwrap(); + assert!(matches!(healthy.strips[0].strip, Some(Strip::EcStrip(_)))); + + hardware + .set_node_status(100, 10, HwStatus::Offline) + .await + .unwrap(); + harness + .topology + .replace(build_snapshot(&hardware).await.expect("replacement topology")); + let source = small_mirror + .segments + .iter() + .find(|segment| segment.disk_id.unwrap().low / 10 == 1000) + .expect("mirror copy on failed node"); + let retained = small_mirror + .segments + .iter() + .copied() + .filter(|segment| segment != source) + .collect::>(); + let replacement = handler + .allocate_replacement_segment(°raded_small.id.unwrap(), source, &retained, &[]) + .await + .expect("replace failed mirror copy on unused survivor"); + assert_eq!(replacement.disk_id.unwrap().low / 10, 1002); +} + +#[tokio::test] +#[allow(clippy::too_many_lines)] +async fn production_degraded_ec_task_repairs_after_node_returns() { + if std::env::var("CROWDB_KV_SERVER_BIN").is_err() && common::cluster::crowdb_kv_server_bin().is_none() { + eprintln!("skipping: crowdb-kv-server binary is unavailable"); + return; + } + if !crowdb_test_harness::diskio::check_diskio_only() { + return; + } + let cluster = KvCluster::start().await; + let hardware = cluster.make_hardware_client(); + let groups = seed_hardware_layout_with_zones( + &hardware, + &[(100, vec![10]), (101, vec![11]), (102, vec![12])], + 32, + ) + .await; + let _diskdb = DiskdbServer::start_with_disk_groups_and_zones(&cluster, &groups, 32).await; + let service = cluster.make_service_registry_client(); + let io = Arc::new(ConversionDiskIo::deferred( + crowdb_chunkdb::chunkdb_config::ConversionIoConfig::default(), + )); + assert!(io.refresh(&service, &hardware).await.is_err()); + + let harness = ChunkdbHarness::start_with_disk_group_count(&cluster, Duration::from_secs(30), 3).await; + hardware + .set_node_status(102, 12, HwStatus::Offline) + .await + .unwrap(); + harness + .topology + .replace(build_snapshot(&hardware).await.expect("outage topology")); + let handler = Arc::new( + LifecycleHandler::new( + Arc::clone(&harness.store), + Arc::clone(&harness.allocator), + harness.topology.clone(), + ) + .with_deployment_mode(DeploymentMode::Production) + .with_layout_validity(Duration::from_millis(1)), + ); + let chunk = handler + .allocate_chunk(None, 1024, 1, StripType::Ec, 2, 1, 0, ChunkType::Repo, 0, 0) + .await + .expect("allocate degraded EC strip"); + assert!(chunk.strips[0].placement_repair_required); + let chunk_id = chunk.id.expect("chunk identity"); + let Some(Strip::EcStrip(ec)) = chunk.strips[0].strip.as_ref() else { + panic!("degraded allocation must remain EC"); + }; + + let diskio = start_diskio_groups( + &cluster, + &[(1000, 100, 10), (1001, 101, 11), (1002, 102, 12)], + 2_000, + ); + let deadline = tokio::time::Instant::now() + Duration::from_secs(15); + loop { + if io.refresh(&service, &hardware).await.is_ok() { + break; + } + assert!( + tokio::time::Instant::now() < deadline, + "DiskIO routes were not published" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + for segment in &ec.segments { + io.write_segment(segment, 1024 * 1024, Bytes::from(vec![0x5a; 1024 * 1024])) + .await + .expect("seed degraded EC fragment"); + } + + let bindings = BindingCache::new(); + bindings.replace(default_binding_table(STORE_ID, DATA_GROUP_ID)); + let tasks = Arc::new(TaskStore::new(cluster.make_crowdb_client(), bindings)); + let coordinator = PlacementRepairCoordinator::new(Arc::clone(&handler), Arc::clone(&tasks)); + assert_eq!(coordinator.scan_batch(256, 100).await.unwrap(), 1); + let mut registry = MetricsRegistry::new(); + let metrics = ChunkdbMetrics::register(&mut registry).placement; + let manager = Arc::new(TaskManager::new(Arc::clone(&tasks), 97, 30_000)); + let executor = Arc::new( + TaskExecutor::new( + Arc::clone(&manager), + 1, + vec![Arc::new(PlacementRepairTaskHandler::new( + Arc::clone(&handler), + Arc::clone(&manager), + Arc::clone(&io), + metrics, + ))], + ) + .unwrap(), + ); + let scanner = TaskScanner::new(Arc::clone(&tasks), manager, executor, 16, Duration::from_secs(1)); + assert_eq!( + scanner.run_once(100).await.unwrap().tasks_completed_or_requeued, + 1 + ); + assert!(handler.query_chunk(&chunk_id).await.unwrap().strips[0].placement_repair_required); + + drop(scanner); + drop(coordinator); + drop(tasks); + let bindings = BindingCache::new(); + bindings.replace(default_binding_table(STORE_ID, DATA_GROUP_ID)); + let tasks = Arc::new(TaskStore::new(cluster.make_crowdb_client(), bindings)); + let mut registry = MetricsRegistry::new(); + let metrics = ChunkdbMetrics::register(&mut registry).placement; + let manager = Arc::new(TaskManager::new(Arc::clone(&tasks), 98, 30_000)); + let executor = Arc::new( + TaskExecutor::new( + Arc::clone(&manager), + 1, + vec![Arc::new(PlacementRepairTaskHandler::new( + Arc::clone(&handler), + Arc::clone(&manager), + Arc::clone(&io), + metrics, + ))], + ) + .unwrap(), + ); + let scanner = TaskScanner::new(Arc::clone(&tasks), manager, executor, 16, Duration::from_secs(1)); + + hardware.set_node_status(102, 12, HwStatus::Up).await.unwrap(); + harness + .topology + .replace(build_snapshot(&hardware).await.expect("recovered topology")); + for _ in 0..6 { + let summary = scanner.run_once(u64::MAX).await.unwrap(); + if !handler.query_chunk(&chunk_id).await.unwrap().strips[0].placement_repair_required { + break; + } + assert!( + summary.tasks_completed_or_requeued > 0, + "placement task did not resume" + ); + } + let repaired = handler.query_chunk(&chunk_id).await.unwrap(); + let assessment = repaired.strips[0].placement_assessment.as_ref().unwrap(); + assert!(assessment.rack_protected && assessment.node_protected && assessment.disk_protected); + assert!(!repaired.strips[0].placement_repair_required); + let Some(Strip::EcStrip(ec)) = repaired.strips[0].strip.as_ref() else { + panic!("repaired strip must remain EC"); + }; + for segment in &ec.segments { + assert_eq!( + io.read_segment(segment, 1024 * 1024).await.unwrap(), + Bytes::from(vec![0x5a; 1024 * 1024]) + ); + } + assert_eq!(diskio.len(), 3); +} + #[tokio::test] async fn active_chunk_creates_one_deadline_indexed_finalizer() { if std::env::var("CROWDB_KV_SERVER_BIN").is_err() && common::cluster::crowdb_kv_server_bin().is_none() { @@ -2910,6 +3329,54 @@ async fn generated_chunk_ids_stay_with_the_serving_range_owner() { } } +#[tokio::test] +async fn allocated_chunk_type_matches_its_id_prefix() { + if std::env::var("CROWDB_KV_SERVER_BIN").is_err() && common::cluster::crowdb_kv_server_bin().is_none() { + eprintln!("skipping: CROWDB_KV_SERVER_BIN not set and binary not found"); + return; + } + let cluster = KvCluster::start().await; + let hw = cluster.make_hardware_client(); + seed_hardware(&hw).await; + let _diskdb = DiskdbServer::start(&cluster).await; + let harness = ChunkdbHarness::start(&cluster).await; + + for chunk_type in [ChunkType::S3, ChunkType::IcebergTable] { + let chunk = harness + .handler + .allocate_chunk(None, 1, 1, StripType::Mirror, 0, 0, 1, chunk_type, 0, 0) + .await + .expect("typed allocation"); + let id = chunk.id.expect("allocated id"); + assert_eq!(id.high >> 56, chunk_type as u64); + assert_eq!(chunk.chunk_type, chunk_type as i32); + let persisted = harness.handler.query_chunk(&id).await.expect("persisted chunk"); + assert_eq!(persisted.chunk_type, chunk_type as i32); + } + + let mismatched = crowdb_protocol::generate_chunk_id(ChunkType::S3 as u8).to_proto(); + let result = harness + .handler + .allocate_chunk( + Some(mismatched), + 1, + 1, + StripType::Mirror, + 0, + 0, + 1, + ChunkType::IcebergTable, + 0, + 0, + ) + .await; + assert!(matches!(result, Err(LifecycleError::InvalidRequest(_)))); + assert!(matches!( + harness.handler.query_chunk(&mismatched).await, + Err(LifecycleError::ChunkNotFound) + )); +} + #[tokio::test] #[allow(clippy::too_many_lines)] async fn conversion_reservation_allocates_joint_plan_and_cleans_every_early_tail() { diff --git a/app/crowdb-chunkdb/tests/selector_test.rs b/app/crowdb-chunkdb/tests/selector_test.rs index 19c517091..b0eec41db 100644 --- a/app/crowdb-chunkdb/tests/selector_test.rs +++ b/app/crowdb-chunkdb/tests/selector_test.rs @@ -231,6 +231,22 @@ fn ec_select_8_4_unsafe_fallback_3_nodes() { assert!(!plan.safe_mode); } +#[test] +fn protected_degraded_ec_uses_both_survivors_and_marks_repair() { + let cache = build_topology(&[(1, &[10]), (2, &[20])]); + let snap = cache.snapshot(); + let constraints = PlacementConstraints::new().allow_degraded_ec(); + let plan = EcPlacement::select(&snap, 4, 2, &constraints).unwrap(); + assert_eq!(plan.entries.len(), 6); + assert_eq!(plan.protection.max_fragments_per_node, 3); + assert!(!plan.safe_mode); + assert!(plan.entries.iter().any(|entry| entry.node_id == 10)); + assert!(plan.entries.iter().any(|entry| entry.node_id == 20)); + + let one_node = build_topology(&[(1, &[10])]); + assert!(EcPlacement::select(&one_node.snapshot(), 4, 2, &constraints).is_err()); +} + #[test] fn ec_select_4_1_unsafe_one_rack_balances_nodes() { let cache = build_topology(&[(1, &[10, 11, 12])]); diff --git a/app/crowdb-diskio/src/dio_config.h b/app/crowdb-diskio/src/dio_config.h index 6a7779b40..25808aa57 100644 --- a/app/crowdb-diskio/src/dio_config.h +++ b/app/crowdb-diskio/src/dio_config.h @@ -19,7 +19,7 @@ namespace crowdb::diskio { // Disk type for dummy disks (when no real block device is configured). -enum class DummyDiskType { +enum class DummyDiskType : std::uint8_t { Null, // memfd, drop-write + pattern read (default, for benchmarks) Mem, // memfd, store + read-back (for correctness tests) }; diff --git a/app/crowdb-diskio/src/disk/block_disk.h b/app/crowdb-diskio/src/disk/block_disk.h index 94cddc43e..2d848506e 100644 --- a/app/crowdb-diskio/src/disk/block_disk.h +++ b/app/crowdb-diskio/src/disk/block_disk.h @@ -23,22 +23,22 @@ class BlockDisk : public Disk bool o_direct); ~BlockDisk() override; - DiskType type() const override + [[nodiscard]] DiskType type() const override { return DiskType::Block; } - int fd() const override + [[nodiscard]] int fd() const override { return fd_; } - bool is_o_direct() const override + [[nodiscard]] bool is_o_direct() const override { return o_direct_; } - size_t block_size() const override + [[nodiscard]] size_t block_size() const override { return block_size_; } @@ -48,7 +48,7 @@ class BlockDisk : public Disk return engine_.get(); } - DiskId id() const override + [[nodiscard]] DiskId id() const override { return id_; } diff --git a/app/crowdb-diskio/src/disk/disk.h b/app/crowdb-diskio/src/disk/disk.h index bcc7a9b7c..2e8b420f9 100644 --- a/app/crowdb-diskio/src/disk/disk.h +++ b/app/crowdb-diskio/src/disk/disk.h @@ -10,13 +10,14 @@ #include "disk/types.h" #include "engine/io_engine.h" +#include #include #include namespace crowdb::diskio { -enum class DiskType { +enum class DiskType : std::uint8_t { Block, Null, Mem, @@ -27,13 +28,13 @@ class Disk public: virtual ~Disk() = default; - virtual DiskType type() const = 0; - virtual int fd() const = 0; - virtual bool is_o_direct() const = 0; - virtual size_t block_size() const = 0; - virtual IoEngine *engine() = 0; - virtual DiskId id() const = 0; - virtual Zone *find_zone(uint32_t zone_index) = 0; + [[nodiscard]] virtual DiskType type() const = 0; + [[nodiscard]] virtual int fd() const = 0; + [[nodiscard]] virtual bool is_o_direct() const = 0; + [[nodiscard]] virtual size_t block_size() const = 0; + virtual IoEngine *engine() = 0; + [[nodiscard]] virtual DiskId id() const = 0; + virtual Zone *find_zone(uint32_t zone_index) = 0; protected: std::vector zones_; diff --git a/app/crowdb-diskio/src/disk/mem_disk.h b/app/crowdb-diskio/src/disk/mem_disk.h index 8598d438f..f34586469 100644 --- a/app/crowdb-diskio/src/disk/mem_disk.h +++ b/app/crowdb-diskio/src/disk/mem_disk.h @@ -32,22 +32,22 @@ class MemDisk : public Disk std::optional props = std::nullopt); ~MemDisk() override; - DiskType type() const override + [[nodiscard]] DiskType type() const override { return DiskType::Mem; } - int fd() const override + [[nodiscard]] int fd() const override { return fd_; } - bool is_o_direct() const override + [[nodiscard]] bool is_o_direct() const override { return false; } - size_t block_size() const override + [[nodiscard]] size_t block_size() const override { return 1; } @@ -57,7 +57,7 @@ class MemDisk : public Disk return engine_.get(); } - DiskId id() const override + [[nodiscard]] DiskId id() const override { return id_; } diff --git a/app/crowdb-diskio/src/disk/null_disk.h b/app/crowdb-diskio/src/disk/null_disk.h index 92eeb3070..26d10dd07 100644 --- a/app/crowdb-diskio/src/disk/null_disk.h +++ b/app/crowdb-diskio/src/disk/null_disk.h @@ -33,22 +33,22 @@ class NullDisk : public Disk std::optional props = std::nullopt); ~NullDisk() override; - DiskType type() const override + [[nodiscard]] DiskType type() const override { return DiskType::Null; } - int fd() const override + [[nodiscard]] int fd() const override { return fd_; } - bool is_o_direct() const override + [[nodiscard]] bool is_o_direct() const override { return false; } - size_t block_size() const override + [[nodiscard]] size_t block_size() const override { return 1; } @@ -58,7 +58,7 @@ class NullDisk : public Disk return wrapper_.get(); } - DiskId id() const override + [[nodiscard]] DiskId id() const override { return id_; } diff --git a/app/crowdb-diskio/src/rpc/dio_server.cpp b/app/crowdb-diskio/src/rpc/dio_server.cpp index c92f46d5a..05d349e49 100644 --- a/app/crowdb-diskio/src/rpc/dio_server.cpp +++ b/app/crowdb-diskio/src/rpc/dio_server.cpp @@ -4,7 +4,6 @@ #include "rpc/dio_server.h" #include "crowdb-common/metrics/metrics.h" -#include "crowdb-protocol/frame.h" #include "crowdb-rpc/server/message.h" #include "crowdb-rpc/server/server.h" #include "disk/disk.h" @@ -182,23 +181,6 @@ crowdb::rpc::OutFrame *DiskioServer::handle_write(crowdb::rpc::Frame *request, c send_error_response(conn, req_id, create_nano, msg_type, static_cast(dproto::FBDiskIoRetCode_IoError)); return nullptr; } - // An EC shard is opaque DiskIO data. It can begin with the same two - // bytes as a public frame because the first data shard carries the - // original prefix, but it is not itself a frame sequence. Without an - // explicit content-kind field, only a single-frame request is - // unambiguously self-describing at this boundary. - if (data_buf != nullptr && size <= crowdb::protocol::kMaxFrameBytes && size >= 2 && - crowdb::protocol::valid_magic(crowdb::protocol::read_u16_le(data_buf->data))) { - const auto frame_status = - crowdb::protocol::validate_frame_sequence(std::span(data_buf->data, size)); - if (frame_status != crowdb::protocol::FrameError::Ok) { - data_buf->release(); - send_error_response(conn, req_id, create_nano, msg_type, - static_cast(dproto::FBDiskIoRetCode_IoError)); - return nullptr; - } - } - uint64_t ordering_phys_offset = zone->base_offset + ordering_zone_offset; auto started = std::chrono::steady_clock::now(); aligned_writer_.submit_ordered(disk, phys_offset, data_buf ? data_buf->data : nullptr, size, ordering_phys_offset, diff --git a/app/crowdb-diskio/tests/aligned_writer_test.cpp b/app/crowdb-diskio/tests/aligned_writer_test.cpp index 4a85ee89e..f1da3a1e3 100644 --- a/app/crowdb-diskio/tests/aligned_writer_test.cpp +++ b/app/crowdb-diskio/tests/aligned_writer_test.cpp @@ -90,22 +90,22 @@ class TestDisk final : public crowdb::diskio::Disk { } - crowdb::diskio::DiskType type() const override + [[nodiscard]] crowdb::diskio::DiskType type() const override { return crowdb::diskio::DiskType::Mem; } - int fd() const override + [[nodiscard]] int fd() const override { return 1; } - bool is_o_direct() const override + [[nodiscard]] bool is_o_direct() const override { return block_size_ > 1; } - size_t block_size() const override + [[nodiscard]] size_t block_size() const override { return block_size_; } @@ -115,7 +115,7 @@ class TestDisk final : public crowdb::diskio::Disk return engine_; } - crowdb::diskio::DiskId id() const override + [[nodiscard]] crowdb::diskio::DiskId id() const override { return {7, 9}; } diff --git a/app/crowdb-diskio/tests/blocking_engine_test.cpp b/app/crowdb-diskio/tests/blocking_engine_test.cpp index 48a83a828..1e1fb72f3 100644 --- a/app/crowdb-diskio/tests/blocking_engine_test.cpp +++ b/app/crowdb-diskio/tests/blocking_engine_test.cpp @@ -65,22 +65,22 @@ class TestDisk : public crowdb::diskio::Disk } } - crowdb::diskio::DiskType type() const override + [[nodiscard]] crowdb::diskio::DiskType type() const override { return crowdb::diskio::DiskType::Block; } - int fd() const override + [[nodiscard]] int fd() const override { return fd_; } - bool is_o_direct() const override + [[nodiscard]] bool is_o_direct() const override { return false; } - size_t block_size() const override + [[nodiscard]] size_t block_size() const override { return 1; } @@ -90,7 +90,7 @@ class TestDisk : public crowdb::diskio::Disk return engine_; } - crowdb::diskio::DiskId id() const override + [[nodiscard]] crowdb::diskio::DiskId id() const override { return id_; } diff --git a/app/crowdb-diskio/tests/dio_config_test.cpp b/app/crowdb-diskio/tests/dio_config_test.cpp index faced6dd1..5ae60700e 100644 --- a/app/crowdb-diskio/tests/dio_config_test.cpp +++ b/app/crowdb-diskio/tests/dio_config_test.cpp @@ -1,8 +1,8 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -#include "dio_config.h" #include "crowdb-common/runtime_path.h" +#include "dio_config.h" #include #include @@ -33,7 +33,7 @@ class TempConfig std::filesystem::remove(path_, error); } - const std::filesystem::path &path() const + [[nodiscard]] const std::filesystem::path &path() const { return path_; } diff --git a/app/crowdb-diskio/tests/dio_server_test.cpp b/app/crowdb-diskio/tests/dio_server_test.cpp index 4b058b256..af9d9bd6b 100644 --- a/app/crowdb-diskio/tests/dio_server_test.cpp +++ b/app/crowdb-diskio/tests/dio_server_test.cpp @@ -223,6 +223,10 @@ TEST(DiskioServerTest, WriteAndReadRoundTrip) for (uint32_t i = 0; i < DATA_SIZE; i++) { payload[i] = static_cast(i % 256); } + // A strip-boundary fragment may start with a frame magic without + // containing the complete frame. DiskIO must store that slice verbatim. + payload[0] = 0x01; + payload[1] = 0x03; uint64_t write_req_id = 10; Buffer *write_ctrl = build_write_request(pool, write_req_id, {1, 1}, 0, 0, DATA_SIZE, wall_time_ms()); diff --git a/app/crowdb-diskio/tests/sq_full_test.cpp b/app/crowdb-diskio/tests/sq_full_test.cpp index b19483e43..349e1536d 100644 --- a/app/crowdb-diskio/tests/sq_full_test.cpp +++ b/app/crowdb-diskio/tests/sq_full_test.cpp @@ -28,6 +28,7 @@ #include #include #include +#include #include #include @@ -74,17 +75,19 @@ TEST(SqFullBackpressureTest, BlockingEngineMoreJobsThanThreads) constexpr int NUM_IOS = 100; constexpr int DATA_SIZE = 4096; - std::atomic completed{0}; - std::vector payload(DATA_SIZE, 0xAB); + std::atomic completed{0}; + auto payload = std::make_shared>(DATA_SIZE, 0xAB); for (int i = 0; i < NUM_IOS; i++) { - std::vector data(payload); // Each write goes to a different offset so they don't overlap. off_t offset = static_cast(i * DATA_SIZE); - engine->submit_write(disk.get(), offset, data.data(), DATA_SIZE, [&completed, DATA_SIZE](int result) { - EXPECT_EQ(result, DATA_SIZE); - completed.fetch_add(1, std::memory_order_release); - }); + // The completion keeps the buffer alive until the asynchronous write finishes. + engine->submit_write(disk.get(), offset, payload->data(), DATA_SIZE, + [payload, &completed, DATA_SIZE](int result) { + static_cast(payload); + EXPECT_EQ(result, DATA_SIZE); + completed.fetch_add(1, std::memory_order_release); + }); } // Wait for all to complete. diff --git a/app/crowdb-diskio/tests/uring_engine_test.cpp b/app/crowdb-diskio/tests/uring_engine_test.cpp index e27c9a582..bcc74679a 100644 --- a/app/crowdb-diskio/tests/uring_engine_test.cpp +++ b/app/crowdb-diskio/tests/uring_engine_test.cpp @@ -73,22 +73,22 @@ class TestDisk : public crowdb::diskio::Disk } } - crowdb::diskio::DiskType type() const override + [[nodiscard]] crowdb::diskio::DiskType type() const override { return crowdb::diskio::DiskType::Block; } - int fd() const override + [[nodiscard]] int fd() const override { return fd_; } - bool is_o_direct() const override + [[nodiscard]] bool is_o_direct() const override { return false; } - size_t block_size() const override + [[nodiscard]] size_t block_size() const override { return 1; } @@ -98,7 +98,7 @@ class TestDisk : public crowdb::diskio::Disk return engine_; } - crowdb::diskio::DiskId id() const override + [[nodiscard]] crowdb::diskio::DiskId id() const override { return id_; } diff --git a/app/crowdb-web/ui/package-lock.json b/app/crowdb-web/ui/package-lock.json index a7ba7bf3a..620eb4869 100644 --- a/app/crowdb-web/ui/package-lock.json +++ b/app/crowdb-web/ui/package-lock.json @@ -1,12 +1,12 @@ { "name": "crowdb-console-frontend", - "version": "0.1.0", + "version": "0.2.0", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "crowdb-console-frontend", - "version": "0.1.0", + "version": "0.2.0", "dependencies": { "clsx": "^2.1.1", "lucide-react": "^0.456.0", diff --git a/app/crowdb-web/ui/package.json b/app/crowdb-web/ui/package.json index 19966ff9c..67e1ca1dd 100644 --- a/app/crowdb-web/ui/package.json +++ b/app/crowdb-web/ui/package.json @@ -1,7 +1,7 @@ { "name": "crowdb-console-frontend", "private": true, - "version": "0.1.0", + "version": "0.2.0", "type": "module", "description": "CrowDB Console SPA. Built with Vite + React + TypeScript + Tailwind. Compiled output in dist/ is served by crowdb-web (Axum) at runtime.", "scripts": { diff --git a/container/crowdb-monitor/Cargo.toml b/container/crowdb-monitor/Cargo.toml index 1ae82c36f..f4c0697aa 100644 --- a/container/crowdb-monitor/Cargo.toml +++ b/container/crowdb-monitor/Cargo.toml @@ -30,4 +30,5 @@ toml = "0.8" uuid = { version = "1", features = ["v4", "v7", "serde"] } [dev-dependencies] +crowdb-rpc-ffi = { path = "../../lib/crowdb-rpc/ffi", features = ["test-util"] } crowdb-test-harness = { path = "../../lib/crowdb-test-harness", features = ["kv-client"] } diff --git a/container/crowdb-monitor/src/bootstrap/iceberg.rs b/container/crowdb-monitor/src/bootstrap/iceberg.rs index b6dab5689..fc87c3f99 100644 --- a/container/crowdb-monitor/src/bootstrap/iceberg.rs +++ b/container/crowdb-monitor/src/bootstrap/iceberg.rs @@ -76,8 +76,8 @@ impl IcebergBootstrap { let service = profile .services .iter() - .find(|service| service.id == "iceberg") - .ok_or(IcebergBootstrapError::Profile("Iceberg service is absent"))?; + .find(|service| service.id == "access") + .ok_or(IcebergBootstrapError::Profile("Access service is absent"))?; let seeds = service .env .get("CROWDB_MANAGEMENT_SEEDS") @@ -247,6 +247,7 @@ impl ManagementCommand<'_> { let output = tokio::time::timeout( COMMAND_TIMEOUT, Command::new(self.program) + .arg("iceberg") .args(arguments) .env("CROWDB_MANAGEMENT_SEEDS", self.seeds) .env("CROWDB_ICEBERG_TOKEN", self.credentials.iceberg_manage_token()) diff --git a/container/crowdb-monitor/src/bootstrap/s3.rs b/container/crowdb-monitor/src/bootstrap/s3.rs index 2bc2b6197..08cb2eea0 100644 --- a/container/crowdb-monitor/src/bootstrap/s3.rs +++ b/container/crowdb-monitor/src/bootstrap/s3.rs @@ -78,8 +78,8 @@ impl S3Bootstrap { let service = profile .services .iter() - .find(|service| service.id == "s3") - .ok_or(S3BootstrapError::Profile("S3 service is missing"))?; + .find(|service| service.id == "access") + .ok_or(S3BootstrapError::Profile("Access service is missing"))?; let seeds = service .env .get("CROWDB_MANAGEMENT_SEEDS") @@ -89,11 +89,9 @@ impl S3Bootstrap { .get("CROWDB_S3_PUBLIC_URI") .cloned() .ok_or(S3BootstrapError::Profile("S3 public URI is missing"))?; - let iceberg_endpoint = profile - .services - .iter() - .find(|service| service.id == "iceberg") - .and_then(|service| service.env.get("CROWDB_ICEBERG_PUBLIC_URI")) + let iceberg_endpoint = service + .env + .get("CROWDB_ICEBERG_PUBLIC_URI") .cloned() .ok_or(S3BootstrapError::Profile("Iceberg public URI is missing"))?; let user = format!("preview-{}", session.manifest().deployment_id()); @@ -101,6 +99,7 @@ impl S3Bootstrap { let output = tokio::time::timeout( COMMAND_TIMEOUT, Command::new(&service.program) + .arg("s3") .args([action, &user]) .env("CROWDB_MANAGEMENT_SEEDS", seeds) .env("CROWDB_S3_MASTER_KEY", credentials.s3_master_key()) diff --git a/container/crowdb-monitor/src/preview.rs b/container/crowdb-monitor/src/preview.rs index 7dc8b7d0b..839fabc1b 100644 --- a/container/crowdb-monitor/src/preview.rs +++ b/container/crowdb-monitor/src/preview.rs @@ -288,20 +288,15 @@ async fn bootstrap_services( .await?; S3Bootstrap::reconcile(session, profile, credentials, supervisor.monitor_log_mut()).await?; IcebergBootstrap::reconcile(session, profile, credentials, supervisor.monitor_log_mut()).await?; - supervisor - .start_service( - "s3", - BTreeMap::from([("CROWDB_S3_MASTER_KEY".into(), credentials.s3_master_key().into())]), - ) - .await?; - let iceberg_environment = credentials + let mut access_environment: BTreeMap = credentials .server_env() .lines() .filter_map(|line| line.split_once('=')) .filter(|(name, _)| name.starts_with("CROWDB_ICEBERG_")) .map(|(name, value)| (name.to_owned(), value.to_owned())) .collect(); - supervisor.start_service("iceberg", iceberg_environment).await?; + access_environment.insert("CROWDB_S3_MASTER_KEY".into(), credentials.s3_master_key().into()); + supervisor.start_service("access", access_environment).await?; supervisor .start_service( "web", @@ -439,8 +434,8 @@ fn management_seed(profile: &DeploymentProfile) -> Result let service = profile .services .iter() - .find(|service| service.id == "s3") - .ok_or(PreviewError::Invalid("S3 service is absent"))?; + .find(|service| service.id == "access") + .ok_or(PreviewError::Invalid("Access service is absent"))?; let seeds = service .env .get("CROWDB_MANAGEMENT_SEEDS") diff --git a/container/crowdb-monitor/src/probe.rs b/container/crowdb-monitor/src/probe.rs index 36677114f..c49f7f6bc 100644 --- a/container/crowdb-monitor/src/probe.rs +++ b/container/crowdb-monitor/src/probe.rs @@ -10,7 +10,7 @@ use thiserror::Error; use tokio::net::TcpStream; use tokio::time::timeout; -use crate::{ProbeKind, ServiceProfile}; +use crate::{ProbeKind, ProbeProfile, ServiceProfile}; #[derive(Debug, Error)] pub enum ProbeError { @@ -112,14 +112,22 @@ impl ProbeExecutor { service: &ServiceProfile, environment: &BTreeMap, ) -> Result<(), ProbeError> { - let duration = Duration::from_millis(service.probe.timeout_ms); - match service.probe.kind { + self.probe_one(&service.probe, environment).await?; + for probe in &service.additional_probes { + self.probe_one(probe, environment).await?; + } + Ok(()) + } + + async fn probe_one( + &self, + probe: &ProbeProfile, + environment: &BTreeMap, + ) -> Result<(), ProbeError> { + let duration = Duration::from_millis(probe.timeout_ms); + match probe.kind { ProbeKind::Tcp => { - let address: SocketAddr = service - .probe - .target - .parse() - .map_err(|_| ProbeError::InvalidTarget)?; + let address: SocketAddr = probe.target.parse().map_err(|_| ProbeError::InvalidTarget)?; timeout(duration, TcpStream::connect(address)) .await .map_err(|_| ProbeError::Timeout)? @@ -127,11 +135,7 @@ impl ProbeExecutor { Ok(()) } ProbeKind::RpcPing => { - let address = service - .probe - .target - .parse() - .map_err(|_| ProbeError::InvalidTarget)?; + let address = probe.target.parse().map_err(|_| ProbeError::InvalidTarget)?; self.rpc .as_ref() .ok_or(ProbeError::Unavailable)? @@ -139,8 +143,8 @@ impl ProbeExecutor { .await } ProbeKind::Http => { - let mut request = self.client.get(&service.probe.target).timeout(duration); - if let Some(name) = &service.probe.bearer_env { + let mut request = self.client.get(&probe.target).timeout(duration); + if let Some(name) = &probe.bearer_env { let token = environment .get(name) .filter(|token| !token.is_empty()) diff --git a/container/crowdb-monitor/src/profile.rs b/container/crowdb-monitor/src/profile.rs index b39a921a1..87ff9b810 100644 --- a/container/crowdb-monitor/src/profile.rs +++ b/container/crowdb-monitor/src/profile.rs @@ -155,6 +155,8 @@ pub struct ServiceProfile { pub fence_listeners: Vec, pub config_template: Option, pub probe: ProbeProfile, + #[serde(default)] + pub additional_probes: Vec, pub restart: RestartProfile, } diff --git a/container/crowdb-monitor/src/profile/validation.rs b/container/crowdb-monitor/src/profile/validation.rs index ced408cad..4c5a310bc 100644 --- a/container/crowdb-monitor/src/profile/validation.rs +++ b/container/crowdb-monitor/src/profile/validation.rs @@ -206,6 +206,15 @@ fn validate_services(profile: &DeploymentProfile) -> Result<(), ProfileError> { } } validate_probe(service)?; + for probe in &service.additional_probes { + validate_probe_profile(&service.id, probe)?; + if probe.failure_threshold != service.probe.failure_threshold { + return invalid(format!( + "service {} probes must share a failure threshold", + service.id + )); + } + } let restart = &service.restart; if restart.max_attempts == 0 || restart.backoff_base_ms == 0 @@ -219,9 +228,12 @@ fn validate_services(profile: &DeploymentProfile) -> Result<(), ProfileError> { } fn validate_probe(service: &ServiceProfile) -> Result<(), ProfileError> { - let probe = &service.probe; + validate_probe_profile(&service.id, &service.probe) +} + +fn validate_probe_profile(service_id: &str, probe: &super::ProbeProfile) -> Result<(), ProfileError> { if probe.timeout_ms == 0 || probe.failure_threshold == 0 { - return invalid(format!("service {} has invalid probe bounds", service.id)); + return invalid(format!("service {service_id} has invalid probe bounds")); } if let Some(name) = &probe.bearer_env { if probe.kind != ProbeKind::Http @@ -231,17 +243,16 @@ fn validate_probe(service: &ServiceProfile) -> Result<(), ProfileError> { .all(|byte| byte.is_ascii_uppercase() || byte.is_ascii_digit() || byte == b'_') { return invalid(format!( - "service {} has an invalid probe credential reference", - service.id + "service {service_id} has an invalid probe credential reference" )); } } match probe.kind { ProbeKind::Http if !(probe.target.starts_with("http://") || probe.target.starts_with("https://")) => { - invalid(format!("service {} has invalid HTTP probe", service.id)) + invalid(format!("service {service_id} has invalid HTTP probe")) } ProbeKind::Tcp | ProbeKind::RpcPing if probe.target.parse::().is_err() => { - invalid(format!("service {} has invalid socket probe", service.id)) + invalid(format!("service {service_id} has invalid socket probe")) } ProbeKind::Http | ProbeKind::Tcp | ProbeKind::RpcPing => Ok(()), } diff --git a/container/crowdb-monitor/src/supervisor.rs b/container/crowdb-monitor/src/supervisor.rs index 1810adc82..9a867d015 100644 --- a/container/crowdb-monitor/src/supervisor.rs +++ b/container/crowdb-monitor/src/supervisor.rs @@ -63,12 +63,13 @@ impl Supervisor { .map(|service| service.id.clone()) .collect(); let processes = ProcessManager::new(log_root.to_owned(), profile.logs.clone()).await?; - let probes = ProbeExecutor::new( - profile - .services - .iter() - .any(|service| service.probe.kind == crate::ProbeKind::RpcPing), - )?; + let probes = ProbeExecutor::new(profile.services.iter().any(|service| { + service.probe.kind == crate::ProbeKind::RpcPing + || service + .additional_probes + .iter() + .any(|probe| probe.kind == crate::ProbeKind::RpcPing) + }))?; let status_store = StatusStore::new(run_root)?; let mut status = MonitorStatus::new(deployment_id, MonitorPhase::Initializing); status_store.publish(&mut status)?; diff --git a/container/crowdb-monitor/tests/access_bootstrap_test.rs b/container/crowdb-monitor/tests/access_bootstrap_test.rs index ce1dd5ef1..23133ca1b 100644 --- a/container/crowdb-monitor/tests/access_bootstrap_test.rs +++ b/container/crowdb-monitor/tests/access_bootstrap_test.rs @@ -39,7 +39,7 @@ fn profile(root: &TestRoot) -> DeploymentProfile { .unwrap(); let program = root.0.join("credential-command"); let script = format!( - "#!/bin/sh\nprintf '%s\\n' \"$1\" >> '{}'\nprintf 'rpc initialization log\\nAWS_ACCESS_KEY_ID=CROW123\\nAWS_SECRET_ACCESS_KEY=secret_123\\n'\n", + "#!/bin/sh\nprintf '%s\\n' \"$2\" >> '{}'\nprintf 'rpc initialization log\\nAWS_ACCESS_KEY_ID=CROW123\\nAWS_SECRET_ACCESS_KEY=secret_123\\n'\n", root.0.join("calls").display() ); fs::write(&program, script).unwrap(); @@ -47,7 +47,7 @@ fn profile(root: &TestRoot) -> DeploymentProfile { profile .services .iter_mut() - .find(|service| service.id == "s3") + .find(|service| service.id == "access") .unwrap() .program = program; profile diff --git a/container/crowdb-monitor/tests/iceberg_bootstrap_test.rs b/container/crowdb-monitor/tests/iceberg_bootstrap_test.rs index 245bd4861..5c5b27a4a 100644 --- a/container/crowdb-monitor/tests/iceberg_bootstrap_test.rs +++ b/container/crowdb-monitor/tests/iceberg_bootstrap_test.rs @@ -38,6 +38,7 @@ fn profile(root: &TestRoot) -> DeploymentProfile { r#"#!/bin/sh set -eu root='{}' +shift printf '%s\n' "$1" >> "$root/calls" if [ "$1" = inspect ]; then if [ ! -f "$root/initialized" ]; then @@ -70,7 +71,7 @@ exit 2 profile .services .iter_mut() - .find(|service| service.id == "iceberg") + .find(|service| service.id == "access") .unwrap() .program = program; profile diff --git a/container/crowdb-monitor/tests/probe_test.rs b/container/crowdb-monitor/tests/probe_test.rs index 78ad7d09b..0babdc18a 100644 --- a/container/crowdb-monitor/tests/probe_test.rs +++ b/container/crowdb-monitor/tests/probe_test.rs @@ -24,6 +24,7 @@ fn service(kind: ProbeKind, target: String) -> ServiceProfile { timeout_ms: 1000, failure_threshold: 1, }, + additional_probes: Vec::new(), restart: RestartProfile { max_attempts: 1, backoff_base_ms: 1, @@ -49,6 +50,24 @@ async fn tcp_probe_requires_a_listener() { .is_err()); } +#[tokio::test] +async fn all_service_probes_must_pass() { + let probes = ProbeExecutor::new(false).unwrap(); + let first = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let second = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let mut service = service(ProbeKind::Tcp, first.local_addr().unwrap().to_string()); + service.additional_probes.push(ProbeProfile { + kind: ProbeKind::Tcp, + target: second.local_addr().unwrap().to_string(), + bearer_env: None, + timeout_ms: 1000, + failure_threshold: 1, + }); + assert!(probes.probe_service(&service, &BTreeMap::new()).await.is_ok()); + drop(second); + assert!(probes.probe_service(&service, &BTreeMap::new()).await.is_err()); +} + #[tokio::test] async fn http_probe_requires_success_status() { let probes = ProbeExecutor::new(false).unwrap(); diff --git a/container/crowdb-monitor/tests/process_test.rs b/container/crowdb-monitor/tests/process_test.rs index 68f6c9107..bacf6c1ba 100644 --- a/container/crowdb-monitor/tests/process_test.rs +++ b/container/crowdb-monitor/tests/process_test.rs @@ -53,6 +53,7 @@ fn service(script: &str) -> ServiceProfile { timeout_ms: 100, failure_threshold: 1, }, + additional_probes: Vec::new(), restart: RestartProfile { max_attempts: 1, backoff_base_ms: 1, diff --git a/container/crowdb-monitor/tests/render_test.rs b/container/crowdb-monitor/tests/render_test.rs index 1dcf470c5..982dbf856 100644 --- a/container/crowdb-monitor/tests/render_test.rs +++ b/container/crowdb-monitor/tests/render_test.rs @@ -52,13 +52,13 @@ fn profile() -> DeploymentProfile { fn renders_profile_paths_and_topology_without_secrets() { let dirs = TestDirs::new(); let outputs = render_configs(&profile(), &dirs.templates(), &dirs.run()).unwrap(); - assert_eq!(outputs.len(), 8); + assert_eq!(outputs.len(), 7); assert_eq!( outputs .iter() .filter(|output| output.path == dirs.run().join("config/access.toml")) .count(), - 2 + 1 ); let diskio = fs::read_to_string(dirs.run().join("config/diskio.toml")).unwrap(); assert!(diskio.contains("path = \"/opt/crowdb/data/disks/disk-0004.img\"")); diff --git a/container/crowdb-monitor/tests/single_node_profile_test.rs b/container/crowdb-monitor/tests/single_node_profile_test.rs index e651e82f3..b73e8c165 100644 --- a/container/crowdb-monitor/tests/single_node_profile_test.rs +++ b/container/crowdb-monitor/tests/single_node_profile_test.rs @@ -39,29 +39,31 @@ fn single_node_preview_has_exact_topology_and_endpoints() { endpoints, BTreeMap::from([("iceberg", 80), ("s3", 81), ("web", 8080)]) ); - let iceberg = profile + let access_service = profile .services .iter() - .find(|service| service.id == "iceberg") + .find(|service| service.id == "access") .unwrap(); assert_eq!( - iceberg.env.get("CROWDB_ICEBERG_PUBLIC_URI"), + access_service.env.get("CROWDB_ICEBERG_PUBLIC_URI"), Some(&"http://localhost".to_owned()) ); - assert_eq!(iceberg.probe.target, "http://127.0.0.1:80/v1/config"); + assert_eq!(access_service.probe.target, "http://127.0.0.1:80/v1/config"); + assert_eq!(access_service.additional_probes.len(), 1); assert_eq!( - iceberg.env.get("CROWDB_MANAGEMENT_SEEDS"), + access_service.additional_probes[0].target, + "http://127.0.0.1:81/_crowdb/health/ready" + ); + assert_eq!( + access_service.additional_probes[0].failure_threshold, + access_service.probe.failure_threshold + ); + assert_eq!( + access_service.env.get("CROWDB_MANAGEMENT_SEEDS"), Some(&"http://127.0.0.1:10000".to_owned()) ); - let s3 = profile - .services - .iter() - .find(|service| service.id == "s3") - .unwrap(); - assert_eq!(s3.probe.target, "http://127.0.0.1:81/_crowdb/health/ready"); - assert_eq!(s3.args[1], "/opt/crowdb/run/config/access.toml"); - assert_eq!(iceberg.args[2], s3.args[1]); - assert_eq!(iceberg.config_template, s3.config_template); + assert_eq!(access_service.args[1], "/opt/crowdb/run/config/access.toml"); + assert_eq!(access_service.fence_listeners.len(), 2); let access = std::fs::read_to_string( Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates/access.toml"), ) @@ -94,7 +96,7 @@ fn single_node_preview_declares_complete_dependency_order() { .collect::>(); assert_eq!( order, - ["kv", "diskdb", "diskio", "chunkdb", "chunk-kv", "s3", "iceberg", "web"] + ["kv", "diskdb", "diskio", "chunkdb", "chunk-kv", "access", "web"] ); for service in &profile.services { if let Some(template) = &service.config_template { diff --git a/container/crowdb-monitor/tests/storage_bootstrap_test.rs b/container/crowdb-monitor/tests/storage_bootstrap_test.rs index 338806038..8ba6ee4a2 100644 --- a/container/crowdb-monitor/tests/storage_bootstrap_test.rs +++ b/container/crowdb-monitor/tests/storage_bootstrap_test.rs @@ -84,26 +84,14 @@ impl TestRoot { .to_string_lossy() .into_owned(), ], - "iceberg" => vec![ - "serve".into(), + "access" => vec![ "--config".into(), format!("{}/run/config/access.toml", self.0.display()), ], _ => unreachable!(), }; - if service.id == "iceberg" { - service.env.insert( - "CROWDB_MANAGEMENT_SEEDS".into(), - format!("http://127.0.0.1:{}", ports.kv_management), - ); - service.env.insert( - "CROWDB_ICEBERG_LISTEN".into(), - format!("127.0.0.1:{}", ports.iceberg), - ); - service.env.insert( - "CROWDB_ICEBERG_PUBLIC_URI".into(), - format!("http://127.0.0.1:{}", ports.iceberg), - ); + if service.id == "access" { + self.configure_access_env(service, ports); } service.fence_listeners = match service.id.as_str() { "kv" => vec![ports.kv_management, ports.kv_rpc], @@ -111,7 +99,7 @@ impl TestRoot { "diskio" => vec![ports.diskio_rpc], "chunkdb" => vec![ports.chunkdb_http, ports.chunkdb_rpc], "chunk-kv" => vec![ports.chunk_kv_http, ports.chunk_kv_rpc], - "iceberg" => vec![ports.iceberg], + "access" => vec![ports.iceberg, ports.s3], _ => unreachable!(), } .into_iter() @@ -123,14 +111,37 @@ impl TestRoot { "diskio" => format!("127.0.0.1:{}", ports.diskio_rpc), "chunkdb" => format!("http://127.0.0.1:{}/ready", ports.chunkdb_http), "chunk-kv" => format!("http://127.0.0.1:{}/ready", ports.chunk_kv_http), - "iceberg" => format!("http://127.0.0.1:{}/v1/config", ports.iceberg), + "access" => format!("http://127.0.0.1:{}/v1/config", ports.iceberg), _ => unreachable!(), }; + if service.id == "access" { + service.additional_probes[0].target = + format!("http://127.0.0.1:{}/_crowdb/health/ready", ports.s3); + } } profile.validate().unwrap(); profile } + fn configure_access_env(&self, service: &mut crowdb_monitor::ServiceProfile, ports: &Ports) { + service.env.insert( + "CROWDB_MANAGEMENT_SEEDS".into(), + format!("http://127.0.0.1:{}", ports.kv_management), + ); + service.env.insert( + "CROWDB_ICEBERG_LISTEN".into(), + format!("127.0.0.1:{}", ports.iceberg), + ); + service.env.insert( + "CROWDB_ICEBERG_PUBLIC_URI".into(), + format!("http://127.0.0.1:{}", ports.iceberg), + ); + service.env.insert( + "CROWDB_ACCESS_LOG_DIR".into(), + self.0.join("data/log/access").to_string_lossy().into_owned(), + ); + } + fn templates(&self, ports: &Ports) { let source = Path::new(env!("CARGO_MANIFEST_DIR")).join("../single-node-container/templates"); for name in [ @@ -155,7 +166,8 @@ impl TestRoot { .replace("127.0.0.1:12200", &format!("127.0.0.1:{}", ports.chunkdb_rpc)) .replace("127.0.0.1:15100", &format!("127.0.0.1:{}", ports.chunk_kv_http)) .replace("127.0.0.1:15200", &format!("127.0.0.1:{}", ports.chunk_kv_rpc)) - .replace("0.0.0.0:80", &format!("127.0.0.1:{}", ports.iceberg)); + .replace("0.0.0.0:80", &format!("127.0.0.1:{}", ports.iceberg)) + .replace("0.0.0.0:81", &format!("127.0.0.1:{}", ports.s3)); fs::write(self.0.join("templates").join(name), body).unwrap(); } } @@ -171,7 +183,7 @@ impl TestRoot { profile .services .iter() - .any(|service| service.id == "iceberg") + .any(|service| service.id == "access") .then(iceberg_step_names) .into_iter() .flatten() @@ -203,12 +215,13 @@ struct Ports { chunk_kv_http: u16, chunk_kv_rpc: u16, iceberg: u16, + s3: u16, } impl Ports { async fn allocate() -> Self { let mut listeners = Vec::new(); - for _ in 0..11 { + for _ in 0..12 { listeners.push(tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap()); } let ports = listeners @@ -227,6 +240,7 @@ impl Ports { chunk_kv_http: ports[8], chunk_kv_rpc: ports[9], iceberg: ports[10], + s3: ports[11], } } } @@ -398,7 +412,7 @@ async fn preview_real_iceberg_catalog_and_listener_survive_restart() { ("diskio", diskio_binary), ("chunkdb", binary_root.join("crowdb-chunkdb")), ("chunk-kv", binary_root.join("crowdb-chunk-kv-server")), - ("iceberg", binary_root.join("crowdb-iceberg")), + ("access", binary_root.join("crowdb-access-server")), ]; if binaries.iter().any(|(_, binary)| !binary.exists()) { eprintln!("skipping real Iceberg bootstrap: storage or Iceberg binary unavailable"); @@ -428,9 +442,10 @@ async fn preview_real_iceberg_catalog_and_listener_survive_restart() { IcebergBootstrap::reconcile(&mut session, &profile, &credentials, supervisor.monitor_log_mut()) .await .unwrap(); - let environment = iceberg_environment(&credentials); + let mut environment = iceberg_environment(&credentials); + environment.insert("CROWDB_S3_MASTER_KEY".into(), credentials.s3_master_key().into()); supervisor - .start_service("iceberg", environment.clone()) + .start_service("access", environment.clone()) .await .unwrap(); session.mark_ready().unwrap(); @@ -456,7 +471,7 @@ async fn preview_real_iceberg_catalog_and_listener_survive_restart() { ) .await .unwrap(); - restarted.start_service("iceberg", environment).await.unwrap(); + restarted.start_service("access", environment).await.unwrap(); restarted.mark_ready().await.unwrap(); restarted.shutdown().await.unwrap(); } diff --git a/container/crowdb-monitor/tests/supervisor_test.rs b/container/crowdb-monitor/tests/supervisor_test.rs index e7efeeb6d..792edc318 100644 --- a/container/crowdb-monitor/tests/supervisor_test.rs +++ b/container/crowdb-monitor/tests/supervisor_test.rs @@ -172,11 +172,7 @@ async fn repeated_exits_exhaust_budget_and_leave_unready() { async fn stable_health_resets_crash_loop_budget() { let roots = TestRoots::new(); let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); - let mut profile = roots.profile( - "sleep 0.3; exit 1".into(), - listener.local_addr().unwrap().port(), - 1, - ); + let mut profile = roots.profile("exec sleep 30".into(), listener.local_addr().unwrap().port(), 1); profile.services[0].restart.stable_after_ms = 50; let mut supervisor = Supervisor::new( profile, @@ -188,14 +184,30 @@ async fn stable_health_resets_crash_loop_budget() { .unwrap(); supervisor.start_service("kv", BTreeMap::new()).await.unwrap(); supervisor.mark_ready().await.unwrap(); - tokio::time::sleep(Duration::from_millis(350)).await; - supervisor.poll_once().await.unwrap(); - assert_eq!(supervisor.status().services["kv"].restart_attempts, 1); - tokio::time::sleep(Duration::from_millis(80)).await; - supervisor.poll_once().await.unwrap(); - assert_eq!(supervisor.status().services["kv"].restart_attempts, 0); - tokio::time::sleep(Duration::from_millis(300)).await; - supervisor.poll_once().await.unwrap(); + for generation in 1..=2 { + let pid = supervisor.status().services["kv"].pid.unwrap(); + let pid = rustix::process::Pid::from_raw(i32::try_from(pid).unwrap()).unwrap(); + rustix::process::kill_process(pid, rustix::process::Signal::KILL).unwrap(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(3); + loop { + let status = fs::read_to_string(format!("/proc/{}/status", pid.as_raw_pid())).unwrap(); + if status + .lines() + .any(|line| line.starts_with("State:") && line.contains('Z')) + { + break; + } + assert!(tokio::time::Instant::now() < deadline); + tokio::task::yield_now().await; + } + supervisor.poll_once().await.unwrap(); + if generation == 1 { + assert_eq!(supervisor.status().services["kv"].restart_attempts, 1); + tokio::time::sleep(Duration::from_millis(80)).await; + supervisor.poll_once().await.unwrap(); + assert_eq!(supervisor.status().services["kv"].restart_attempts, 0); + } + } assert_eq!(supervisor.status().services["kv"].generation, 3); supervisor.shutdown().await.unwrap(); let body = fs::read_to_string(roots.0.join("data/log/monitor/monitor.log")).unwrap(); diff --git a/container/single-node-container/Dockerfile b/container/single-node-container/Dockerfile index c32d125b2..c752da17a 100644 --- a/container/single-node-container/Dockerfile +++ b/container/single-node-container/Dockerfile @@ -5,7 +5,6 @@ RUN apt-get update && apt-get install -y --no-install-recommends ca-certificates RUN --mount=type=bind,source=.,target=/staged,ro \ mkdir -p /opt/crowdb \ && cp -a /staged/bin /staged/lib /opt/crowdb/ \ - && setcap cap_net_bind_service=+ep /opt/crowdb/bin/crowdb-iceberg \ && setcap cap_net_bind_service=+ep /opt/crowdb/bin/crowdb-access-server \ && mkdir -p /opt/crowdb/data /opt/crowdb/run \ && chown -R crowdb:crowdb /opt/crowdb/data /opt/crowdb/run diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index 92d40f85d..c2aab676d 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -6,6 +6,21 @@ This page describes building and running CROWDB from source on a Linux amd64 development or CI host. +This profile explicitly selects `test_single_node`. KV groups have one voter; +chunk writes use one 1 MiB mirror copy with no EC or conversion. The profile +provides no data protection, and a failed copy returns an I/O error. Production +requires at least three voting nodes and protected placement. + +The profile keeps chunk capacity separate from strip size. Tree chunks use +`storage.tree_chunk_capacity_bytes` in `chunk-kv.toml`; a 16 MiB tree chunk +contains multiple 1 MiB mirror strips. Stream chunks use +`storage.stream_chunk_capacity_bytes`. S3 and Iceberg each accept their own +`s3.small_write.chunk_capacity_bytes` or +`iceberg.small_write.chunk_capacity_bytes` and large `max_chunk_size` values +in `access.toml`. The shared `small_write` section remains a fallback. The +profile also sets RPC worker and client connection counts explicitly so local +resource use can be tuned without changing production defaults. + ```sh pixi run build-single-node-container pixi run test-single-node-container @@ -26,7 +41,7 @@ pixi run test-single-node-container `pixi run stage-single-node-container` produces the runtime directory without building a Docker image. Work on a `release/` branch whose `VERSION` -matches the branch name, such as `release/0.1.0`. After pushing each candidate +matches the branch name, such as `release/0.2.0`. After pushing each candidate commit, select that branch in the GitHub Actions manual run form, or dispatch it from a clean checkout that matches the remote branch: @@ -35,6 +50,11 @@ pixi run -- python tools/release.py --dry-run pixi run -- python tools/release.py --execute ``` +After the image passes container verification, publication updates both +`crowdb/crowdb-iceberg:` and `crowdb/crowdb-iceberg:latest` to the +same image digest. Rerunning publication from an older release branch also +updates `latest`, so use the newest release branch for the blog's moving tag. + The script only dispatches the workflow; it does not change files or push. The dry run does not contact GitHub. diff --git a/container/single-node-container/collect-libs.sh b/container/single-node-container/collect-libs.sh index 4276ad68a..c876d4fc3 100644 --- a/container/single-node-container/collect-libs.sh +++ b/container/single-node-container/collect-libs.sh @@ -8,7 +8,7 @@ mkdir -p "$output/bin" "$output/lib" for binary in \ crowdb-monitor crowdb-kv-server crowdb-diskdb crowdb-diskio \ crowdb-chunkdb crowdb-chunk-kv-server crowdb-access-server \ - crowdb-iceberg crowdb-web; do + crowdb-web; do if [[ "$binary" == crowdb-diskio ]]; then source="$build_root/app/crowdb-diskio/build/crowdb-diskio" else @@ -51,6 +51,10 @@ if [[ ! -f "$output/lib/libcrowdb_kv_client.so" ]]; then echo 'DiskIO FFI library was not collected' >&2 exit 1 fi +if [[ ! -f "$output/lib/libcrypto.so.3" ]]; then + echo 'the pixi OpenSSL runtime was not collected' >&2 + exit 1 +fi for library in "$output"/lib/*; do patchelf --set-rpath '/opt/crowdb/lib' "$library" done diff --git a/container/single-node-container/profile.toml b/container/single-node-container/profile.toml index de9b299aa..284de9356 100644 --- a/container/single-node-container/profile.toml +++ b/container/single-node-container/profile.toml @@ -171,35 +171,22 @@ backoff_base_ms = 250 backoff_max_ms = 5000 [[services]] -id = "s3" +id = "access" program = "/opt/crowdb/bin/crowdb-access-server" args = ["--config", "/opt/crowdb/run/config/access.toml"] -env = { CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_S3_REGION = "us-east-1", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } +env = { CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_S3_REGION = "us-east-1", CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_ICEBERG_GC_ENABLED = "0", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000", CROWDB_ACCESS_LOG_DIR = "/opt/crowdb/data/log/access" } dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] -fence_listeners = ["127.0.0.1:81"] +fence_listeners = ["127.0.0.1:80", "127.0.0.1:81"] config_template = "/opt/crowdb/etc/templates/access.toml" [services.probe] kind = "http" -target = "http://127.0.0.1:81/_crowdb/health/ready" +target = "http://127.0.0.1:80/v1/config" +bearer_env = "CROWDB_ICEBERG_READ_TOKEN" timeout_ms = 1000 failure_threshold = 5 -[services.restart] -max_attempts = 5 -backoff_base_ms = 250 -backoff_max_ms = 5000 - -[[services]] -id = "iceberg" -program = "/opt/crowdb/bin/crowdb-iceberg" -args = ["serve", "--config", "/opt/crowdb/run/config/access.toml"] -env = { CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_ICEBERG_GC_ENABLED = "0", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } -dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] -fence_listeners = ["127.0.0.1:80"] -config_template = "/opt/crowdb/etc/templates/access.toml" -[services.probe] +[[services.additional_probes]] kind = "http" -target = "http://127.0.0.1:80/v1/config" -bearer_env = "CROWDB_ICEBERG_READ_TOKEN" +target = "http://127.0.0.1:81/_crowdb/health/ready" timeout_ms = 1000 failure_threshold = 5 [services.restart] @@ -211,7 +198,7 @@ backoff_max_ms = 5000 id = "web" program = "/opt/crowdb/bin/crowdb-web" args = ["--config", "/opt/crowdb/run/config/crowdb-web.toml"] -dependencies = ["s3", "iceberg"] +dependencies = ["access"] fence_listeners = ["127.0.0.1:8080"] config_template = "/opt/crowdb/etc/templates/crowdb-web.toml" [services.probe] diff --git a/container/single-node-container/templates/access.toml b/container/single-node-container/templates/access.toml index 4eb6885f9..57c19e3bd 100644 --- a/container/single-node-container/templates/access.toml +++ b/container/single-node-container/templates/access.toml @@ -1,10 +1,14 @@ -# Access configuration shared by S3 and Iceberg processes. +# Access configuration shared by the S3 and Iceberg listeners. # Secrets stay in /opt/crowdb/data/secrets/server.env. +[deployment] +mode = "test_single_node" +max_node_failures = 0 + [common] management_seeds = ["http://127.0.0.1:10000"] -diskio_connections_per_endpoint = 2 -diskio_rpc_workers = 2 +diskio_connections_per_endpoint = 1 +diskio_rpc_workers = 1 [read] stream_window_bytes = 1048576 @@ -15,7 +19,8 @@ recovery_memory_bytes = 268435456 [small_write] threshold_ratio = 0.9 disk_block_bytes = 1048576 -conversion_enabled = true +conversion_enabled = false +mirror_copies = 1 ec_data = 2 ec_code = 1 memory_budget_bytes = 1342177280 @@ -23,8 +28,10 @@ queue_capacity = 1024 min_pipelines = 1 max_pipelines = 32 max_batch_bytes = 1048576 +chunk_capacity_bytes = 67108864 [s3] +large_mirror_copies = 1 listen = "0.0.0.0:81" tenant = "preview" region = "us-east-1" @@ -36,15 +43,33 @@ native_budget_bytes = 268435456 cleanup_backlog_limit = 10000 ec_data = 2 ec_code = 1 -max_chunk_size = 1073741824 +max_chunk_size = 67108864 + +[s3.small_write] +conversion_enabled = false +mirror_copies = 1 +disk_block_bytes = 1048576 +ec_data = 2 +ec_code = 1 +chunk_capacity_bytes = 67108864 [iceberg] +large_mirror_copies = 1 listen = "0.0.0.0:80" native_budget_bytes = 268435456 +max_chunk_size = 67108864 + +[iceberg.small_write] +conversion_enabled = false +mirror_copies = 1 +disk_block_bytes = 1048576 +ec_data = 2 +ec_code = 1 +chunk_capacity_bytes = 33554432 [iceberg.gc] enabled = false -interval_ms = 1000 +interval_ms = 60000 step_bytes = 8388608 step_ms = 1000 page_items = 64 diff --git a/container/single-node-container/templates/chunk-kv.toml b/container/single-node-container/templates/chunk-kv.toml index 8e9770005..1b64650b0 100644 --- a/container/single-node-container/templates/chunk-kv.toml +++ b/container/single-node-container/templates/chunk-kv.toml @@ -4,6 +4,7 @@ rpc_advertise_addr = "127.0.0.1:15200" http_listen_addr = "127.0.0.1:15100" group0_mgmt_seeds = ["http://127.0.0.1:10000"] catalog_refresh_interval_ms = 200 +rpc_workers = 1 [balance] enabled = false @@ -16,6 +17,10 @@ max_owner_request_rate = 0 [storage] metadata_store_id = 0 stream_mirror_copies = 1 +tree_chunk_capacity_bytes = 16777216 +stream_chunk_capacity_bytes = 16777216 +diskio_connections_per_endpoint = 1 +diskio_rpc_workers = 1 [bootstrap_partition] partition_id = { high = 1, low = 1 } diff --git a/container/single-node-container/templates/chunkdb.toml b/container/single-node-container/templates/chunkdb.toml index 1f5db6f6f..c40ad5e09 100644 --- a/container/single-node-container/templates/chunkdb.toml +++ b/container/single-node-container/templates/chunkdb.toml @@ -1,14 +1,23 @@ +[deployment] +mode = "test_single_node" +max_node_failures = 0 + [server] -rpc_workers = 2 +rpc_workers = 1 http_listen_addr = "127.0.0.1:12100" rpc_listen_addr = "127.0.0.1:12200" instance_id = "1" kv_server_mgmt_seeds = ["http://127.0.0.1:10000"] keepalive_interval_secs = 1 kv_pool_size = 1 -kv_rpc_workers = 2 +kv_rpc_workers = 1 diskdb_pool_size = 1 -diskdb_rpc_workers = 2 +diskdb_rpc_workers = 1 + +[conversion_io] +normal_connections_per_endpoint = 1 +priority_connections_per_endpoint = 1 +rpc_workers = 1 [topology] refresh_interval_secs = 1 @@ -23,5 +32,3 @@ lock_hold_warn_threshold_ms = 1000 [placement] mode = "unsafe_colocated" -allow_unsafe_ec = true -allow_degraded_failure_domains = true diff --git a/container/single-node-container/templates/diskdb.toml b/container/single-node-container/templates/diskdb.toml index 6ff07eac6..039ab6e49 100644 --- a/container/single-node-container/templates/diskdb.toml +++ b/container/single-node-container/templates/diskdb.toml @@ -1,5 +1,7 @@ [server] -rpc_workers = 2 +rpc_workers = 1 +kv_pool_size = 1 +kv_rpc_workers = 1 listen_addr = "127.0.0.1:11000" http_listen_addr = "127.0.0.1:11100" rpc_listen_addr = "127.0.0.1:11200" diff --git a/container/single-node-container/templates/diskio.toml b/container/single-node-container/templates/diskio.toml index c8f8e8989..854290858 100644 --- a/container/single-node-container/templates/diskio.toml +++ b/container/single-node-container/templates/diskio.toml @@ -1,13 +1,13 @@ [server] bind_address = "127.0.0.1" listen_port = 13000 -rpc_workers = 4 +rpc_workers = 1 node_id = {{node.0.id}} dummy_disk_type = "null" o_direct = true [engine] -thread_pool_size = 4 +thread_pool_size = 1 sq_entries = 256 [group0] diff --git a/container/single-node-container/templates/kv.toml b/container/single-node-container/templates/kv.toml index 0187ac8f6..54006f518 100644 --- a/container/single-node-container/templates/kv.toml +++ b/container/single-node-container/templates/kv.toml @@ -2,4 +2,5 @@ wal_early_ack = true async_engine_apply = true [server] -rpc_workers = 2 +rpc_workers = 1 +peer_pool_size = 1 diff --git a/container/single-node-container/tests/container-e2e.sh b/container/single-node-container/tests/container-e2e.sh index 15fc35b01..3e441f5e7 100644 --- a/container/single-node-container/tests/container-e2e.sh +++ b/container/single-node-container/tests/container-e2e.sh @@ -3,6 +3,7 @@ set -euo pipefail image=${CROWDB_CONTAINER_IMAGE:-crowdb-iceberg-single-node:dev} root=$(mktemp -d /tmp/crowdb-preview-e2e.XXXXXX) +layout_binary=$(mktemp /tmp/crowdb-chunk-layout.XXXXXX) name="crowdb-preview-e2e-$$" chmod 0777 "$root" @@ -27,9 +28,15 @@ cleanup() { --mount "type=bind,source=$root,target=/data" \ --entrypoint /bin/chmod "$image" -R 0777 /data >/dev/null 2>&1 || true rm -rf "$root" + rm -f "$layout_binary" } trap cleanup EXIT +cargo build --locked --release -p crowdb-chunkdb-client --example single_node_chunk_layout +cp target/release/examples/single_node_chunk_layout "$layout_binary" +patchelf --set-rpath /opt/crowdb/lib "$layout_binary" +chmod 0755 "$layout_binary" + start_container() { local storage_mode=${1:-bind} local mount_args=() @@ -98,6 +105,46 @@ verify_clients() { pixi run -e iceberg-e2e python container/single-node-container/tests/iceberg-client.py "$operation" } +verify_chunk_layouts() { + docker cp "$layout_binary" "$name:/tmp/chunk-layout-check" + docker exec "$name" /tmp/chunk-layout-check +} + +verify_listener_failure_propagation() { + local failed=$1 output status + if output=$(timeout 30 docker exec "$name" /bin/sh -ec ' + failed=$1 + config=/tmp/crowdb-access-listener-failure.toml + case "$failed" in + s3) + sed "s/0.0.0.0:80/127.0.0.1:18080/" /opt/crowdb/run/config/access.toml > "$config" + ;; + iceberg) + sed "s/0.0.0.0:81/127.0.0.1:18181/" /opt/crowdb/run/config/access.toml > "$config" + ;; + *) exit 2 ;; + esac + set -a + . /opt/crowdb/data/secrets/server.env + set +a + export CROWDB_ACCESS_LOG_DIR=/tmp/crowdb-access-listener-failure-log + export CROWDB_ICEBERG_PUBLIC_URI=http://127.0.0.1:80 + export CROWDB_S3_PUBLIC_URI=http://127.0.0.1:81 + exec /opt/crowdb/bin/crowdb-access-server --config "$config" + ' _ "$failed" 2>&1); then + echo "combined access process accepted an occupied $failed listener" >&2 + return 1 + else + status=$? + fi + if (( status == 124 )) || [[ "$output" != *'Address already in use'* ]]; then + echo "combined access $failed failure did not terminate as expected: status=$status" >&2 + printf '%s\n' "$output" >&2 + return 1 + fi + docker exec "$name" crowdb-monitor readiness +} + verify_web_logical() { local web_port manage_token status web_port=$(port 8080) @@ -292,13 +339,18 @@ verify_public_services node container/single-node-container/tests/web-ui.cjs "http://127.0.0.1:$(port 8080)" "$name" echo "checking S3 and Iceberg client writes" verify_clients write +echo "checking single-node protocol chunk layouts" +verify_chunk_layouts +echo "checking combined access listener failure propagation" +verify_listener_failure_propagation s3 +verify_listener_failure_propagation iceberg echo "checking Web logical writes" verify_web_logical -for service in kv diskdb diskio chunkdb chunk-kv s3 iceberg web; do +for service in kv diskdb diskio chunkdb chunk-kv access web; do echo "checking $service crash recovery" verify_child_recovery "$service" KILL child_exited done -for service in kv diskdb diskio chunkdb chunk-kv s3 iceberg web; do +for service in kv diskdb diskio chunkdb chunk-kv access web; do echo "checking $service hang recovery" verify_child_recovery "$service" STOP probe_failed done diff --git a/container/single-node-container/tests/iceberg-client.py b/container/single-node-container/tests/iceberg-client.py index 5003a13cb..fccee670c 100644 --- a/container/single-node-container/tests/iceberg-client.py +++ b/container/single-node-container/tests/iceberg-client.py @@ -1,3 +1,4 @@ +import hashlib import os import sys @@ -11,6 +12,8 @@ NAMESPACE = ("crowdb-preview-e2e",) TABLE = NAMESPACE + ("events",) ORDERS = NAMESPACE + ("orders",) +LARGE = NAMESPACE + ("large",) +LARGE_PAYLOAD = hashlib.shake_256(b"crowdb-single-node-large-file").digest(9 * 1024 * 1024) def main(): @@ -41,11 +44,16 @@ def main(): arrow_orders = pa.Table.from_pandas(orders, preserve_index=False) table = catalog.create_table(ORDERS, schema=arrow_orders.schema) table.append(arrow_orders) + large_schema = pa.schema([pa.field("payload", pa.binary())]) + large = catalog.create_table(LARGE, schema=large_schema) + large.append(pa.Table.from_pylist([{"payload": LARGE_PAYLOAD}], schema=large_schema)) assert catalog.namespace_exists(NAMESPACE) assert NAMESPACE in catalog.list_namespaces() assert catalog.load_namespace_properties(NAMESPACE) == {"preview": "persisted"} assert TABLE in catalog.list_tables(NAMESPACE) + assert LARGE in catalog.list_tables(NAMESPACE) assert catalog.load_table(TABLE).properties["preview"] == "persisted" + assert catalog.load_table(LARGE).scan().to_arrow().column("payload")[0].as_py() == LARGE_PAYLOAD saved = catalog.load_table(ORDERS).scan().to_pandas() revenue = ( saved[saved["status"] == "paid"] diff --git a/container/single-node-container/tests/image-smoke.sh b/container/single-node-container/tests/image-smoke.sh index 584d1507b..35658aa18 100644 --- a/container/single-node-container/tests/image-smoke.sh +++ b/container/single-node-container/tests/image-smoke.sh @@ -24,6 +24,9 @@ done docker run --rm --network none --entrypoint /opt/crowdb/bin/crowdb-monitor "$image" validate /opt/crowdb/etc/profile.toml docker run --rm --network none --entrypoint /bin/sh "$image" -ec ' + test -f /opt/crowdb/lib/libcrypto.so.3 + LD_LIBRARY_PATH=/opt/crowdb/lib ldd /opt/crowdb/bin/crowdb-access-server | + grep -F "libcrypto.so.3 => /opt/crowdb/lib/libcrypto.so.3" >/dev/null for tool in pixi cargo rustc gcc g++ cmake npm; do if command -v "$tool" >/dev/null 2>&1; then echo "Build tool was packaged into the runtime image: $tool" >&2 @@ -37,12 +40,10 @@ docker run --rm --network none --entrypoint /bin/sh "$image" -ec ' fi done ' -for binary in crowdb-iceberg crowdb-access-server; do - capability=$(docker run --rm --network none --entrypoint /sbin/getcap "$image" "/opt/crowdb/bin/$binary") - [[ "$capability" == *'cap_net_bind_service=ep' ]] -done +capability=$(docker run --rm --network none --entrypoint /sbin/getcap "$image" /opt/crowdb/bin/crowdb-access-server) +[[ "$capability" == *'cap_net_bind_service=ep' ]] -iceberg_output=$(docker run --rm --network none --entrypoint /opt/crowdb/bin/crowdb-iceberg "$image" 2>&1) && { +iceberg_output=$(docker run --rm --network none --entrypoint /opt/crowdb/bin/crowdb-access-server "$image" iceberg 2>&1) && { echo "Iceberg started without required configuration" >&2 exit 1 } diff --git a/container/single-node-container/tests/s3-client.py b/container/single-node-container/tests/s3-client.py index 0f76e9538..b82c87031 100644 --- a/container/single-node-container/tests/s3-client.py +++ b/container/single-node-container/tests/s3-client.py @@ -10,9 +10,11 @@ BUCKET = "crowdb-preview-e2e" KEY = "objects/persisted.parquet" +LARGE_KEY = "objects/large.bin" fixture = Path(__file__).resolve().parents[3] / "lib/crowdb-access-iceberg/tests/common/parquet_scalar_official.rs" encoded = re.search(r'pub const PARQUET_1_0_FALSE: &str = "([^"]+)"', fixture.read_text()).group(1) BODY = base64.b64decode(encoded) +LARGE_BODY = bytes(range(256)) * (9 * 1024 * 1024 // 256) assert BODY.startswith(b"PAR1") and BODY.endswith(b"PAR1") @@ -32,14 +34,17 @@ def main(): if sys.argv[1] == "write": client.create_bucket(Bucket=BUCKET) client.put_object(Bucket=BUCKET, Key=KEY, Body=BODY) + client.put_object(Bucket=BUCKET, Key=LARGE_KEY, Body=LARGE_BODY) assert BUCKET in {item["Name"] for item in client.list_buckets()["Buckets"]} listed = client.list_objects_v2(Bucket=BUCKET, Prefix="objects/") - assert [item["Key"] for item in listed["Contents"]] == [KEY] + assert {item["Key"] for item in listed["Contents"]} == {KEY, LARGE_KEY} head = client.head_object(Bucket=BUCKET, Key=KEY) assert head["ContentLength"] == len(BODY) assert head["LastModified"] is not None assert client.get_object(Bucket=BUCKET, Key=KEY)["Body"].read() == BODY assert client.get_object(Bucket=BUCKET, Key=KEY, Range="bytes=5-13")["Body"].read() == BODY[5:14] + assert client.head_object(Bucket=BUCKET, Key=LARGE_KEY)["ContentLength"] == len(LARGE_BODY) + assert client.get_object(Bucket=BUCKET, Key=LARGE_KEY)["Body"].read() == LARGE_BODY if __name__ == "__main__": diff --git a/container/single-node-container/tests/web-ui.cjs b/container/single-node-container/tests/web-ui.cjs index 805704a04..3ab3bd635 100644 --- a/container/single-node-container/tests/web-ui.cjs +++ b/container/single-node-container/tests/web-ui.cjs @@ -20,7 +20,7 @@ async function main() { await expect(page.getByTestId('managed-readonly')).toHaveText('Hardware topology is read-only', { timeout: 3000 }); await expect(page.getByRole('region', { name: 'Preview summary' })).toBeVisible({ timeout: 3000 }); await expect(page.getByTestId('managed-monitor-phase')).toContainText('Phase: ready', { timeout: 3000 }); - for (const service of ['kv', 'diskdb', 'diskio', 'chunkdb', 'chunk-kv', 's3', 'iceberg', 'web']) { + for (const service of ['kv', 'diskdb', 'diskio', 'chunkdb', 'chunk-kv', 'access', 'web']) { await expect(page.getByTestId(`managed-process-${service}`)).toContainText(/PID \d+ · generation \d+/, { timeout: 3000 }); } await expect(page.getByTestId('managed-unavailable')).toHaveCount(0, { timeout: 3000 }); diff --git a/doc/backlog/R193-chunkdb-node-failure-budget.md b/doc/backlog/R193-chunkdb-node-failure-budget.md new file mode 100644 index 000000000..e2c5159f5 --- /dev/null +++ b/doc/backlog/R193-chunkdb-node-failure-budget.md @@ -0,0 +1,47 @@ + + + +### R193: chunkdb — Configurable node failure budget and EC placement + +#### Status + +Planned. The explicit single-node and three-node profiles, persisted strip layouts, and degraded-placement repair baseline are complete. + +#### Problem + +The current system defines two deployment contracts: one test node with no node-failure tolerance and three production nodes tolerating one failed node. The current EC selector checks the maximum fragments on one node against the parity count. That check cannot express a larger failure budget: with six nodes and two allowed failures, `4+2` and `8+4` can survive any two nodes, while `2+1` cannot. Mirror copy counts and journal/tree write policies are also configured independently rather than derived from one system protection contract. See [chunkdb placement](../design/chunkdb/design-crowdb-chunkdb.md) and [chunk IO](../design/chunkio/design-crowdb-chunkio.md). + +Operators need to choose a node failure budget for a deployment without accidentally admitting an EC shape or mirror layout that loses committed data within that budget. When nodes fail, new placement must use the remaining budget and record any loss of the full-cluster protection target for repair after recovery. + +#### Solution + +The system property `max_node_failures` is the number of unavailable nodes the configured deployment promises to tolerate from its complete topology. It is a protection target, not an automatic instruction to reduce copies each time a node fails. The remaining budget is the target minus the nodes already unavailable. A production profile must have enough voting KV replicas to retain quorum at that target and enough distinct storage nodes to place its selected layouts. The explicit test-single-node profile has a zero budget. Startup and allocation reject mismatched service policy, impossible budgets, and unsafe layouts; failure never silently switches deployment mode. + +For a mirror strip, the full protection target requires at least `max_node_failures + 1` copies on distinct nodes. Continue to use the full copy count after a failure when placement permits it. For an EC strip with `k` data and `m` parity fragments, sort per-node fragment counts descending; the sum of the largest `max_node_failures` counts must be at most `m` in a healthy topology. Apply the corresponding remaining-budget check to new allocations after failures. Keep fragments spread across available nodes even when no further node-failure budget remains. Persist actual strip geometry and its full protection target separately so readers use the real layout and repair can restore the target. + +1. Validate the deployment property and KV/storage topology consistently in deployment configuration, ChunkDB startup, and access/chunk writer policy. Preserve the existing one-node/zero-failure and three-node/one-failure profiles. +2. Replace the one-node EC bound in ChunkDB placement and physical validation with the worst-case sum across the configured number of failed nodes. Select mirror copy counts from the deployment contract, while retaining explicit per-strip policy only when it meets or exceeds the target. +3. During an outage within the configured budget, try full protection first. If it cannot fit, permit a layout that meets the remaining budget, persist a degraded-placement marker, and create a durable placement task. Reject allocation if even the remaining-budget layout or KV quorum is unavailable. Never claim full protection for a degraded strip. +4. After capacity returns, use fenced placement tasks to move or rebuild fragments until the original target holds. Reads and writes follow each strip's persisted geometry throughout migration; a restart resumes unfinished tasks without accepting stale placement. + +#### Dependencies + +- The existing two deployment profiles, strip dispatch, and placement-repair baseline are the starting point for generalized validation. +- R103 is responsible for ChunkDB range-owner migration after a ChunkDB instance failure. This requirement's storage placement budget does not replace metadata service failover; end-to-end availability depends on both. +- R139 may later distribute the system property through Group 0. Until then, startup must reject inconsistent local configuration rather than assume a remote configuration service exists. + +#### Acceptance + +- Given six voting KV/storage nodes and `max_node_failures = 2`, start services; startup accepts the budget and reports the configured target. Given an impossible budget or too few voters, startup rejects it before writes. Integration test. +- Given six healthy nodes with budget two, allocate `4+2` and `8+4` EC strips; every pair of nodes owns at most two and four fragments respectively. Request `2+1`; allocation rejects it because a pair can own more than one fragment. Integration test. +- Given six healthy nodes with budget two, allocate mirror strips used by object data, journal, and tree pages; each new strip has at least three copies on distinct nodes and reads from its persisted layout. Integration test. +- Given a six-node budget-two cluster, stop one node and allocate new mirror and EC strips; placement retains the full target where possible, otherwise uses the remaining one-failure budget and persists a repair task without changing deployment mode. E2E test. +- Given the same cluster with two nodes unavailable, allocate while KV quorum and a valid remaining-budget placement exist; operations succeed without claiming that a third failure is tolerated. Remove enough further capacity or quorum; allocation returns an error. E2E test. +- Given a degraded EC strip and recovered nodes, restart the repair worker and complete its task; data remains readable during movement, the task survives restart, and the final placement again passes the full two-node-failure check. E2E test. +- Given the existing single-node and three-node configurations, run their write/read and one-node-out suites after introducing the generalized policy; their configured budgets and persisted layouts remain valid. E2E test. + +Run `pixi run rs-fmt-check`, `pixi run rs-lint`, `pixi run cargo test -p crowdb-kv`, `pixi run cargo test -p crowdb-chunkdb`, `pixi run cargo test -p crowdb-chunk-client`, and `pixi run cargo test -p crowdb-chunk-stream` for the implemented scope. + +#### Open Questions + +None. diff --git a/doc/backlog/R194-access-iceberg-object-listing.md b/doc/backlog/R194-access-iceberg-object-listing.md new file mode 100644 index 000000000..43863b81a --- /dev/null +++ b/doc/backlog/R194-access-iceberg-object-listing.md @@ -0,0 +1,53 @@ + + + +### R194: access-iceberg — Native object listing and S3-style address semantics + +#### Status + +Deferred until the client-use and address-model research below is complete. Exact-object FileIO already supports the small TPC-H and TPC-DS loader flow; listing is not a prerequisite for that flow. + +#### Problem + +The native Iceberg file endpoint accepts signed operations on exact immutable objects, but rejects `ListObjectsV2`. [PyIceberg's PyArrow S3 FileIO](https://py.iceberg.apache.org/reference/pyiceberg/io/pyarrow/) calls PyArrow `get_file_info` for `exists` and length, which can issue `ListObjectsV2` even when the caller has an exact object path. In a local SF 0.01 loader run, that call failed before upload with `InvalidRequest`. A CrowDB-specific FileIO can use exact-object HEAD-backed reads and complete this load, but clients that intentionally list prefixes still lack an answer. [Iceberg's FileIO guide](https://iceberg.apache.org/docs/latest/fileio/) names read, write, and seek as essential file operations and tracks data paths in table metadata; it does not make S3 prefix listing an essential FileIO operation. This requirement must establish which clients actually need S3-style listing before extending the native file endpoint. + +The current S3-shaped file location uses an encoded catalog ID in the URI authority field and a `t//` key prefix. This is a routing convention, not a declared Iceberg bucket. The native service has no general S3 bucket authority or file DELETE, and the general S3 service has separate authority. See [native Iceberg design](../design/access-server/iceberge/design-crowdb-iceberg.md) sections 3 and 6. + +#### Solution + +Research and record the exact API call sequences of PyIceberg, Arrow, DuckDB, and any engine accepted under R189. Distinguish incidental listing used for exact-object existence from intentional prefix discovery. Check the Iceberg FileIO and REST Catalog contracts separately; compatibility with an S3-shaped URI alone does not make S3 bucket/list semantics part of Iceberg. + +Choose an address model only after that research. The S3 request's bucket field could represent a catalog, a table, or an opaque native routing scope. Preserve catalog/table IDs as first-class Iceberg authorities and avoid creating general S3 bucket records or granting cross-table discovery by default. Document what each choice means for existing table locations, catalog isolation, credentials, pagination, and future multiple-catalog deployments. There is no historical-data compatibility requirement, but an address change must still be atomic for active tables and clients. + +If intentional listing is needed, implement only the chosen native listing contract: + +1. Extend `app/crowdb-access-server/src/iceberg/file_request.rs` and the native route selection to parse and validate `ListObjectsV2` parameters, including prefix, delimiter, continuation token, encoding, and page limit. Reject unsupported or ambiguous requests before storage access. +2. Add authorized, bounded prefix scans over published file records in `lib/crowdb-access-iceberg/src/file/repository.rs` and its catalog storage. Return only files visible to the caller's table scope; exclude drafts, uncommitted uploads, losing CAS candidates, and retired files according to a documented visibility rule. +3. Extend `lib/crowdb-access-iceberg/src/file/credentials.rs` and credential vending only if a distinct list permission is needed. Bind it to the chosen routing scope and table authorization, with no privilege inherited from the general S3 service. +4. Make pagination stable under concurrent publication and deletion. Bind continuation tokens to the caller, scope, prefix, delimiter, and listing generation or equivalent consistent cursor. Bound scan work and response size; reject malformed, expired, or cross-scope tokens. +5. Add protocol and official-client tests for clients shown by the research to require listing. Keep exact-object FileIO functional without listing and keep the separate general S3 authority unchanged. + +#### Dependencies + +- R189 client and engine acceptance supplies the observed call sequences and determines which listing cases have user value. Until this requirement is implemented, use an exact-object FileIO for clients that only need Iceberg table files. +- The native file, catalog, and credential contracts in the [Iceberg design](../design/access-server/iceberge/design-crowdb-iceberg.md) define the present authority boundary. General S3 listing is not a fallback for native Iceberg files. +- If research finds no client that needs intentional prefix listing, close this requirement with the client evidence and retain only exact-object FileIO adapters. + +#### Acceptance + +- Given the chosen supported clients and a current container, trace an exact-file open and an intentional prefix query; record which client issues each request and which Iceberg or S3 interface it relies on. Assert the decision does not infer an Iceberg listing requirement from a PyArrow existence probe alone. Integration test. +- Given each proposed address model, resolve two catalogs and two tables with different principals; assert routing and authorization cannot expose another table's names or files. Select and document one model before implementing the endpoint. Integration test. +- Given an authorized table with committed, staged, abandoned, and retired file records, request a native listing prefix; assert only the chosen visible set appears, while an unauthorized principal sees none. Integration test. +- Given more matching files than one page, request successive pages with prefix and delimiter; assert bounded pages have no duplicate or omitted eligible keys and tokens cannot be replayed under another principal or scope. Integration test. +- Given concurrent file publication or reclamation during pagination, continue listing; assert the documented snapshot or cursor rule, with bounded work and no cross-table leakage. Integration test. +- Given malformed parameters, foreign catalog/table routes, and attempts to use general S3 credentials, issue a native list request; assert fail-closed responses before scanning. E2E test. +- Given exact-object PyIceberg reads and writes, run the existing native FileIO tests after any routing change; assert they still work without `ListObjectsV2`. E2E test. + +Run `pixi run rs-fmt-check`, `pixi run rs-lint`, `pixi run cargo test -p crowdb-access-iceberg`, and `pixi run cargo test -p crowdb-access-server` for the implemented scope. + +#### Open Questions + +- Which supported client operations require intentional prefix listing rather than an exact-object existence or length check? Is the behavior required by the Iceberg FileIO or REST Catalog specification, or by a particular S3 client implementation? +- Should the S3 bucket field map to a catalog, a table, or an opaque native routing scope? A catalog keeps existing locations compact but makes per-table isolation rely on key prefixes; a table makes isolation explicit but affects location and credential vending; an opaque scope allows routing evolution but is less readable to clients. +- Should listing include only files reachable from current table snapshots, all retained snapshots, or every published immutable file awaiting reclamation? How should that choice interact with namespace/table deletion and concurrent commits? +- Is a native `ListObjectsV2` endpoint worth maintaining if supported clients can instead use exact-object FileIO and Iceberg metadata enumeration? diff --git a/doc/backlog/R195-access-shared-large-upload-flow.md b/doc/backlog/R195-access-shared-large-upload-flow.md new file mode 100644 index 000000000..62b2faea1 --- /dev/null +++ b/doc/backlog/R195-access-shared-large-upload-flow.md @@ -0,0 +1,142 @@ + + + +### R195: access server — TPC Iceberg object upload performance + +#### Problem + +The TPC loader writes Parquet objects through Iceberg FileIO. A measured +100-MiB upload to the single-node container took about 5 seconds, mostly +while closing its multipart output stream. A focused direct PUT in the +small-cluster test took about 1.36 seconds. These are different client paths, +but the gap requires tracing the real TPC route. The Iceberg HTTP loop awaits +each `ChunkIoWriter::on_framed_data` call before polling the next body buffer; +the mirror writer can await DiskIO before accepting more data. Receive, +digest, and durable writing therefore overlap poorly. Multipart session and +publication work may add further latency. See the [Iceberg upload-flow +analysis](../design/access-server/iceberge/design-crowdb-iceberg-upload-flow.md), +[access server design](../design/access-server/design-crowdb-access-server.md), +and [chunk IO design](../design/chunkio/design-crowdb-chunkio.md). + +#### Solution + +Implement the object-scoped, bounded producer/consumer write flow first, then +measure the complete 100-MiB TPC FileIO upload, including multipart part +transfers and CompleteMultipart, and reduce its dominant costs. The goal is +the fastest practical upload on the documented single-node profile without +changing durability, integrity, or publication semantics. There is no fixed +seconds threshold: compare before and after results on the same host and +profile, retain stage evidence, and stop when further changes add complexity +without a measured benefit. Parquet generation and the later Iceberg table +snapshot commit are reported separately from FileIO upload latency. + +One object-scoped Iceberg write owner retains the parsed request context, +body bounds, digest state, writer, and terminal result for each large file PUT +or multipart part. Large requests may overlap receive, digest, and +chunk writes under bounded write-flow backpressure. The chunk writer keeps +exclusive mutable ownership of its state; do not introduce a hot-path lock, +per-frame virtual dispatch, or a kernel wake for every buffer. The digest +consumes zero-copy logical payload views; SHA-256 is computed only when the +request requires it. The write owner waits for body validation, digest, and +durable writer completion before Iceberg publishes a part or file record. + +Implement the flow regardless of the baseline timing. Apply further +optimizations only where supported by the measured stage breakdown. +S3 and Iceberg PUT and multipart UploadPart use one shared producer/consumer +driver below protocol publication policy. The same writer handoff applies to +small objects after body receive; their distinct shared small-write pipeline +remains responsible for durable chunk placement. Multi-node EC throughput and +small-write performance parity are outside this work. + +The following invariants define the work: + +- **I1 — Bounded overlap.** Only one task fetches an object's socket body. + It offers each completed owner to the write flow and immediately drives an + idle writer in the same upload task. Four completed owners may be held while + up to four independent mirror-strip writes are in flight by default; both + limits are separately configurable. The writer's next dequeue resumes paused fetch + without timer polling or a wake on every frame. The global native buffer + budget bounds retained receive memory, including digest references. +- **I2 — Correct bytes.** The fetch layer prepares frame headers and CRC32C + over placement-independent bytes. The writer fills the actual chunk ID + after placement; chunk ID is excluded from CRC32C. The digest sees ordered + logical payload, never + frame headers or footers. Partial frames and chunk rotation remain valid + without an object-sized copy. +- **I3 — Durable publication.** Accepted buffers stay owned until writer + and digest views finish. A completed strip keeps its buffer until every + preceding strip commits in order. A failed mirror segment is replaced and + replayed before later results can commit. A part or file becomes visible only after decoded + body length, digest, all required mirror/EC writes, fsyncs, seals, and + metadata preconditions succeed. Failed or ambiguous publication follows + the existing authoritative recovery rules. +- **I4 — Measured costs.** Record attempts, completions, errors, bytes, + current and peak owners and writes in flight, plus counts and cumulative + wait time for socket input, write-flow pause, digest capacity, writer capacity, + DiskIO completion, and metadata publication. Record end-to-end latency + separately because concurrent stage times overlap. Use fixed metric + dimensions and avoid one shared atomic update per 64-KiB frame. Preserve + failure and cancellation measurements without routine buffer logs. + +Work items: + +1. Consolidate Iceberg `file_http` request write state and completion into an + object-scoped owner. Share the body handoff, digest worker, and writer + scheduling between S3 and Iceberg PUT and UploadPart, while retaining + separate publication and authorization rules. +2. Add bounded receive/digest/write overlap and ordered concurrent mirror + strip completion. Use the same write-consumer handoff for small objects, + retaining their separate shared small-write pipeline. Maintain buffer + lifetime, frame integrity, and failure fencing. +3. Use the real TPC loader/FileIO route and R196's benchmark to compare + client preparation, UploadPart, CompleteMultipart, digest, chunk writes, + and metadata publication. Record part size, concurrency, topology, + durability settings, software revision, and host with every result. +4. Expose fixed-stage metrics through the Iceberg metrics surface and retain + before/after snapshots with raw samples. Explain any remaining dominant + cost when the chosen implementation stops improving. + +#### Dependencies + +- R196 provides a reusable HTTP benchmark and regression result format. A + focused existing FileIO test may be used while R196 is implemented, but + R195 completion requires a reproducible measurement of the real TPC route. +- The current native receive provider, `FramedWriteBuffer`, Iceberg + multipart authority, and chunk writer are the baseline. Include only the + bounded digest behavior needed here if its current implementation is + unmerged. +- S3 and multi-node EC performance remain observable through R196 and their + existing tests, but are not completion gates for R195. + +#### Acceptance + +- Given a prepared 100-MiB TPC Parquet object and one documented single-node + profile, run repeated warm FileIO uploads before and after the change; + assert zero failed or incomplete operations, correct publication, retained + raw samples and stage breakdown, and a material improvement in the + dominant measured stage without a regression in total upload time + (I1–I4). E2E test. +- Given delayed digest and DiskIO completions during a large part upload, + continue receiving while offer returns continue; assert receive and write + overlap, memory stays within the native budget, actual waits have counts + and durations, and no + referenced buffer is freed early (I1, I4). Integration test. +- Given a delayed first mirror failure after later writes complete, retain + later buffers and their order, replay the failed segment into a replacement, + then seal and read back the exact object (I1–I3). Integration test. +- Given S3 and Iceberg multipart parts and ordinary PUTs, upload equal payloads + through the shared handoff, validate their MD5 and optional SHA-256, and + assert both protocols publish only their own completed locations (I1–I3). + Integration test. +- Given a small S3 or Iceberg object, feed the same write consumer and finish + through its shared small-write pipeline without changing publication or + digest behavior (I1–I3). Integration test. +- Given a wrong digest, truncated body, failed write, or ambiguous part + publication, stop or drain the upload; assert no invalid part or file + becomes visible, authoritative metadata is checked before cleanup, and + metrics retain the failed stage and elapsed time (I2–I4). Integration test. +- Given a completed 100-MiB upload, read a bounded first, middle, and final + range after timing; assert bytes match the source and no per-upload + readback was included in latency (I2, I3). E2E test. + +Run `pixi run rs-fmt-check`, `pixi run rs-lint`, `pixi run cargo test -p crowdb-access-server --features iceberg-e2e`, and the focused R196 Iceberg upload regression through `pixi run` for the implemented scope. diff --git a/doc/backlog/R196-access-upload-benchmark-regression.md b/doc/backlog/R196-access-upload-benchmark-regression.md new file mode 100644 index 000000000..7cca08e31 --- /dev/null +++ b/doc/backlog/R196-access-upload-benchmark-regression.md @@ -0,0 +1,155 @@ + + + +### R196: access server — S3 and Iceberg upload benchmark regression + +#### Problem + +The current `crowdb-cli bench s3` measures an invocation-owned memory-backed +cluster, not an HTTP request through a durable access server. The S3 E2E Python +benchmark captures request samples inside a full-stack test but has no reusable +regression command or failure sentinel. Iceberg has focused upload timings, but +no equivalent repeatable benchmark. The chunk IO regression script exercises +the writer below HTTP decoding, request integrity, and protocol publication. +Consequently, a change can improve chunk IO while slowing S3 or Iceberg PUT, +multipart part upload, or metadata publication without a comparable measurement. +See the [access server design](../design/access-server/design-crowdb-access-server.md), +[Iceberg upload-flow analysis](../design/access-server/iceberge/design-crowdb-iceberg-upload-flow.md), +and [shared upload-flow requirement](R195-access-shared-large-upload-flow.md). + +#### Solution + +Extend `crowdb-cli bench` with one real HTTP upload workload driver that uses +the same measurement, payload generation, concurrency, and result schema for +S3 and Iceberg. Keep the existing memory-backed `bench s3` contract unchanged. +Protocol adapters perform authentication, target setup, direct PUT or multipart +requests, response validation, and cleanup. A timed sample starts immediately +before request submission and ends only when the server acknowledges the +durable write and the protocol's corresponding object, file, or part record is +published. Multipart completion is a separate timed phase; an Iceberg table +snapshot commit is not included in FileIO upload latency and is reported +separately if exercised. + +The workload reuses a prepared, deterministic 1-MiB payload block for large +objects and computes expected integrity values before timing. Streaming does +not allocate or copy an object-sized client buffer, hash the full object in the +timed loop, or read the object back after every successful upload. A bounded +sample outside the timed interval verifies stored bytes and range boundaries. +The CLI reports client preparation and transfer timings separately so client +work is not mistaken for server time. + +The regression scripts own a reproducible local single-node storage and +access-server deployment by default, using the project's existing deployment +and configuration facilities. They retain service logs, metrics snapshots, +machine-readable per-operation samples, and a summary under one run directory. +An explicit remote-endpoint mode uses the same workload against a supplied +server without claiming local process or storage metrics. The scripts compare +only like-for-like topology and configuration profiles. + +The following invariants define the benchmark: + +- **I1 — Real protocol path.** Every timed S3 and Iceberg sample crosses the + HTTP access server, request integrity checks, chunk writer, and the relevant + protocol publication. The report distinguishes direct PUT, multipart part, + multipart completion, and optional table commit. +- **I2 — Controlled input.** Size, concurrency, operation count or duration, + warmup, payload seed, request integrity mode, and deployment profile are + recorded. Client payload memory is bounded independently of object size. + Preparation, namespace/table/session setup, final verification, and cleanup + are outside the timed upload samples. +- **I3 — Honest completion.** The driver records admitted, completed, failed, + and incomplete operations, drains admitted work before exiting, validates + success responses and ETags/digests, and exits nonzero for any failed, + missing, or unverified operation. A timeout or metrics-collection failure + remains visible in the retained result. +- **I4 — Comparable measurements.** Both protocols emit the same JSON and TSV + fields for object size, concurrency, request count, logical bytes, elapsed + time, throughput, average, p50, p95, p99 when sample counts support them, + and error classes. Percentiles with too few samples are absent, not inferred. + The report includes client CPU/RSS and, for local runs, access-server + CPU/RSS and before/after server metric deltas. The shared write-flow stage + counters from R195 are included when available, with concurrent stage + durations reported separately from wall-clock latency. +- **I5 — Regression decision.** The default sentinel gates correctness, + completion, timeout, artifact presence, and required metric availability. + Throughput and latency comparisons use an explicitly selected baseline for + the same profile and a documented tolerance; they do not use one fixed + hardware-dependent number. The result states whether a case was measured, + comparable, regressed, or invalid, and names the failing condition. + +Work items: + +1. Add a real HTTP upload verb and shared result model under + `app/crowdb-cli/src/commands/bench/` and its workload implementation under + `lib/crowdb-console-shared/src/ops/`. The CLI accepts protocol, endpoint, + direct or multipart mode, workload bounds, output path, and credentials via + existing environment/config conventions. Reuse the same input producer and + percentile/accounting logic for both protocol adapters. Do not route this + workload through the memory-backed `bench s3` engine. +2. Implement S3 and Iceberg upload adapters. For S3, cover direct PUT and + UploadPart plus CompleteMultipart. For Iceberg, cover native FileIO direct + PUT and multipart part plus completion against an authorized table/file + scope. Preserve each protocol's signing, integrity, ETag, and publication + rules; never count a successful part as a published final object. Report + client-side signing or hashing work separately if the selected integrity + mode requires it. +3. Add `tools/benchmark/` regression scripts modeled on + `bench-chunkio-write-regression.sh`. Use a named local deployment profile, + build required binaries through `pixi run`, start and clean up the stack, + execute isolated S3 and Iceberg cases, retain full output and service + metrics, and print a compact per-case summary. Provide bounded timeouts + and a case filter for focused runs. Support explicit remote endpoints + without trying to destroy a remote deployment. +4. Include a small-path case below the configured threshold, a boundary + case, direct 100-MiB PUT, and multipart 100-MiB upload for each protocol. + Run at least one single-client and one concurrent-client case. Record the + exact threshold, part size, mirror/EC policy, durability setting, and + software revision with each run so changed profiles are not silently + compared. Keep setup and verification outside timed samples. +5. Save a baseline artifact and comparison contract for each supported local + profile. Accept an optional reference result and tolerance in the runner; + reject comparisons across incompatible profiles. Preserve raw samples and + metrics for diagnosis even when a case fails. + +#### Dependencies + +- The existing local deployment facilities, access-server S3 and Iceberg + HTTP routes, protocol credentials, and chunk IO benchmark artifact + conventions are the baseline. This requirement can produce an HTTP + performance baseline before R195 lands. +- R195 supplies finer write-flow stage metrics. Until then, the benchmark + records available request and chunk IO metrics and marks absent R195 fields + unavailable rather than fabricating stage measurements. Once R195 lands, + the local sentinel requires those fields for both protocols. +- Official client compatibility remains covered by the existing E2E suites; + this benchmark's common producer isolates server upload performance from + PyArrow or boto3 buffering. A separate client-inclusive profile may reuse + the same result schema without replacing the controlled regression profile. + +#### Acceptance + +- Given a local single-node deployment and the same prepared payload, run S3 + and Iceberg direct PUT cases; assert each traverses the HTTP server, + validates integrity, publishes the correct object or file record, and emits + comparable latency and byte fields (I1, I2, I4). E2E test. +- Given S3 and Iceberg multipart workloads, upload parts and complete them; + assert part and completion latencies and outcomes are distinct, and a part + alone does not count as a published final object (I1, I3). E2E test. +- Given small, threshold-boundary, and 100-MiB cases at one and multiple + clients, run the CLI with a fixed seed; assert the recorded case profile, + bounded client payload memory, complete drain, verified bytes, and retained + JSON/TSV samples match the submitted operations (I2–I4). E2E test. +- Given a rejected digest, failed publication, timeout, or interrupted client, + run a case; assert the command exits nonzero, records the correct failure or + incomplete count, and retains logs and metrics for diagnosis (I3, I5). + Integration test. +- Given a matching baseline and a mismatched profile, compare both with the + same result; assert the matching comparison applies its configured tolerance + and reports a regression when exceeded, while the mismatched comparison is + rejected without a performance verdict (I4, I5). Unit test. +- Given R195 metrics on S3 and Iceberg servers, run the local scripts; assert + before/after deltas include the shared stage counts and waits, have no + object-derived labels, and keep overlapping durations separate from total + request latency (I4, I5). E2E test. + +Run `pixi run rs-fmt-check`, `pixi run rs-lint`, `pixi run cargo test -p crowdb-console-shared`, `pixi run cargo test -p crowdb-cli`, and the focused S3 and Iceberg benchmark regression scripts through `pixi run` for the implemented scope. diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 4f57be269..4a2e7936e 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -11,7 +11,7 @@ complexity, and dependency. Before implementation, follow the ## Item Index -**Next R number: R191** — Bump this line in the same commit when adding a new item. +**Next R number: R197** — Bump this line in the same commit when adding a new item. ### Next Milestone — Chunk-backed range KV @@ -40,6 +40,24 @@ R152–R166 delivered the limited basic S3 service, including the restart acceptance baseline. Multipart upload is available; R168–R169 defer shared-storage GC without blocking basic large-object deletion. R170 adds optional cuObject/RDMA acceleration after the TCP baseline is correct and measured. +- **[R195](R195-access-shared-large-upload-flow.md)** — shared bounded large + HTTP upload flow — Area: access server / S3 / Iceberg / chunk IO — Run socket + fetch and chunk writes as separate producer/consumer stages with three + owner credits, transition-only wakeups, placement-safe frame completion, + and bounded mirror/EC writes. S3 and Iceberg keep separate authority and + publication while sharing the transport flow. +- **[R196](R196-access-upload-benchmark-regression.md)** — S3 and Iceberg HTTP + upload benchmark regression — Area: CLI / access server / benchmark — Add a + shared real-protocol CLI workload and retained local regression scripts for + direct and multipart small/large uploads, with correctness gates, metrics, + and profile-matched performance baselines. +- **[R193](R193-chunkdb-node-failure-budget.md)** — configurable node failure + budget and EC placement — Area: KV / chunkdb / chunk IO / deployment — + Generalize the fixed one-node and three-node profiles to larger clusters. + Validate mirror copies and EC against the worst configured set of failed + nodes. A six-node cluster with a two-node budget admits `4+2` and `8+4` EC, + but rejects `2+1`. Persist degraded placement and restore full protection + after capacity returns. - **[R168](R168-s3-shared-object-reclamation.md)** — shared small-object reclamation — Area: access server / S3 / chunkdb — **Deferred on R95.** Turn exact pending shared ranges into qualified, restart-safe physical deletion. @@ -76,6 +94,12 @@ Caches, selected ORC and container engine workflows remain separate. container verification.** Verify Python dataframe, local SQL, distributed engine and optional ingest scenarios against the single-node image; publish only tested compatibility recipes. +- **[R194](R194-access-iceberg-object-listing.md)** — native object listing and + S3-style address semantics — Area: Iceberg / native FileIO / clients — + **Deferred pending client and address-model research.** Determine which clients + need intentional prefix listing, whether the bucket field should identify a + catalog, table, or opaque scope, and implement a bounded authorized listing + contract only if that evidence warrants it. ### High Priority diff --git a/doc/design/access-server/design-crowdb-access-server.md b/doc/design/access-server/design-crowdb-access-server.md index 048ea4bdf..1edb3170f 100644 --- a/doc/design/access-server/design-crowdb-access-server.md +++ b/doc/design/access-server/design-crowdb-access-server.md @@ -110,11 +110,49 @@ authentication policy, admission budget, metrics, and lifecycle. Shared utilities may manage buffers, credentials, errors, and shutdown, but cannot reinterpret model semantics. +The `crowdb-access-server` executable starts the S3 and Iceberg listeners +together by default. The container supervises one access process for both +ports. Explicit `s3` and `iceberg` commands are reserved for focused tests +and management operations. +The combined entry point signals the other listener when either service +returns, then waits for both services to drain their owned small-write pools +before exiting. An unexpected Iceberg recovery or GC worker exit fails the +Iceberg listener, so the combined process also stops the S3 listener. A +terminated small-write manager fails its owning listener; an empty pipeline +route set remains recoverable while the manager continues restarting pipelines. + +The listeners own separate `ChunkIoClient` instances and small-write pools. +New S3 chunks use type `S3`; new Iceberg file chunks use type `IcebergTable`. +Each protocol may override the legacy common small-write policy and select +its own large-write EC, memory, prefetch, and mirror settings. Each protocol +library constructs its own storage clients and chooses foreground writers; +the executable handles HTTP framing, listener startup, and process shutdown. +Iceberg file-record construction also lives in the Iceberg library, including +the format rule shared by ordinary uploads and multipart completion. +Iceberg GC uses a separate chunk client with Iceberg's small-write policy. +The Iceberg library owns its GC storage budget and file-block adapters. +Each small-write policy also chooses its chunk capacity; each protocol's +large-write policy chooses its own maximum chunk size. The deployment profile +sets RPC workers and DiskIO connections independently of these data limits. + ## 5. Data paths The ordinary path streams bounded data through the Access Server over HTTP. It is the universal path and remains available without specialized hardware. +S3 and Iceberg PUT and multipart part uploads share an object-scoped transfer +driver. One task polls the protocol's body decoder and a bounded write +consumer; it can continue receiving while earlier strips are in flight. +Completed receive owners are offered to the writer before their logical +payload views enter a separate MD5 and optional SHA-256 worker. The writer +uses separately bounded held-buffer and in-flight strip windows, and +processes strip completion in submission order. Small objects use the same +handoff and digest completion but retain their protocol-owned shared +small-write pipeline after the handoff. Authentication, checksum declarations, +metadata publication, multipart authority, and cleanup remain with each +protocol. The [Iceberg upload-flow design](iceberge/design-crowdb-iceberg-upload-flow.md) +describes scheduling and stage measurements in detail. + The Dataset native path embeds routing, retry, bounded planning, streaming, and buffer ownership in the application. It resolves one immutable dataset generation and distributes work directly to responsible CROWDB services. A @@ -180,6 +218,13 @@ table, or dataset size. descriptors are not capabilities by themselves. - **AS-I9 — Completion lifetime:** every buffer and registration outlives all socket, RPC, storage, NIC, and GPU operations that reference it. +- **AS-I10 — Protocol storage ownership:** S3 and Iceberg use distinct chunk + types and independently admitted foreground write pools; each library owns + its file or object authority and storage policy. +- **AS-I11 — Shared upload progress:** one body reader and one write consumer + own each S3 or Iceberg PUT or multipart part. Socket readiness and write + completion can each resume an idle transfer without changing protocol + publication authority. ## 9. Direction and risks diff --git a/doc/design/access-server/iceberge/design-crowdb-iceberg-upload-flow.md b/doc/design/access-server/iceberge/design-crowdb-iceberg-upload-flow.md new file mode 100644 index 000000000..3854da5ec --- /dev/null +++ b/doc/design/access-server/iceberge/design-crowdb-iceberg-upload-flow.md @@ -0,0 +1,232 @@ + + + +# CROWDB - Design: Iceberg File Upload Flow + +This document defines the FileIO upload path used by Iceberg clients and records +measurements from the single-node profile. It covers file publication, not table +snapshot publication. + +Depends on: [Native Iceberg Storage](design-crowdb-iceberg.md) and +[Chunk I/O](../../chunkio/design-crowdb-chunkio.md). + +## Table of contents + +1. [Client and server flow](#1-client-and-server-flow) +2. [Large write ownership and scheduling](#2-large-write-ownership-and-scheduling) +3. [Integrity and durable publication](#3-integrity-and-durable-publication) +4. [Measured costs](#4-measured-costs) + +## 1. Client and server flow + +The TPC loader writes one local Parquet part at a time through `CrowdbFileIO`. +It copies bounded input buffers into the PyArrow output stream while computing +a local SHA-256 digest. It does not read the uploaded object back. Closing the +stream waits for the FileIO transfer to finish. Only after all files for a +table are uploaded does the loader import them and commit the table once. + +PyArrow currently uses multipart FileIO even for the measured 64-KiB object: + +1. FileIO checks whether the exact target exists, then starts a multipart + session. Neither step publishes a table snapshot. +2. Each `UploadPart` authenticates and validates its body, prepares a chunk + writer, streams bytes to Chunk I/O, finishes the writer, and persists the + part's locations and session progress. The write path verifies the supplied + body integrity information before accepting it. +3. `CompleteMultipart` freezes the selected parts and assembles their chunk + locations into one file record. Native stream parts are composed logically; + completion does not copy the object payload. The file mapping is then + published. Recovery can resume an interrupted completion. +4. A later Iceberg table commit publishes metadata that references this file. + Uploaded files remain outside the table snapshot until that commit. + +Writer selection uses the decoded length of each HTTP request, when available. +Payloads below the small-object threshold can use the small-object writer for +one frame or the shared-object writer for a longer request. Other requests use +the large writer. The total logical file size alone does not select the writer. + +## 2. Large write ownership and scheduling + +The single-use Iceberg `WriteObject` owns the request body, bounds, file +identity, writer, digest pipe, measurements, and final publication action. S3 +PUT and `UploadPart` use the same transfer driver and checksum worker while +retaining S3 authorization and publication. Its transfer coroutine polls the +body receiver and write consumer with one task waker. Only +the receiver fetches the socket body. When it offers a prepared buffer, the +same task immediately polls the consumer. If both sides are pending, the task +yields; a Hyper body-read event or a write completion wakes it again. An idle +writer therefore resumes when a slow socket supplies the next buffer. + +The receiver assembles up to 1 MiB of payload from body frames. The native +receive path prepares frame headers and CRC32C on the received owner before +handoff. CRC32C excludes the chunk ID; after placement the writer fills that +field and writes the frame without an object-sized copy. The receiver offers +the owner to a bounded channel with four held-buffer slots. A full channel +pauses further body reads until the consumer removes an owner. The digest +worker receives borrowed payload views after the write offer, so checksum work +can overlap receive and DiskIO without controlling write backpressure. Small +objects use this same handoff, then enter the shared small-write pipeline; +their data path does not submit independent large-write strip tasks. + +The large chunk writer prepares strips ahead of demand. For a known object +size, it batches up to the configured strip-prefetch limit and requests the +next batch when half of the current one has been consumed. Mirror strips are +submitted to independent tasks, with at most four strip writes in flight by +default. Later writes may finish first, but completion is consumed in strip +order. Each completed task keeps its buffer until every preceding strip has +committed. A failed mirror segment is replaced and replayed from those retained +bytes before its strip completes; later completions remain held meanwhile. +If replacement is exhausted, the writer drains submitted IO, reads the +committed prefix from the old chunk one strip at a time, and replays it with +the retained suffix into a new chunk. Frame footers receive the new chunk ID; +the CRC does not cover that field. The old chunk is deleted only after the new +copy is durable. Rotation attempts are bounded. +At a full write window the coroutine awaits the oldest completion; +the completed owner queue can still retain four prepared buffers. The +`large_parallel_strip_writes` and `large_held_buffers` settings are separate. + +The following invariants apply: + +- **I1 — One body reader.** No second task reads an object's HTTP body. +- **I2 — Bounded ownership.** Receive buffers remain owned until the writer and + digest have consumed their views; the write queue controls backpressure. +- **I3 — Ordered durability.** A later strip result cannot make an earlier + failed strip successful or release its replay buffer. Chunk sealing waits + for every submitted strip, replacement, and required fsync. +- **I4 — Event-driven progress.** Socket readiness and write completion wake + the suspended coroutine. The write path does not spin or poll a timer for + capacity. + +The upload metrics distinguish elapsed body-frame polling from actual body +waits. A body wait is counted only when the body future returns `Pending`; its +duration ends when that frame or EOF becomes ready. Immediate frames add no +wait count, while writer-capacity and write-flow pause time remain separate. +`queued_buffers_peak` is the highest number of owners waiting in the handoff +queue during one upload. `mirror_uncommitted_peak` counts submitted strip +writes that have not yet committed in order, including writes whose DiskIO task +already finished. The counters describe different stages and are not added. +`mirror_active_write_peak` counts mirror data writes actually inside DiskIO; +it excludes completed tasks waiting for an earlier strip. +The native allocator exports current and peak retained bytes. The chunk client +also counts repair-driven rotations, replayed bytes, and rotation time. + +## 3. Integrity and durable publication + +The object-scoped OpenSSL worker computes MD5 over ordered logical payload +views. It also computes SHA-256 when a signed payload requires it. It never +hashes frame headers or footers. The writer fills placement-dependent chunk +IDs after the receiver has calculated placement-independent CRC32C. Digest +failure, declared-length mismatch, failed DiskIO, or seal failure prevents +file and part publication. An upload is published only after the body has +ended, the digest has been verified, all writes and fsyncs have completed, +and chunk locations have been sealed. Multipart completion and table commit +are later, distinct authority operations. + +## 4. Measured costs + +On 2026-10-01, a local single-node container received a 100-MiB generated +object through the same `CrowdbFileIO` output API as the loader. The object was +not registered in an Iceberg table. Two runs took 6.23 s and 5.90 s. The second +run's stage and server-counter deltas were: + +| Measurement | Result | +| ------------------------------------- | ----------: | +| Create output stream, including probe | 0.322 s | +| Thirteen 8-MiB client writes | 0.075 s | +| Close and finish multipart transfer | 5.507 s | +| End-to-end elapsed time | 5.904 s | +| Effective logical throughput | 16.9 MiB/s | +| Successful FileIO HTTP requests | 12 | +| Catalog GET operations | 144 | +| Catalog compare-exchange operations | 16 | +| Small-write completions | 0 | + +The 12 successful requests are consistent with creating a session, uploading +parts, and completing it. The existence probe returns not found and is not in +the success count. The summed server dispatch time was 31.61 s across requests; +multipart part requests overlap, so this sum is not wall-clock latency. + +The client writes returned in 75 ms because PyArrow buffers or schedules the +transfers. The 5.5-s `close()` wait is the dominant observed client stage. +The zero small-write delta rules out the small-object queue as the path for +this 100-MiB sample. Catalog GET and CAS counts show metadata work remains +per multipart session and part, rather than a single final table update. +These counters do not yet isolate network transfer, chunk allocation, DiskIO, +or sealing within the 5.5-s wait. + +A second container run with the checksum worker used one prepared 1-MiB block +100 times through the same FileIO output stream. The 100-MiB upload took +5.014 s: 0.034 s to open, 0.104 s in client writes, and 4.876 s in `close()`. +The FileIO path still uses multipart. This result shows that checksum work is +not the only cost in its close stage; it does not isolate the remaining +multipart, transport, or storage costs. + +### Small files + +The same container and API were used for three generated objects. Each used +three successful FileIO requests, 45 catalog GETs, and seven catalog CAS +operations, even though the payload sizes differed. + +| Object size | Open | Client write | Close | Total | Small-write completions | +| ----------- | ------: | -----------: | ------: | ------: | ----------------------: | +| 64 KiB | 0.345 s | <0.001 s | 1.242 s | 1.587 s | 1 | +| 1 MiB | 0.328 s | <0.001 s | 1.213 s | 1.541 s | 0 | +| 5 MiB | 0.344 s | 0.001 s | 1.498 s | 1.844 s | 0 | + +The 64-KiB transfer completed through the small-write queue. The 1-MiB and +5-MiB requests did not increment that counter; the available metrics do not +separate shared-object from large-writer completions. Fixed FileIO session, +part, and completion work is material for a small object. The table-level +commit is not included in these timings. + +### Focused direct PUT + +The single-node small-cluster fixture uses one mirror copy and null DiskIO. +Its producer yields 100 references to one prepared 1-MiB buffer, with +Content-MD5 computed before timing and `UNSIGNED-PAYLOAD` in the request. The +direct 100-MiB PUT completed in 335.9 ms. The upload observation was 315.4 ms, +including 224.0 ms in body-frame polling and waiting, 47.4 ms across 30 +writer-capacity waits, 23.1 ms across two strip-preparation waits, 23.0 ms in +writer finish, and 47.9 ms in metadata publication. The sum of 101 strip-write +durations was 558.0 ms and digest CPU time was 234.3 ms. These stages overlap +and must not be added to estimate wall-clock time. Body-frame time includes +Hyper delivery and coroutine scheduling, not just socket reads. + +The container's multipart FileIO timings above have different transport, +storage, and publication work from this focused direct PUT. A comparable +FileIO profile must record part size, concurrency, topology, durability, +software revision, and raw stage samples before attributing its close time +to a specific server stage. + +### Frame-aligned FileIO follow-up + +The loader's PyArrow S3 output stream split a 100-MiB file into ten multipart +parts. On the local single-node container, the first current-tree run took +4.668 s. Its non-native receive path handed the writer 1 MiB of logical bytes +at a time, and the writer pushed each 64-KiB physical frame separately. +Consequently the mirror-strip batch path never ran: the measured peak was one +active mirror write, zero queued full mirror strips, and 1,610 individual +strip-push completions. Aligning receive batches to 16 frame payloads alone +left the run at 4.638 s because the writer still split them into single-frame +pushes. + +After the large async writer grouped those 16 complete frames into one 1-MiB +physical strip push, the same FileIO path completed in 0.993 s: 0.109 s to +open, 0.034 s in client write calls, and 0.850 s in close. The server recorded +ten completed part uploads, 100 MiB of logical data, 110 strip-push +completions, and peaks of four queued full mirror strips and four active +DiskIO writes. This is one local sample, not a stable throughput distribution. +The loader now chooses a single PUT below 256 MiB; one 100-MiB direct loader +upload took 1.091 s including local SHA-256, and a 256-MiB four-part loader +upload took 3.042 s. These timings include different client preparation and +must not be compared as pure server write latency. + +The updated loader and image then completed a fresh TPC-H SF1 load in 24.94 s +and SF10 in 81.30 s. Those wall times include generation, upload, eight table +snapshot commits, and remote manifest/footer/sample-scan verification. SF10 +uploaded 3,635,609,132 data bytes; all eight tables and 86,586,082 rows were +verified. The access upload metrics recorded zero failed requests and a peak of +four active mirror writes. Previous runs on this host took about 50 s and +232 s, respectively. These single-run totals show a material end-to-end +improvement, but they do not isolate upload-only wall time or constitute a +repeatable benchmark distribution. diff --git a/doc/design/chunkdb/design-crowdb-chunkdb-mirror-to-ec.md b/doc/design/chunkdb/design-crowdb-chunkdb-mirror-to-ec.md index 02d2cbfed..622d8671a 100644 --- a/doc/design/chunkdb/design-crowdb-chunkdb-mirror-to-ec.md +++ b/doc/design/chunkdb/design-crowdb-chunkdb-mirror-to-ec.md @@ -138,9 +138,10 @@ cannot join or allocate. ## 3. Foreground Conversion Each small-write pipeline owns one special reservation group containing eight -three-copy mirror candidate sets and four parity segments. The group is chosen -from one placement snapshot and is hidden from `Chunk.strips` until individual -mirrors are confirmed. A group is not started when the current chunk cannot +mirror candidate sets with the configured copy count and four parity segments. +The group is chosen from one placement snapshot and is hidden from +`Chunk.strips` until individual mirrors are confirmed. A group is not started +when the current chunk cannot contain eight remaining full strips. After a mirror strip is durable, its existing one-MiB shadow updates four diff --git a/doc/design/chunkdb/design-crowdb-chunkdb.md b/doc/design/chunkdb/design-crowdb-chunkdb.md index 9dbe0257d..378a16f5d 100644 --- a/doc/design/chunkdb/design-crowdb-chunkdb.md +++ b/doc/design/chunkdb/design-crowdb-chunkdb.md @@ -238,11 +238,14 @@ definition; Rust code works with proto types directly. ### 3.9 Chunk types for different use cases -Four chunk types are defined for CROWDB's storage hierarchy: -- **Repo chunk**: User data storage. +Seven chunk types are defined for CROWDB's storage hierarchy: +- **Repo chunk**: Historical general user data storage. - **WAL chunk**: Write-ahead log entries. - **BTree page chunk**: B-tree page storage for the crowdb-tree engine. - **Page index chunk**: Page index metadata. +- **Stream chunk**: Native chunk stream data. +- **S3 chunk**: S3 object data. +- **Iceberg table chunk**: Iceberg immutable file data. **Rationale:** Different storage components have different redundancy and performance requirements. Chunk types allow optimization for each component's @@ -320,13 +323,13 @@ zone_offset, size, tag }` (from diskdb proto). A **strip** is the atomic redundancy unit. Two strip types: -**Mirror Strip**: One disk block capacity, replicated across N nodes -(configurable copy count, default 3). Each replica is a full copy on a -different node. Data capacity = 1 × disk_block_size. +**Mirror Strip**: A configured number of disk allocation units replicated +across N nodes (configurable copy count, production default 2). Each segment +is a full copy on a different node. Data capacity = unit_count × unit_size. -**EC Strip**: `data_num` data blocks + `code_num` parity blocks, -distributed across different nodes. Data capacity = `data_num × -disk_block_size`. For example: +**EC Strip**: `data_num` data segments + `code_num` parity segments, +distributed across nodes under the failure-domain placement rule. Data +capacity = `data_num × unit_count × unit_size`. For example: - 6+3 EC with 1 MB blocks → 6 MB data capacity, 9 MB total. - 8+4 EC with 1 MB blocks → 8 MB data capacity, 12 MB total. @@ -346,9 +349,9 @@ Each strip tracks: A **chunk** is a container for strips. Chunk properties: - **128-bit ID**: Chunk type (8 bits) + Timestamp (48 bits) + Randomness (72 bits). - **State**: `Init` → `Active` → `Sealed` → `Deleted`. -- **Type**: Repo, WAL, B-tree page, or page index. Shared versus - dedicated is a client-side packing and ownership policy for Repo - chunks, not a wire-level chunk type. +- **Type**: Repo, WAL, B-tree page, page index, stream, S3, or Iceberg table. + The ID's high-byte prefix and the stored type must agree. Historical Repo + references remain readable by both access protocols. - **Capacity**: Total data capacity across all strips. - **Write granularity**: Minimum write alignment (e.g., 4 KB). - **Strips**: Ordered list of strips (mirror or EC). @@ -536,12 +539,15 @@ BucketMigrationState: ### 5.5 Chunk types | Type | Chunk Type Value | Description | -|---------------|------------------|--------------------------------------| -| Repo | 0 | User data storage | +| ------------- | ---------------- | ------------------------------------ | +| Repo | 0 | Historical general user data | | WAL | 1 | Write-ahead log entries | | BTree page | 2 | B-tree page storage | | Page index | 3 | Page index metadata | -| Reserved | 4-255 | Reserved for future use | +| Stream | 4 | Native chunk streams | +| S3 | 5 | S3 object data | +| Iceberg table | 6 | Iceberg immutable file data | +| Reserved | 7-255 | Reserved for future use | **Note:** Chunk type is independent of strip type. Any chunk type can use either mirror or EC strips based on configuration and requirements. @@ -611,6 +617,21 @@ projected usable utilization, `(used + in_flight + planned) / capacity`, only among candidates that meet that safety constraint. Equal scores use stable topology identifiers, making retries deterministic. +`deployment.mode` is `production` or `test_single_node` in release builds. Production startup +requires at least three distinct voting nodes for every KV group and protected +placement, with `deployment.max_node_failures = 1`. Test-single-node startup +requires one voting node per group and `deployment.max_node_failures = 0`, with +explicit colocated placement. The mode never changes in response to topology +loss. In test-single-node mode, new strips must be one-copy 1 MiB mirrors; +EC, extra copies, and mirror-to-EC conversion are rejected. Production rejects +new mirror strips with fewer than two copies. The production profile normally +places two mirror copies on distinct nodes, including for journal and tree-page +data. + +Debug builds also accept `test_unsafe_placement` for legacy colocated EC +integration fixtures. Release builds reject it during configuration loading; +it is separate from the single-node deployment profile. + `placement.mode` selects one placement strategy at process construction. The allocator depends on the `ChunkPlacementStrategy` interface and does not branch on the mode while allocating, converting, repairing, or deciding whether a @@ -618,18 +639,18 @@ degraded disk result may be published. Each mode is a separate strategy type: - `protected` uses failure-domain-aware mirror and EC selectors. The granular degraded-placement settings below remain available only within this mode. -- `unsafe_colocated` deliberately selects one healthy disk group and may place - every mirror copy or EC fragment in that same group, on the same physical - disk, and in the same zone. This mode supports the minimum container topology - of one rack, one node, one disk group, one disk, and one zone. It preserves - strip geometry and encoding but provides no node-, disk-, or zone-failure - durability; losing the colocated resource may lose every fragment. +- `unsafe_colocated` selects one healthy disk group for the explicit + test-single-node deployment. This deployment uses one mirror copy and has + no data protection; a read or write error reaches the caller. The mode is an explicit deployment property, not an automatic fallback. A protected deployment never changes to `unsafe_colocated` because topology is -small or unavailable. New placement policies are added as strategy -implementations and selected at the composition root, keeping policy branches -out of the allocation hot path. +small or unavailable. With two healthy nodes remaining, a new EC strip retains +its requested data and parity geometry. Its fragments may span both survivors +with a degraded-placement marker and a durable repair task. A new mirror strip +still places two copies across the survivors. New placement +policies are added as strategy implementations and selected at the composition +root, keeping policy branches out of the allocation hot path. For an EC `data_num + code_num` strip, a protected rack, node, or physical disk contains at most `code_num` fragments. For a mirror strip, losing a @@ -659,7 +680,7 @@ rack failures: **Negative hints**: Nodes can be excluded from placement (e.g., during recovery to avoid re-using failed nodes). -**Example**: 3-copy mirror on 3-rack cluster → 3 replicas on 3 distinct racks. +**Example**: 2-copy mirror on 3-rack cluster → 2 replicas on 2 distinct racks. On insufficient topology, normal placement returns a typed failure before any DiskDB allocation. An explicitly degraded result identifies the missing protection instead of claiming rack safety. @@ -685,8 +706,10 @@ that exceeds the normal recovery budget when the cluster is too small. It does not select colocated placement. Insufficient topology otherwise returns a typed placement error without allocating blocks. -**Example**: 8+4 EC on 12-node cluster → 12 blocks across ≥3 racks, max 4 -blocks per node. On 3-node cluster (unsafe mode) → 12 blocks, 4 per node. +**Example**: 8+4 EC on a healthy 3-node cluster places 12 fragments with at +most 4 per node, so loss of any one node leaves 8 fragments. If one node is +already unavailable, the same geometry may span the two survivors and is +marked for placement repair when the third node returns. ### 7.3 Physical validation and degraded-placement repair @@ -1235,7 +1258,8 @@ Key configuration parameters: | Parameter | Default | Description | |---------------------------------------------------|------------|---------------------------------------------------------------| | disk_block_size | 1 MB | Size of disk blocks from diskdb | -| mirror_copy_count | 3 | Number of replicas for mirror strips | +| deployment.max_node_failures | 1 | Protected production node-failure budget | +| mirror_copy_count | 2 | Production mirror copies on distinct nodes | | default_ec_scheme | 6+3 | Default EC scheme (data+parity) | | topology_refresh_interval | 30 s | Topology cache refresh interval | | placement.mode | protected | Select `protected` or explicit `unsafe_colocated` strategy | diff --git a/doc/design/chunkio/design-crowdb-chunkio-small-object-writer.md b/doc/design/chunkio/design-crowdb-chunkio-small-object-writer.md index c09d12e06..b41863d9f 100644 --- a/doc/design/chunkio/design-crowdb-chunkio-small-object-writer.md +++ b/doc/design/chunkio/design-crowdb-chunkio-small-object-writer.md @@ -115,9 +115,10 @@ reservations and removes attached strips beyond the written length. Objects and batches never straddle a strip or chunk. When automatic conversion is enabled and at least eight strips remain, one -special reservation allocates eight three-copy mirror sets and four parity -segments from a joint placement plan. Each completed strip updates the four -incremental parity accumulators and releases its input image. After strip eight, +special reservation allocates eight mirror sets with the configured copy +count and four parity segments from a joint placement plan. Each completed +strip updates the four incremental parity accumulators and releases its input +image. After strip eight, the client writes and fsyncs only the parity segments. ChunkDB then reselects one healthy survivor per mirror set against current topology and atomically publishes the 8+4 EC strip. If optimal publication is unavailable, mirrors stay diff --git a/doc/design/chunkio/design-crowdb-chunkio.md b/doc/design/chunkio/design-crowdb-chunkio.md index ea9829eb8..da475b43d 100644 --- a/doc/design/chunkio/design-crowdb-chunkio.md +++ b/doc/design/chunkio/design-crowdb-chunkio.md @@ -4,7 +4,7 @@ # CROWDB - Design: Chunk IO Data Path (Overview) The chunk IO data path is the client-side layer that writes and reads -large-object data as EC-encoded strips across diskio servers, using chunkdb +large-object data as mirror or EC strips across diskio servers, using chunkdb for chunk lifecycle management (allocate, append, seal, delete). It lives in the `crowdb-chunk-client` crate and is consumed by object store layers and application upload handlers. The chunkdb server design @@ -19,6 +19,14 @@ model, and the design choices that make a 1 TB upload cost the same specified in the [small-object writer design](design-crowdb-chunkio-small-object-writer.md). +`ChunkWriter` selects a mirror or EC strip writer from each persisted strip, +so consecutive strips in one chunk can have different layouts and capacities. +The mirror writer sends each write to every mirror segment and synchronizes +them before strip completion; any failed copy returns a write error. The EC +writer derives its shard scheme from that strip's metadata and retains its +separate parity and repair flow. `ChunkReader` likewise dispatches each strip +to mirror or EC recovery based on stored geometry. + ## Table of Contents - [1. Non-Goals](#1-non-goals) @@ -199,10 +207,11 @@ The flow, step by step: ### 3.1 Partial Last Strip Partial strips occur only at EOF, never mid-chunk. When EOF arrives -before all `data_num` blocks of the current strip are filled, the main -write task writes only the filled data blocks, releases the empty ones, -hands the partial set off to parity for partial EC (§5), and records -`sealed_length` for `seal_chunk`. +before the current strip is full, the writer persists only the filled +data. EC strips also release empty blocks and hand the partial set to +parity (§5). At chunk seal, ChunkDB records the durable `sealed_length` +of every written strip, whether mirror or EC, so a reader can cross +strip boundaries and read the partial final strip. ## 4. Backpressure and Memory Budget @@ -371,6 +380,15 @@ Edge cases: ## 9. Tunables and Defaults +Chunk capacity is a write-policy limit, while each persisted strip records its +own size and protection layout. A single-node test chunk may contain many +1 MiB one-copy mirror strips. The S3 and Iceberg access policies can select +different small and large chunk capacities in their configuration. Tree page +storage separately uses `storage.tree_chunk_capacity_bytes` from the chunk KV +server configuration and splits a single-copy tree chunk into 1 MiB strips. +Chunk KV stream storage uses `storage.stream_chunk_capacity_bytes` for the +`Stream` chunk type, also independently of its 1 MiB strip geometry. + | Knob | Default | Role | | --- | --- | --- | | `max_chunk_size` | 1 GB | Chunk rotation threshold. | diff --git a/doc/design/config/design-crowdb-config.md b/doc/design/config/design-crowdb-config.md index 58d339210..e056d2628 100644 --- a/doc/design/config/design-crowdb-config.md +++ b/doc/design/config/design-crowdb-config.md @@ -23,7 +23,7 @@ validated, and activated. ## 1. Scope The `crowdb-kv-server`, `crowdb-diskdb`, `crowdb-chunkdb`, `crowdb-diskio`, -`crowdb-access-server`, and `crowdb-iceberg` processes accept typed TOML startup +and `crowdb-access-server` processes accept typed TOML startup configuration. A service may require a file or make it optional, but a supplied file follows the same resolution and failure rules in every process. diff --git a/doc/design/kv/design-crowdb-kv-reconfiguration.md b/doc/design/kv/design-crowdb-kv-reconfiguration.md index 90331f584..a744abd1c 100644 --- a/doc/design/kv/design-crowdb-kv-reconfiguration.md +++ b/doc/design/kv/design-crowdb-kv-reconfiguration.md @@ -31,11 +31,11 @@ CROWDB supports membership changes within a single group. Specifically: - **Replace a member.** Implemented as add-then-remove (or vice versa), each as a single-member change. - **Change the leadership of the group.** Triggered as a side effect when removing the current leader. -Out of scope (design-crowdb-kv.md §2](design-crowdb-kv.md)): +Out of scope ([design-crowdb-kv.md §2](design-crowdb-kv.md)): - Changing `num_groups` (the total number of groups in the cluster) — fixed at cluster creation. - Splitting or merging groups — not supported. -- Going below 3 voting members — not supported (a 1-member group has no fault tolerance). +- Reconfiguring a protected group below 3 voting members — not supported. The explicit single-node test deployment starts with one voter and has no fault tolerance; it is not entered through membership reconfiguration. - Going above 7 voting members — not in the initial scope; quorum size grows linearly with membership and the marginal availability gain past 7 is small. **Granularity:** every reconfiguration moves *exactly one member* in or out at a time. To go 3 → 5, do two single-member additions in sequence. Under the exact-match `membership_epoch` fence (§6) this single-member-at-a-time rule is no longer required for safety, but it is still recommended because it minimizes the propagation window during which writes stall. diff --git a/doc/design/kv/design-crowdb-kv.md b/doc/design/kv/design-crowdb-kv.md index 9f08c384e..123690cc3 100644 --- a/doc/design/kv/design-crowdb-kv.md +++ b/doc/design/kv/design-crowdb-kv.md @@ -148,6 +148,12 @@ btree can replay the WAL on crash. A write is acknowledged to the client only after a quorum of acceptors have durably flushed. Multi- disk WAL segments are tagged by slot index for parallelism. +The protected three-node deployment keeps three voting replicas in each KV +group, so one unavailable node leaves a two-voter quorum. The explicit +single-node test deployment uses one voting replica and has no node-failure +protection. ChunkDB checks the voting topology at startup before accepting +either deployment profile; it never switches profiles after a node failure. + ### 3.9 Plaintext transport, TLS hooks reserved Node-to-node and client-to-node channels are plaintext initially. The diff --git a/doc/dev/crash_debugging.md b/doc/dev/crash_debugging.md index 52ba03aae..3b39d281c 100644 --- a/doc/dev/crash_debugging.md +++ b/doc/dev/crash_debugging.md @@ -61,9 +61,9 @@ cat /proc/sys/kernel/core_pattern does not keep the file pattern after a restart. Verify `core_pattern` again after reboot. `fs.suid_dumpable=0` permits an ordinary process to write a relative core; executables with file capabilities may still be excluded. -The container currently gives file capabilities to `crowdb-iceberg` and -`crowdb-access-server` for low ports, so do not assume those two will produce -cores under this setting. Verify the particular crashed service. +The container currently gives file capabilities to `crowdb-access-server` for +low ports, so do not assume it will produce cores under this setting. Verify +the particular crashed service. For a one-time investigation, stop Apport and apply the two `sysctl -w` commands without creating the sysctl file or disabling the service. Restart diff --git a/doc/doc_index.md b/doc/doc_index.md index f57984c7a..48dea52e8 100644 --- a/doc/doc_index.md +++ b/doc/doc_index.md @@ -32,16 +32,17 @@ listed document or section needed by the task. ## Working & Flow-Analysis Docs Temporary plans live under `doc/working/`; flow analyses live under -`doc/design/{kv,chunkio,rpc}/`. - -| Doc | When to read | -| ------------------------------------------------------ | ---------------------------------------------- | -| `doc/design/kv/kv-read-flow-analysis.md` | KV point-read flow and benchmarks. | -| `doc/design/kv/kv-scan-flow-analysis.md` | KV scan flow and benchmarks. | -| `doc/design/kv/kv-write-flow-analysis.md` | KV write flow and optimization evidence. | -| `doc/design/chunkio/chunkio-write-flow-analysis.md` | Chunk I/O large-write flow and benchmarks. | -| `doc/design/chunkio/chunkio-small-io-flow-analysis.md` | Chunk I/O small-I/O flow and benchmarks. | -| `doc/design/rpc/rpc-flow-analysis.md` | RPC flow, benchmarks, and performance history. | +`doc/design/{access-server,kv,chunkio,rpc}/`. + +| Doc | When to read | +| ----------------------------------------------------------------------- | ------------------------------------------------------- | +| `doc/design/kv/kv-read-flow-analysis.md` | KV point-read flow and benchmarks. | +| `doc/design/kv/kv-scan-flow-analysis.md` | KV scan flow and benchmarks. | +| `doc/design/kv/kv-write-flow-analysis.md` | KV write flow and optimization evidence. | +| `doc/design/chunkio/chunkio-write-flow-analysis.md` | Chunk I/O large-write flow and benchmarks. | +| `doc/design/chunkio/chunkio-small-io-flow-analysis.md` | Chunk I/O small-I/O flow and benchmarks. | +| `doc/design/access-server/iceberge/design-crowdb-iceberg-upload-flow.md` | Iceberg FileIO upload path and container measurements. | +| `doc/design/rpc/rpc-flow-analysis.md` | RPC flow, benchmarks, and performance history. | ## Dev Environment (`doc/dev/`) diff --git a/doc/working/plan-tpc-iceberg-upload-performance.md b/doc/working/plan-tpc-iceberg-upload-performance.md new file mode 100644 index 000000000..0ed797e19 --- /dev/null +++ b/doc/working/plan-tpc-iceberg-upload-performance.md @@ -0,0 +1,47 @@ + + + +# TPC Iceberg Object Upload Performance Plan + +Upstream: [R195](../backlog/R195-access-shared-large-upload-flow.md); benchmark contract: [R196](../backlog/R196-access-upload-benchmark-regression.md). +Goal: implement the agreed object-scoped large-write flow, then measure and improve the real 100-MiB TPC FileIO upload while preserving integrity, bounded memory, and durable publication. + +## Measurement + +- [x] **Retain existing baseline evidence**: The direct small-cluster PUT passed in 677 ms on 2026-10-01; its 33-second test duration includes cluster startup. The 13-part profile (four concurrent uploads) completed in 613 ms: 48 ms session creation, 404 ms part phase, 162 ms completion. Across 13 parts, server metrics summed 905 ms writer feed, 198 ms part publication, 98 ms digest enqueue, and 83 ms body polls; these sums overlap. One earlier profile run returned HTTP 200 for completion but got a 404 on subsequent range read, while two later runs passed; determine whether the completion body carried an embedded error. Baseline does not gate the flow implementation. +- [x] **Measure strip readiness and write completion**: The focused single-node, one-mirror, null-DiskIO 100-MiB PUT passed in 2.056 s after adding the counters. Writer feed occupied 1.920 s; waiting for the next strip occupied 0.795 s across 52 waits, while 101 successful `strip.push` calls occupied 0.309 s total (6.35 ms maximum). These stages overlap with receive and digest. Default `prefetch_strips_per_chunk` is 1; inspect prefetch runway before changing write concurrency. +- [x] **Batch known-size large-write strip prefetch**: Keep the initial chunk allocation and ordinary prefetch depth at one strip. For a known-size large write, cap each append batch by the object's remaining framed bytes, the chunk's strip capacity, and configurable `large_prefetch_max_strips_per_batch` (default 32). Start the next append after half of the prior batch has been consumed. The same 100-MiB PUT passed in 0.836 s and 1.035 s in two local runs; strip preparation wait fell to 2 waits/17 ms and 1 wait/43 ms respectively, while `strip.push` success time stayed near 255–259 ms. Treat these as samples, not a stable throughput distribution. +- [x] **Measure the four-write coroutine flow**: A focused 100-MiB PUT on the single-node null-DiskIO fixture passed in 335.9 ms after forwarding capacity waits through `PreparedLargeWrite`. The upload observation was 315.4 ms, including 224.0 ms in body-frame polls, 47.4 ms across 30 writer-capacity waits, 23.1 ms across 2 strip-preparation waits, 23.0 ms writer finish, 0.95 ms digest finish, and 47.9 ms publication. The 558.0 ms sum of 101 strip-write durations and 234.3 ms digest CPU time overlap other stages and are not additive wall time. A missing `wait_for_capacity` delegation first caused a capacity-loop livelock; the same test passed after the fix. +- [x] **Verify active-write measurement after rotation work**: A later 100-MiB direct PUT passed in 381.6 ms on the same one-node null-DiskIO fixture. The upload observation was 362.4 ms; active mirror writes peaked at 4, uncommitted writes peaked at 4, and queued owners peaked at 4. Body waits accounted for 252.6 ms and writer-capacity waits 65.2 ms; overlapping stages are not additive. The first attempt spent its 60-second shell allowance compiling and starting the fixture, so the compiled test was run again and passed in 36.2 seconds including startup. +- [x] **Verify multipart after rotation work**: The same fixture passed the 100-MiB, 13-part multipart profile. Four concurrent part uploads took 455.6 ms; completion took 381.3 ms; the slowest part took 158.4 ms. The upload metrics recorded 13 completed transfers, 100 MiB of logical data, one multipart completion, and an active mirror-write peak of 4. These are local samples with null DiskIO, not production latency claims. +- [x] **Profile the real loader FileIO path**: The current-tree single-node image initially took 4.668 s for a 100-MiB PyArrow FileIO stream split into ten parts. Aligning non-native receive batches to 16 payload frames alone took 4.638 s; `LargeAsyncObjectWriter` still passed each frame separately. Grouping 16 frames into one 1-MiB physical strip push reduced the same path to 0.993 s (open 0.109 s, client writes 0.034 s, close 0.850 s). Strip-push completions fell from 1,610 to 110, and actual mirror write peak rose from one to four. One direct loader PUT of 100 MiB took 1.091 s including local SHA-256; a 256-MiB, four-part loader MPU took 3.042 s. These are individual local samples; retain the benchmark gate for repeated samples. +- [x] **Run full TPC-H loads with the updated tool and image**: SF1 took 24.94 s and SF10 took 81.30 s for generation, upload, eight ordered commits, and remote sample verification. SF10 transferred 3,635,609,132 data bytes; all eight tables and 86,586,082 rows were verified. The image recorded zero failed upload requests and an active mirror-write peak of four. Earlier runs took about 50 s and 232 s on this host. These runs do not separately time the overlapping upload phase; R196 still owns reproducible repeated measurements. +- [x] **Expose fixed-stage counters**: Per-upload observations aggregate on success, failure, and cancellation. Body waits count only an actual `Pending` poll and are distinct from elapsed body polling; writer-capacity and flow-pause waits are separate. Queue peak counts waiting owners; mirror uncommitted peak includes tasks whose DiskIO work has finished, while mirror active-write peak counts data writes actually executing in DiskIO. The native allocator exports current and peak retained bytes. Chunk repair exports rotation count, completed rotations, replayed physical bytes, and rotation duration. The benchmark output can compare these separate stages. Files: `app/crowdb-access-server/src/iceberg/file_http/metrics.rs`, `app/crowdb-access-server/src/iceberg/file_http/stream.rs`, `app/crowdb-access-server/src/upload_flow.rs`, `lib/crowdb-chunk-client/src/chunk/chunk_writer.rs`. + +## Write flow + +- [x] **Own the Iceberg write**: A single-use `WriteObject` owns body, writer, digest, object identity, bounds, and terminal result. Direct-file and multipart-part publication remain separate. Files: `app/crowdb-access-server/src/iceberg/file_http.rs`, `app/crowdb-access-server/src/iceberg/file_http/stream.rs`, `app/crowdb-access-server/src/iceberg/file_http/multipart.rs`. +- [x] **Implement bounded coroutine receive and mirror writes**: One upload task polls the socket receiver and write consumer with the same waker. Four held owners and four independently configured mirror-strip writes permit receive-ahead. The oldest completion is awaited only when the write window is full; later results are processed in submission order. The digest runs on its own bounded worker queue. A two-block delayed-socket test confirms that a new readable event restarts the idle upload after earlier disk writes have drained. Files: `app/crowdb-access-server/src/iceberg/file_http/stream.rs`, `app/crowdb-access-server/src/upload_flow/digest_pipe.rs`, `lib/crowdb-chunk-client/src/chunk/chunk_writer.rs`, `lib/crowdb-chunk-client/src/client.rs`. +- [x] **Share the transfer driver across protocols and writer sizes**: The access server now owns the coroutine scheduler, bounded offer queue, write consumer, and OpenSSL worker. S3 PUT and UploadPart use the same driver as Iceberg PUT and UploadPart. Small writers enter the same write consumer and keep their distinct shared small-write pipeline for durability. Digest validation precedes final writer seal. Focused S3 multipart replacement/publication, S3 ordinary PUT size matrix (10 KiB, 1 MiB, 12 MiB, 100 MiB), S3 integrity tests, and Iceberg direct 100-MiB PUT pass. Files: `app/crowdb-access-server/src/upload_flow.rs`, `app/crowdb-access-server/src/s3/operations/upload.rs`, `app/crowdb-access-server/src/iceberg/file_http/stream.rs`, `lib/crowdb-access-s3/src/integrity.rs`. +- [~] **Handle failed concurrent strip writes**: Mirror strips retain fragment views until finish, replace a failed segment before ordered completion, and keep later completed buffers until preceding writes commit. After replacement exhaustion, the writer drains submitted DiskIO, reads the committed prefix one strip at a time, and replays it with retained failed/later buffers into a new chunk. Tests cover rotation while a 225-frame input view still has an untouched tail, a failed partial final strip, persistent failure, and delayed abort before deleting the old chunk. The one-node fixture cannot allocate two mirror copies; validate multi-copy behavior with a multi-node fixture separately. Files: `lib/crowdb-chunk-client/src/chunk/chunk_writer.rs`, `lib/crowdb-chunk-client/src/chunk/chunk_writer/recovery.rs`, `lib/crowdb-chunk-client/src/chunk/mirror_strip_writer.rs`, affected chunk-client tests. + + - Rotation contract: keep one logical chunk location. Drain submitted writes, read the already committed prefix from old durable mirror strips one bounded strip at a time, and use retained failed/later buffers for the rest. Rewrite each frame footer's chunk ID on this rare recovery path; CRC excludes that ID. Checkpoint replayed bytes before deleting the old unsealed chunk. A partially processed input view keeps its accepted offset and rewrites its untouched tail for the new chunk. Candidate placement is tried a bounded number of times, including the original disk because a block-local failure may recover in a different allocation there. A failed candidate is deleted before another attempt. + +## Verification and closeout + +- [ ] **Verify large-path boundaries and errors**: Run 100-MiB direct, multipart, digest failure, cancellation, and chunk rotation cases. Confirm owner credits return and no unpublished record becomes visible. Files: `app/crowdb-access-server/tests/iceberg_file_http_test.rs`, affected chunk-client tests. +- [ ] **Compare real FileIO**: Use the R196 benchmark when available, or the focused FileIO path until then; retain raw samples and stage counters for before/after comparison on the same profile. Update the permanent upload-flow analysis with measured outcome. Files: `doc/design/access-server/iceberge/design-crowdb-iceberg-upload-flow.md`. +- [ ] **Run gates and clean up**: Run affected tests, `pixi run rs-fmt-check`, and `pixi run rs-lint` separately; remove the completed requirement, index entry, and this plan after acceptance. +- [~] **Check packaged OpenSSL**: The current working tree built `crowdb-iceberg-single-node:dev` successfully. `image-smoke.sh` passed, including packaged `libcrypto.so.3` linkage and runtime entrypoint checks. The 100-MiB direct small-cluster test exercises valid MD5, invalid MD5, and signed SHA-256 requests, but image-level digest upload smoke remains. Files: `container/single-node-container/collect-libs.sh`, relevant container smoke test. + +## Files + +- HTTP upload and metrics: `app/crowdb-access-server/src/upload_flow.rs`, `upload_flow/digest_pipe.rs`, `iceberg/file_http.rs`, `iceberg/file_http/stream.rs`, `iceberg/file_http/multipart.rs`, `iceberg/metrics.rs`, `s3/operations/upload.rs`. +- Native receive and chunk writer, only where measured: `lib/crowdb-access-s3/src/native_buffer.rs`, `lib/crowdb-chunk-client/src/chunk/mirror_strip_writer.rs`, `lib/crowdb-chunk-client/src/writer/large_async_object.rs`. +- Tests and evidence: `app/crowdb-access-server/tests/iceberg_file_http_test.rs`, `doc/design/access-server/iceberge/design-crowdb-iceberg-upload-flow.md`. + +## Tests + +- Unit: offer/pause/resume, digest order and lifetime, metric wait count and duration, frame finalization where changed. +- Integration: delayed large writer/digest, failed body/digest/write/publication. +- E2E: focused 100-MiB direct and multipart FileIO, range verification after timing, S3 smoke if shared chunk or native receive code changes. diff --git a/lib/crowdb-access-iceberg/Cargo.toml b/lib/crowdb-access-iceberg/Cargo.toml index a029acced..a0aa10ef5 100644 --- a/lib/crowdb-access-iceberg/Cargo.toml +++ b/lib/crowdb-access-iceberg/Cargo.toml @@ -26,6 +26,7 @@ lz4_flex = { version = "0.11", default-features = false, features = ["std", "saf md-5 = "0.10" crowdb-chunk-kv-client = { path = "../crowdb-chunk-kv-client" } crowdb-chunk-client = { path = "../crowdb-chunk-client" } +crowdb-kv-client = { path = "../crowdb-kv-client" } crowdb-access-multipart = { path = "../crowdb-access-multipart" } crowdb-common = { workspace = true } crowdb-protocol = { path = "../crowdb-protocol" } diff --git a/lib/crowdb-access-iceberg/src/catalog.rs b/lib/crowdb-access-iceberg/src/catalog.rs index 8fef30d80..0fcaacdb5 100644 --- a/lib/crowdb-access-iceberg/src/catalog.rs +++ b/lib/crowdb-access-iceberg/src/catalog.rs @@ -16,5 +16,6 @@ pub use repository::{CatalogError, CatalogRepository, ManagementPrivilege}; pub use root::{ActiveCatalogRecord, RootState}; pub use state::{CatalogAuthority, CatalogContext, CatalogLifecycle, ClearTransition}; pub use storage::{ - CasOutcome, CatalogStore, CatalogStoreOperationCounts, RoutedCatalogStore, StoreError, StoredValue, + CasOutcome, CatalogStore, CatalogStoreOperationCounts, CatalogStoreOperationMeter, RoutedCatalogStore, + StoreError, StoredValue, }; diff --git a/lib/crowdb-access-iceberg/src/catalog/storage.rs b/lib/crowdb-access-iceberg/src/catalog/storage.rs index af3613f57..96d4a9db3 100644 --- a/lib/crowdb-access-iceberg/src/catalog/storage.rs +++ b/lib/crowdb-access-iceberg/src/catalog/storage.rs @@ -56,6 +56,30 @@ struct OperationCounters { conditional_delete: AtomicU64, } +#[derive(Clone, Default)] +pub struct CatalogStoreOperationMeter(Arc); + +impl CatalogStoreOperationMeter { + #[must_use] + pub fn snapshot(&self) -> CatalogStoreOperationCounts { + self.0.snapshot() + } + + pub async fn observe(&self, future: F) -> F::Output { + REQUEST_OPERATIONS.scope(self.clone(), future).await + } +} + +tokio::task_local! { + static REQUEST_OPERATIONS: CatalogStoreOperationMeter; +} + +fn count_request(operation: fn(&OperationCounters) -> &AtomicU64) { + let _ = REQUEST_OPERATIONS.try_with(|meter| { + operation(&meter.0).fetch_add(1, Ordering::Relaxed); + }); +} + impl OperationCounters { fn snapshot(&self) -> CatalogStoreOperationCounts { CatalogStoreOperationCounts { @@ -132,6 +156,7 @@ impl RoutedCatalogStore { } validate_value(expected)?; self.counters.conditional_delete.fetch_add(1, Ordering::Relaxed); + count_request(|counters| &counters.conditional_delete); let response = self .client .execute_with_identity( @@ -183,6 +208,7 @@ impl RoutedCatalogStore { return Err(StoreError::Response); } self.counters.scan.fetch_add(1, Ordering::Relaxed); + count_request(|counters| &counters.scan); let page = self.client.scan(request).await?; if let Some(failure) = page.terminal_failure { return Err(StoreError::Rejected(failure)); @@ -207,6 +233,7 @@ impl CatalogStore for RoutedCatalogStore { async fn get(&self, key: &[u8]) -> Result, StoreError> { IcebergKey::decode(key)?; self.counters.get.fetch_add(1, Ordering::Relaxed); + count_request(|counters| &counters.get); let response = self.client.get(key.to_vec(), None).await?; match response.result.map_err(StoreError::Rejected)? { OperationResult::Value(value) => value.map(|value| stored(key, value)).transpose(), @@ -238,6 +265,7 @@ impl CatalogStore for RoutedCatalogStore { }, }; self.counters.compare_exchange.fetch_add(1, Ordering::Relaxed); + count_request(|counters| &counters.compare_exchange); let response = self .client .execute_with_identity(operation, None, identity) diff --git a/lib/crowdb-access-iceberg/src/commit/publication/completion.rs b/lib/crowdb-access-iceberg/src/commit/publication/completion.rs index 9351ce113..b0f4c890c 100644 --- a/lib/crowdb-access-iceberg/src/commit/publication/completion.rs +++ b/lib/crowdb-access-iceberg/src/commit/publication/completion.rs @@ -52,11 +52,15 @@ impl Publisher { mut operation: TableCommitOperation, ) -> Result { self.current(&operation).await?; + let mut prepared_body = None; if operation.phase == Phase::Publishing { - operation = self.select(&operation).await?; + (operation, prepared_body) = self.select(&operation).await?; } if operation.phase == Phase::Published { - let body = self.success_body(&operation).await?; + let body = match prepared_body { + Some(body) => body, + None => self.success_body(&operation).await?, + }; let mut next = advance(&operation, Phase::Complete)?; next.outcome = Some(TableCommitOutcome { status: 200, body }); self.change(&operation, &next).await?; @@ -74,9 +78,12 @@ impl Publisher { Ok(outcome) } - async fn select(&self, operation: &TableCommitOperation) -> Result { + async fn select( + &self, + operation: &TableCommitOperation, + ) -> Result<(TableCommitOperation, Option), Error> { let candidate = operation.candidate.as_ref().ok_or(ValidationError::Record)?; - self.success_body(operation).await?; + let prepared_body = self.success_body(operation).await?; self.current(operation).await?; let key = head_key(candidate.catalog, candidate.table).encode()?; let before = StorageRecord::TableHead(Box::new(operation.before.clone())).encode()?; @@ -111,7 +118,7 @@ impl Publisher { next.outcome = Some(TableCommitOutcome { status: 409, body }); } self.change(operation, &next).await?; - Ok(next) + Ok((next, published.then_some(prepared_body))) } async fn success_body( diff --git a/lib/crowdb-access-iceberg/src/file.rs b/lib/crowdb-access-iceberg/src/file.rs index 74aa2bd3a..5daee7a23 100644 --- a/lib/crowdb-access-iceberg/src/file.rs +++ b/lib/crowdb-access-iceberg/src/file.rs @@ -39,7 +39,8 @@ pub use avro::{ AvroRecordArray, AvroRecords, AvroScalar, AvroScalarType, AvroSchema, AvroTuple, AvroTupleField, }; pub use blocks::{ - FileBlockStore, FileIoError, NativeFileBlocks, MAX_FILE_BLOCK_BYTES, NATIVE_FILE_BLOCK_BYTES, + FileBlockStore, FileIoError, FileLocationStream, NativeFileBlocks, MAX_FILE_BLOCK_BYTES, + NATIVE_FILE_BLOCK_BYTES, }; pub use content::{ChunkRoot, FileContent, InlineCodec, MAX_COMPRESSION_INPUT_BYTES, MAX_INLINE_BYTES}; pub use credentials::{ diff --git a/lib/crowdb-access-iceberg/src/file/blocks.rs b/lib/crowdb-access-iceberg/src/file/blocks.rs index 86e3ac7d1..880878bb8 100644 --- a/lib/crowdb-access-iceberg/src/file/blocks.rs +++ b/lib/crowdb-access-iceberg/src/file/blocks.rs @@ -1,6 +1,9 @@ use async_trait::async_trait; use bytes::Bytes; -use crowdb_chunk_client::{ChunkIoClient, ChunkIoWriter}; +use crowdb_chunk_client::{ + ChunkIoClient, ChunkIoWriter, ChunkReadStream, LargeWritePolicy, ReadFlowMetricsSnapshot, + SmallWriteMetricsSnapshot, +}; use crowdb_protocol::chunkdb::rpc::Location; use crowdb_protocol::frame::MAX_FRAME_PAYLOAD_BYTES; use sha2::{Digest, Sha256}; @@ -29,12 +32,80 @@ pub enum FileIoError { Finished, } +pub struct FileLocationStream { + inner: ChunkReadStream, +} + +impl FileLocationStream { + pub async fn next_chunk(&mut self) -> Option> { + self.inner + .next_chunk() + .await + .map(|result| result.map_err(FileIoError::from)) + } +} + #[async_trait] pub trait FileBlockStore: Send + Sync { fn stream_client(&self) -> Option<&ChunkIoClient> { None } + fn supports_stream_io(&self) -> bool { + self.stream_client().is_some() + } + + fn chunk_metrics(&self) -> Option<(ReadFlowMetricsSnapshot, SmallWriteMetricsSnapshot)> { + self.stream_client() + .map(|client| (client.read_flow_metrics(), client.small_write_metrics())) + } + + /// Prepares an upload writer when this store owns native chunk storage. + /// + /// # Errors + /// Returns a chunk write admission or preparation error. + async fn prepare_upload_writer( + &self, + location_key: &str, + small_length: Option, + declared_length: Option, + large_write: &LargeWritePolicy, + ) -> Result, FileIoError> { + let Some(chunks) = self.stream_client() else { + return Ok(None); + }; + Ok(Some( + crate::storage::prepare_file_writer( + chunks, + location_key, + small_length, + declared_length, + large_write, + ) + .await?, + )) + } + + /// Opens a native pull stream when this store owns chunk locations. + /// + /// # Errors + /// Returns an invalid range or chunk read preparation error. + fn stream_locations( + &self, + locations: &[Location], + start: u64, + end: u64, + ) -> Result, FileIoError> { + self.stream_client() + .map(|client| { + client + .read_range_stream(locations, start, end) + .map(|inner| FileLocationStream { inner }) + .map_err(FileIoError::from) + }) + .transpose() + } + async fn read_locations( &self, _locations: &[Location], diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs index 8b334b6c6..178b0ed47 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs @@ -1,8 +1,8 @@ use crate::catalog::CatalogError; use crate::error::ValidationError; use crate::file::{ - file_key, ContentFormat, FileContent, FileKind, FileRecord, FileRepository, FileTree, MultipartPhase, - MultipartSelection, MultipartSession, MultipartStreamPart, SelectedPart, SelectedStreamPart, + file_key, FileContent, FileRecord, FileRepository, FileTree, MultipartPhase, MultipartSelection, + MultipartSession, MultipartStreamPart, SelectedPart, SelectedStreamPart, }; use crate::operation::PayloadStore; use crate::record::StorageRecord; @@ -64,34 +64,13 @@ impl MultipartRepository { .map_err(|_| ValidationError::Record)?; } let assembled = composer.finish().map_err(|_| ValidationError::Record)?; - let content = FileContent::from_locations(&assembled.locations, assembled.length, assembled.etag)?; - let path = session.location.relative_key(); - let extension = std::path::Path::new(path).extension(); - let has_extension = |wanted: &str| extension.is_some_and(|value| value.eq_ignore_ascii_case(wanted)); - let (kind, format) = if has_extension("json") { - (FileKind::Metadata, ContentFormat::Json) - } else if has_extension("avro") { - (FileKind::Unbound, ContentFormat::Avro) - } else if has_extension("parquet") { - (FileKind::Unbound, ContentFormat::Parquet) - } else if has_extension("orc") { - (FileKind::Unbound, ContentFormat::Orc) - } else if has_extension("puffin") { - (FileKind::Unbound, ContentFormat::Puffin) - } else { - (FileKind::Unbound, ContentFormat::Opaque) - }; - let record = FileRecord { - file: session.owner.file, - location: session.location.clone(), - kind, - format, - length: assembled.length, - digest: [0; 32], - content, - hint: None, - }; - record.validate()?; + let record = FileRecord::from_uploaded_locations( + session.owner.file, + session.location.clone(), + &assembled.locations, + assembled.length, + assembled.etag, + )?; let value = StorageRecord::File(Box::new(record)).encode()?; let publication = PayloadStore::new(self.store.clone()) .put(session.context.catalog, session.upload, &value) diff --git a/lib/crowdb-access-iceberg/src/file/parquet.rs b/lib/crowdb-access-iceberg/src/file/parquet.rs index a01045d16..06e98db00 100644 --- a/lib/crowdb-access-iceberg/src/file/parquet.rs +++ b/lib/crowdb-access-iceberg/src/file/parquet.rs @@ -102,7 +102,7 @@ impl ParquetMetadataLimits { if self.footer_bytes == 0 || self.footer_bytes > 1024 * 1024 || self.values == 0 - || self.values > 100_000 + || self.values > 500_000 || self.depth == 0 || self.depth > 32 || self.schema_elements == 0 diff --git a/lib/crowdb-access-iceberg/src/file/record.rs b/lib/crowdb-access-iceberg/src/file/record.rs index 705b4ea69..584a6983d 100644 --- a/lib/crowdb-access-iceberg/src/file/record.rs +++ b/lib/crowdb-access-iceberg/src/file/record.rs @@ -1,5 +1,6 @@ use crate::error::ValidationError; use crate::key::FileId; +use crowdb_protocol::chunkdb::rpc::Location; use super::{FileContent, FileLocation}; @@ -54,6 +55,46 @@ pub struct FileRecord { } impl FileRecord { + /// Builds one unbound file authority from completed chunk locations. + /// # Errors + /// Rejects invalid locations, `ETag`, or file record contents. + pub fn from_uploaded_locations( + file: FileId, + location: FileLocation, + locations: &[Location], + length: u64, + etag: String, + ) -> Result { + let content = FileContent::from_locations(locations, length, etag)?; + let extension = std::path::Path::new(location.relative_key()).extension(); + let has_extension = |wanted: &str| extension.is_some_and(|value| value.eq_ignore_ascii_case(wanted)); + let (kind, format) = if has_extension("json") { + (FileKind::Metadata, ContentFormat::Json) + } else if has_extension("avro") { + (FileKind::Unbound, ContentFormat::Avro) + } else if has_extension("parquet") { + (FileKind::Unbound, ContentFormat::Parquet) + } else if has_extension("orc") { + (FileKind::Unbound, ContentFormat::Orc) + } else if has_extension("puffin") { + (FileKind::Unbound, ContentFormat::Puffin) + } else { + (FileKind::Unbound, ContentFormat::Opaque) + }; + let record = Self { + file, + location, + kind, + format, + length, + digest: [0; 32], + content, + hint: None, + }; + record.validate()?; + Ok(record) + } + /// # Errors /// Rejects invalid format/kind pairs and inconsistent bounded storage variants. pub fn validate(&self) -> Result<(), ValidationError> { diff --git a/lib/crowdb-access-iceberg/src/file/repository.rs b/lib/crowdb-access-iceberg/src/file/repository.rs index 6d348cb74..91b242238 100644 --- a/lib/crowdb-access-iceberg/src/file/repository.rs +++ b/lib/crowdb-access-iceberg/src/file/repository.rs @@ -86,13 +86,6 @@ impl FileRepository { async fn stage(&self, candidate: &FileRecord) -> Result<(), CatalogError> { let key = file_key(candidate.location.table().catalog, candidate.file).encode()?; let bytes = StorageRecord::File(Box::new(candidate.clone())).encode()?; - if let Some(existing) = self.store.get(&key).await? { - return if existing.bytes == bytes { - Ok(()) - } else { - Err(CatalogError::Conflict) - }; - } match self .store .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) diff --git a/lib/crowdb-access-iceberg/src/gc.rs b/lib/crowdb-access-iceberg/src/gc.rs index 934b24a58..f6e58b01a 100644 --- a/lib/crowdb-access-iceberg/src/gc.rs +++ b/lib/crowdb-access-iceberg/src/gc.rs @@ -1,6 +1,7 @@ //! Durable, bounded reclamation after reachability and retention proof. mod admission; +mod budget; mod candidate; mod claim; mod discovery; @@ -18,6 +19,7 @@ mod task; mod tree; mod worker; +pub use budget::{BudgetedGcBlocks, BudgetedGcStore, GcIoBudget}; pub use candidate::{CandidatePhase, GcCandidate}; pub use limits::GcLimits; pub use mark::GcMarkError; diff --git a/app/crowdb-access-server/src/iceberg/gc_runtime/budget.rs b/lib/crowdb-access-iceberg/src/gc/budget.rs similarity index 88% rename from app/crowdb-access-server/src/iceberg/gc_runtime/budget.rs rename to lib/crowdb-access-iceberg/src/gc/budget.rs index e7f0ff506..ea74da130 100644 --- a/app/crowdb-access-server/src/iceberg/gc_runtime/budget.rs +++ b/lib/crowdb-access-iceberg/src/gc/budget.rs @@ -1,22 +1,23 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + use std::sync::{ atomic::{AtomicU32, AtomicU64, Ordering}, Arc, }; -use async_trait::async_trait; -use crowdb_access_iceberg::{ +use crate::{ catalog::{CasOutcome, CatalogStore, StoreError, StoredValue}, - file::{ChunkRoot, FileBlockStore, FileIdentity, FileIoError}, + file::{ChunkRoot, FileBlockStore, FileIdentity, FileIoError, NativeFileBlocks}, gc::{GcScan, GcStore, GcSystemScan}, key::{CatalogScope, IcebergKey}, record::MAX_RECORD_BYTES, }; -use crowdb_chunk_client::ReclaimOutcome; +use async_trait::async_trait; +use crowdb_chunk_client::{ChunkIoClient, ReclaimOutcome}; use crowdb_chunk_kv_client::MultiScanPage; use crowdb_protocol::chunk_kv::ClientRequestId; -use super::GcRuntimeConfig; - pub struct GcIoBudget { kv_bytes: AtomicU64, kv_requests: AtomicU32, @@ -31,24 +32,8 @@ pub struct GcIoBudget { } impl GcIoBudget { - pub(super) fn new(config: &GcRuntimeConfig) -> Self { - Self { - kv_bytes: AtomicU64::new(0), - kv_requests: AtomicU32::new(0), - chunk_bytes: AtomicU64::new(0), - chunk_requests: AtomicU32::new(0), - recovery_bytes: AtomicU64::new(0), - recovery_requests: AtomicU32::new(0), - max_kv_bytes: config.kv_bytes, - max_kv_requests: config.kv_requests, - max_chunk_bytes: config.chunk_bytes, - max_chunk_requests: config.chunk_requests, - } - } - - #[cfg(feature = "test-util")] #[must_use] - pub fn for_tests(kv_bytes: u64, kv_requests: u32, chunk_bytes: u64, chunk_requests: u32) -> Self { + pub fn new(kv_bytes: u64, kv_requests: u32, chunk_bytes: u64, chunk_requests: u32) -> Self { Self { kv_bytes: AtomicU64::new(0), kv_requests: AtomicU32::new(0), @@ -206,6 +191,11 @@ impl BudgetedGcBlocks { pub fn new(inner: Arc, budget: Arc) -> Self { Self { inner, budget } } + + #[must_use] + pub fn native(client: ChunkIoClient, store: Arc, budget: Arc) -> Self { + Self::new(Arc::new(NativeFileBlocks::new(client, store)), budget) + } } #[async_trait] diff --git a/lib/crowdb-access-iceberg/src/lib.rs b/lib/crowdb-access-iceberg/src/lib.rs index 4cbab3037..7aca6269d 100644 --- a/lib/crowdb-access-iceberg/src/lib.rs +++ b/lib/crowdb-access-iceberg/src/lib.rs @@ -11,5 +11,6 @@ pub mod metadata_projection; pub mod namespace; pub mod operation; pub mod record; +pub mod storage; pub mod table; pub mod wire; diff --git a/lib/crowdb-access-iceberg/src/storage.rs b/lib/crowdb-access-iceberg/src/storage.rs new file mode 100644 index 000000000..4a83b049c --- /dev/null +++ b/lib/crowdb-access-iceberg/src/storage.rs @@ -0,0 +1,267 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Iceberg catalog and foreground chunk-client wiring. + +use std::sync::Arc; + +use bytes::Bytes; +use crowdb_chunk_client::{ + ChunkClientConfig, ChunkIoClient, ChunkIoClientConfig, ChunkIoWriter, ChunkReadPolicy, ChunkWriteTiming, + FramedWriteBuffer, IoError, LargeWritePolicy, SmallWritePolicy, +}; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ChunkKvRpcTransport, ClientConfig, Group0ChunkKvRangeCatalogSource, +}; +use crowdb_common::ec::EcScheme; +use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; +use crowdb_protocol::chunkdb::rpc::{ChunkType, Location}; +use crowdb_protocol::frame::MAX_FRAME_PAYLOAD_BYTES; + +use crate::catalog::{CatalogRepository, ClearBounds, RoutedCatalogStore}; +use crate::file::{FileBlockStore, NativeFileBlocks}; + +pub type IcebergStorageError = Box; + +pub fn own_large_write(policy: &mut LargeWritePolicy) { + Arc::make_mut(&mut policy.client).chunk_type = ChunkType::IcebergTable; +} + +#[must_use] +pub fn default_large_write() -> LargeWritePolicy { + let mut policy = LargeWritePolicy { + ec_scheme: EcScheme::new(8, 4), + client: Arc::new(ChunkClientConfig::default()), + }; + own_large_write(&mut policy); + policy +} + +#[must_use] +pub fn foreground_blocks(chunks: ChunkIoClient, store: Arc) -> Arc { + Arc::new(NativeFileBlocks::new(chunks, store)) +} + +pub struct IcebergFileWriter { + inner: Box, +} + +#[async_trait::async_trait] +impl ChunkIoWriter for IcebergFileWriter { + async fn on_data(&mut self, buffer: Bytes) -> Result { + self.inner.on_data(buffer).await + } + + async fn on_framed_data( + &mut self, + buffer: Box, + ) -> Result { + self.inner.on_framed_data(buffer).await + } + + async fn on_finish(&mut self) -> Result, IoError> { + self.inner.on_finish().await + } + + async fn on_error(&mut self) -> Result, IoError> { + self.inner.on_error().await + } + + fn require_data(&self) -> bool { + self.inner.require_data() + } + + fn input_complete(&self) -> bool { + self.inner.input_complete() + } + + fn write_timing(&self) -> Option { + self.inner.write_timing() + } + + async fn wait_for_capacity(&mut self) { + self.inner.wait_for_capacity().await; + } +} + +impl IcebergFileWriter { + #[must_use] + pub fn require_data(&self) -> bool { + self.inner.require_data() + } + + #[must_use] + pub fn input_complete(&self) -> bool { + self.inner.input_complete() + } + + #[must_use] + pub fn write_timing(&self) -> Option { + self.inner.write_timing() + } + + pub async fn wait_for_capacity(&mut self) { + self.inner.wait_for_capacity().await; + } + + /// # Errors + /// Returns a chunk write failure. + pub async fn on_data(&mut self, bytes: Bytes) -> Result<(), IoError> { + self.inner.on_data(bytes).await.map(|_| ()) + } + + /// # Errors + /// Returns a chunk write failure. + pub async fn on_framed_data(&mut self, buffer: Box) -> Result<(), IoError> { + self.inner.on_framed_data(buffer).await.map(|_| ()) + } + + /// # Errors + /// Returns a chunk seal failure. + pub async fn on_finish(&mut self) -> Result, IoError> { + self.inner.on_finish().await + } + + /// # Errors + /// Returns a chunk cleanup failure. + pub async fn on_error(&mut self) -> Result<(), IoError> { + self.inner.on_error().await.map(|_| ()) + } +} + +/// Prepares the chunk writer for one foreground Iceberg file upload. +/// +/// # Errors +/// Returns a chunk admission or preparation failure. +pub async fn prepare_file_writer( + chunks: &ChunkIoClient, + location_key: &str, + small_length: Option, + declared_length: Option, + large_write: &LargeWritePolicy, +) -> Result { + if let Some(length) = small_length { + if length <= MAX_FRAME_PAYLOAD_BYTES { + let mut writer = chunks + .prepare_small_write_for_key(length, location_key.as_bytes()) + .await?; + writer.require_durable_completion(); + return Ok(IcebergFileWriter { + inner: Box::new(writer), + }); + } + return Ok(IcebergFileWriter { + inner: Box::new( + chunks + .prepare_shared_object_write_for_key(length, location_key.as_bytes()) + .await?, + ), + }); + } + let mut writer = chunks.prepare_large_write(declared_length, large_write.clone()); + writer.wait_until_prepared().await?; + Ok(IcebergFileWriter { + inner: Box::new(writer), + }) +} + +pub struct IcebergLargeWriteSettings { + pub ec_data: usize, + pub ec_code: usize, + pub disk_block_bytes: usize, + pub mirror_copies: Option, + pub max_chunk_size: Option, + pub memory_budget_bytes: Option, + pub prefetch_strips_per_chunk: Option, + pub prefetch_max_strips_per_batch: Option, + pub parallel_strip_writes: Option, + pub held_buffers: Option, + pub chunk_preparation_depth: Option, +} + +impl IcebergLargeWriteSettings { + /// # Errors + /// Rejects invalid Iceberg large-write geometry before connecting storage. + pub fn policy(self) -> Result { + if self.ec_data == 0 || self.ec_data > 32 || self.ec_code == 0 { + return Err("Iceberg EC data and code counts are invalid".into()); + } + let mut client = ChunkClientConfig { + chunk_type: ChunkType::IcebergTable, + large_mirror_copies: self.mirror_copies, + read_buffer_size: self.disk_block_bytes, + ..ChunkClientConfig::default() + }; + if let Some(value) = self.max_chunk_size { + client.max_chunk_size = value; + } + if let Some(value) = self.memory_budget_bytes { + client.memory_budget = value; + } + if let Some(value) = self.prefetch_strips_per_chunk { + client.prefetch_strips_per_chunk = value; + } + if let Some(value) = self.prefetch_max_strips_per_batch { + client.large_prefetch_max_strips_per_batch = value; + } + if let Some(value) = self.parallel_strip_writes { + client.large_parallel_strip_writes = value; + } + if let Some(value) = self.held_buffers { + client.large_held_buffers = value; + } + if let Some(value) = self.chunk_preparation_depth { + client.chunk_preparation_depth = value; + } + client.validate().map_err(|error| error.to_string())?; + Ok(LargeWritePolicy { + ec_scheme: EcScheme::new(self.ec_data, self.ec_code), + client: Arc::new(client), + }) + } +} + +/// Connects Iceberg metadata and a separately budgeted foreground chunk pool. +/// +/// # Errors +/// Returns a discovery, catalog, or chunk-client configuration failure. +pub async fn connect( + seeds: Vec, + read_policy: ChunkReadPolicy, + mut small_write: SmallWritePolicy, + diskio_connections_per_endpoint: usize, + diskio_rpc_workers: u32, +) -> Result<(Arc, Arc, ChunkIoClient), IcebergStorageError> { + small_write.chunk_type = ChunkType::IcebergTable; + let control = Arc::new(CrowdbKvClient::new(KvConfig::new(seeds.clone()))); + let client_config = ClientConfig::default(); + let source = Arc::new(Group0ChunkKvRangeCatalogSource::from_shared(Arc::clone(&control))); + let transport = Arc::new(ChunkKvRpcTransport::new( + client_config.max_owner_connections, + 1, + 2, + )); + let client = Arc::new(ChunkKvClient::new(client_config, source, transport)?); + client.refresh_catalog().await?; + let store = Arc::new(RoutedCatalogStore::new(client)); + let repository = Arc::new(CatalogRepository::new( + store.clone(), + ClearBounds { + request_ms: 300_000, + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + )?); + let chunks = ChunkIoClient::connect_with_kv_read_policy( + ChunkIoClientConfig { + management_seeds: seeds, + diskio_connections_per_endpoint, + diskio_rpc_workers, + small_write, + }, + control, + read_policy, + ) + .await?; + Ok((repository, store, chunks)) +} diff --git a/lib/crowdb-access-iceberg/tests/avro_projection_test.rs b/lib/crowdb-access-iceberg/tests/avro_projection_test.rs index b31bb4e35..75da64d2a 100644 --- a/lib/crowdb-access-iceberg/tests/avro_projection_test.rs +++ b/lib/crowdb-access-iceberg/tests/avro_projection_test.rs @@ -1,5 +1,5 @@ use crowdb_access_iceberg::file::{ - AvroContainerError, AvroDatumLimits, AvroProjection, AvroScalar, AvroSchema, + AvroContainerError, AvroDatumLimits, AvroProjection, AvroScalar, AvroScalarType, AvroSchema, }; fn limits() -> AvroDatumLimits { @@ -46,16 +46,13 @@ fn projection_uses_ids_and_request_order_with_borrowed_strings_and_nullable_valu } #[test] -fn projection_rejects_ambiguous_ids_and_non_scalar_layouts_without_changing_avro_validation() { +fn projection_rejects_ambiguous_ids_and_unsupported_layouts_without_changing_avro_validation() { let schema = schema(); - for ids in [ - vec![], - vec![500; 65], - vec![500, 500], - vec![-1], - vec![999], - vec![507], - ] { + assert_eq!( + AvroProjection::new(&schema, &[507]).unwrap().field_types(), + &[Some(AvroScalarType::IntList)] + ); + for ids in [vec![], vec![500; 65], vec![500, 500], vec![-1], vec![999]] { assert!(AvroProjection::new(&schema, &ids).is_err()); } for fields in [ @@ -66,6 +63,7 @@ fn projection_rejects_ambiguous_ids_and_non_scalar_layouts_without_changing_avro r#"[{"name":"a","field-id":0,"type":"long"},{"name":"b","field-id":0,"type":"long"}]"#, r#"[{"name":"a","field-id":0,"type":["long","int"]}]"#, r#"[{"name":"a","field-id":0,"type":["null","long","int"]}]"#, + r#"[{"name":"a","field-id":0,"type":{"type":"array","items":"string"}}]"#, ] { let bytes = format!(r#"{{"type":"record","name":"R","fields":{fields}}}"#); let schema = AvroSchema::parse(bytes.as_bytes()).unwrap(); diff --git a/lib/crowdb-access-iceberg/tests/parquet_metadata_test.rs b/lib/crowdb-access-iceberg/tests/parquet_metadata_test.rs index b7a951a34..54b64e3fb 100644 --- a/lib/crowdb-access-iceberg/tests/parquet_metadata_test.rs +++ b/lib/crowdb-access-iceberg/tests/parquet_metadata_test.rs @@ -20,6 +20,28 @@ async fn apache_alltypes_plain_footer_decodes_standard_delta_field_headers() { assert_eq!(metadata.schema[11].physical_type, Some(3)); } +#[tokio::test] +async fn many_row_groups_fit_within_a_bounded_footer() { + let groups = vec![row_group(10, &column()); 6_000]; + let mut fields = footer(); + set(&mut fields, 3, number(60_000)); + set(&mut fields, 4, list(12, &groups)); + let bytes = structure(&fields); + assert!(bytes.len() < 1024 * 1024); + let (store, record) = stored(&bytes, 64).await; + let mut limits = limits(); + limits.footer_bytes = 1024 * 1024; + limits.row_groups = 10_000; + limits.values = 100_000; + assert!(matches!( + read_parquet_metadata(store.clone(), &record, limits).await, + Err(Error::Bounds) + )); + limits.values = 500_000; + let metadata = read_parquet_metadata(store, &record, limits).await.unwrap(); + assert_eq!((metadata.rows, metadata.row_groups), (60_000, 6_000)); +} + #[tokio::test] async fn canonical_footer_decodes_long_form_fields_and_ignores_stored_hints() { let mut fields = footer(); diff --git a/lib/crowdb-access-iceberg/tests/storage_policy_test.rs b/lib/crowdb-access-iceberg/tests/storage_policy_test.rs new file mode 100644 index 000000000..2fe3666ac --- /dev/null +++ b/lib/crowdb-access-iceberg/tests/storage_policy_test.rs @@ -0,0 +1,50 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_iceberg::storage::IcebergLargeWriteSettings; +use crowdb_protocol::chunkdb::rpc::ChunkType; + +#[test] +fn iceberg_large_policy_owns_type_and_capacity() { + let policy = IcebergLargeWriteSettings { + ec_data: 2, + ec_code: 1, + disk_block_bytes: 1024 * 1024, + mirror_copies: Some(1), + max_chunk_size: Some(32 * 1024 * 1024), + memory_budget_bytes: Some(16 * 1024 * 1024), + prefetch_strips_per_chunk: Some(2), + prefetch_max_strips_per_batch: Some(20), + parallel_strip_writes: Some(4), + held_buffers: Some(4), + chunk_preparation_depth: Some(2), + } + .policy() + .unwrap(); + + assert_eq!(policy.ec_scheme.data_num, 2); + assert_eq!(policy.client.chunk_type, ChunkType::IcebergTable); + assert_eq!(policy.client.large_mirror_copies, Some(1)); + assert_eq!(policy.client.max_chunk_size, 32 * 1024 * 1024); + assert_eq!(policy.client.prefetch_strips_per_chunk, 2); + assert_eq!(policy.client.large_prefetch_max_strips_per_batch, 20); +} + +#[test] +fn iceberg_large_policy_rejects_zero_prefetch() { + let result = IcebergLargeWriteSettings { + ec_data: 2, + ec_code: 1, + disk_block_bytes: 1024 * 1024, + mirror_copies: None, + max_chunk_size: None, + memory_budget_bytes: None, + prefetch_strips_per_chunk: Some(0), + prefetch_max_strips_per_batch: None, + parallel_strip_writes: None, + held_buffers: None, + chunk_preparation_depth: None, + } + .policy(); + assert!(result.is_err()); +} diff --git a/lib/crowdb-access-s3/Cargo.toml b/lib/crowdb-access-s3/Cargo.toml index 80ded8296..fb85c4e6a 100644 --- a/lib/crowdb-access-s3/Cargo.toml +++ b/lib/crowdb-access-s3/Cargo.toml @@ -19,8 +19,10 @@ base64 = "0.22" bincode = "1" chrono = { version = "0.4", default-features = false, features = ["std"] } crowdb-chunk-client = { path = "../crowdb-chunk-client" } +crowdb-common = { workspace = true } crowdb-access-multipart = { path = "../crowdb-access-multipart" } crowdb-chunk-kv-client = { path = "../crowdb-chunk-kv-client" } +crowdb-kv-client = { path = "../crowdb-kv-client" } crowdb-protocol = { path = "../crowdb-protocol" } flatbuffers = { workspace = true } hyper = { workspace = true, features = ["http1", "server"] } diff --git a/lib/crowdb-access-s3/src/integrity.rs b/lib/crowdb-access-s3/src/integrity.rs index ecaf782fb..ff9294c76 100644 --- a/lib/crowdb-access-s3/src/integrity.rs +++ b/lib/crowdb-access-s3/src/integrity.rs @@ -7,6 +7,7 @@ use base64::engine::general_purpose::STANDARD; use base64::Engine as _; use hyper::body::Bytes; use sha2::{Digest as _, Sha256}; +use std::fmt::Write as _; #[derive(Clone, Copy, Debug, Eq, PartialEq, thiserror::Error)] pub enum IntegrityError { @@ -87,30 +88,52 @@ impl SinglePartIntegrity { ) -> Result<(String, Vec), IntegrityError> { let Self { md5, sha256 } = self; let digest = md5.compute(); - let result = (format!("{digest:x}"), digest.0.to_vec()); - if let Some(expected) = expected_content_md5 { - let expected = STANDARD - .decode(expected) - .map_err(|_| IntegrityError::InvalidDigest)?; - if expected.len() != 16 { - return Err(IntegrityError::InvalidDigest); - } - if expected != result.1 { - return Err(IntegrityError::Mismatch); - } + let sha256 = sha256.map(|sha256| sha256.finalize().into()); + validate_completed_digests(digest.0, sha256, expected_content_md5, expected_payload_sha256) + } +} + +/// Validates checksums produced by a separate object-scoped digest worker. +/// +/// # Errors +/// Rejects malformed or mismatched declared digests. +pub fn validate_completed_digests( + md5: [u8; 16], + sha256: Option<[u8; 32]>, + expected_content_md5: Option<&str>, + expected_payload_sha256: Option<&str>, +) -> Result<(String, Vec), IntegrityError> { + let result = (hex_digest(&md5), md5.to_vec()); + if let Some(expected) = expected_content_md5 { + let expected = STANDARD + .decode(expected) + .map_err(|_| IntegrityError::InvalidDigest)?; + if expected.len() != 16 { + return Err(IntegrityError::InvalidDigest); + } + if expected != result.1 { + return Err(IntegrityError::Mismatch); } - if let Some(expected) = expected_payload_sha256 { - if expected.len() != 64 || !expected.bytes().all(|byte| byte.is_ascii_hexdigit()) { - return Err(IntegrityError::InvalidPayloadDigest); - } - let sha256 = sha256.ok_or(IntegrityError::InvalidPayloadDigest)?; - let actual = format!("{:x}", sha256.finalize()); - if !actual.eq_ignore_ascii_case(expected) { - return Err(IntegrityError::PayloadMismatch); - } + } + if let Some(expected) = expected_payload_sha256 { + if expected.len() != 64 || !expected.bytes().all(|byte| byte.is_ascii_hexdigit()) { + return Err(IntegrityError::InvalidPayloadDigest); + } + let sha256 = sha256.ok_or(IntegrityError::InvalidPayloadDigest)?; + let actual = hex_digest(&sha256); + if !actual.eq_ignore_ascii_case(expected) { + return Err(IntegrityError::PayloadMismatch); } - Ok(result) } + Ok(result) +} + +fn hex_digest(bytes: &[u8]) -> String { + let mut value = String::with_capacity(bytes.len() * 2); + for byte in bytes { + write!(&mut value, "{byte:02x}").expect("writing to String cannot fail"); + } + value } /// Encodes a composite multipart `ETag` as a distinct 18-byte metadata diff --git a/lib/crowdb-access-s3/src/lib.rs b/lib/crowdb-access-s3/src/lib.rs index 1619f3a27..e2ceaa332 100644 --- a/lib/crowdb-access-s3/src/lib.rs +++ b/lib/crowdb-access-s3/src/lib.rs @@ -17,6 +17,7 @@ pub mod publication; pub mod range; pub mod retrieval; pub mod route; +pub mod storage; pub mod streaming; pub mod wire; diff --git a/lib/crowdb-access-s3/src/metrics.rs b/lib/crowdb-access-s3/src/metrics.rs index 473430e6d..b576d546e 100644 --- a/lib/crowdb-access-s3/src/metrics.rs +++ b/lib/crowdb-access-s3/src/metrics.rs @@ -315,6 +315,7 @@ fn append_request_metrics(output: &mut String, snapshot: &S3MetricsSnapshot) { fn append_native_metrics(output: &mut String, native: NativeBufferMetricsSnapshot) { for (name, value) in [ ("crowdb_s3_native_retained_bytes", native.retained_bytes), + ("crowdb_s3_native_peak_retained_bytes", native.peak_retained_bytes), ("crowdb_s3_native_direct_bytes_total", native.direct_bytes), ( "crowdb_s3_native_prefix_copy_bytes_total", diff --git a/lib/crowdb-access-s3/src/native_buffer.rs b/lib/crowdb-access-s3/src/native_buffer.rs index 90b0caa63..ba092929c 100644 --- a/lib/crowdb-access-s3/src/native_buffer.rs +++ b/lib/crowdb-access-s3/src/native_buffer.rs @@ -14,15 +14,15 @@ use std::ptr::NonNull; use std::sync::atomic::{AtomicBool, AtomicPtr, AtomicUsize, Ordering}; use std::sync::{Arc, Weak}; use std::task::{Context, Poll}; -use std::time::Instant; +use std::time::{Instant, SystemTime, UNIX_EPOCH}; use arc_swap::ArcSwap; use atomic_waker::AtomicWaker; use crowdb_chunk_client::FramedWriteBuffer; use crowdb_protocol::common::ChunkId; use crowdb_protocol::frame::{ - encode_frame_regions, FrameError, FrameMagic, FRAME_FOOTER_BYTES, FRAME_HEADER_PREFIX_BYTES, - MAX_FRAME_BYTES, MAX_FRAME_PAYLOAD_BYTES, + encode_frame_regions, prepare_frame_regions, set_frame_chunk_id, FrameError, FrameMagic, + FRAME_FOOTER_BYTES, FRAME_HEADER_PREFIX_BYTES, MAX_FRAME_BYTES, MAX_FRAME_PAYLOAD_BYTES, }; use hyper::body::{Bytes, Http1BodyReceiveBuffer, Http1BodyReceiveProvider}; @@ -46,6 +46,8 @@ pub struct NativeFramedOwner { payload_lengths: Box<[u16]>, logical_len: u64, physical_len: usize, + prepared_slots: usize, + prepared: bool, } // SAFETY: Hyper invokes one provider serially for one Incoming body. The @@ -58,6 +60,7 @@ struct ReceiverState { next_slot: usize, issued: Option, completed_slots: usize, + prepared_slots: usize, payload_lengths: Vec, append_slot: Option, credit_wait_started: Option, @@ -74,6 +77,7 @@ struct AllocatorState { budget_bytes: usize, owner_bytes: usize, retained_bytes: AtomicUsize, + peak_retained_bytes: AtomicUsize, allocations: AtomicUsize, direct_bytes: AtomicUsize, prefix_copy_bytes: AtomicUsize, @@ -87,6 +91,7 @@ pub struct NativeBufferMetricsSnapshot { pub budget_bytes: usize, pub owner_bytes: usize, pub retained_bytes: usize, + pub peak_retained_bytes: usize, pub allocations: usize, pub direct_bytes: usize, pub prefix_copy_bytes: usize, @@ -120,6 +125,7 @@ impl NativeBodyAllocator { budget_bytes, owner_bytes, retained_bytes: AtomicUsize::new(0), + peak_retained_bytes: AtomicUsize::new(0), allocations: AtomicUsize::new(0), direct_bytes: AtomicUsize::new(0), prefix_copy_bytes: AtomicUsize::new(0), @@ -156,6 +162,7 @@ impl NativeBodyAllocator { budget_bytes: self.state.budget_bytes, owner_bytes: self.state.owner_bytes, retained_bytes: self.state.retained_bytes.load(Ordering::Acquire), + peak_retained_bytes: self.state.peak_retained_bytes.load(Ordering::Acquire), allocations: self.state.allocations.load(Ordering::Relaxed), direct_bytes: self.state.direct_bytes.load(Ordering::Relaxed), prefix_copy_bytes: self.state.prefix_copy_bytes.load(Ordering::Relaxed), @@ -185,6 +192,7 @@ impl NativeBodyAllocator { next_slot: 0, issued: None, completed_slots: 0, + prepared_slots: 0, payload_lengths: vec![0; self.state.owner_bytes / MAX_FRAME_BYTES], append_slot: None, credit_wait_started: None, @@ -196,14 +204,22 @@ impl NativeBodyAllocator { } fn try_reserve(&self, bytes: usize) -> bool { - self.state - .retained_bytes - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { - current - .checked_add(bytes) - .filter(|next| *next <= self.state.budget_bytes) - }) - .is_ok() + let reserved = + self.state + .retained_bytes + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { + current + .checked_add(bytes) + .filter(|next| *next <= self.state.budget_bytes) + }); + if let Ok(previous) = reserved { + self.state + .peak_retained_bytes + .fetch_max(previous + bytes, Ordering::Relaxed); + true + } else { + false + } } fn wake_credit_waiters(&self) { @@ -222,6 +238,45 @@ impl AllocatorState { } impl NativeBodyReceiver { + fn prepare_full_slot(state: &mut ReceiverState, owner: &NativeOwner, slot: usize) -> io::Result<()> { + if state.prepared_slots != slot { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + "native frames completed out of order", + )); + } + let frame_offset = slot * MAX_FRAME_BYTES; + let payload_offset = frame_offset + FRAME_HEADER_PREFIX_BYTES; + let write_time_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |duration| { + u64::try_from(duration.as_millis()).unwrap_or(u64::MAX) + }); + // SAFETY: a full receive slot is initialized and its reserved header + // and footer do not overlap the payload or any other slot. + unsafe { + let header = std::slice::from_raw_parts_mut( + owner.pointer.as_ptr().add(frame_offset), + FRAME_HEADER_PREFIX_BYTES, + ); + let payload = std::slice::from_raw_parts( + owner.pointer.as_ptr().add(payload_offset), + MAX_FRAME_PAYLOAD_BYTES, + ); + let footer = std::slice::from_raw_parts_mut( + owner + .pointer + .as_ptr() + .add(payload_offset + MAX_FRAME_PAYLOAD_BYTES), + FRAME_FOOTER_BYTES, + ); + prepare_frame_regions(FrameMagic::RepoLargeV1, payload, write_time_ms, header, footer) + .map_err(|error| io::Error::new(io::ErrorKind::InvalidData, error))?; + } + state.prepared_slots += 1; + Ok(()) + } + fn poll_prepare_owner(&self, state: &mut ReceiverState, cx: &mut Context<'_>) -> Poll> { if state.owner.is_some() { return Poll::Ready(Ok(())); @@ -289,6 +344,9 @@ impl NativeBodyReceiver { io::Error::new(io::ErrorKind::InvalidData, "native payload length exceeds u16") })?; state.completed_slots += 1; + if payload.len() == MAX_FRAME_PAYLOAD_BYTES { + Self::prepare_full_slot(state, &owner, slot)?; + } } state.next_slot = state.completed_slots; if let Some(last) = state.completed_slots.checked_sub(1) { @@ -447,6 +505,8 @@ fn completed_owner(state: &mut ReceiverState, owner: Arc) -> io::Re + usize::from(last_payload) + FRAME_FOOTER_BYTES; state.completed_slots = 0; + let prepared_slots = std::mem::take(&mut state.prepared_slots); + let prepared = prepared_slots == lengths.len(); state.append_slot = None; state.payload_lengths.fill(0); Ok(NativeFramedOwner { @@ -454,6 +514,8 @@ fn completed_owner(state: &mut ReceiverState, owner: Arc) -> io::Re payload_lengths: lengths, logical_len, physical_len, + prepared_slots, + prepared, }) } @@ -480,6 +542,36 @@ impl FramedWriteBuffer for NativeFramedOwner { self.payload_lengths.get(index).copied().map(usize::from) } + fn prepare_frames(&mut self, magic: FrameMagic, write_time_ms: u64) -> Result<(), FrameError> { + if self.prepared { + return Ok(()); + } + for index in self.prepared_slots..self.payload_lengths.len() { + let payload_len = usize::from(self.payload_lengths[index]); + let frame_offset = index + .checked_mul(MAX_FRAME_BYTES) + .ok_or(FrameError::LengthOverflow)?; + let payload_offset = frame_offset + FRAME_HEADER_PREFIX_BYTES; + // SAFETY: socket receive has completed this slot. Header and + // footer regions are disjoint from immutable payload views. + unsafe { + let header = std::slice::from_raw_parts_mut( + self.owner.pointer.as_ptr().add(frame_offset), + FRAME_HEADER_PREFIX_BYTES, + ); + let payload = + std::slice::from_raw_parts(self.owner.pointer.as_ptr().add(payload_offset), payload_len); + let footer = std::slice::from_raw_parts_mut( + self.owner.pointer.as_ptr().add(payload_offset + payload_len), + FRAME_FOOTER_BYTES, + ); + prepare_frame_regions(magic, payload, write_time_ms, header, footer)?; + } + } + self.prepared = true; + Ok(()) + } + fn finalize_frame( &mut self, index: usize, @@ -509,7 +601,11 @@ impl FramedWriteBuffer for NativeFramedOwner { self.owner.pointer.as_ptr().add(payload_offset + payload_len), FRAME_FOOTER_BYTES, ); - encode_frame_regions(magic, chunk_id, payload, write_time_ms, header, footer)?; + if self.prepared { + set_frame_chunk_id(chunk_id, footer)?; + } else { + encode_frame_regions(magic, chunk_id, payload, write_time_ms, header, footer)?; + } } Ok(frame_offset..frame_offset + frame_len) } @@ -563,6 +659,7 @@ impl Http1BodyReceiveProvider for NativeBodyReceiver { if state.owner.is_none() { state.next_slot = 0; state.completed_slots = 0; + state.prepared_slots = 0; state.payload_lengths.fill(0); state.append_slot = None; match self.poll_prepare_owner(state, cx) { @@ -576,6 +673,7 @@ impl Http1BodyReceiveProvider for NativeBodyReceiver { if state.owner.is_none() { state.next_slot = 0; state.completed_slots = 0; + state.prepared_slots = 0; state.payload_lengths.fill(0); state.append_slot = None; match self.poll_prepare_owner(state, cx) { @@ -661,6 +759,8 @@ impl Http1BodyReceiveProvider for NativeBodyReceiver { } if payload_len < MAX_FRAME_PAYLOAD_BYTES { state.append_slot = Some(issued.slot); + } else { + Self::prepare_full_slot(state, &owner, issued.slot)?; } if state.completed_slots == state.payload_lengths.len() && state.append_slot.is_none() { state.owner = None; diff --git a/lib/crowdb-access-s3/src/storage.rs b/lib/crowdb-access-s3/src/storage.rs new file mode 100644 index 000000000..69ece34a4 --- /dev/null +++ b/lib/crowdb-access-s3/src/storage.rs @@ -0,0 +1,194 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Production client wiring for stateless S3 handlers. + +use std::sync::Arc; + +use crate::metadata::ChunkKvMetadataStore; +use crowdb_chunk_client::{ + ChunkClientConfig, ChunkIoClient, ChunkIoClientConfig, ChunkReadPolicy, LargeWritePolicy, + SmallWritePolicy, +}; +use crowdb_chunk_kv_client::{ + ChunkKvClient, ChunkKvRpcTransport, ClientConfig as ChunkKvConfig, Group0ChunkKvRangeCatalogSource, +}; +use crowdb_common::ec::EcScheme; +use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; +use crowdb_protocol::chunkdb::rpc::ChunkType; + +#[derive(Clone)] +pub struct S3StorageClients { + pub control: Arc, + pub metadata: Arc, + pub chunks: Arc, +} + +#[derive(Debug, thiserror::Error)] +pub enum StorageConnectError { + #[error("Chunk-KV client configuration failed: {0}")] + ChunkKv(String), + #[error("chunk I/O discovery failed: {0}")] + ChunkIo(String), +} + +pub fn own_large_write(policy: &mut LargeWritePolicy) { + Arc::make_mut(&mut policy.client).chunk_type = ChunkType::S3; +} + +pub struct S3WriteSettings { + pub small: SmallWritePolicy, + pub threshold_ratio: f64, + pub disk_block_bytes: usize, + pub ec_data: usize, + pub ec_code: usize, + pub large: S3LargeWriteSettings, +} + +#[derive(Default)] +pub struct S3LargeWriteSettings { + pub mirror_copies: Option, + pub max_chunk_size: Option, + pub memory_budget_bytes: Option, + pub prefetch_strips_per_chunk: Option, + pub prefetch_max_strips_per_batch: Option, + pub parallel_strip_writes: Option, + pub held_buffers: Option, + pub chunk_preparation_depth: Option, +} + +pub struct S3WritePolicies { + pub small: SmallWritePolicy, + pub large: LargeWritePolicy, + pub small_threshold: usize, +} + +impl S3WriteSettings { + /// # Errors + /// Rejects invalid S3 admission or large-write geometry before connecting storage. + pub fn policies(self) -> Result { + if self.ec_data == 0 || self.ec_data > 32 || self.ec_code == 0 { + return Err("S3 EC data and code counts are invalid".into()); + } + if !self.threshold_ratio.is_finite() + || !(0.0..=1.0).contains(&self.threshold_ratio) + || self.threshold_ratio == 0.0 + { + return Err("S3 small-object threshold ratio is invalid".into()); + } + let mut small = self.small; + small.chunk_type = ChunkType::S3; + small.conversion_data_num = self.ec_data; + small.conversion_code_num = self.ec_code; + small.validate().map_err(|error| error.to_string())?; + let data_shards = if small.conversion_enabled { self.ec_data } else { 1 }; + #[allow( + clippy::cast_possible_truncation, + clippy::cast_sign_loss, + clippy::cast_precision_loss + )] + let small_threshold = + (self.threshold_ratio * data_shards.saturating_mul(self.disk_block_bytes) as f64).ceil() as usize; + if small_threshold == 0 || small_threshold > small.object_limit { + return Err("S3 small-object threshold exceeds the shared writer limit".into()); + } + let mut client = ChunkClientConfig { + chunk_type: ChunkType::S3, + large_mirror_copies: self.large.mirror_copies, + read_buffer_size: self.disk_block_bytes, + ..ChunkClientConfig::default() + }; + if let Some(value) = self.large.max_chunk_size { + client.max_chunk_size = value; + } + if let Some(value) = self.large.memory_budget_bytes { + client.memory_budget = value; + } + if let Some(value) = self.large.prefetch_strips_per_chunk { + client.prefetch_strips_per_chunk = value; + } + if let Some(value) = self.large.prefetch_max_strips_per_batch { + client.large_prefetch_max_strips_per_batch = value; + } + if let Some(value) = self.large.parallel_strip_writes { + client.large_parallel_strip_writes = value; + } + if let Some(value) = self.large.held_buffers { + client.large_held_buffers = value; + } + if let Some(value) = self.large.chunk_preparation_depth { + client.chunk_preparation_depth = value; + } + client.validate().map_err(|error| error.to_string())?; + Ok(S3WritePolicies { + small, + large: LargeWritePolicy { + ec_scheme: EcScheme::new(self.ec_data, self.ec_code), + client: Arc::new(client), + }, + small_threshold, + }) + } +} + +impl S3StorageClients { + /// Connects metadata and chunk clients through one discovery client. + /// + /// # Errors + /// + /// Returns before readiness on configuration or discovery failure. + pub async fn connect( + management_seeds: Vec, + diskio_connections_per_endpoint: usize, + diskio_rpc_workers: u32, + small_write: SmallWritePolicy, + ) -> Result { + Self::connect_with_read_policy( + management_seeds, + diskio_connections_per_endpoint, + diskio_rpc_workers, + small_write, + ChunkReadPolicy::default(), + ) + .await + } + + /// # Errors + /// Returns an error when the management or `DiskIO` connection cannot be established. + pub async fn connect_with_read_policy( + management_seeds: Vec, + diskio_connections_per_endpoint: usize, + diskio_rpc_workers: u32, + mut small_write: SmallWritePolicy, + read_policy: ChunkReadPolicy, + ) -> Result { + small_write.chunk_type = ChunkType::S3; + let kv = Arc::new(CrowdbKvClient::new(KvConfig::new(management_seeds.clone()))); + let config = ChunkKvConfig::default(); + let catalog = Arc::new(Group0ChunkKvRangeCatalogSource::from_shared(Arc::clone(&kv))); + let transport = Arc::new(ChunkKvRpcTransport::new(config.max_owner_connections, 1, 2)); + let metadata = ChunkKvClient::new(config, catalog, transport) + .map_err(|error| StorageConnectError::ChunkKv(error.to_string()))?; + metadata + .refresh_catalog() + .await + .map_err(|error| StorageConnectError::ChunkKv(error.to_string()))?; + let chunks = ChunkIoClient::connect_with_kv_read_policy( + ChunkIoClientConfig { + management_seeds, + diskio_connections_per_endpoint, + diskio_rpc_workers, + small_write, + }, + Arc::clone(&kv), + read_policy, + ) + .await + .map_err(|error| StorageConnectError::ChunkIo(error.to_string()))?; + Ok(Self { + control: kv, + metadata: Arc::new(ChunkKvMetadataStore::new(Arc::new(metadata))), + chunks: Arc::new(chunks), + }) + } +} diff --git a/lib/crowdb-access-s3/src/streaming.rs b/lib/crowdb-access-s3/src/streaming.rs index adf54f514..e2242ceaa 100644 --- a/lib/crowdb-access-s3/src/streaming.rs +++ b/lib/crowdb-access-s3/src/streaming.rs @@ -3,17 +3,19 @@ use std::future::poll_fn; use std::pin::Pin; +use std::time::{SystemTime, UNIX_EPOCH}; use crowdb_protocol::chunkdb::rpc::Location; use crowdb_protocol::common::ChunkId; +use crowdb_protocol::frame::FrameMagic; -use crowdb_chunk_client::ChunkIoWriter; +use crowdb_chunk_client::{ChunkIoWriter, FramedWriteBuffer}; use crowdb_chunk_kv_client::ClientError; use hyper::body::{Body, Bytes}; use crate::integrity::SinglePartIntegrity; use crate::metadata::{ChunkKvMetadataStore, MetadataStoreError, ObjectRecord}; -use crate::native_buffer::NativeBodyReceiver; +use crate::native_buffer::{NativeBodyReceiver, NativeFramedOwner}; use crate::publication::{publish, PublicationError, PublicationRequest}; #[derive(Debug, thiserror::Error)] @@ -318,11 +320,12 @@ where let frame = poll_fn(|context| Pin::new(&mut *body).poll_frame(context)).await; let Some(frame) = frame else { if receiver.owner_handoff_active() { - if let Some(owner) = receiver + if let Some(mut owner) = receiver .finish_owner_when_ready() .await .map_err(|error| put_error(PutErrorCode::BodyRead, error))? { + prepare_native_owner(&mut owner)?; writer .on_framed_data(Box::new(owner)) .await @@ -340,7 +343,8 @@ where } integrity.update(&data); if receiver.owner_handoff_active() { - if let Some(owner) = receiver.take_ready_owner() { + if let Some(mut owner) = receiver.take_ready_owner() { + prepare_native_owner(&mut owner)?; writer .on_framed_data(Box::new(owner)) .await @@ -355,6 +359,17 @@ where } } +fn prepare_native_owner(owner: &mut NativeFramedOwner) -> Result<(), PutOutcome> { + let write_time_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |duration| { + u64::try_from(duration.as_millis()).unwrap_or(u64::MAX) + }); + owner + .prepare_frames(FrameMagic::RepoLargeV1, write_time_ms) + .map_err(|error| put_error(PutErrorCode::ChunkWrite, error)) +} + fn finish_integrity( integrity: SinglePartIntegrity, expected_content_md5: Option<&str>, @@ -474,9 +489,12 @@ pub async fn cleanup_after_definite_error( fn classify_publication_error(error: PublicationError) -> PutOutcome { match error { - PublicationError::Store(MetadataStoreError::Client( + timeout @ PublicationError::Store(MetadataStoreError::Client( ClientError::Deadline | ClientError::Transport(_), - )) => PutOutcome::Timeout, + )) => { + tracing::warn!(%timeout, "S3 object publication outcome is uncertain"); + PutOutcome::Timeout + } PublicationError::Metadata(error) => PutOutcome::Error { code: PutErrorCode::MetadataEncoding, message: error.to_string(), diff --git a/lib/crowdb-access-s3/tests/integrity_test.rs b/lib/crowdb-access-s3/tests/integrity_test.rs index c03c729e9..aa50715f5 100644 --- a/lib/crowdb-access-s3/tests/integrity_test.rs +++ b/lib/crowdb-access-s3/tests/integrity_test.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -use crowdb_access_s3::integrity::{IntegrityError, SinglePartIntegrity}; +use crowdb_access_s3::integrity::{validate_completed_digests, IntegrityError, SinglePartIntegrity}; use hyper::body::Bytes; #[test] @@ -70,3 +70,33 @@ fn signed_payload_sha256_is_checked_incrementally() { Err(IntegrityError::PayloadMismatch) ); } + +#[test] +fn completed_digest_worker_values_keep_s3_checksum_contract() { + let md5 = [ + 0x90, 0x01, 0x50, 0x98, 0x3c, 0xd2, 0x4f, 0xb0, 0xd6, 0x96, 0x3f, 0x7d, 0x28, 0xe1, 0x7f, 0x72, + ]; + let sha256 = [ + 0xba, 0x78, 0x16, 0xbf, 0x8f, 0x01, 0xcf, 0xea, 0x41, 0x41, 0x40, 0xde, 0x5d, 0xae, 0x22, 0x23, 0xb0, + 0x03, 0x61, 0xa3, 0x96, 0x17, 0x7a, 0x9c, 0xb4, 0x10, 0xff, 0x61, 0xf2, 0x00, 0x15, 0xad, + ]; + let expected_sha = "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"; + assert_eq!( + validate_completed_digests( + md5, + Some(sha256), + Some("kAFQmDzST7DWlj99KOF/cg=="), + Some(expected_sha) + ), + Ok(("900150983cd24fb0d6963f7d28e17f72".into(), md5.to_vec())) + ); + assert_eq!( + validate_completed_digests( + md5, + Some(sha256), + None, + Some("aa7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad") + ), + Err(IntegrityError::PayloadMismatch) + ); +} diff --git a/lib/crowdb-access-s3/tests/native_buffer_test.rs b/lib/crowdb-access-s3/tests/native_buffer_test.rs index cda1ffe77..c0d781a83 100644 --- a/lib/crowdb-access-s3/tests/native_buffer_test.rs +++ b/lib/crowdb-access-s3/tests/native_buffer_test.rs @@ -27,6 +27,7 @@ async fn native_frame_retains_and_releases_allocator_credit() { drop(bytes); assert_eq!(allocator.retained_bytes(), 0); + assert_eq!(allocator.metrics_snapshot().peak_retained_bytes, MAX_FRAME_BYTES); } #[tokio::test] diff --git a/lib/crowdb-access-s3/tests/storage_policy_test.rs b/lib/crowdb-access-s3/tests/storage_policy_test.rs new file mode 100644 index 000000000..cd22ef5fc --- /dev/null +++ b/lib/crowdb-access-s3/tests/storage_policy_test.rs @@ -0,0 +1,50 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_access_s3::storage::{S3LargeWriteSettings, S3WriteSettings}; +use crowdb_chunk_client::SmallWritePolicy; +use crowdb_protocol::chunkdb::rpc::ChunkType; + +#[test] +fn s3_policies_own_both_chunk_types_and_independent_limits() { + let policies = S3WriteSettings { + small: SmallWritePolicy::default(), + threshold_ratio: 0.9, + disk_block_bytes: 1024 * 1024, + ec_data: 4, + ec_code: 2, + large: S3LargeWriteSettings { + max_chunk_size: Some(64 * 1024 * 1024), + prefetch_strips_per_chunk: Some(3), + memory_budget_bytes: Some(32 * 1024 * 1024), + ..S3LargeWriteSettings::default() + }, + } + .policies() + .unwrap(); + + assert_eq!(policies.small.chunk_type, ChunkType::S3); + assert_eq!(policies.small.conversion_data_num, 4); + assert_eq!(policies.small_threshold, 3_774_874); + assert_eq!(policies.large.ec_scheme.data_num, 4); + assert_eq!(policies.large.client.chunk_type, ChunkType::S3); + assert_eq!(policies.large.client.max_chunk_size, 64 * 1024 * 1024); + assert_eq!(policies.large.client.prefetch_strips_per_chunk, 3); +} + +#[test] +fn s3_policy_rejects_invalid_large_capacity() { + let result = S3WriteSettings { + small: SmallWritePolicy::default(), + threshold_ratio: 0.9, + disk_block_bytes: 1024 * 1024, + ec_data: 8, + ec_code: 4, + large: S3LargeWriteSettings { + max_chunk_size: Some(0), + ..S3LargeWriteSettings::default() + }, + } + .policies(); + assert!(result.is_err()); +} diff --git a/lib/crowdb-chunk-client/src/chunk/chunk_prefetch.rs b/lib/crowdb-chunk-client/src/chunk/chunk_prefetch.rs index 1712121de..8acc0e7bd 100644 --- a/lib/crowdb-chunk-client/src/chunk/chunk_prefetch.rs +++ b/lib/crowdb-chunk-client/src/chunk/chunk_prefetch.rs @@ -67,7 +67,11 @@ impl ChunkPrefetch { async fn run(self, object_size: Option, tx: &mpsc::Sender>) -> Result<()> { let write_granularity_kb = (self.config.read_buffer_size / 1024) as u32; let unit_bytes = u64::from(write_granularity_kb) * 1024; - let strip_data_capacity = self.ec_scheme.data_num as u64 * unit_bytes; + let strip_data_capacity = if self.config.large_mirror_copies.is_some() { + unit_bytes + } else { + self.ec_scheme.data_num as u64 * unit_bytes + }; let strips_per_chunk = (self.config.max_chunk_size / strip_data_capacity).max(1) as u32; let chunk_data_capacity = strip_data_capacity * u64::from(strips_per_chunk); @@ -93,6 +97,7 @@ impl ChunkPrefetch { write_granularity_kb, self.chunk_type_byte, self.config.prefetch_strips_per_chunk, + self.config.large_mirror_copies, ) .await?; @@ -112,6 +117,7 @@ impl ChunkPrefetch { write_granularity_kb, self.chunk_type_byte, self.config.prefetch_strips_per_chunk, + self.config.large_mirror_copies, ) .await } @@ -124,17 +130,31 @@ pub(crate) async fn allocate_new_chunk( write_granularity_kb: u32, chunk_type_byte: u8, prefetch_strips_per_chunk: usize, + mirror_copies: Option, ) -> Result { let chunk_id = crowdb_protocol::generate_chunk_id(chunk_type_byte).to_proto(); let req = AllocateChunkRequest { chunk_id: Some(chunk_id), write_granularity: write_granularity_kb, strip_count: u32::try_from(prefetch_strips_per_chunk).unwrap_or(u32::MAX), - strip_type: StripType::Ec as i32, - data_num: ec_scheme.data_num as u32, - code_num: ec_scheme.code_num as u32, - copy_count: 0, - chunk_type: ChunkType::Repo as i32, + strip_type: if mirror_copies.is_some() { + StripType::Mirror as i32 + } else { + StripType::Ec as i32 + }, + data_num: if mirror_copies.is_some() { + 0 + } else { + ec_scheme.data_num as u32 + }, + code_num: if mirror_copies.is_some() { + 0 + } else { + ec_scheme.code_num as u32 + }, + copy_count: mirror_copies.unwrap_or(0), + chunk_type: ChunkType::try_from(i32::from(chunk_type_byte)) + .map_err(|()| IoError::Internal("invalid chunk type".into()))? as i32, writer_epoch: 0, writer_lease_ms: 0, owner_key: Vec::new(), diff --git a/lib/crowdb-chunk-client/src/chunk/chunk_writer.rs b/lib/crowdb-chunk-client/src/chunk/chunk_writer.rs index 6be1b69d1..22f5b3afd 100644 --- a/lib/crowdb-chunk-client/src/chunk/chunk_writer.rs +++ b/lib/crowdb-chunk-client/src/chunk/chunk_writer.rs @@ -18,6 +18,7 @@ use tokio::task::JoinHandle; use tracing::warn; use crate::chunk::ec_strip_writer::EcStripWriter; +use crate::chunk::mirror_strip_writer::{MirrorIoConcurrency, MirrorStripWriter}; use crate::chunk::segment_writer::{FailedSegmentWrite, SegmentRepair}; use crate::chunk::strip::{StripResult, StripWriter}; use crate::config::ChunkClientConfig; @@ -29,10 +30,15 @@ use crate::traits::ChunkAllocator; use crate::{IoError, Result}; use crowdb_common::ec::EcScheme; use crowdb_protocol::chunkdb::rpc::{ - AppendChunkRequest, Chunk, DeleteChunkRequest, Location as ProtoLocation, SealChunkRequest, StripType, + Chunk, DeleteChunkRequest, Location as ProtoLocation, SealChunkRequest, Strip, }; use crowdb_protocol::common::ChunkId; +mod prefetch; +mod recovery; + +use prefetch::{append_strips, compute_strips_remaining}; + /// Chunk wrapper + write ability. Owns `Arc`; the strip-level /// drive loop is in `push` (auto-rotates strips). Collects write /// completion handles from each `finish_strip` and joins them at @@ -51,17 +57,65 @@ pub struct ChunkWriter { pub(crate) strips_remaining: Option, pub(crate) current_strip: Option, pub(crate) completion_handles: VecDeque>>>, + mirror_completions: VecDeque, pub(crate) prefetch_handle: Option>, pub(crate) prefetch_rx: Option>>, + prefetch_plan: Option, + prefetch_trigger: Option>, + prefetch_trigger_index: Option, pub(crate) preparation_stalls: u64, pub(crate) preparation_stall_time: Duration, + pub(crate) strip_write_successes: u64, + pub(crate) strip_write_success_time: Duration, + pub(crate) strip_write_success_max: Duration, + pub(crate) mirror_uncommitted_peak: u64, + committed_mirror_strips: u32, + replaying_mirror: bool, + framed_input: bool, pub(crate) ec_encode_time: Duration, pub(crate) completion_wait_time: Duration, pub(crate) failed_disks: Arc, pub(crate) repair_metrics: Arc, + mirror_io_concurrency: Arc, +} + +struct MirrorCompletion { + result: StripResult, + elapsed: Duration, + failures: Vec, + // Later completions retain their data until every preceding strip commits. + buffer: Bytes, +} + +enum MirrorPending { + Running { + handle: JoinHandle>, + buffer: Bytes, + }, + Completed(MirrorCompletion), + Failed(Bytes), +} + +impl MirrorPending { + fn is_finished(&self) -> bool { + match self { + Self::Running { handle, .. } => handle.is_finished(), + Self::Completed(_) | Self::Failed(_) => true, + } + } +} + +#[derive(Clone, Copy)] +pub(crate) struct StripPrefetchPlan { + pub total_strips: u32, + pub batch_max: u32, } impl ChunkWriter { + pub(crate) fn set_framed_input(&mut self) { + self.framed_input = true; + } + /// Construct a new chunk writer (no chunk open yet). pub fn new( allocator: Arc, @@ -99,14 +153,26 @@ impl ChunkWriter { strips_remaining: None, current_strip: None, completion_handles: VecDeque::new(), + mirror_completions: VecDeque::new(), prefetch_handle: None, prefetch_rx: None, + prefetch_plan: None, + prefetch_trigger: None, + prefetch_trigger_index: None, preparation_stalls: 0, preparation_stall_time: Duration::ZERO, + strip_write_successes: 0, + strip_write_success_time: Duration::ZERO, + strip_write_success_max: Duration::ZERO, + mirror_uncommitted_peak: 0, + committed_mirror_strips: 0, + replaying_mirror: false, + framed_input: false, ec_encode_time: Duration::ZERO, completion_wait_time: Duration::ZERO, failed_disks, repair_metrics, + mirror_io_concurrency: Arc::default(), } } @@ -119,6 +185,15 @@ impl ChunkWriter { /// allocated; unknown-size objects pre-append up to /// `strips_per_chunk`. pub fn open(&mut self, chunk: Chunk, object_size: Option) -> Result<()> { + self.open_with_prefetch_plan(chunk, object_size, None) + } + + pub(crate) fn open_with_prefetch_plan( + &mut self, + chunk: Chunk, + object_size: Option, + plan: Option, + ) -> Result<()> { if chunk.id.is_none() { return Err(IoError::AllocationFailed("open: chunk missing id".into())); } @@ -126,14 +201,19 @@ impl ChunkWriter { return Err(IoError::AllocationFailed("open: chunk has no strips".into())); } self.object_size = object_size; - self.strips_remaining = - compute_strips_remaining(object_size, chunk.strips.len(), &self.ec_scheme, &self.config); + self.strips_remaining = plan.map_or_else( + || compute_strips_remaining(object_size, &chunk), + |plan| Some((plan.total_strips as usize).saturating_sub(chunk.strips.len())), + ); + self.prefetch_plan = plan; + self.prefetch_trigger_index = None; let chunk = Arc::new(chunk); - let strip = EcStripWriter::new(Arc::clone(&chunk), 0, self.disk_writer.clone(), self.ec_scheme); + let strip = self.make_strip_writer(Arc::clone(&chunk), 0)?; self.chunk = Some(chunk); self.write_cursor = 0; self.bytes_in_chunk = 0; - self.current_strip = Some(StripWriter::Ec(strip)); + self.committed_mirror_strips = 0; + self.current_strip = Some(strip); // Start the internal strip-prefetch task. self.start_strip_prefetch(); Ok(()) @@ -153,15 +233,10 @@ impl ChunkWriter { } let next_index = self.write_cursor + 1; let chunk = Arc::new(chunk); - let strip = EcStripWriter::new( - Arc::clone(&chunk), - next_index, - self.disk_writer.clone(), - self.ec_scheme, - ); + let strip = self.make_strip_writer(Arc::clone(&chunk), next_index)?; self.chunk = Some(chunk); self.write_cursor = next_index; - self.current_strip = Some(StripWriter::Ec(strip)); + self.current_strip = Some(strip); Ok(()) } @@ -172,22 +247,36 @@ impl ChunkWriter { /// block to the new strip. Returns `Pause` if the chunk is full /// after finishing the current strip — the block is NOT pushed /// (caller rotates chunks, then re-pushes). - pub async fn push(&mut self, buffer: Bytes) -> Result { - if self.current_strip.is_none() { + pub async fn push(&mut self, mut buffer: Bytes) -> Result { + if self.current_strip.is_none() && self.chunk.is_none() { return Err(IoError::Internal("push with no open strip".into())); } // A public frame can span several EC data blocks. Feed each strip only // the bytes it owns; never leave an overflow tail in a completed // strip, which would otherwise be encoded as an extra data shard. let mut offset = 0usize; + let mut input_chunk_id = self.current_chunk_id(); while offset < buffer.len() { if self.is_strip_full() { - self.finish_strip().await?; + if self.current_strip.is_some() { + self.finish_strip().await?; + } if self.is_full() { return Ok(FeedStatus::Pause); } self.open_next_strip().await?; } + if self.current_chunk_id() != input_chunk_id { + let old_id = input_chunk_id + .ok_or_else(|| IoError::Internal("mirror replay lost source chunk ID".into()))?; + let new_id = self + .current_chunk_id() + .ok_or_else(|| IoError::Internal("mirror replay lost destination chunk ID".into()))?; + if self.framed_input { + buffer = recovery::rewrite_frames(&buffer, old_id, new_id)?; + } + input_chunk_id = Some(new_id); + } let strip = self .current_strip .as_mut() @@ -198,12 +287,157 @@ impl ChunkWriter { continue; } let end = offset.saturating_add(remaining).min(buffer.len()); + if matches!(strip, StripWriter::Mirror(_)) + && strip.accepted_bytes() == 0 + && end - offset == remaining + { + self.ensure_mirror_capacity().await?; + if self.current_chunk_id() != input_chunk_id { + let old_id = input_chunk_id + .ok_or_else(|| IoError::Internal("mirror replay lost source chunk ID".into()))?; + let new_id = self + .current_chunk_id() + .ok_or_else(|| IoError::Internal("mirror replay lost destination chunk ID".into()))?; + if self.framed_input { + buffer = recovery::rewrite_frames(&buffer, old_id, new_id)?; + } + input_chunk_id = Some(new_id); + continue; + } + let mut strip = self + .current_strip + .take() + .ok_or_else(|| IoError::Internal("mirror strip vanished before dispatch".into()))?; + let bytes = buffer.slice(offset..end); + let retained = bytes.clone(); + let handle = tokio::spawn(async move { + let started = Instant::now(); + let StripWriter::Mirror(mirror) = &mut strip else { + return Err(IoError::Internal("mirror dispatch changed strip type".into())); + }; + let retained = bytes.clone(); + let (result, failures) = mirror.write_full_repairable(bytes).await?; + Ok(MirrorCompletion { + result, + elapsed: started.elapsed(), + failures, + buffer: retained, + }) + }); + self.mirror_completions.push_back(MirrorPending::Running { + handle, + buffer: retained, + }); + self.mirror_uncommitted_peak = self + .mirror_uncommitted_peak + .max(u64::try_from(self.mirror_completions.len()).unwrap_or(u64::MAX)); + self.bytes_in_chunk += remaining as u64; + offset = end; + continue; + } + let started = Instant::now(); strip.push(buffer.slice(offset..end)).await?; + let elapsed = started.elapsed(); + self.strip_write_successes += 1; + self.strip_write_success_time += elapsed; + self.strip_write_success_max = self.strip_write_success_max.max(elapsed); offset = end; } Ok(FeedStatus::Continue) } + pub(crate) fn mirror_write_capacity(&self) -> bool { + self.mirror_completions.len() < self.config.large_parallel_strip_writes + } + + pub(crate) async fn ensure_mirror_capacity(&mut self) -> Result<()> { + self.commit_ready_mirrors().await?; + if !self.mirror_write_capacity() { + self.commit_oldest_mirror().await?; + self.commit_ready_mirrors().await?; + } + Ok(()) + } + + async fn commit_oldest_mirror(&mut self) -> Result<()> { + // Keep the handle in the queue across cancellation of this await. + let pending = self + .mirror_completions + .front_mut() + .ok_or_else(|| IoError::Internal("missing mirror completion".into()))?; + if let MirrorPending::Running { handle, buffer } = pending { + match handle.await { + Ok(Ok(completion)) => *pending = MirrorPending::Completed(completion), + Ok(Err(error)) => { + *pending = MirrorPending::Failed(buffer.clone()); + return Err(error); + } + Err(error) => { + *pending = MirrorPending::Failed(buffer.clone()); + return Err(IoError::Internal(format!("mirror write task panicked: {error}"))); + } + } + } + let MirrorPending::Completed(completion) = self + .mirror_completions + .front() + .ok_or_else(|| IoError::Internal("missing mirror completion".into()))? + else { + return Err(IoError::Internal( + "mirror write task failed before completion".into(), + )); + }; + if !completion.result.completion_handles.is_empty() { + return Err(IoError::Internal( + "mirror strip returned unexpected completion handles".into(), + )); + } + // Retain this strip ahead of later completions if replacement fails. + let failures = completion.failures.clone(); + let elapsed = completion.elapsed; + if let Err(error) = self.repair_mirror_failures(failures).await { + if matches!(error, IoError::ReplicaRepairExhausted(_)) && !self.replaying_mirror { + self.rotate_failed_mirror(None).await?; + return Ok(()); + } + return Err(error); + } + self.mirror_completions.pop_front(); + self.committed_mirror_strips += 1; + self.strip_write_successes += 1; + self.strip_write_success_time += elapsed; + self.strip_write_success_max = self.strip_write_success_max.max(elapsed); + Ok(()) + } + + async fn commit_ready_mirrors(&mut self) -> Result<()> { + while self + .mirror_completions + .front() + .is_some_and(MirrorPending::is_finished) + { + self.commit_oldest_mirror().await?; + } + Ok(()) + } + + async fn repair_mirror_failures(&mut self, failures: Vec) -> Result<()> { + let chunk_id = self + .current_chunk_id() + .ok_or_else(|| IoError::Internal("mirror repair has no active chunk".into()))?; + for failure in failures { + let repair = SegmentRepair { + allocator: &self.allocator, + disk_writer: &self.disk_writer, + failed_disks: &self.failed_disks, + metrics: &self.repair_metrics, + attempts: self.config.large_write_repair_attempts, + }; + self.chunk = Some(Arc::new(repair.repair(chunk_id, failure).await?)); + } + Ok(()) + } + /// Open the next strip on the current chunk. First drains the /// prefetch channel (non-blocking) to pick up any pre-appended /// chunks. If the next strip is in `chunk.strips`, opens it @@ -230,14 +464,10 @@ impl ChunkWriter { .chunk .as_ref() .ok_or_else(|| IoError::Internal("open_next_strip with no chunk".into()))?; - let strip = EcStripWriter::new( - Arc::clone(chunk), - next_index, - self.disk_writer.clone(), - self.ec_scheme, - ); + let strip = self.make_strip_writer(Arc::clone(chunk), next_index)?; self.write_cursor = next_index; - self.current_strip = Some(StripWriter::Ec(strip)); + self.current_strip = Some(strip); + self.maybe_trigger_prefetch(); return Ok(()); } // Next strip not ready — wait for the prefetch task to @@ -259,7 +489,7 @@ impl ChunkWriter { self.preparation_stall_time += started.elapsed(); match result { Some(Ok(new_chunk)) => { - self.chunk = Some(Arc::new(new_chunk)); + self.accept_prefetched_chunk(new_chunk); // Loop back: check if the next strip is now available. } Some(Err(e)) => return Err(e), @@ -277,25 +507,48 @@ impl ChunkWriter { /// Drain the prefetch channel (non-blocking) and Arc-swap to the /// latest cumulative `Chunk` from the prefetch task. fn drain_prefetch(&mut self) { - if let Some(rx) = self.prefetch_rx.as_mut() { - while let Ok(result) = rx.try_recv() { - match result { - Ok(new_chunk) => { - self.chunk = Some(Arc::new(new_chunk)); - } - Err(e) => { - warn!("strip prefetch error: {e}"); - break; - } + loop { + let result = self.prefetch_rx.as_mut().and_then(|rx| rx.try_recv().ok()); + match result { + Some(Ok(new_chunk)) => self.accept_prefetched_chunk(new_chunk), + Some(Err(error)) => { + warn!("strip prefetch error: {error}"); + break; } + None => break, } } } + fn accept_prefetched_chunk(&mut self, chunk: Chunk) { + if self.prefetch_plan.is_some() { + let previous = self.chunk.as_ref().map_or(0, |current| current.strips.len()); + let batch = chunk.strips.len().saturating_sub(previous); + let half = batch.div_ceil(2); + self.prefetch_trigger_index = + Some(u32::try_from(chunk.strips.len().saturating_sub(half)).unwrap_or(u32::MAX)); + } + self.chunk = Some(Arc::new(chunk)); + } + + fn maybe_trigger_prefetch(&mut self) { + if self + .prefetch_trigger_index + .is_some_and(|index| self.write_cursor >= index) + { + if let Some(trigger) = &self.prefetch_trigger { + let _ = trigger.try_send(()); + } + self.prefetch_trigger_index = None; + } + } + /// Stop the strip-prefetch task: drop the receiver (task's /// `tx.send` fails → task exits) + abort the handle. fn stop_prefetch(&mut self) { self.prefetch_rx.take(); + self.prefetch_trigger.take(); + self.prefetch_trigger_index = None; if let Some(handle) = self.prefetch_handle.take() { handle.abort(); } @@ -311,14 +564,28 @@ impl ChunkWriter { let Some(mut chunk) = self.chunk.as_deref().cloned() else { return; }; - let (tx, rx) = mpsc::channel::>(self.config.prefetch_strips_per_chunk); + let plan = self.prefetch_plan; + let capacity = if plan.is_some() { + 1 + } else { + self.config.prefetch_strips_per_chunk + }; + let (tx, rx) = mpsc::channel::>(capacity); self.prefetch_rx = Some(rx); + let (trigger_tx, mut trigger_rx) = mpsc::channel::<()>(1); + self.prefetch_trigger = plan.map(|_| trigger_tx); let allocator = Arc::clone(&self.allocator); let ec_scheme = self.ec_scheme; let config = Arc::clone(&self.config); let max_chunk_size = config.max_chunk_size; let unit_bytes = u64::from((config.read_buffer_size / 1024) as u32) * 1024; - let strip_data_bytes = ec_scheme.data_num as u64 * unit_bytes; + let strip_data_bytes = chunk + .strips + .first() + .map_or(ec_scheme.data_num as u64 * unit_bytes, |strip| { + u64::from(strip.capacity) * 1024 + }) + .max(1); let strips_per_chunk = (max_chunk_size / strip_data_bytes) as u32; let mut strips_remaining = self.strips_remaining; let mut next_strip_index = chunk.strips.len() as u32; @@ -344,15 +611,18 @@ impl ChunkWriter { // For larger objects (more strips to allocate), batch 2 // strips per append to reduce RPC count. For smaller objects, // allocate 1 at a time so the first strip is ready sooner. - let batch = match strips_remaining.as_ref() { - Some(total) if *total > 4 => 2u32, - _ => 1u32, - }; + let batch = plan.map_or_else( + || match strips_remaining.as_ref() { + Some(total) if *total > 4 => 2u32, + _ => 1u32, + }, + |plan| plan.batch_max, + ); let strip_count = batch.min(runway).min(remaining); if strip_count == 0 { break; } - let result = append_strips(&*allocator, chunk, ec_scheme, strip_count).await; + let result = append_strips(&*allocator, chunk, strip_count).await; match result { Ok(new_chunk) => { chunk = new_chunk.clone(); @@ -361,6 +631,9 @@ impl ChunkWriter { if let Some(remaining) = strips_remaining.as_mut() { *remaining = remaining.saturating_sub(strip_count as usize); } + if plan.is_some() && trigger_rx.recv().await.is_none() { + break; + } } Err(e) => { permit.send(Err(e)); @@ -376,13 +649,48 @@ impl ChunkWriter { /// write completion handles (joined at `seal` time, not here). pub async fn finish_strip(&mut self) -> Result { self.await_parity_capacity().await?; + if matches!(self.current_strip, Some(StripWriter::Mirror(_))) { + // A partial mirror strip completes inline. Commit every earlier + // detached strip first so its repair and release stay ordered. + while !self.mirror_completions.is_empty() { + self.commit_oldest_mirror().await?; + self.commit_ready_mirrors().await?; + } + } let mut strip = self .current_strip .take() .ok_or_else(|| IoError::Internal("finish_strip with no open strip".into()))?; - let mut strip_result = strip.finish().await?; + let partial_replay = match &strip { + StripWriter::Mirror(mirror) => mirror.replay_views(), + StripWriter::Ec(_) => Vec::new(), + }; + let (mut strip_result, mirror_failures) = match &mut strip { + StripWriter::Mirror(mirror) => { + let result = mirror.finish().await?; + (result, mirror.take_failures()?) + } + StripWriter::Ec(_) => (strip.finish().await?, Vec::new()), + }; self.ec_encode_time += strip_result.ec_encode_time; self.bytes_in_chunk += strip_result.bytes_written; + if let Err(error) = self.repair_mirror_failures(mirror_failures).await { + if matches!(error, IoError::ReplicaRepairExhausted(_)) && !self.replaying_mirror { + self.rotate_failed_mirror(Some(partial_replay)).await?; + return Box::pin(self.finish_strip()).await; + } + return Err(error); + } + if matches!(strip, StripWriter::Mirror(_)) + && strip_result.bytes_written + == self + .chunk + .as_ref() + .and_then(|chunk| chunk.strips.get(strip_result.strip_index_in_chunk as usize)) + .map_or(0, |strip| u64::from(strip.capacity) * 1024) + { + self.committed_mirror_strips += 1; + } // One queue entry represents one completed strip. This keeps // `parity_depth` expressed in strips instead of accidentally counting // every data and parity shard as an independent depth unit. @@ -436,6 +744,10 @@ impl ChunkWriter { (self.preparation_stalls, self.preparation_stall_time) } + pub(crate) fn active_mirror_write_peak(&self) -> u64 { + self.mirror_io_concurrency.peak() + } + /// Append a new strip to the current chunk via `append_chunk` RPC. /// Returns the full cumulative `Chunk` (with the new strip /// appended). Used by the internal strip prefetch + the inline @@ -446,7 +758,7 @@ impl ChunkWriter { .as_deref() .cloned() .ok_or_else(|| IoError::Internal("append_strip with no open chunk".into()))?; - append_strips(&*self.allocator, chunk, self.ec_scheme, 1).await + append_strips(&*self.allocator, chunk, 1).await } /// Seal the chunk: finish the current strip (if open with data), @@ -465,6 +777,10 @@ impl ChunkWriter { self.finish_strip().await?; } } + while !self.mirror_completions.is_empty() { + self.commit_oldest_mirror().await?; + self.commit_ready_mirrors().await?; + } let chunk_id = self.current_chunk_id(); let bytes_in_chunk = self.bytes_in_chunk; @@ -549,6 +865,11 @@ impl ChunkWriter { for handle in self.completion_handles.drain(..) { let _ = handle.await; } + for pending in self.mirror_completions.drain(..) { + if let MirrorPending::Running { handle, .. } = pending { + let _ = handle.await; + } + } // Delete the chunk if it was opened and has any data — either // finished strips (bytes_in_chunk > 0), an in-progress strip // (had_strip), or prior finished strips (write_cursor > 0). @@ -592,6 +913,33 @@ impl ChunkWriter { self.chunk.as_ref().and_then(|c| c.id) } + fn make_strip_writer(&self, chunk: Arc, index: u32) -> Result { + let strip = chunk + .strips + .get(index as usize) + .ok_or_else(|| IoError::AllocationFailed("strip index is absent".into()))?; + match &strip.strip { + Some(Strip::MirrorStrip(_)) => { + Ok(StripWriter::Mirror(MirrorStripWriter::new_with_io_concurrency( + chunk, + index, + Arc::clone(&self.disk_writer), + Arc::clone(&self.mirror_io_concurrency), + ))) + } + Some(Strip::EcStrip(ec)) if ec.data_num > 0 && ec.code_num > 0 => { + let scheme = EcScheme::new(ec.data_num as usize, ec.code_num as usize); + Ok(StripWriter::Ec(EcStripWriter::new( + chunk, + index, + Arc::clone(&self.disk_writer), + scheme, + ))) + } + _ => Err(IoError::AllocationFailed("unsupported strip layout".into())), + } + } + /// Strips opened in the current chunk so far (= write_cursor + 1 /// when a chunk is open). pub fn strips_in_chunk(&self) -> u32 { @@ -602,86 +950,3 @@ impl ChunkWriter { } } } - -/// Compute the number of strips not yet allocated for a known-size -/// object. Returns `None` for unknown-size objects. Used by the -/// internal strip prefetch task for planning. -fn compute_strips_remaining( - object_size: Option, - allocated_strips: usize, - ec_scheme: &EcScheme, - config: &ChunkClientConfig, -) -> Option { - let total = object_size?; - let unit_bytes = u64::from((config.read_buffer_size / 1024) as u32) * 1024; - let strip_data_capacity = ec_scheme.data_num as u64 * unit_bytes; - let total_strips = total.div_ceil(strip_data_capacity) as usize; - Some(total_strips.saturating_sub(allocated_strips)) -} - -/// Append one strip and merge the incremental response into the local chunk. -/// A stale revision response carries the current full chunk; retry once with -/// that revision so concurrent metadata changes do not duplicate an append. -async fn append_strips( - chunkdb: &dyn ChunkAllocator, - mut chunk: Chunk, - ec_scheme: EcScheme, - strip_count: u32, -) -> Result { - let chunk_id = chunk - .id - .ok_or_else(|| IoError::AllocationFailed("append_chunk: chunk missing id".into()))?; - let unit_count = chunk - .strips - .first() - .and_then(|strip| match strip.strip.as_ref() { - Some(crowdb_protocol::chunkdb::rpc::Strip::EcStrip(ec)) => ec.segments.first(), - Some(crowdb_protocol::chunkdb::rpc::Strip::MirrorStrip(mirror)) => mirror.segments.first(), - None => None, - }) - .map(|segment| segment.unit_count) - .filter(|count| *count > 0) - .ok_or_else(|| { - IoError::AllocationFailed("append_chunk: existing strip has no segment geometry".into()) - })?; - for attempt in 0..2 { - let resp = chunkdb - .append_chunk(AppendChunkRequest { - chunk_id: Some(chunk_id), - modify_ts: chunk.modify_ts, - strip_size: unit_count, - strip_count, - strip_type: StripType::Ec as i32, - data_num: ec_scheme.data_num as u32, - code_num: ec_scheme.code_num as u32, - copy_count: 0, - }) - .await?; - if let Some(current) = resp.chunk { - if current.id != Some(chunk_id) { - return Err(IoError::AllocationFailed( - "append_chunk refresh returned a different chunk".into(), - )); - } - chunk = current; - if attempt == 0 { - continue; - } - return Err(IoError::AllocationFailed( - "append_chunk revision changed twice".into(), - )); - } - if resp.strips.is_empty() { - return Err(IoError::AllocationFailed( - "append_chunk response missing appended strips".into(), - )); - } - chunk.modify_ts = resp.modify_ts; - chunk.capacity = chunk - .capacity - .saturating_add(resp.strips.iter().map(|strip| strip.capacity).sum::()); - chunk.strips.extend(resp.strips); - return Ok(chunk); - } - unreachable!() -} diff --git a/lib/crowdb-chunk-client/src/chunk/chunk_writer/prefetch.rs b/lib/crowdb-chunk-client/src/chunk/chunk_writer/prefetch.rs new file mode 100644 index 000000000..ef3ca792c --- /dev/null +++ b/lib/crowdb-chunk-client/src/chunk/chunk_writer/prefetch.rs @@ -0,0 +1,98 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Known-size strip planning and revision-aware metadata append. + +use crowdb_protocol::chunkdb::rpc::{AppendChunkRequest, Chunk, Strip, StripType}; + +use crate::traits::ChunkAllocator; +use crate::{IoError, Result}; + +/// Number of strips not yet allocated for a known-size object. +pub(super) fn compute_strips_remaining(object_size: Option, chunk: &Chunk) -> Option { + let total = object_size?; + let strip_data_capacity = u64::from(chunk.strips.first()?.capacity) * 1024; + let total_strips = total.div_ceil(strip_data_capacity.max(1)) as usize; + Some(total_strips.saturating_sub(chunk.strips.len())) +} + +/// Append strips and merge the incremental response into the local chunk. +/// A stale revision carries the current full chunk; retry once with that +/// revision so concurrent metadata changes do not duplicate an append. +pub(super) async fn append_strips( + chunkdb: &dyn ChunkAllocator, + mut chunk: Chunk, + strip_count: u32, +) -> Result { + let chunk_id = chunk + .id + .ok_or_else(|| IoError::AllocationFailed("append_chunk: chunk missing id".into()))?; + let unit_count = chunk + .strips + .first() + .and_then(|strip| match strip.strip.as_ref() { + Some(Strip::EcStrip(ec)) => ec.segments.first(), + Some(Strip::MirrorStrip(mirror)) => mirror.segments.first(), + None => None, + }) + .map(|segment| segment.unit_count) + .filter(|count| *count > 0) + .ok_or_else(|| { + IoError::AllocationFailed("append_chunk: existing strip has no segment geometry".into()) + })?; + let (strip_type, data_num, code_num, copy_count) = + match chunk.strips.last().and_then(|strip| strip.strip.as_ref()) { + Some(Strip::MirrorStrip(mirror)) => ( + StripType::Mirror as i32, + 0, + 0, + u32::try_from(mirror.segments.len()).unwrap_or(u32::MAX), + ), + Some(Strip::EcStrip(ec)) => (StripType::Ec as i32, ec.data_num, ec.code_num, 0), + None => { + return Err(IoError::AllocationFailed( + "append_chunk: missing strip layout".into(), + )) + } + }; + for attempt in 0..2 { + let resp = chunkdb + .append_chunk(AppendChunkRequest { + chunk_id: Some(chunk_id), + modify_ts: chunk.modify_ts, + strip_size: unit_count, + strip_count, + strip_type, + data_num, + code_num, + copy_count, + }) + .await?; + if let Some(current) = resp.chunk { + if current.id != Some(chunk_id) { + return Err(IoError::AllocationFailed( + "append_chunk refresh returned a different chunk".into(), + )); + } + chunk = current; + if attempt == 0 { + continue; + } + return Err(IoError::AllocationFailed( + "append_chunk revision changed twice".into(), + )); + } + if resp.strips.is_empty() { + return Err(IoError::AllocationFailed( + "append_chunk response missing appended strips".into(), + )); + } + chunk.modify_ts = resp.modify_ts; + chunk.capacity = chunk + .capacity + .saturating_add(resp.strips.iter().map(|strip| strip.capacity).sum::()); + chunk.strips.extend(resp.strips); + return Ok(chunk); + } + unreachable!() +} diff --git a/lib/crowdb-chunk-client/src/chunk/chunk_writer/recovery.rs b/lib/crowdb-chunk-client/src/chunk/chunk_writer/recovery.rs new file mode 100644 index 000000000..01d994fe9 --- /dev/null +++ b/lib/crowdb-chunk-client/src/chunk/chunk_writer/recovery.rs @@ -0,0 +1,311 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Bounded mirror-chunk replay after in-place segment replacement is exhausted. + +use std::collections::VecDeque; +use std::sync::Arc; +use std::time::Instant; + +use bytes::{Bytes, BytesMut}; +use crowdb_protocol::chunkdb::rpc::{Chunk, DeleteChunkRequest, QueryChunkRequest, Strip}; +use crowdb_protocol::common::ChunkId; +use crowdb_protocol::frame::{parse_frame, set_frame_chunk_id, FrameError, FRAME_FOOTER_BYTES}; + +use super::{ChunkWriter, MirrorPending}; +use crate::chunk::chunk_prefetch::allocate_new_chunk; +use crate::chunk::strip::StripWriter; +use crate::io::FeedStatus; +use crate::{IoError, Result}; + +impl ChunkWriter { + /// Replace the whole active mirror chunk after all ordered writes have + /// stopped. The committed prefix is read one strip at a time; only the + /// bounded set of uncommitted writes remains in memory. + pub(super) async fn rotate_failed_mirror(&mut self, partial: Option>) -> Result<()> { + let started = Instant::now(); + self.repair_metrics.chunk_rotations.inc(); + let replayed_bytes = self + .bytes_in_chunk + .saturating_add(self.current_strip.as_ref().map_or(0, StripWriter::accepted_bytes)); + let old_id = self + .current_chunk_id() + .ok_or_else(|| IoError::Internal("mirror rotation has no chunk".into()))?; + let (old_chunk, pending) = self.collect_mirror_replay(old_id, partial).await?; + let copy_count = old_chunk + .strips + .first() + .and_then(|strip| match &strip.strip { + Some(Strip::MirrorStrip(mirror)) => u32::try_from(mirror.segments.len()).ok(), + _ => None, + }) + .filter(|copies| *copies > 0) + .ok_or_else(|| IoError::Internal("mirror rotation has no copy layout".into()))?; + + let attempts = self.config.large_write_repair_attempts.max(1); + let mut last_error = IoError::ReplicaRepairExhausted("mirror chunk rotation exhausted".into()); + for _ in 0..attempts { + let candidate = allocate_new_chunk( + &*self.allocator, + self.ec_scheme, + u32::try_from(self.config.read_buffer_size / 1024).unwrap_or(u32::MAX), + self.config.chunk_type as u8, + self.config.prefetch_strips_per_chunk, + Some(copy_count), + ) + .await?; + let new_id = candidate + .id + .ok_or_else(|| IoError::AllocationFailed("replacement chunk has no ID".into()))?; + let mut replacement = ChunkWriter::new_with_repair( + Arc::clone(&self.allocator), + Arc::clone(&self.disk_writer), + self.ec_scheme, + Arc::clone(&self.config), + Arc::clone(&self.failed_disks), + Arc::clone(&self.repair_metrics), + ); + replacement.replaying_mirror = true; + replacement.framed_input = self.framed_input; + replacement.mirror_io_concurrency = Arc::clone(&self.mirror_io_concurrency); + if let Err(error) = replacement.open(candidate, self.object_size) { + let _ = self + .allocator + .delete_chunk(DeleteChunkRequest { + chunk_id: Some(new_id), + }) + .await; + return Err(error); + } + let replay = replay_chunk( + &old_chunk, + self.committed_mirror_strips, + &pending, + old_id, + new_id, + self, + &mut replacement, + ) + .await; + match replay { + Ok(()) => { + replacement.replaying_mirror = false; + replacement.inherit_write_metrics(self); + // The new physical bytes are durable. The old unsealed + // chunk can no longer be published by this writer. + if let Err(error) = self + .allocator + .delete_chunk(DeleteChunkRequest { + chunk_id: Some(old_id), + }) + .await + { + tracing::warn!(%error, "retired mirror chunk deletion failed"); + } + self.repair_metrics.rotated_chunks.inc(); + self.repair_metrics.replayed_bytes.inc_by(replayed_bytes); + self.repair_metrics + .rotation_ns + .inc_by(u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)); + *self = replacement; + return Ok(()); + } + Err(error) => { + last_error = error; + let _ = replacement.abort().await; + } + } + } + self.repair_metrics + .rotation_ns + .inc_by(u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)); + Err(IoError::ReplicaRepairExhausted(format!( + "mirror chunk rotation exhausted: {last_error}" + ))) + } + + async fn collect_mirror_replay( + &mut self, + old_id: ChunkId, + partial: Option>, + ) -> Result<(Chunk, VecDeque)> { + let current_bytes = self.current_strip.as_ref().map_or(0, StripWriter::accepted_bytes); + let expected_bytes = self.bytes_in_chunk.saturating_add(current_bytes); + self.prefetch_rx.take(); + self.prefetch_trigger.take(); + self.prefetch_trigger_index = None; + if let Some(handle) = self.prefetch_handle.take() { + handle.abort(); + let _ = handle.await; + } + + let old_chunk = self + .allocator + .query_chunk(QueryChunkRequest { + chunk_id: Some(old_id), + }) + .await? + .chunk + .ok_or_else(|| IoError::ChunkNotFound(format!("{old_id:?}")))?; + let mut pending = VecDeque::new(); + for completion in self.mirror_completions.drain(..) { + match completion { + MirrorPending::Running { handle, buffer } => { + // Submitted DiskIO cannot be cancelled. The buffer remains + // owned until the task exits, even when an earlier write + // already failed. + let _ = handle.await; + pending.push_back(buffer); + } + MirrorPending::Completed(completion) => pending.push_back(completion.buffer), + MirrorPending::Failed(buffer) => pending.push_back(buffer), + } + } + if let Some(views) = partial { + pending.extend(views); + } else if let Some(StripWriter::Mirror(mirror)) = &self.current_strip { + pending.extend(mirror.replay_views()); + } + self.current_strip.take(); + + let committed_bytes = old_chunk + .strips + .iter() + .take(self.committed_mirror_strips as usize) + .try_fold(0_u64, |sum, strip| { + sum.checked_add(u64::from(strip.capacity) * 1024) + .ok_or_else(|| IoError::WriteFailed("mirror replay length overflow".into())) + })?; + let pending_bytes = pending.iter().try_fold(0_u64, |sum, bytes| { + sum.checked_add(bytes.len() as u64) + .ok_or_else(|| IoError::WriteFailed("mirror replay length overflow".into())) + })?; + if committed_bytes.saturating_add(pending_bytes) != expected_bytes { + return Err(IoError::Internal(format!( + "mirror replay lost bytes: committed={committed_bytes} pending={pending_bytes} expected={expected_bytes}" + ))); + } + Ok((old_chunk, pending)) + } + + fn inherit_write_metrics(&mut self, previous: &Self) { + self.preparation_stalls += previous.preparation_stalls; + self.preparation_stall_time += previous.preparation_stall_time; + self.strip_write_successes += previous.strip_write_successes; + self.strip_write_success_time += previous.strip_write_success_time; + self.strip_write_success_max = self.strip_write_success_max.max(previous.strip_write_success_max); + self.mirror_uncommitted_peak = self.mirror_uncommitted_peak.max(previous.mirror_uncommitted_peak); + self.ec_encode_time += previous.ec_encode_time; + self.completion_wait_time += previous.completion_wait_time; + } +} + +async fn replay_chunk( + old_chunk: &Chunk, + committed_strips: u32, + pending: &VecDeque, + old_id: ChunkId, + new_id: ChunkId, + old: &ChunkWriter, + replacement: &mut ChunkWriter, +) -> Result<()> { + let mut frames = FrameReplay::default(); + for strip in old_chunk.strips.iter().take(committed_strips as usize) { + let Some(Strip::MirrorStrip(mirror)) = &strip.strip else { + return Err(IoError::Internal("committed strip is not a mirror".into())); + }; + let length = strip.capacity.saturating_mul(1024); + let unit_bytes = u64::from(strip.unit_kb) * 1024; + let mut read_error = None; + let mut data = None; + for segment in &mirror.segments { + match old.disk_writer.read(segment, unit_bytes, 0, length).await { + Ok(bytes) if bytes.len() == length as usize => { + data = Some(bytes); + break; + } + Ok(_) => read_error = Some(IoError::ReadFailed("short mirror replay read".into())), + Err(error) => read_error = Some(error), + } + } + let bytes = data.ok_or_else(|| { + read_error.unwrap_or_else(|| IoError::ReadFailed("mirror replay has no readable copy".into())) + })?; + if old.framed_input { + frames.feed(bytes, old_id, new_id, replacement).await?; + } else if Box::pin(replacement.push(bytes)).await? == FeedStatus::Pause { + return Err(IoError::WriteFailed( + "mirror replay exceeded chunk capacity".into(), + )); + } + } + for bytes in pending { + if old.framed_input { + frames.feed(bytes.clone(), old_id, new_id, replacement).await?; + } else if Box::pin(replacement.push(bytes.clone())).await? == FeedStatus::Pause { + return Err(IoError::WriteFailed( + "mirror replay exceeded chunk capacity".into(), + )); + } + } + if !frames.pending.is_empty() { + return Err(IoError::WriteFailed("mirror replay ended inside a frame".into())); + } + while !replacement.mirror_completions.is_empty() { + Box::pin(replacement.commit_oldest_mirror()).await?; + } + if let Some(StripWriter::Mirror(mirror)) = &replacement.current_strip { + if mirror.has_data() { + mirror.checkpoint().await?; + } + } + Ok(()) +} + +#[derive(Default)] +struct FrameReplay { + pending: BytesMut, +} + +impl FrameReplay { + async fn feed( + &mut self, + bytes: Bytes, + old_id: ChunkId, + new_id: ChunkId, + replacement: &mut ChunkWriter, + ) -> Result<()> { + self.pending.extend_from_slice(&bytes); + loop { + let length = match parse_frame(&self.pending, old_id) { + Ok(frame) => frame.physical_length, + Err(FrameError::Incomplete { .. }) => break, + Err(error) => return Err(IoError::WriteFailed(format!("mirror replay frame: {error}"))), + }; + let mut frame = self.pending.split_to(length); + set_frame_chunk_id(new_id, &mut frame[length - FRAME_FOOTER_BYTES..]) + .map_err(|error| IoError::WriteFailed(format!("mirror replay footer: {error}")))?; + if Box::pin(replacement.push(frame.freeze())).await? == FeedStatus::Pause { + return Err(IoError::WriteFailed( + "mirror replay exceeded chunk capacity".into(), + )); + } + } + Ok(()) + } +} + +pub(super) fn rewrite_frames(bytes: &Bytes, old_id: ChunkId, new_id: ChunkId) -> Result { + let mut rewritten = BytesMut::from(bytes.as_ref()); + let mut offset = 0; + while offset < rewritten.len() { + let frame = parse_frame(&rewritten[offset..], old_id) + .map_err(|error| IoError::WriteFailed(format!("mirror input frame: {error}")))?; + let length = frame.physical_length; + let footer = offset + length - FRAME_FOOTER_BYTES; + set_frame_chunk_id(new_id, &mut rewritten[footer..footer + FRAME_FOOTER_BYTES]) + .map_err(|error| IoError::WriteFailed(format!("mirror input footer: {error}")))?; + offset += length; + } + Ok(rewritten.freeze()) +} diff --git a/lib/crowdb-chunk-client/src/chunk/mirror_chunk_writer.rs b/lib/crowdb-chunk-client/src/chunk/mirror_chunk_writer.rs index 6f9b1342a..b6cf86767 100644 --- a/lib/crowdb-chunk-client/src/chunk/mirror_chunk_writer.rs +++ b/lib/crowdb-chunk-client/src/chunk/mirror_chunk_writer.rs @@ -34,7 +34,7 @@ pub struct MirrorChunkWriter { } impl MirrorChunkWriter { - /// Allocates one three-copy WAL chunk with a fixed logical capacity. + /// Allocates one WAL chunk with the default mirror policy. /// /// # Errors /// @@ -53,7 +53,7 @@ impl MirrorChunkWriter { stream_name, writer_epoch, writer_lease_ms, - 3, + 2, ) .await } @@ -71,7 +71,7 @@ impl MirrorChunkWriter { writer_lease_ms: u64, copy_count: u32, ) -> Result { - if writer_epoch == 0 || writer_lease_ms == 0 || copy_count == 0 { + if writer_epoch == 0 || writer_lease_ms == 0 || !(1..=5).contains(©_count) { return Err(IoError::AllocationFailed( "stream mirror writer requires a nonzero epoch and lease".into(), )); @@ -125,7 +125,7 @@ impl MirrorChunkWriter { stream_name, writer_epoch, writer_lease_ms, - 3, + 2, ) } @@ -165,7 +165,13 @@ impl MirrorChunkWriter { "stream chunk does not contain a mirror strip".into(), )); }; - if mirror.segments.len() != copy_count as usize || strip.unit_kb == 0 || strip.capacity == 0 { + let actual_copies = mirror.segments.len(); + let protected = if copy_count == 1 { + actual_copies == 1 + } else { + (2..=copy_count as usize).contains(&actual_copies) + }; + if !protected || strip.unit_kb == 0 || strip.capacity == 0 { return Err(IoError::MetadataConflict( "stream chunk mirror geometry is invalid".into(), )); diff --git a/lib/crowdb-chunk-client/src/chunk/mirror_flow/repair.rs b/lib/crowdb-chunk-client/src/chunk/mirror_flow/repair.rs index 613eb7731..81b071eeb 100644 --- a/lib/crowdb-chunk-client/src/chunk/mirror_flow/repair.rs +++ b/lib/crowdb-chunk-client/src/chunk/mirror_flow/repair.rs @@ -184,9 +184,7 @@ impl MirrorStripFlow { } self.mark_unavailable(chunk, committed_cursor, strip_sequence, failed) .await?; - Err(IoError::WriteFailed(format!( - "mirror replica repair exhausted: {last_error}" - ))) + Err(IoError::ReplicaRepairExhausted(last_error)) } async fn publish(&self, request: ReplaceChunkStripRangeRequest, replacement: Segment) -> Result { diff --git a/lib/crowdb-chunk-client/src/chunk/mirror_strip_writer.rs b/lib/crowdb-chunk-client/src/chunk/mirror_strip_writer.rs index 1f5c2a57a..421db3d12 100644 --- a/lib/crowdb-chunk-client/src/chunk/mirror_strip_writer.rs +++ b/lib/crowdb-chunk-client/src/chunk/mirror_strip_writer.rs @@ -1,65 +1,284 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -#![allow(clippy::unused_async)] +//! Durable writes for one persisted mirror strip. -//! `MirrorStripWriter` — placeholder stub for mirror strips. -//! -//! Declared so the `StripWriter` enum shape is fixed. The large-write -//! flow never constructs it. Filled in by R93 (mirror-to-EC -//! conversion) and R106. Mirror strips have no parity, so a -//! `MirrorStripWriter` owns no `EcWorker`. +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::Duration; use bytes::Bytes; +use crowdb_protocol::chunkdb::rpc::{Chunk, Strip}; +use crowdb_protocol::diskdb::rpc::Segment; +use tokio::task::JoinSet; +use crate::chunk::segment_writer::FailedSegmentWrite; use crate::chunk::strip::StripResult; +use crate::disk_io::DiskWriter; use crate::io::FeedStatus; use crate::{IoError, Result}; -/// Mirror strip writer — placeholder. Methods return -/// `IoError::Internal` until R93/R106 fills in the impl. pub struct MirrorStripWriter { - _placeholder: (), + chunk: Arc, + strip_index: u32, + disk_writer: Arc, + accepted: u64, + finished: bool, + history: Vec, + failed_segments: Vec<(Segment, String)>, + io_concurrency: Arc, +} + +#[derive(Default)] +pub(crate) struct MirrorIoConcurrency { + active: AtomicU64, + peak: AtomicU64, +} + +impl MirrorIoConcurrency { + pub(crate) fn peak(&self) -> u64 { + self.peak.load(Ordering::Relaxed) + } + + fn begin(self: &Arc) -> MirrorIoGuard { + let active = self.active.fetch_add(1, Ordering::AcqRel) + 1; + self.peak.fetch_max(active, Ordering::Relaxed); + MirrorIoGuard(Arc::clone(self)) + } +} + +struct MirrorIoGuard(Arc); + +impl Drop for MirrorIoGuard { + fn drop(&mut self) { + self.0.active.fetch_sub(1, Ordering::AcqRel); + } } impl MirrorStripWriter { - /// Construct a new mirror strip writer (placeholder). #[must_use] - pub fn new() -> Self { - Self { _placeholder: () } + pub fn new(chunk: Arc, strip_index: u32, disk_writer: Arc) -> Self { + Self::new_with_io_concurrency(chunk, strip_index, disk_writer, Arc::default()) + } + + pub(crate) fn new_with_io_concurrency( + chunk: Arc, + strip_index: u32, + disk_writer: Arc, + io_concurrency: Arc, + ) -> Self { + Self { + chunk, + strip_index, + disk_writer, + accepted: 0, + finished: false, + history: Vec::new(), + failed_segments: Vec::new(), + io_concurrency, + } + } + + fn geometry(&self) -> Result<(u64, u64, u32, Vec)> { + let strip = self + .chunk + .strips + .get(self.strip_index as usize) + .ok_or_else(|| IoError::Internal("mirror strip index is missing".into()))?; + let Some(Strip::MirrorStrip(mirror)) = &strip.strip else { + return Err(IoError::Internal("expected persisted mirror strip".into())); + }; + let unit_bytes = u64::from(strip.unit_kb) * 1024; + let capacity = u64::from(strip.capacity) * 1024; + if mirror.segments.is_empty() || unit_bytes == 0 || capacity == 0 { + return Err(IoError::Internal("invalid mirror strip geometry".into())); + } + Ok(( + unit_bytes, + capacity, + strip.strip_sequence, + mirror.segments.clone(), + )) } - /// Push a data block to the strip. - #[allow(clippy::unused_async_trait_impl)] - pub async fn push(&mut self, _buffer: Bytes) -> Result { - Err(IoError::Internal("MirrorStripWriter not yet implemented".into())) + pub async fn push(&mut self, buffer: Bytes) -> Result { + if self.finished { + return Err(IoError::Finished); + } + let (unit_bytes, capacity, _, segments) = self.geometry()?; + let length = u64::try_from(buffer.len()) + .map_err(|_| IoError::WriteFailed("mirror write is too large".into()))?; + if length > capacity.saturating_sub(self.accepted) { + return Err(IoError::WriteFailed("mirror strip capacity exceeded".into())); + } + self.history.push(buffer.clone()); + if segments.len() == 1 { + let segment = segments[0]; + if !self.failed_segments.iter().any(|(failed, _)| *failed == segment) { + let _active = self.io_concurrency.begin(); + if let Err(error) = self + .disk_writer + .write_at_byte_offset(&segment, unit_bytes, self.accepted, buffer) + .await + { + self.failed_segments.push((segment, error.to_string())); + } + } + } else { + let mut writes = JoinSet::new(); + for segment in segments { + if self.failed_segments.iter().any(|(failed, _)| *failed == segment) { + continue; + } + let disk_io = Arc::clone(&self.disk_writer); + let io_concurrency = Arc::clone(&self.io_concurrency); + let bytes = buffer.clone(); + let offset = self.accepted; + writes.spawn(async move { + let _active = io_concurrency.begin(); + ( + segment, + disk_io + .write_at_byte_offset(&segment, unit_bytes, offset, bytes) + .await, + ) + }); + } + while let Some(result) = writes.join_next().await { + let (segment, write) = result + .map_err(|error| IoError::WriteFailed(format!("mirror replica task failed: {error}")))?; + if let Err(error) = write { + self.failed_segments.push((segment, error.to_string())); + } + } + } + self.accepted += length; + Ok(if self.accepted == capacity { + FeedStatus::Pause + } else { + FeedStatus::Continue + }) + } + + /// Write a complete strip while retaining the failed replica and its data + /// for ordered replacement before the chunk can be sealed. + pub(crate) async fn write_full_repairable( + &mut self, + buffer: Bytes, + ) -> Result<(StripResult, Vec)> { + if self.finished || self.accepted != 0 { + return Err(IoError::Finished); + } + let (_, capacity, _, _) = self.geometry()?; + if buffer.len() as u64 != capacity { + return Err(IoError::WriteFailed("full mirror strip length mismatch".into())); + } + self.push(buffer).await?; + let result = self.finish().await?; + Ok((result, self.take_failures()?)) } - /// End of strip: return the strip result. - #[allow(clippy::unused_async_trait_impl)] pub async fn finish(&mut self) -> Result { - Err(IoError::Internal("MirrorStripWriter not yet implemented".into())) + if self.finished { + return Err(IoError::Finished); + } + let (unit_bytes, _, _, segments) = self.geometry()?; + self.finished = true; + let mut syncs = JoinSet::new(); + for segment in segments { + if self.failed_segments.iter().any(|(failed, _)| *failed == segment) { + continue; + } + let writer = Arc::clone(&self.disk_writer); + syncs.spawn(async move { (segment, writer.fsync(&segment).await) }); + } + while let Some(result) = syncs.join_next().await { + let (segment, sync) = + result.map_err(|error| IoError::WriteFailed(format!("mirror sync task failed: {error}")))?; + if let Err(error) = sync { + self.failed_segments.push((segment, error.to_string())); + } + } + Ok(StripResult { + chunk_id: self.chunk.id.unwrap_or_default(), + strip_index_in_chunk: self.strip_index, + data_blocks_written: u32::try_from(self.accepted.div_ceil(unit_bytes)).unwrap_or(u32::MAX), + bytes_written: self.accepted, + partial: self.accepted % unit_bytes != 0, + ec_encode_time: Duration::ZERO, + completion_handles: Vec::new(), + }) + } + + pub(crate) fn take_failures(&mut self) -> Result> { + if !self.finished { + return Err(IoError::Internal("mirror repair requested before finish".into())); + } + let (unit_bytes, _, strip_sequence, _) = self.geometry()?; + let data = std::mem::take(&mut self.history); + Ok(std::mem::take(&mut self.failed_segments) + .into_iter() + .map(|(segment, error)| FailedSegmentWrite { + strip_sequence, + segment, + unit_bytes, + data: data.clone(), + error, + }) + .collect()) + } + + pub(crate) fn replay_views(&self) -> Vec { + self.history.clone() + } + + pub(crate) async fn checkpoint(&self) -> Result<()> { + let (_, _, _, segments) = self.geometry()?; + if !self.failed_segments.is_empty() { + return Err(IoError::WriteFailed( + "mirror checkpoint has failed replicas".into(), + )); + } + for segment in segments { + self.disk_writer.fsync(&segment).await?; + } + Ok(()) } - /// Abort: return already-durable state. - #[allow(clippy::unused_async_trait_impl)] - pub async fn abort(&mut self) -> Result { - Err(IoError::Internal("MirrorStripWriter not yet implemented".into())) + pub fn abort(&mut self) -> Result { + self.finished = true; + Ok(StripResult { + chunk_id: self.chunk.id.unwrap_or_default(), + strip_index_in_chunk: self.strip_index, + data_blocks_written: 0, + bytes_written: self.accepted, + partial: false, + ec_encode_time: Duration::ZERO, + completion_handles: Vec::new(), + }) } - /// Non-async capacity hint. + #[must_use] pub fn ready(&self) -> bool { - false + !self.finished && self.remaining_capacity() > 0 } - /// True if the strip has any data blocks written. + #[must_use] pub fn has_data(&self) -> bool { - false + self.accepted > 0 } -} -impl Default for MirrorStripWriter { - fn default() -> Self { - Self::new() + #[must_use] + pub fn remaining_capacity(&self) -> u64 { + self.chunk + .strips + .get(self.strip_index as usize) + .map_or(0, |strip| u64::from(strip.capacity) * 1024) + .saturating_sub(self.accepted) + } + + #[must_use] + pub fn accepted_bytes(&self) -> u64 { + self.accepted } } diff --git a/lib/crowdb-chunk-client/src/chunk/segment_writer.rs b/lib/crowdb-chunk-client/src/chunk/segment_writer.rs index bd7192ba2..8f9d86f52 100644 --- a/lib/crowdb-chunk-client/src/chunk/segment_writer.rs +++ b/lib/crowdb-chunk-client/src/chunk/segment_writer.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! Durable EC-segment writes and placement-safe in-line replacement. +//! Durable segment writes and placement-safe in-line replacement. use std::sync::Arc; @@ -18,7 +18,7 @@ use crate::metrics::LargeWriteRepairMetrics; use crate::negative_list::FailedDiskList; use crate::{ChunkAllocator, DiskWriter, IoError, Result}; -#[derive(Debug)] +#[derive(Debug, Clone)] pub(crate) struct FailedSegmentWrite { pub strip_sequence: u32, pub segment: Segment, @@ -82,21 +82,21 @@ impl SegmentRepair<'_> { .position(|strip| strip.strip_sequence == failure.strip_sequence) else { return Err(IoError::MetadataConflict( - "failed EC strip disappeared during replacement".into(), + "failed strip disappeared during replacement".into(), )); }; let old_strip = chunk.strips[strip_index].clone(); - let Some(Strip::EcStrip(mut ec)) = old_strip.strip.clone() else { - return Err(IoError::MetadataConflict( - "failed EC strip changed type during replacement".into(), - )); + let Some(strip_body) = old_strip.strip.clone() else { + return Err(IoError::MetadataConflict("failed strip lost its body".into())); }; - let Some(segment_index) = ec.segments.iter().position(|segment| *segment == failure.segment) - else { + let segments = match &strip_body { + Strip::EcStrip(ec) => &ec.segments, + Strip::MirrorStrip(mirror) => &mirror.segments, + }; + let Some(segment_index) = segments.iter().position(|segment| *segment == failure.segment) else { return Ok(chunk); }; - let survivors = ec - .segments + let survivors = segments .iter() .copied() .filter(|segment| *segment != failure.segment) @@ -120,21 +120,25 @@ impl SegmentRepair<'_> { let Some(replacement) = allocation.segment else { continue; }; - if self - .disk_writer - .write_views(&replacement, failure.unit_bytes, failure.data.clone()) - .await - .is_err() - { + let replay = self.replay_replacement(&strip_body, &replacement, &failure).await; + if replay.is_err() { if let Some(disk_id) = replacement.disk_id { self.failed_disks.insert(disk_id); } self.discard(chunk_id, replacement).await; continue; } - ec.segments[segment_index] = replacement; let mut replacement_strip = old_strip.clone(); - replacement_strip.strip = Some(Strip::EcStrip(ec)); + replacement_strip.strip = Some(match strip_body { + Strip::EcStrip(mut ec) => { + ec.segments[segment_index] = replacement; + Strip::EcStrip(ec) + } + Strip::MirrorStrip(mut mirror) => { + mirror.segments[segment_index] = replacement; + Strip::MirrorStrip(mirror) + } + }); replacement_strip .unavailable_segments .retain(|segment| *segment != failure.segment); @@ -158,12 +162,42 @@ impl SegmentRepair<'_> { } } self.metrics.exhausted.inc(); - Err(IoError::WriteFailed(format!( - "EC segment repair exhausted after durable write failure: {}", + Err(IoError::ReplicaRepairExhausted(format!( + "segment repair exhausted after durable write failure: {}", failure.error ))) } + async fn replay_replacement( + &self, + strip: &Strip, + replacement: &Segment, + failure: &FailedSegmentWrite, + ) -> Result<()> { + match strip { + Strip::EcStrip(_) => { + self.disk_writer + .write_views(replacement, failure.unit_bytes, failure.data.clone()) + .await + } + Strip::MirrorStrip(_) => { + if failure.data.is_empty() { + return Err(IoError::Internal("mirror repair lost its buffer".into())); + } + let mut offset = 0u64; + for bytes in &failure.data { + self.disk_writer + .write_at_byte_offset(replacement, failure.unit_bytes, offset, bytes.clone()) + .await?; + offset = offset + .checked_add(bytes.len() as u64) + .ok_or_else(|| IoError::WriteFailed("mirror replay offset overflow".into()))?; + } + self.disk_writer.fsync(replacement).await + } + } + } + async fn query_chunk(&self, chunk_id: ChunkId) -> Result { self.allocator .query_chunk(QueryChunkRequest { diff --git a/lib/crowdb-chunk-client/src/chunk/strip.rs b/lib/crowdb-chunk-client/src/chunk/strip.rs index d08962b66..d4555f918 100644 --- a/lib/crowdb-chunk-client/src/chunk/strip.rs +++ b/lib/crowdb-chunk-client/src/chunk/strip.rs @@ -29,8 +29,7 @@ pub struct StripResult { } /// Strip writer enum — Rust enum (not trait object) for monomorphic -/// dispatch. `Ec` variant is used by the large-write flow; `Mirror` -/// is a placeholder stub. +/// dispatch for the persisted strip layout. #[allow(clippy::large_enum_variant)] // avoid one allocation and indirection per hot-path EC strip pub enum StripWriter { Ec(crate::chunk::ec_strip_writer::EcStripWriter), @@ -61,7 +60,7 @@ impl StripWriter { pub async fn abort(&mut self) -> Result { match self { Self::Ec(w) => w.abort().await, - Self::Mirror(w) => w.abort().await, + Self::Mirror(w) => w.abort(), } } @@ -85,15 +84,14 @@ impl StripWriter { pub fn remaining_capacity(&self) -> u64 { match self { Self::Ec(w) => w.remaining_capacity(), - // Mirror strips are not a large-object write target yet. - Self::Mirror(_) => 0, + Self::Mirror(w) => w.remaining_capacity(), } } pub fn accepted_bytes(&self) -> u64 { match self { Self::Ec(w) => w.accepted_bytes(), - Self::Mirror(_) => 0, + Self::Mirror(w) => w.accepted_bytes(), } } } diff --git a/lib/crowdb-chunk-client/src/client.rs b/lib/crowdb-chunk-client/src/client.rs index daaaa26a9..833a2e44f 100644 --- a/lib/crowdb-chunk-client/src/client.rs +++ b/lib/crowdb-chunk-client/src/client.rs @@ -393,6 +393,24 @@ impl ChunkIoClient { self.small_pool.shutdown().await } + /// Wait until the small-write manager stops without a requested shutdown. + pub async fn wait_for_small_write_manager_failure(&self) { + let mut ticker = tokio::time::interval(std::time::Duration::from_millis(250)); + ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip); + loop { + ticker.tick().await; + if self.small_pool.manager_failed() { + return; + } + } + } + + /// Stop the manager without a shutdown request to exercise process supervision. + #[cfg(feature = "test-util")] + pub async fn stop_small_write_manager_for_test(&self) -> Result<()> { + self.small_pool.stop_manager_for_test().await + } + /// Snapshot lock-free shared small-write counters and gauges. pub fn small_write_metrics(&self) -> SmallWriteMetricsSnapshot { let mut snapshot = self.small_pool.metrics.snapshot(); @@ -849,6 +867,14 @@ impl ChunkIoWriter for PreparedLargeWrite { fn require_data(&self) -> bool { self.writer.require_data() } + + async fn wait_for_capacity(&mut self) { + self.writer.wait_for_capacity().await; + } + + fn write_timing(&self) -> Option { + Some(self.writer.write_timing()) + } } fn build_large_write_result( @@ -860,6 +886,25 @@ fn build_large_write_result( let logical_bytes: u64 = locations.iter().map(|location| location.logical_length).sum(); let data_bytes: u64 = locations.iter().map(|location| location.length).sum(); let block_bytes = policy.client.read_buffer_size as u64; + if let Some(copies) = policy.client.large_mirror_copies { + return LargeWriteResult { + chunks: locations.len(), + locations, + logical_bytes, + physical_bytes: data_bytes.saturating_mul(u64::from(copies)), + strips: data_bytes.div_ceil(block_bytes.max(1)), + elapsed, + preparation_stalls: writer.preparation_stalls(), + preparation_stall_time: writer.preparation_stall_time(), + source_reads: writer.source_reads, + source_read_time: writer.source_read_time, + assembly_copies: writer.assembly_copies, + assembly_copy_bytes: writer.assembly_copy_bytes, + assembly_copy_time: writer.assembly_copy_time, + ec_encode_time: writer.ec_encode_time, + completion_wait_time: writer.completion_wait_time, + }; + } let strip_data_bytes = block_bytes * policy.ec_scheme.data_num as u64; let full_strips = data_bytes / strip_data_bytes; let tail_bytes = data_bytes % strip_data_bytes; diff --git a/lib/crowdb-chunk-client/src/config.rs b/lib/crowdb-chunk-client/src/config.rs index 46ae37120..d524fa383 100644 --- a/lib/crowdb-chunk-client/src/config.rs +++ b/lib/crowdb-chunk-client/src/config.rs @@ -11,12 +11,15 @@ use std::time::Duration; use crowdb_common::ec::EcScheme; +use crowdb_protocol::chunkdb::rpc::ChunkType; use crate::IoError; /// Bounded aggregation and elasticity policy for shared small writes. #[derive(Debug, Clone)] pub struct SmallWritePolicy { + /// Type assigned to every chunk owned by this pool. + pub chunk_type: ChunkType, pub object_limit: usize, pub memory_budget: usize, pub queue_capacity: usize, @@ -50,6 +53,7 @@ impl Default for SmallWritePolicy { fn default() -> Self { const MIB: usize = 1024 * 1024; Self { + chunk_type: ChunkType::Repo, object_limit: 8 * MIB, // 1,000 concurrent 1 MiB objects are a normal S3 small-object // workload. 3,000 and 5,000 require roughly 3.25 GiB and 5.25 @@ -69,7 +73,7 @@ impl Default for SmallWritePolicy { cooldown: Duration::from_millis(100), chunk_capacity: 1024 * 1024 * 1024, small_strip_prefetch_count: 4, - mirror_copies: 3, + mirror_copies: 2, conversion_enabled: true, conversion_data_num: 8, conversion_code_num: 4, @@ -118,7 +122,7 @@ impl SmallWritePolicy { || self.scale_out_queue_objects > self.queue_capacity || self.chunk_capacity < self.object_limit as u64 || self.small_strip_prefetch_count == 0 - || self.mirror_copies == 0 + || !(1..=5).contains(&self.mirror_copies) || self.conversion_data_num == 0 || self.conversion_code_num == 0 || self.batch_watchdog.is_zero() @@ -155,6 +159,10 @@ impl SmallWritePolicy { /// Configuration for the chunk data path. Shared by all writers. #[derive(Debug, Clone)] pub struct ChunkClientConfig { + /// Type assigned to every chunk prepared by a large-write session. + pub chunk_type: ChunkType, + /// Mirror copies for large-write strips; `None` selects EC. + pub large_mirror_copies: Option, // ── write path ────────────────────────────────────────────── /// Fetch read granularity / block size (bytes). Default 1 MB. pub read_buffer_size: usize, @@ -166,6 +174,12 @@ pub struct ChunkClientConfig { pub max_chunk_size: u64, /// Strip-prefetch results buffered ahead of the write cursor. Default 1. pub prefetch_strips_per_chunk: usize, + /// Maximum strips in one known-size large-write prefetch batch. Default 32. + pub large_prefetch_max_strips_per_batch: usize, + /// Maximum mirror-strip writes in flight for one large object. + pub large_parallel_strip_writes: usize, + /// Completed body owners retained before a large-object writer consumes them. + pub large_held_buffers: usize, /// Maximum completed-strip parity/finalization tasks in flight. Default 2. pub parity_depth: usize, /// Chunks allocated ahead. Default 1. @@ -184,10 +198,15 @@ impl Default for ChunkClientConfig { const MB: usize = 1024 * 1024; const GB: usize = 1024 * 1024 * 1024; Self { + chunk_type: ChunkType::Repo, + large_mirror_copies: None, read_buffer_size: MB, max_cached_buffer: 4 * MB, max_chunk_size: GB as u64, prefetch_strips_per_chunk: 1, + large_prefetch_max_strips_per_batch: 32, + large_parallel_strip_writes: 4, + large_held_buffers: 4, parity_depth: 2, chunk_preparation_depth: 1, large_write_repair_attempts: 3, @@ -213,6 +232,16 @@ impl ChunkClientConfig { if self.prefetch_strips_per_chunk == 0 { return Err(IoError::Internal("prefetch_strips_per_chunk must be > 0".into())); } + if self.large_prefetch_max_strips_per_batch == 0 { + return Err(IoError::Internal( + "large_prefetch_max_strips_per_batch must be > 0".into(), + )); + } + if self.large_parallel_strip_writes == 0 || self.large_held_buffers == 0 { + return Err(IoError::Internal( + "large write parallel and held buffer counts must be > 0".into(), + )); + } if self.parity_depth == 0 { return Err(IoError::Internal("parity_depth must be > 0".into())); } @@ -224,6 +253,14 @@ impl ChunkClientConfig { "large_write_repair_attempts must be > 0".into(), )); } + if self + .large_mirror_copies + .is_some_and(|copies| !(1..=5).contains(&copies)) + { + return Err(IoError::Internal( + "large mirror copy count must be between one and five".into(), + )); + } Ok(()) } diff --git a/lib/crowdb-chunk-client/src/error.rs b/lib/crowdb-chunk-client/src/error.rs index bf01d5982..c3b44dbd7 100644 --- a/lib/crowdb-chunk-client/src/error.rs +++ b/lib/crowdb-chunk-client/src/error.rs @@ -12,6 +12,8 @@ pub enum IoError { AllocationFailed(String), #[error("disk write failed: {0}")] WriteFailed(String), + #[error("mirror replica repair exhausted: {0}")] + ReplicaRepairExhausted(String), #[error("disk read failed: {0}")] ReadFailed(String), #[error("transient disk read failed: {0}")] diff --git a/lib/crowdb-chunk-client/src/io.rs b/lib/crowdb-chunk-client/src/io.rs index a588f2ba5..b68383527 100644 --- a/lib/crowdb-chunk-client/src/io.rs +++ b/lib/crowdb-chunk-client/src/io.rs @@ -5,6 +5,7 @@ //! data-path writers. use std::ops::Range; +use std::time::Duration; use bytes::Bytes; @@ -22,6 +23,15 @@ pub trait FramedWriteBuffer: Send { fn frame_count(&self) -> usize; /// Logical payload bytes in one frame slot. fn frame_payload_len(&self, index: usize) -> Option; + /// Prepare frame headers and CRC before placement. Implementations that + /// cannot prepare early may use the default and finalize in the writer. + fn prepare_frames( + &mut self, + _magic: FrameMagic, + _write_time_ms: u64, + ) -> std::result::Result<(), FrameError> { + Ok(()) + } /// Fill one slot's reserved frame bytes for its actual destination chunk. fn finalize_frame( &mut self, @@ -44,6 +54,20 @@ pub enum FeedStatus { Pause, } +/// Timings observed by one large writer after its completed chunk writes. +#[derive(Clone, Copy, Debug, Default)] +pub struct ChunkWriteTiming { + pub strip_prepare_waits: u64, + pub strip_prepare_wait_time: Duration, + pub strip_write_successes: u64, + pub strip_write_success_time: Duration, + pub strip_write_success_max: Duration, + /// Submitted mirror strips awaiting ordered commit, including completed tasks. + pub mirror_uncommitted_peak: u64, + /// Highest number of mirror data writes actually inside DiskIO. + pub mirror_active_write_peak: u64, +} + /// Caller-side backpressure strategy. Selects how to react when /// `require_data()` returns false. Not a property of the writer. #[derive(Debug, Clone, Copy)] @@ -99,6 +123,10 @@ pub trait ChunkIoWriter: Send { fn input_complete(&self) -> bool { false } + /// Per-writer strip preparation and successful push timings, if available. + fn write_timing(&self) -> Option { + None + } /// Wait for a capacity change without polling more network input. Writers /// with no external notifier use the short default recheck. async fn wait_for_capacity(&mut self) { diff --git a/lib/crowdb-chunk-client/src/lib.rs b/lib/crowdb-chunk-client/src/lib.rs index 72371e3d7..65765ba5d 100644 --- a/lib/crowdb-chunk-client/src/lib.rs +++ b/lib/crowdb-chunk-client/src/lib.rs @@ -50,7 +50,7 @@ pub use client::{ pub use config::{ChunkClientConfig, SmallWritePolicy}; pub use disk_io::{DiskWriter, RoutedDiskWriter}; pub use error::{IoError, ReadError, ReadResult, Result}; -pub use io::{BackpressurePolicy, ChunkIoWriter, FeedStatus, FramedWriteBuffer}; +pub use io::{BackpressurePolicy, ChunkIoWriter, ChunkWriteTiming, FeedStatus, FramedWriteBuffer}; pub use metrics::{ ChunkClientMetrics, LargeWriteBufferMetricsSnapshot, LargeWriteRepairMetricsSnapshot, ReadFlowMetricsSnapshot, SmallWriteMetricsSnapshot, diff --git a/lib/crowdb-chunk-client/src/metrics.rs b/lib/crowdb-chunk-client/src/metrics.rs index f95ac5555..886b6a7ac 100644 --- a/lib/crowdb-chunk-client/src/metrics.rs +++ b/lib/crowdb-chunk-client/src/metrics.rs @@ -355,6 +355,10 @@ pub struct LargeWriteRepairMetrics { pub(crate) exhausted: Arc, pub(crate) negative_list_hits: Arc, pub(crate) discarded_segments: Arc, + pub(crate) chunk_rotations: Arc, + pub(crate) rotated_chunks: Arc, + pub(crate) replayed_bytes: Arc, + pub(crate) rotation_ns: Arc, } impl Default for LargeWriteRepairMetrics { @@ -369,6 +373,12 @@ impl Default for LargeWriteRepairMetrics { discarded_segments: Arc::new(Counter::new( "chunkio.large_write.repair.discarded_segments.c".into(), )), + chunk_rotations: Arc::new(Counter::new( + "chunkio.large_write.repair.chunk_rotations.c".into(), + )), + rotated_chunks: Arc::new(Counter::new("chunkio.large_write.repair.rotated_chunks.c".into())), + replayed_bytes: Arc::new(Counter::new("chunkio.large_write.repair.replayed_bytes.c".into())), + rotation_ns: Arc::new(Counter::new("chunkio.large_write.repair.rotation_ns.c".into())), } } } @@ -380,6 +390,10 @@ pub struct LargeWriteRepairMetricsSnapshot { pub exhausted: u64, pub negative_list_hits: u64, pub discarded_segments: u64, + pub chunk_rotations: u64, + pub rotated_chunks: u64, + pub replayed_bytes: u64, + pub rotation_ns: u64, } impl LargeWriteRepairMetrics { @@ -390,6 +404,10 @@ impl LargeWriteRepairMetrics { exhausted: registry.register_counter("chunkio.large_write.repair.exhausted.c"), negative_list_hits: registry.register_counter("chunkio.large_write.repair.negative_list_hits.c"), discarded_segments: registry.register_counter("chunkio.large_write.repair.discarded_segments.c"), + chunk_rotations: registry.register_counter("chunkio.large_write.repair.chunk_rotations.c"), + rotated_chunks: registry.register_counter("chunkio.large_write.repair.rotated_chunks.c"), + replayed_bytes: registry.register_counter("chunkio.large_write.repair.replayed_bytes.c"), + rotation_ns: registry.register_counter("chunkio.large_write.repair.rotation_ns.c"), } } @@ -401,6 +419,10 @@ impl LargeWriteRepairMetrics { exhausted: self.exhausted.snapshot().total, negative_list_hits: self.negative_list_hits.snapshot().total, discarded_segments: self.discarded_segments.snapshot().total, + chunk_rotations: self.chunk_rotations.snapshot().total, + rotated_chunks: self.rotated_chunks.snapshot().total, + replayed_bytes: self.replayed_bytes.snapshot().total, + rotation_ns: self.rotation_ns.snapshot().total, } } } diff --git a/lib/crowdb-chunk-client/src/writer/large_async_object.rs b/lib/crowdb-chunk-client/src/writer/large_async_object.rs index 82778e877..818884615 100644 --- a/lib/crowdb-chunk-client/src/writer/large_async_object.rs +++ b/lib/crowdb-chunk-client/src/writer/large_async_object.rs @@ -23,7 +23,7 @@ use tokio::sync::mpsc; use tokio::task::JoinHandle; use crate::chunk::chunk_prefetch::ChunkPrefetch; -use crate::chunk::chunk_writer::ChunkWriter; +use crate::chunk::chunk_writer::{ChunkWriter, StripPrefetchPlan}; use crate::config::ChunkClientConfig; use crate::disk_io::DiskWriter; use crate::io::{ChunkIoWriter, FeedStatus, FramedWriteBuffer}; @@ -52,8 +52,16 @@ pub struct LargeAsyncObjectWriter { pub(crate) frame_tail: BytesMut, pub(crate) object_size: Option, pub(crate) finished: bool, + pub(crate) deferred_write_error: Option, pub(crate) preparation_stalls: u64, pub(crate) preparation_stall_time: Duration, + pub(crate) strip_prepare_waits: u64, + pub(crate) strip_prepare_wait_time: Duration, + pub(crate) strip_write_successes: u64, + pub(crate) strip_write_success_time: Duration, + pub(crate) strip_write_success_max: Duration, + pub(crate) mirror_uncommitted_peak: u64, + pub(crate) mirror_active_write_peak: u64, pub(crate) source_reads: u64, pub(crate) source_read_time: Duration, pub(crate) assembly_copies: u64, @@ -109,8 +117,16 @@ impl LargeAsyncObjectWriter { frame_tail: BytesMut::new(), object_size: None, finished: false, + deferred_write_error: None, preparation_stalls: 0, preparation_stall_time: Duration::ZERO, + strip_prepare_waits: 0, + strip_prepare_wait_time: Duration::ZERO, + strip_write_successes: 0, + strip_write_success_time: Duration::ZERO, + strip_write_success_max: Duration::ZERO, + mirror_uncommitted_peak: 0, + mirror_active_write_peak: 0, source_reads: 0, source_read_time: Duration::ZERO, assembly_copies: 0, @@ -139,6 +155,18 @@ impl LargeAsyncObjectWriter { self.preparation_stall_time } + pub fn write_timing(&self) -> crate::ChunkWriteTiming { + crate::ChunkWriteTiming { + strip_prepare_waits: self.strip_prepare_waits, + strip_prepare_wait_time: self.strip_prepare_wait_time, + strip_write_successes: self.strip_write_successes, + strip_write_success_time: self.strip_write_success_time, + strip_write_success_max: self.strip_write_success_max, + mirror_uncommitted_peak: self.mirror_uncommitted_peak, + mirror_active_write_peak: self.mirror_active_write_peak, + } + } + /// Snapshot owner-view and payload-copy accounting for this writer's /// shared metric set. pub fn buffer_metrics(&self) -> crate::LargeWriteBufferMetricsSnapshot { @@ -152,7 +180,7 @@ impl LargeAsyncObjectWriter { self.allocator.clone(), self.ec_scheme, self.config.clone(), - crowdb_protocol::chunk_id::CHUNK_TYPE_REPO, + self.config.chunk_type as u8, ); let (chunk_rx, prefetch_handle) = prefetch.spawn(object_size); self.chunk_prefetch_rx = Some(chunk_rx); @@ -195,6 +223,13 @@ impl LargeAsyncObjectWriter { let (stalls, stall_time) = cw.preparation_metrics(); self.preparation_stalls += stalls; self.preparation_stall_time += stall_time; + self.strip_prepare_waits += stalls; + self.strip_prepare_wait_time += stall_time; + self.strip_write_successes += cw.strip_write_successes; + self.strip_write_success_time += cw.strip_write_success_time; + self.strip_write_success_max = self.strip_write_success_max.max(cw.strip_write_success_max); + self.mirror_uncommitted_peak = self.mirror_uncommitted_peak.max(cw.mirror_uncommitted_peak); + self.mirror_active_write_peak = self.mirror_active_write_peak.max(cw.active_mirror_write_peak()); self.ec_encode_time += cw.ec_encode_time; self.completion_wait_time += cw.completion_wait_time; if location.length > 0 { @@ -243,7 +278,7 @@ impl LargeAsyncObjectWriter { self.allocator.clone(), self.ec_scheme, self.config.clone(), - crowdb_protocol::chunk_id::CHUNK_TYPE_REPO, + self.config.chunk_type as u8, ); let started = Instant::now(); self.preparation_stalls += 1; @@ -270,11 +305,42 @@ impl LargeAsyncObjectWriter { Arc::clone(&self.failed_disks), Arc::clone(&self.repair_metrics), ); - cw.open(chunk, self.object_size)?; + cw.set_framed_input(); + let plan = self.strip_prefetch_plan(&chunk); + let remaining_size = self + .object_size + .map(|size| size.saturating_sub(self.logical_offset)); + cw.open_with_prefetch_plan(chunk, remaining_size, plan)?; self.chunk_writer = Some(cw); Ok(()) } + fn strip_prefetch_plan(&self, chunk: &Chunk) -> Option { + let remaining_bytes = self.object_size?.saturating_sub(self.logical_offset); + let frames = remaining_bytes.div_ceil(MAX_FRAME_PAYLOAD_BYTES as u64); + let physical_bytes = remaining_bytes + .saturating_add(frames.saturating_mul((FRAME_HEADER_PREFIX_BYTES + FRAME_FOOTER_BYTES) as u64)); + let strip_bytes = u64::from(chunk.strips.first()?.capacity) + .checked_mul(1024)? + .max(1); + let chunk_limit = (self.config.max_chunk_size / strip_bytes) + .max(chunk.strips.len() as u64) + .max(1); + let needed = physical_bytes + .div_ceil(strip_bytes) + .max(chunk.strips.len() as u64) + .min(chunk_limit) + .min(u64::from(u32::MAX)); + let total_strips = u32::try_from(needed).ok()?; + Some(StripPrefetchPlan { + total_strips, + batch_max: u32::try_from(self.config.large_prefetch_max_strips_per_batch) + .unwrap_or(u32::MAX) + .min(total_strips) + .max(1), + }) + } + /// Rotate: seal the current chunk, pull the next `Chunk`, open a /// new `ChunkWriter`. pub(crate) async fn rotate_chunk(&mut self) -> Result<()> { @@ -400,6 +466,9 @@ impl LargeAsyncObjectWriter { #[async_trait::async_trait] impl ChunkIoWriter for LargeAsyncObjectWriter { async fn on_data(&mut self, buffer: Bytes) -> Result { + if let Some(error) = self.deferred_write_error.take() { + return Err(error); + } if self.finished { return Err(IoError::Finished); } @@ -409,7 +478,7 @@ impl ChunkIoWriter for LargeAsyncObjectWriter { self.allocator.clone(), self.ec_scheme, self.config.clone(), - crowdb_protocol::chunk_id::CHUNK_TYPE_REPO, + self.config.chunk_type as u8, ); let (rx, handle) = prefetch.spawn(None); self.chunk_prefetch_rx = Some(rx); @@ -420,6 +489,9 @@ impl ChunkIoWriter for LargeAsyncObjectWriter { } async fn on_framed_data(&mut self, mut buffer: Box) -> Result { + if let Some(error) = self.deferred_write_error.take() { + return Err(error); + } if self.finished { return Err(IoError::Finished); } @@ -433,6 +505,9 @@ impl ChunkIoWriter for LargeAsyncObjectWriter { } async fn on_finish(&mut self) -> Result> { + if let Some(error) = self.deferred_write_error.take() { + return Err(error); + } if self.finished { return Err(IoError::Finished); } @@ -452,9 +527,20 @@ impl ChunkIoWriter for LargeAsyncObjectWriter { } fn require_data(&self) -> bool { - // The next push rotates a full strip or chunk. Waiting for a - // background capacity change here would deadlock at that boundary. !self.finished + && (self.deferred_write_error.is_some() + || self + .chunk_writer + .as_ref() + .map_or(true, ChunkWriter::mirror_write_capacity)) + } + + async fn wait_for_capacity(&mut self) { + if let Some(chunk) = self.chunk_writer.as_mut() { + if let Err(error) = chunk.ensure_mirror_capacity().await { + self.deferred_write_error = Some(error); + } + } } } @@ -561,12 +647,74 @@ impl LargeAsyncObjectWriter { self.buffer_metrics.payload_copy_bytes.inc_by(buffer.len() as u64); self.frame_tail.extend_from_slice(&buffer); while self.frame_tail.len() >= MAX_FRAME_PAYLOAD_BYTES { - let payload = self.frame_tail.split_to(MAX_FRAME_PAYLOAD_BYTES).freeze(); - self.push_payload_frame(payload).await?; + self.push_full_payload_frames().await?; } Ok(()) } + async fn push_full_payload_frames(&mut self) -> Result<()> { + const FRAMES_PER_BATCH: usize = 16; + const FRAME_BYTES: usize = FRAME_HEADER_PREFIX_BYTES + MAX_FRAME_PAYLOAD_BYTES + FRAME_FOOTER_BYTES; + + loop { + self.ensure_open().await?; + let writer = self + .chunk_writer + .as_ref() + .ok_or_else(|| IoError::Internal("large async writer has no chunk writer".into()))?; + let remaining = writer.remaining_capacity(); + let available_frames = usize::try_from(remaining / FRAME_BYTES as u64).unwrap_or(usize::MAX); + if available_frames == 0 { + if FRAME_BYTES as u64 > self.config.max_chunk_size { + return Err(IoError::WriteFailed("large frame exceeds chunk capacity".into())); + } + self.rotate_chunk().await?; + continue; + } + let frame_count = (self.frame_tail.len() / MAX_FRAME_PAYLOAD_BYTES) + .min(FRAMES_PER_BATCH) + .min(available_frames); + let chunk_id = writer + .current_chunk_id() + .ok_or_else(|| IoError::Internal("large async writer has no chunk ID".into()))?; + let write_time_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |duration| { + u64::try_from(duration.as_millis()).unwrap_or(u64::MAX) + }); + let mut framed = Vec::with_capacity(frame_count * FRAME_BYTES); + for payload in + self.frame_tail[..frame_count * MAX_FRAME_PAYLOAD_BYTES].chunks_exact(MAX_FRAME_PAYLOAD_BYTES) + { + framed.extend_from_slice( + &encode_frame(FrameMagic::RepoLargeV1, chunk_id, payload, write_time_ms) + .map_err(|error| IoError::WriteFailed(error.to_string()))?, + ); + } + self.buffer_metrics + .payload_copy_operations + .inc_by(u64::try_from(frame_count).unwrap_or(u64::MAX)); + self.buffer_metrics + .payload_copy_bytes + .inc_by(u64::try_from(frame_count * MAX_FRAME_PAYLOAD_BYTES).unwrap_or(u64::MAX)); + let status = self + .chunk_writer + .as_mut() + .ok_or_else(|| IoError::Internal("large async writer has no chunk writer".into()))? + .push(Bytes::from(framed)) + .await?; + if status == FeedStatus::Pause { + self.rotate_chunk().await?; + continue; + } + let _ = self.frame_tail.split_to(frame_count * MAX_FRAME_PAYLOAD_BYTES); + self.logical_bytes_in_chunk = self + .logical_bytes_in_chunk + .saturating_add((frame_count * MAX_FRAME_PAYLOAD_BYTES) as u64); + return Ok(()); + } + } + async fn push_payload_frame(&mut self, payload: Bytes) -> Result<()> { loop { self.ensure_open().await?; diff --git a/lib/crowdb-chunk-client/src/writer/large_object.rs b/lib/crowdb-chunk-client/src/writer/large_object.rs index 28cb058ce..5ec90a3f3 100644 --- a/lib/crowdb-chunk-client/src/writer/large_object.rs +++ b/lib/crowdb-chunk-client/src/writer/large_object.rs @@ -107,7 +107,7 @@ impl LargeObjectWriter { self.allocator.clone(), self.ec_scheme, self.config.clone(), - crowdb_protocol::chunk_id::CHUNK_TYPE_REPO, + self.config.chunk_type as u8, ); let (rx, handle) = prefetch.spawn(object_size); self.chunk_prefetch_rx = Some(rx); @@ -131,7 +131,7 @@ impl LargeObjectWriter { self.allocator.clone(), self.ec_scheme, self.config.clone(), - crowdb_protocol::chunk_id::CHUNK_TYPE_REPO, + self.config.chunk_type as u8, ); let chunk = pf.on_demand().await?; Ok(Some(chunk)) @@ -155,6 +155,7 @@ impl LargeObjectWriter { Arc::clone(&self.failed_disks), Arc::clone(&self.repair_metrics), ); + cw.set_framed_input(); cw.open(chunk, self.object_size)?; self.chunk_writer = Some(cw); Ok(()) diff --git a/lib/crowdb-chunk-client/src/writer/shared_object.rs b/lib/crowdb-chunk-client/src/writer/shared_object.rs index 8ab6add8f..f02d90b78 100644 --- a/lib/crowdb-chunk-client/src/writer/shared_object.rs +++ b/lib/crowdb-chunk-client/src/writer/shared_object.rs @@ -292,12 +292,11 @@ impl ChunkIoWriter for SharedObjectWriter { return; }; let notified = route.capacity_changed.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); if route.has_capacity() { return; } - tokio::select! { - () = notified => {}, - () = tokio::time::sleep(std::time::Duration::from_millis(5)) => {}, - } + notified.await; } } diff --git a/lib/crowdb-chunk-client/src/writer/small_manager.rs b/lib/crowdb-chunk-client/src/writer/small_manager.rs index 9a7de77b6..5bfd15431 100644 --- a/lib/crowdb-chunk-client/src/writer/small_manager.rs +++ b/lib/crowdb-chunk-client/src/writer/small_manager.rs @@ -16,6 +16,8 @@ use super::small_pool::{SmallPoolRuntime, SmallWritePool}; pub(crate) enum ManagerCommand { Shutdown(oneshot::Sender>), + #[cfg(feature = "test-util")] + StopForTest(oneshot::Sender<()>), } pub(crate) async fn start(pool: Arc) -> Result> { @@ -70,18 +72,30 @@ async fn run( let mut next_id = pipelines.len() as u64; loop { tokio::select! { - command = commands.recv() => { - let Some(ManagerCommand::Shutdown(done)) = command else { break; }; - runtime.publish(&[]); - runtime.metrics.draining_pipelines.set(pipelines.len() as u64); - for pipeline in &pipelines { - pipeline.begin_retire(); + command = commands.recv() => match command { + Some(ManagerCommand::Shutdown(done)) => { + runtime.publish(&[]); + runtime.metrics.draining_pipelines.set(pipelines.len() as u64); + for pipeline in &pipelines { + pipeline.begin_retire(); + } + let result = join_all(pipelines).await; + runtime.metrics.draining_pipelines.set(0); + let _ = done.send(result); + return; } - let result = join_all(pipelines).await; - runtime.metrics.draining_pipelines.set(0); - let _ = done.send(result); - return; - } + #[cfg(feature = "test-util")] + Some(ManagerCommand::StopForTest(done)) => { + runtime.publish(&[]); + for pipeline in &pipelines { + pipeline.begin_retire(); + } + let _ = join_all(pipelines).await; + let _ = done.send(()); + return; + } + None => break, + }, _ = ticker.tick() => { let mut failed_pipelines = reap_finished(&runtime, &mut pipelines).await; while pipelines.len() < runtime.policy.min_pipelines { diff --git a/lib/crowdb-chunk-client/src/writer/small_pipeline.rs b/lib/crowdb-chunk-client/src/writer/small_pipeline.rs index 55c573de8..1b848bab8 100644 --- a/lib/crowdb-chunk-client/src/writer/small_pipeline.rs +++ b/lib/crowdb-chunk-client/src/writer/small_pipeline.rs @@ -12,16 +12,15 @@ use bytes::{Bytes, BytesMut}; use crowdb_common::ec::{EcScheme, IncrementalParity}; use crowdb_protocol::chunkdb::rpc::{ AdvanceChunkWriteRequest, AllocateChunkRequest, AppendChunkRequest, Chunk, ChunkState, ChunkStrip, - ChunkType, DeleteChunkRequest, Location, MutateStripReservationRequest, - PrepareMirrorToEcConversionRequest, QueryChunkRequest, ReserveStripGroupRequest, SealChunkRequest, Strip, - StripReservationAction, StripType, + DeleteChunkRequest, Location, MutateStripReservationRequest, PrepareMirrorToEcConversionRequest, + QueryChunkRequest, ReserveStripGroupRequest, SealChunkRequest, Strip, StripReservationAction, StripType, }; use crowdb_protocol::common::ChunkId; use crowdb_protocol::diskdb::rpc::Segment; use crowdb_protocol::frame::{ encode_frame, FrameMagic, FRAME_FOOTER_BYTES, FRAME_HEADER_PREFIX_BYTES, MAX_FRAME_PAYLOAD_BYTES, }; -use crowdb_protocol::{generate_chunk_id, CHUNK_TYPE_REPO}; +use crowdb_protocol::generate_chunk_id; use tokio::sync::{mpsc, Notify, OwnedSemaphorePermit}; use crate::chunk::mirror_flow::MirrorStripFlow; @@ -175,7 +174,7 @@ impl PipelineWorker { let logical_bytes: usize = batch.iter().map(|object| object.len).sum(); let watchdog = self.runtime.policy.batch_watchdog; let metrics = Arc::clone(&self.runtime.metrics); - let write = self.chunk.write_batch(batch, &metrics); + let write = self.chunk.write_batch(batch, &metrics, &self.runtime); tokio::pin!(write); let mut elapsed = Duration::ZERO; loop { @@ -797,7 +796,7 @@ impl OwnedChunk { data_num: 0, code_num: 0, copy_count: runtime.policy.mirror_copies, - chunk_type: ChunkType::Repo as i32, + chunk_type: runtime.policy.chunk_type as i32, writer_epoch, writer_lease_ms: lease_ms, owner_key: Vec::new(), @@ -1006,7 +1005,7 @@ impl OwnedChunk { .chunk .id .ok_or_else(|| IoError::AllocationFailed("shared chunk missing id".into()))?; - let group_id = generate_chunk_id(CHUNK_TYPE_REPO).to_proto(); + let group_id = generate_chunk_id(self.policy.chunk_type as u8).to_proto(); let unit_count = last.map_or(1, |strip| { u32::try_from(strip_kb) .unwrap_or(u32::MAX) @@ -1117,14 +1116,51 @@ impl OwnedChunk { &mut self, mut batch: Vec, metrics: &SmallWriteMetrics, + runtime: &SmallPoolRuntime, ) -> Result<()> { - let result = if batch.len() == 1 && batch[0].len > MAX_FRAME_PAYLOAD_BYTES { + let stream_object = batch.len() == 1 && batch[0].len > MAX_FRAME_PAYLOAD_BYTES; + let first = if stream_object { self.try_write_stream_object(&mut batch[0], metrics) .await .map(|location| vec![location]) } else { self.try_write_batch(&batch, metrics).await }; + // Stream sources cannot be replayed, and durable intents already name an exact location. + let can_relocate = !stream_object && batch.iter().all(|object| object.intent.is_none()); + let result = if can_relocate && matches!(first, Err(IoError::ReplicaRepairExhausted(_))) { + let prior = first.unwrap_err(); + match self.finish().await { + Ok(()) => match Self::allocate(runtime, Arc::clone(&self.conversion_active)).await { + Ok(next) => { + *self = next; + self.try_write_batch(&batch, metrics).await.map_err(|error| { + if let IoError::ReplicaRepairExhausted(message) = error { + IoError::WriteFailed(format!( + "mirror replica repair exhausted after chunk rotation: {message}" + )) + } else { + error + } + }) + } + Err(error) => Err(IoError::WriteFailed(format!( + "{prior}; chunk rotation allocation failed: {error}" + ))), + }, + Err(error) => Err(IoError::WriteFailed(format!( + "{prior}; failed to seal previous chunk: {error}" + ))), + } + } else { + first.map_err(|error| { + if let IoError::ReplicaRepairExhausted(message) = error { + IoError::WriteFailed(format!("mirror replica repair exhausted: {message}")) + } else { + error + } + }) + }; match result { Ok(locations) => { for (object, location) in batch.into_iter().zip(locations) { diff --git a/lib/crowdb-chunk-client/src/writer/small_pool.rs b/lib/crowdb-chunk-client/src/writer/small_pool.rs index 140cb06a8..e2b393347 100644 --- a/lib/crowdb-chunk-client/src/writer/small_pool.rs +++ b/lib/crowdb-chunk-client/src/writer/small_pool.rs @@ -188,8 +188,7 @@ impl SmallPoolRuntime { } let routes = self.routes.load_full(); if routes.is_empty() { - tokio::task::yield_now().await; - continue; + return Err(IoError::WriteFailed("small-write pipelines unavailable".into())); } let route = Arc::clone(&object.route); route.accepted(object.len); @@ -232,6 +231,12 @@ pub(crate) struct SmallWritePool { } impl SmallWritePool { + pub fn manager_failed(&self) -> bool { + self.runtime + .get() + .is_some_and(|runtime| !runtime.closed.load(Ordering::Acquire) && runtime.manager_tx.is_closed()) + } + pub fn new( allocator: Arc, disk_writer: Arc, @@ -283,6 +288,19 @@ impl SmallWritePool { .await .map_err(|_| IoError::Internal("small-write manager shutdown was lost".into()))? } + + #[cfg(feature = "test-util")] + pub async fn stop_manager_for_test(self: &Arc) -> Result<()> { + let runtime = self.runtime().await?; + let (done_tx, done_rx) = oneshot::channel(); + runtime + .manager_tx + .send(ManagerCommand::StopForTest(done_tx)) + .map_err(|_| IoError::Internal("small-write manager already stopped".into()))?; + done_rx + .await + .map_err(|_| IoError::Internal("small-write manager test stop was lost".into())) + } } impl Drop for SmallWritePool { diff --git a/lib/crowdb-chunk-client/tests/chunk_writer_test.rs b/lib/crowdb-chunk-client/tests/chunk_writer_test.rs index a29569547..7bcee3098 100644 --- a/lib/crowdb-chunk-client/tests/chunk_writer_test.rs +++ b/lib/crowdb-chunk-client/tests/chunk_writer_test.rs @@ -20,13 +20,17 @@ use crowdb_test_harness::test_dirs; use async_trait::async_trait; use bytes::Bytes; -use crowdb_chunk_client::{ChunkAllocator, ChunkClientConfig, ChunkWriter, DiskWriter, IoError, Result}; +use crowdb_chunk_client::{ + ChunkAllocator, ChunkClientConfig, ChunkReadPolicy, ChunkReader, ChunkWriter, DiskWriter, IoError, Result, +}; use crowdb_common::ec::EcScheme; +use crowdb_diskio_client::DiskId; use crowdb_protocol::chunkdb::rpc::Strip as StripOneof; use crowdb_protocol::chunkdb::rpc::{ - AllocateChunkRequest, AllocateChunkResponse, AppendChunkRequest, AppendChunkResponse, Chunk, ChunkStrip, - ChunkType, DeleteChunkRequest, DeleteChunkResponse, EcStrip, QueryChunkRequest, QueryChunkResponse, - SealChunkRequest, SealChunkResponse, StripType, UpdateChunkStripRequest, UpdateChunkStripResponse, + AllocateChunkRequest, AllocateChunkResponse, AppendChunkRequest, AppendChunkResponse, Chunk, ChunkState, + ChunkStrip, ChunkType, DeleteChunkRequest, DeleteChunkResponse, EcStrip, MirrorStrip, QueryChunkRequest, + QueryChunkResponse, SealChunkRequest, SealChunkResponse, StripType, UpdateChunkStripRequest, + UpdateChunkStripResponse, }; use crowdb_protocol::common::{ChunkId, DiskId as ProtoDiskId}; use crowdb_protocol::diskdb::rpc::Segment; @@ -49,6 +53,78 @@ struct ConcurrentDiskWriter { max_inflight: AtomicUsize, } +#[derive(Default)] +struct RejectingDiskWriter { + attempts: AtomicUsize, +} + +#[derive(Debug)] +struct OrderedMirrorDiskWriter { + first_release: tokio::sync::Semaphore, + first_four_started: tokio::sync::Barrier, + inflight: AtomicUsize, + max_inflight: AtomicUsize, + completed: Mutex>, +} + +impl Default for OrderedMirrorDiskWriter { + fn default() -> Self { + Self { + first_release: tokio::sync::Semaphore::new(0), + first_four_started: tokio::sync::Barrier::new(4), + inflight: AtomicUsize::new(0), + max_inflight: AtomicUsize::new(0), + completed: Mutex::new(Vec::new()), + } + } +} + +#[async_trait] +impl DiskWriter for OrderedMirrorDiskWriter { + async fn write(&self, segment: &Segment, unit_bytes: u64, data: Bytes) -> Result<()> { + self.write_at_byte_offset(segment, unit_bytes, 0, data).await + } + + async fn write_at_byte_offset( + &self, + segment: &Segment, + _unit_bytes: u64, + _byte_offset: u64, + _data: Bytes, + ) -> Result<()> { + let inflight = self.inflight.fetch_add(1, Ordering::Relaxed) + 1; + self.max_inflight.fetch_max(inflight, Ordering::Relaxed); + if segment.unit_offset < 4 { + self.first_four_started.wait().await; + } + if segment.unit_offset == 0 { + self.first_release.acquire().await.unwrap().forget(); + } + self.completed.lock().unwrap().push(segment.unit_offset); + self.inflight.fetch_sub(1, Ordering::Relaxed); + Ok(()) + } +} + +#[async_trait] +impl DiskWriter for RejectingDiskWriter { + async fn write(&self, _seg: &Segment, _unit_bytes: u64, _data: Bytes) -> Result<()> { + self.attempts.fetch_add(1, Ordering::Relaxed); + Err(IoError::WriteFailed("injected single-copy failure".into())) + } + + async fn write_at_byte_offset( + &self, + _seg: &Segment, + _unit_bytes: u64, + _byte_offset: u64, + _data: Bytes, + ) -> Result<()> { + self.attempts.fetch_add(1, Ordering::Relaxed); + Err(IoError::WriteFailed("injected single-copy failure".into())) + } +} + #[async_trait] impl DiskWriter for ConcurrentDiskWriter { async fn write(&self, _seg: &Segment, _unit_bytes: u64, _data: Bytes) -> Result<()> { @@ -217,7 +293,7 @@ impl ChunkAllocator for MockChunkAllocator { capacity: strips.iter().map(|strip| strip.capacity).sum(), sealed_length: 0, strips: strips.clone(), - chunk_type: ChunkType::Repo as i32, + chunk_type: req.chunk_type, writer_epoch: req.writer_epoch, acknowledged_cursor: 0, closed_strip_sequence: None, @@ -297,10 +373,29 @@ impl ChunkAllocator for MockChunkAllocator { Ok(UpdateChunkStripResponse { chunk: None }) } - async fn query_chunk(&self, _req: QueryChunkRequest) -> Result { + async fn query_chunk(&self, req: QueryChunkRequest) -> Result { + let chunk_id = req.chunk_id.unwrap(); + let state = self.state.lock().unwrap(); + let (strips, length, deleted) = state + .chunks + .get(&(chunk_id.high, chunk_id.low)) + .ok_or_else(|| IoError::MetadataConflict("chunk is missing".into()))?; + let chunk = Chunk { + id: Some(chunk_id), + strips: strips.clone(), + capacity: strips.iter().map(|strip| strip.capacity).sum(), + state: if *deleted { + ChunkState::Deleted as i32 + } else { + ChunkState::Sealed as i32 + }, + sealed_length: *length, + acknowledged_cursor: u64::from(*length) * 1024, + ..Chunk::default() + }; Ok(QueryChunkResponse { - chunk: None, - layout_validity_ms: 0, + chunk: Some(chunk), + layout_validity_ms: 30_000, }) } } @@ -309,8 +404,13 @@ impl ChunkAllocator for MockChunkAllocator { fn test_config(max_chunk_size: u64) -> Arc { Arc::new(ChunkClientConfig { + chunk_type: crowdb_protocol::chunkdb::rpc::ChunkType::default(), + large_mirror_copies: None, max_chunk_size, prefetch_strips_per_chunk: 2, + large_prefetch_max_strips_per_batch: 32, + large_parallel_strip_writes: 4, + large_held_buffers: 4, parity_depth: 2, chunk_preparation_depth: 1, large_write_repair_attempts: 3, @@ -320,6 +420,266 @@ fn test_config(max_chunk_size: u64) -> Arc { }) } +#[tokio::test] +async fn large_chunk_prefetch_preserves_type_in_id_and_metadata() { + let prefetch = crowdb_chunk_client::ChunkPrefetch::new( + Arc::new(MockChunkAllocator::new()), + ec_4_1(), + test_config(1024 * 1024), + crowdb_protocol::CHUNK_TYPE_S3, + ); + let chunk = prefetch.on_demand().await.unwrap(); + assert_eq!( + chunk.id.unwrap().high >> 56, + u64::from(crowdb_protocol::CHUNK_TYPE_S3) + ); + assert_eq!(chunk.chunk_type, ChunkType::S3 as i32); + + let (mut prepared, task) = prefetch.spawn(Some(2 * 1024 * 1024)); + let first = prepared.recv().await.unwrap().unwrap(); + let second = prepared.recv().await.unwrap().unwrap(); + assert_ne!(first.id, second.id); + for prepared_chunk in [first, second] { + assert_eq!( + prepared_chunk.id.unwrap().high >> 56, + u64::from(crowdb_protocol::CHUNK_TYPE_S3) + ); + assert_eq!(prepared_chunk.chunk_type, ChunkType::S3 as i32); + } + task.await.unwrap(); +} + +#[tokio::test] +async fn chunk_writer_crosses_mirror_and_ec_strip_boundaries() { + let allocator = Arc::new(MockChunkAllocator::new()); + let temp = test_dirs::tempdir_in_test_data("chunk-client"); + let disk = Arc::new(LocalFileDiskWriter::new(temp.path())); + let chunk_id = ChunkId { high: 1, low: 9 }; + let mut offset = 0; + let mut mirror_segments = make_segments(chunk_id, 1, &mut offset); + mirror_segments[0].unit_count = 2; + let mirror = ChunkStrip { + unit_kb: 4, + capacity: 8, + strip_sequence: 0, + strip_type: StripType::Mirror as i32, + strip: Some(StripOneof::MirrorStrip(MirrorStrip { + segments: mirror_segments.clone(), + })), + ..ChunkStrip::default() + }; + let ec_segments = make_segments(chunk_id, 2, &mut offset); + let mut ec = make_strip(1, 1, 1, ec_segments.clone()); + ec.chunk_offset = 8; + ec.capacity = 4; + let strips = vec![mirror, ec]; + allocator + .state + .lock() + .unwrap() + .chunks + .insert((chunk_id.high, chunk_id.low), (strips.clone(), 0, false)); + let chunk = Chunk { + id: Some(chunk_id), + strips, + capacity: 12, + modify_ts: 1, + ..Chunk::default() + }; + let mut config = (*test_config(16 * 1024)).clone(); + config.read_buffer_size = 4 * 1024; + let mut writer = ChunkWriter::new( + allocator.clone(), + disk.clone(), + EcScheme::new(1, 1), + Arc::new(config), + ); + writer.open(chunk, Some(12 * 1024)).unwrap(); + writer.push(Bytes::from(vec![5; 12 * 1024])).await.unwrap(); + let location = writer.seal().await.unwrap(); + assert_eq!(location.length, 12 * 1024); + assert_eq!( + disk.read_block( + DiskId::new( + mirror_segments[0].disk_id.unwrap().high, + mirror_segments[0].disk_id.unwrap().low + ), + 0, + 8 * 1024, + ) + .unwrap(), + vec![5; 8 * 1024] + ); + assert_eq!( + disk.read_block( + DiskId::new( + ec_segments[0].disk_id.unwrap().high, + ec_segments[0].disk_id.unwrap().low + ), + 4 * 1024, + 4 * 1024, + ) + .unwrap(), + vec![5; 4 * 1024] + ); + drop(writer); + drop(disk); + let reopened_disk = Arc::new(LocalFileDiskWriter::new(temp.path())); + let reader = ChunkReader::new(allocator, reopened_disk, ChunkReadPolicy::default()).unwrap(); + assert_eq!( + reader + .read_object(std::slice::from_ref(&location)) + .await + .unwrap() + .concat(), + vec![5; 12 * 1024] + ); +} + +#[tokio::test] +async fn single_copy_mirror_write_reports_error_at_finish() { + let chunk_id = ChunkId { high: 1, low: 10 }; + let mut offset = 0; + let strip = ChunkStrip { + unit_kb: 4, + capacity: 4, + strip_type: StripType::Mirror as i32, + strip: Some(StripOneof::MirrorStrip(MirrorStrip { + segments: make_segments(chunk_id, 1, &mut offset), + })), + ..ChunkStrip::default() + }; + let chunk = Chunk { + id: Some(chunk_id), + strips: vec![strip], + capacity: 4, + ..Chunk::default() + }; + let disk = Arc::new(RejectingDiskWriter::default()); + let mut writer = ChunkWriter::new( + Arc::new(MockChunkAllocator::new()), + disk.clone(), + EcScheme::new(2, 1), + test_config(4 * 1024), + ); + writer.open(chunk, Some(1024)).unwrap(); + writer.push(Bytes::from(vec![7; 1024])).await.unwrap(); + assert!(writer.seal().await.is_err()); + assert_eq!(disk.attempts.load(Ordering::Relaxed), 1); +} + +#[tokio::test] +async fn mirror_strip_writes_keep_four_in_flight_and_commit_in_order() { + let chunk_id = ChunkId { high: 1, low: 11 }; + let mut offset = 0; + let strips = (0..5) + .map(|index| ChunkStrip { + chunk_offset: index * 4, + strip_sequence: index, + unit_kb: 4, + capacity: 4, + strip_type: StripType::Mirror as i32, + strip: Some(StripOneof::MirrorStrip(MirrorStrip { + segments: make_segments(chunk_id, 1, &mut offset), + })), + ..ChunkStrip::default() + }) + .collect::>(); + let allocator = MockChunkAllocator::new(); + allocator + .state + .lock() + .unwrap() + .chunks + .insert((chunk_id.high, chunk_id.low), (strips.clone(), 0, false)); + let disk = Arc::new(OrderedMirrorDiskWriter::default()); + let mut writer = ChunkWriter::new( + Arc::new(allocator.clone()), + disk.clone(), + EcScheme::new(2, 1), + test_config(5 * UNIT_BYTES), + ); + writer + .open( + Chunk { + id: Some(chunk_id), + strips, + capacity: 20, + ..Chunk::default() + }, + Some(5 * UNIT_BYTES), + ) + .unwrap(); + for index in 0..4 { + writer.push(block(index, UNIT_BYTES as usize)).await.unwrap(); + } + { + let fifth = writer.push(block(4, UNIT_BYTES as usize)); + tokio::pin!(fifth); + tokio::select! { + result = &mut fifth => panic!("fifth write passed the four-write window: {result:?}"), + () = tokio::time::sleep(std::time::Duration::from_millis(10)) => {} + } + assert_eq!(disk.max_inflight.load(Ordering::Relaxed), 4); + let mut completed = disk.completed.lock().unwrap().clone(); + completed.sort_unstable(); + assert_eq!(completed, vec![1, 2, 3]); + assert_eq!(allocator.snapshot().seal_calls, 0); + disk.first_release.add_permits(1); + fifth.await.unwrap(); + } + assert_eq!(writer.seal().await.unwrap().length, 5 * UNIT_BYTES); + assert_eq!(disk.max_inflight.load(Ordering::Relaxed), 4); + assert_eq!(allocator.snapshot().seal_calls, 1); +} + +#[tokio::test] +async fn failed_full_mirror_strip_cannot_seal_after_async_dispatch() { + let chunk_id = ChunkId { high: 1, low: 12 }; + let mut offset = 0; + let strip = ChunkStrip { + unit_kb: 4, + capacity: 4, + strip_type: StripType::Mirror as i32, + strip: Some(StripOneof::MirrorStrip(MirrorStrip { + segments: make_segments(chunk_id, 1, &mut offset), + })), + ..ChunkStrip::default() + }; + let allocator = MockChunkAllocator::new(); + allocator + .state + .lock() + .unwrap() + .chunks + .insert((chunk_id.high, chunk_id.low), (vec![strip.clone()], 0, false)); + let disk = Arc::new(RejectingDiskWriter::default()); + let mut writer = ChunkWriter::new( + Arc::new(allocator.clone()), + disk.clone(), + EcScheme::new(2, 1), + test_config(UNIT_BYTES), + ); + writer + .open( + Chunk { + id: Some(chunk_id), + strips: vec![strip], + capacity: 4, + ..Chunk::default() + }, + Some(UNIT_BYTES), + ) + .unwrap(); + writer.push(block(7, UNIT_BYTES as usize)).await.unwrap(); + let result = writer.seal().await; + assert!(result.is_err(), "faulty mirror unexpectedly sealed: {result:?}"); + assert_eq!(allocator.snapshot().seal_calls, 0); + writer.abort().await.unwrap(); + assert!(allocator.snapshot().delete_calls >= 1); + assert_eq!(disk.attempts.load(Ordering::Relaxed), 1); +} + fn ec_4_1() -> EcScheme { EcScheme::new(4, 1) } diff --git a/lib/crowdb-chunk-client/tests/common/e2e_stack.rs b/lib/crowdb-chunk-client/tests/common/e2e_stack.rs index 0ca4ffdfd..22a6a836a 100644 --- a/lib/crowdb-chunk-client/tests/common/e2e_stack.rs +++ b/lib/crowdb-chunk-client/tests/common/e2e_stack.rs @@ -43,7 +43,7 @@ pub struct E2eStack { _permit: OwnedSemaphorePermit, pub cluster: KvCluster, _diskdb: DiskdbProcess, - _diskio: DiskioProcess, + diskio: DiskioProcess, #[allow(dead_code)] chunkdb: Option, #[allow(dead_code)] @@ -59,6 +59,23 @@ impl E2eStack { Self::start_with_disk_and_chunkdb_options( small_write, "mem", + 0.0, + ChunkdbStartOptions { + allow_unsafe_ec: true, + allow_degraded_failure_domains: true, + repair_allow_unsafe_placement: true, + ..ChunkdbStartOptions::default() + }, + ) + .await + } + + #[allow(dead_code)] + pub async fn start_with_diskio_fault_rate(small_write: SmallWritePolicy, fault_error_rate: f64) -> Self { + Self::start_with_disk_and_chunkdb_options( + small_write, + "mem", + fault_error_rate, ChunkdbStartOptions { allow_unsafe_ec: true, allow_degraded_failure_domains: true, @@ -74,6 +91,7 @@ impl E2eStack { Self::start_with_disk_and_chunkdb_options( small_write, "null", + 0.0, ChunkdbStartOptions { allow_unsafe_ec: true, allow_degraded_failure_domains: true, @@ -89,12 +107,13 @@ impl E2eStack { small_write: SmallWritePolicy, chunkdb_options: ChunkdbStartOptions, ) -> Self { - Self::start_with_disk_and_chunkdb_options(small_write, "mem", chunkdb_options).await + Self::start_with_disk_and_chunkdb_options(small_write, "mem", 0.0, chunkdb_options).await } async fn start_with_disk_and_chunkdb_options( small_write: SmallWritePolicy, dummy_disk: &str, + fault_error_rate: f64, chunkdb_options: ChunkdbStartOptions, ) -> Self { let permit = E2E_STACK_PERMITS @@ -113,7 +132,7 @@ impl E2eStack { dummy_disk, kv_seeds: &cluster.mgmt_endpoints, disks: &[], - fault_error_rate: 0.0, + fault_error_rate, fault_latency_ms: None, no_o_direct: false, }); @@ -157,7 +176,7 @@ impl E2eStack { _permit: permit, cluster, _diskdb: diskdb, - _diskio: diskio, + diskio, chunkdb: Some(chunkdb), chunkdb_options, client, @@ -204,6 +223,12 @@ impl E2eStack { .await; } + #[allow(dead_code)] + pub fn crash_diskio(&mut self) { + self.diskio.child.kill().expect("kill DiskIO process"); + self.diskio.child.wait().expect("reap DiskIO process"); + } + #[allow(dead_code)] pub async fn crash_and_restart_chunkdb_with_options(&mut self, options: ChunkdbStartOptions) { let mut chunkdb = self.chunkdb.take().expect("chunkdb is running"); @@ -226,7 +251,17 @@ impl E2eStack { ))); let service = ServiceRegistryClient::from_shared(kv); let chunkdb = ChunkdbClient::new(service, Arc::new(ChunkdbRpcTransport::new())); - chunkdb.refresh_endpoints().await.unwrap(); + let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(10); + loop { + match chunkdb.refresh_endpoints().await { + Ok(()) => break, + Err(error) if tokio::time::Instant::now() < deadline => { + eprintln!("waiting for ChunkDB discovery after KV recovery: {error}"); + tokio::time::sleep(std::time::Duration::from_millis(100)).await; + } + Err(error) => panic!("ChunkDB discovery did not recover: {error}"), + } + } let response = chunkdb .query_chunk(QueryChunkRequest { chunk_id: location.chunk_id, diff --git a/lib/crowdb-chunk-client/tests/common/mod.rs b/lib/crowdb-chunk-client/tests/common/mod.rs index ee733662e..0e8d8f274 100644 --- a/lib/crowdb-chunk-client/tests/common/mod.rs +++ b/lib/crowdb-chunk-client/tests/common/mod.rs @@ -116,4 +116,16 @@ impl DiskWriter for LocalFileDiskWriter { "byte-offset writes not supported by this writer".into(), )) } + + async fn read(&self, seg: &Segment, unit_bytes: u64, offset: u64, length: u32) -> Result { + let disk_id = seg + .disk_id + .ok_or_else(|| IoError::ReadFailed("segment missing disk_id".into()))?; + let bytes = self.read_block( + DiskId::new(disk_id.high, disk_id.low), + seg.unit_offset * unit_bytes + offset, + usize::try_from(length).unwrap(), + )?; + Ok(Bytes::from(bytes)) + } } diff --git a/lib/crowdb-chunk-client/tests/common/small_durable.rs b/lib/crowdb-chunk-client/tests/common/small_durable.rs index 859740885..9f7d558bf 100644 --- a/lib/crowdb-chunk-client/tests/common/small_durable.rs +++ b/lib/crowdb-chunk-client/tests/common/small_durable.rs @@ -7,6 +7,16 @@ struct TestWriteIntent { length: AtomicU64, } +struct CountingIntent(AtomicUsize); + +#[async_trait] +impl crowdb_chunk_client::SmallWriteIntent for CountingIntent { + async fn before_write(&self, _location: &crowdb_protocol::chunkdb::rpc::Location) -> Result<()> { + self.0.fetch_add(1, Ordering::Relaxed); + Ok(()) + } +} + #[async_trait] impl crowdb_chunk_client::SmallWriteIntent for TestWriteIntent { async fn before_write(&self, location: &crowdb_protocol::chunkdb::rpc::Location) -> Result<()> { @@ -66,6 +76,25 @@ async fn failed_intent_never_writes_object_bytes_or_returns_a_location() { } } +#[tokio::test] +async fn durable_intent_is_not_relocated_after_mirror_repair_exhaustion() { + let (client, _, disk) = client(policy()); + disk.fail.store(true, Ordering::Relaxed); + let intent = Arc::new(CountingIntent(AtomicUsize::new(0))); + let mut writer = client.prepare_small_write(16).await.unwrap(); + writer + .on_data(Bytes::from_static(b"0123456789abcdef")) + .await + .unwrap(); + assert!(matches!( + writer.finish_durable_with_intent(intent.clone()).await, + Err(IoError::WriteFailed(_)) + )); + assert_eq!(intent.0.load(Ordering::Relaxed), 1); + assert_eq!(client.small_write_metrics().exhausted_repairs, 1); + let _ = client.shutdown_small_writes().await; +} + #[tokio::test] async fn durable_completion_does_not_publish_a_location_after_cursor_failure() { let (client, allocator, disk) = client(policy()); diff --git a/lib/crowdb-chunk-client/tests/config_validation_test.rs b/lib/crowdb-chunk-client/tests/config_validation_test.rs new file mode 100644 index 000000000..938032bf2 --- /dev/null +++ b/lib/crowdb-chunk-client/tests/config_validation_test.rs @@ -0,0 +1,19 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_chunk_client::ChunkClientConfig; + +#[test] +fn large_mirror_copy_count_is_bounded_to_five() { + let mut config = ChunkClientConfig { + large_mirror_copies: Some(5), + ..ChunkClientConfig::default() + }; + assert!(config.validate().is_ok()); + + config.large_mirror_copies = Some(6); + assert!(config.validate().is_err()); + + config.large_mirror_copies = Some(0); + assert!(config.validate().is_err()); +} diff --git a/lib/crowdb-chunk-client/tests/large_object_writer_e2e.rs b/lib/crowdb-chunk-client/tests/large_object_writer_e2e.rs index 009ddff64..b1b94eec1 100644 --- a/lib/crowdb-chunk-client/tests/large_object_writer_e2e.rs +++ b/lib/crowdb-chunk-client/tests/large_object_writer_e2e.rs @@ -6,26 +6,82 @@ #[path = "common/e2e_stack.rs"] mod e2e_stack; -use std::sync::atomic::{AtomicUsize, Ordering}; +use std::ops::Range; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use std::sync::{Arc, Mutex}; +use std::time::Duration; use async_trait::async_trait; use bytes::Bytes; use crowdb_chunk_client::{ - ChunkClientConfig, ChunkIoClient, ChunkReadPolicy, DiskWriter, IoError, LargeWritePolicy, Result, - RoutedDiskWriter, SmallWritePolicy, + ChunkClientConfig, ChunkIoClient, ChunkIoWriter, ChunkReadPolicy, DiskWriter, FramedWriteBuffer, IoError, + LargeWritePolicy, Result, RoutedDiskWriter, SmallWritePolicy, }; use crowdb_chunkdb_client::{ChunkdbClient, ChunkdbRpcTransport}; use crowdb_common::ec::{encode_parity_from_shards, EcScheme}; use crowdb_kv_client::{ClientConfig, CrowdbKvClient, HardwareClient, ServiceRegistryClient}; use crowdb_protocol::chunkdb::rpc::{Chunk, ChunkState, EcState, Location, Strip}; use crowdb_protocol::diskdb::rpc::Segment; -use crowdb_protocol::frame::{ChunkLocation, MAX_FRAME_PAYLOAD_BYTES}; +use crowdb_protocol::frame::{ + encode_frame_regions, ChunkLocation, FrameError, FrameMagic, FRAME_FOOTER_BYTES, + FRAME_HEADER_PREFIX_BYTES, MAX_FRAME_BYTES, MAX_FRAME_PAYLOAD_BYTES, +}; use e2e_stack::{all_binaries_available, E2eStack}; const MIB: usize = 1024 * 1024; +struct FullFramedOwner { + bytes: Vec, + frames: usize, +} + +impl FullFramedOwner { + fn new(frames: usize) -> Self { + let mut bytes = vec![0; frames * MAX_FRAME_BYTES]; + for index in 0..frames { + let start = index * MAX_FRAME_BYTES + FRAME_HEADER_PREFIX_BYTES; + bytes[start..start + MAX_FRAME_PAYLOAD_BYTES].fill(0x5a); + } + Self { bytes, frames } + } +} + +impl FramedWriteBuffer for FullFramedOwner { + fn logical_len(&self) -> u64 { + (self.frames * MAX_FRAME_PAYLOAD_BYTES) as u64 + } + + fn frame_count(&self) -> usize { + self.frames + } + + fn frame_payload_len(&self, index: usize) -> Option { + (index < self.frames).then_some(MAX_FRAME_PAYLOAD_BYTES) + } + + fn finalize_frame( + &mut self, + index: usize, + magic: FrameMagic, + chunk_id: crowdb_protocol::common::ChunkId, + write_time_ms: u64, + ) -> std::result::Result, FrameError> { + let start = index * MAX_FRAME_BYTES; + let end = start + MAX_FRAME_BYTES; + let frame = &mut self.bytes[start..end]; + let (header, remainder) = frame.split_at_mut(FRAME_HEADER_PREFIX_BYTES); + let (payload, footer) = remainder.split_at_mut(MAX_FRAME_PAYLOAD_BYTES); + debug_assert_eq!(footer.len(), FRAME_FOOTER_BYTES); + encode_frame_regions(magic, chunk_id, payload, write_time_ms, header, footer)?; + Ok(start..end) + } + + fn views(&self, range: Range) -> std::result::Result, FrameError> { + Ok(vec![Bytes::copy_from_slice(&self.bytes[range])]) + } +} + struct FailWriteCall { inner: Arc, calls: AtomicUsize, @@ -33,6 +89,9 @@ struct FailWriteCall { persistent: bool, failed_segment: Mutex>, segments: Mutex>, + failure_delay: Duration, + failure_pending: AtomicBool, + successes_during_failure: AtomicUsize, } impl FailWriteCall { @@ -66,12 +125,21 @@ impl DiskWriter for FailWriteCall { byte_offset: u64, data: Bytes, ) -> Result<()> { - if byte_offset % unit_bytes == 0 { - return self.write_at(seg, unit_bytes, byte_offset, data).await; + let injected = self.inject_failure(seg); + if injected.is_err() && !self.failure_delay.is_zero() { + self.failure_pending.store(true, Ordering::Release); + tokio::time::sleep(self.failure_delay).await; + self.failure_pending.store(false, Ordering::Release); + } + injected?; + let result = self + .inner + .write_at_byte_offset(seg, unit_bytes, byte_offset, data) + .await; + if result.is_ok() && self.failure_pending.load(Ordering::Acquire) { + self.successes_during_failure.fetch_add(1, Ordering::AcqRel); } - Err(IoError::WriteFailed( - "byte-offset writes not supported by this writer".into(), - )) + result } async fn read( @@ -83,6 +151,76 @@ impl DiskWriter for FailWriteCall { ) -> Result { self.inner.read(segment, unit_bytes, segment_offset, length).await } + + async fn fsync(&self, segment: &Segment) -> Result<()> { + self.inner.fsync(segment).await + } +} + +struct FailLaterWritesInFirstChunk { + inner: Arc, + first_chunk: Mutex>, + first_chunk_writes: AtomicUsize, + successful_writes: usize, +} + +impl FailLaterWritesInFirstChunk { + fn check(&self, segment: &Segment) -> Result<()> { + let chunk_id = segment + .owner_chunk + .ok_or_else(|| IoError::WriteFailed("fault segment has no chunk".into()))?; + let mut first = self.first_chunk.lock().unwrap(); + let first_id = *first.get_or_insert(chunk_id); + drop(first); + if chunk_id == first_id + && self.first_chunk_writes.fetch_add(1, Ordering::AcqRel) >= self.successful_writes + { + return Err(IoError::WriteFailed( + "old chunk is persistently unwritable".into(), + )); + } + Ok(()) + } +} + +#[async_trait] +impl DiskWriter for FailLaterWritesInFirstChunk { + async fn write(&self, segment: &Segment, unit_bytes: u64, data: Bytes) -> Result<()> { + self.check(segment)?; + self.inner.write(segment, unit_bytes, data).await + } + + async fn write_views(&self, segment: &Segment, unit_bytes: u64, data: Vec) -> Result<()> { + self.check(segment)?; + self.inner.write_views(segment, unit_bytes, data).await + } + + async fn write_at_byte_offset( + &self, + segment: &Segment, + unit_bytes: u64, + byte_offset: u64, + data: Bytes, + ) -> Result<()> { + self.check(segment)?; + self.inner + .write_at_byte_offset(segment, unit_bytes, byte_offset, data) + .await + } + + async fn read( + &self, + segment: &Segment, + unit_bytes: u64, + segment_offset: u64, + length: u32, + ) -> Result { + self.inner.read(segment, unit_bytes, segment_offset, length).await + } + + async fn fsync(&self, segment: &Segment) -> Result<()> { + self.inner.fsync(segment).await + } } fn ec_4_1() -> EcScheme { @@ -103,8 +241,13 @@ fn policy(max_chunk_size: u64) -> LargeWritePolicy { LargeWritePolicy { ec_scheme: ec_4_1(), client: Arc::new(ChunkClientConfig { + chunk_type: crowdb_protocol::chunkdb::rpc::ChunkType::default(), + large_mirror_copies: None, max_chunk_size, prefetch_strips_per_chunk: 2, + large_prefetch_max_strips_per_batch: 32, + large_parallel_strip_writes: 4, + large_held_buffers: 4, parity_depth: 2, chunk_preparation_depth: 1, large_write_repair_attempts: 3, @@ -226,6 +369,312 @@ async fn large_write_multi_strip_persists_data_metadata_and_parity() { ); } +#[tokio::test] +async fn large_one_copy_mirror_reads_across_strips() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start(small_policy()).await; + let data = make_test_data(9 * MIB); + let mut configured = policy(16 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().large_mirror_copies = Some(1); + let result = stack + .client + .prepare_large_write(Some(data.len() as u64), configured) + .write_stream(data.as_slice()) + .await + .unwrap(); + let chunk = stack.query_chunk(&result.locations[0]).await; + assert!(chunk.strips.len() >= 9); + let Strip::MirrorStrip(mirror) = chunk.strips[0].strip.as_ref().unwrap() else { + panic!("large mirror policy allocated a different strip"); + }; + assert_eq!(mirror.segments.len(), 1); + assert!(chunk.strips.iter().take(9).all(|strip| strip.sealed_length > 0)); + assert_eq!( + stack + .client + .read_object(&result.locations) + .await + .unwrap() + .concat(), + data + ); + assert_eq!( + stack + .client + .read_range(&result.locations, MIB as u64 - 100, MIB as u64 + 100) + .await + .unwrap() + .concat(), + data[MIB - 100..MIB + 100] + ); +} + +#[tokio::test] +async fn large_raw_buffers_submit_complete_mirror_strips_concurrently() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start(small_policy()).await; + let mut configured = policy(16 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().large_mirror_copies = Some(1); + let mut writer = stack.client.prepare_large_write(None, configured); + let block = Bytes::from(vec![0x5a; 16 * MAX_FRAME_PAYLOAD_BYTES]); + for _ in 0..8 { + assert_eq!( + writer.on_data(block.clone()).await.unwrap(), + crowdb_chunk_client::FeedStatus::Continue + ); + } + let locations = writer.on_finish().await.unwrap(); + let timing = writer.write_timing().unwrap(); + assert_eq!(timing.strip_write_successes, 8); + assert!(timing.mirror_uncommitted_peak > 0); + assert!(timing.mirror_active_write_peak > 0); + assert_eq!( + stack.client.read_object(&locations).await.unwrap().concat(), + block.repeat(8) + ); +} + +#[tokio::test] +async fn large_mirror_replaces_failed_replica_before_ordered_completion() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start(small_policy()).await; + let (allocator, disk_writer) = real_parts(&stack).await; + for fail_on in [1, 3, 4] { + let fault = Arc::new(FailWriteCall { + inner: disk_writer.clone(), + calls: AtomicUsize::new(0), + fail_on, + persistent: false, + failed_segment: Mutex::new(None), + segments: Mutex::new(Vec::new()), + failure_delay: Duration::ZERO, + failure_pending: AtomicBool::new(false), + successes_during_failure: AtomicUsize::new(0), + }); + let client = + ChunkIoClient::from_parts_with_small_policy(allocator.clone(), fault.clone(), small_policy()) + .unwrap(); + let data = make_test_data(4 * MIB); + let mut configured = policy(16 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().large_mirror_copies = Some(1); + let result = client + .prepare_large_write(Some(data.len() as u64), configured) + .write_stream(data.as_slice()) + .await + .unwrap(); + let failed = fault.failed_segment.lock().unwrap().expect("injected segment"); + let chunk = stack.query_chunk(&result.locations[0]).await; + assert!(chunk.strips.iter().all(|strip| { + let Some(Strip::MirrorStrip(mirror)) = &strip.strip else { + return false; + }; + !mirror.segments.contains(&failed) + })); + assert_eq!(client.large_write_repair_metrics().repaired_segments, 1); + assert_eq!( + client.read_object(&result.locations).await.unwrap().concat(), + data + ); + } +} + +#[tokio::test] +async fn large_mirror_retains_later_framed_buffers_until_failed_first_strip_is_repaired() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start(small_policy()).await; + let (allocator, disk_writer) = real_parts(&stack).await; + let fault = Arc::new(FailWriteCall { + inner: disk_writer, + calls: AtomicUsize::new(0), + fail_on: 1, + persistent: false, + failed_segment: Mutex::new(None), + segments: Mutex::new(Vec::new()), + failure_delay: Duration::from_millis(80), + failure_pending: AtomicBool::new(false), + successes_during_failure: AtomicUsize::new(0), + }); + let client = + ChunkIoClient::from_parts_with_small_policy(allocator, fault.clone(), small_policy()).unwrap(); + let mut configured = policy(16 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().large_mirror_copies = Some(1); + let frames = 65; + let expected = vec![0x5a; frames * MAX_FRAME_PAYLOAD_BYTES]; + let mut writer = client.prepare_large_write(Some(expected.len() as u64), configured); + writer + .on_framed_data(Box::new(FullFramedOwner::new(frames))) + .await + .unwrap(); + let locations = writer.on_finish().await.unwrap(); + assert!(fault.successes_during_failure.load(Ordering::Acquire) > 0); + assert_eq!(client.large_write_repair_metrics().repaired_segments, 1); + assert_eq!(client.read_object(&locations).await.unwrap().concat(), expected); +} + +#[tokio::test] +async fn large_mirror_rotates_after_repair_exhaustion_and_replays_ordered_data() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start(small_policy()).await; + let (allocator, disk_writer) = real_parts(&stack).await; + let fault = Arc::new(FailLaterWritesInFirstChunk { + inner: disk_writer, + first_chunk: Mutex::new(None), + first_chunk_writes: AtomicUsize::new(0), + successful_writes: 2, + }); + let client = + ChunkIoClient::from_parts_with_small_policy(allocator, fault.clone(), small_policy()).unwrap(); + let mut configured = policy(16 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().large_mirror_copies = Some(1); + let frames = 225; + let expected = vec![0x5a; frames * MAX_FRAME_PAYLOAD_BYTES]; + let mut writer = client.prepare_large_write(Some(expected.len() as u64), configured); + writer + .on_framed_data(Box::new(FullFramedOwner::new(frames))) + .await + .unwrap(); + let locations = writer.on_finish().await.unwrap(); + let old_id = fault.first_chunk.lock().unwrap().expect("first chunk"); + assert_ne!(locations[0].chunk_id, Some(old_id)); + let repair = client.large_write_repair_metrics(); + assert_eq!(repair.chunk_rotations, 1); + assert_eq!(repair.rotated_chunks, 1); + assert!(repair.replayed_bytes > 0); + assert!(repair.replayed_bytes < (frames * MAX_FRAME_BYTES) as u64); + assert_eq!(client.read_object(&locations).await.unwrap().concat(), expected); + assert_eq!( + stack + .query_chunk(&Location { + chunk_id: Some(old_id), + ..Location::default() + }) + .await + .state, + ChunkState::Deleted as i32 + ); +} + +#[tokio::test] +async fn large_mirror_rotates_failed_partial_final_strip() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start(small_policy()).await; + let (allocator, disk_writer) = real_parts(&stack).await; + let fault = Arc::new(FailLaterWritesInFirstChunk { + inner: disk_writer, + first_chunk: Mutex::new(None), + first_chunk_writes: AtomicUsize::new(0), + successful_writes: 0, + }); + let client = + ChunkIoClient::from_parts_with_small_policy(allocator, fault.clone(), small_policy()).unwrap(); + let mut configured = policy(16 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().large_mirror_copies = Some(1); + let expected = vec![0x5a; MAX_FRAME_PAYLOAD_BYTES]; + let mut writer = client.prepare_large_write(Some(expected.len() as u64), configured); + writer + .on_framed_data(Box::new(FullFramedOwner::new(1))) + .await + .unwrap(); + let locations = writer.on_finish().await.unwrap(); + let old_id = fault.first_chunk.lock().unwrap().expect("first chunk"); + assert_ne!(locations[0].chunk_id, Some(old_id)); + assert_eq!(client.read_object(&locations).await.unwrap().concat(), expected); +} + +#[tokio::test] +async fn large_mirror_abort_drains_delayed_submitted_write_before_delete() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start(small_policy()).await; + let (allocator, disk_writer) = real_parts(&stack).await; + let fault = Arc::new(FailWriteCall { + inner: disk_writer, + calls: AtomicUsize::new(0), + fail_on: 1, + persistent: false, + failed_segment: Mutex::new(None), + segments: Mutex::new(Vec::new()), + failure_delay: Duration::from_millis(80), + failure_pending: AtomicBool::new(false), + successes_during_failure: AtomicUsize::new(0), + }); + let client = + ChunkIoClient::from_parts_with_small_policy(allocator, fault.clone(), small_policy()).unwrap(); + let mut configured = policy(16 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().large_mirror_copies = Some(1); + let mut writer = client.prepare_large_write(Some((16 * MAX_FRAME_PAYLOAD_BYTES) as u64), configured); + writer + .on_framed_data(Box::new(FullFramedOwner::new(16))) + .await + .unwrap(); + let started = std::time::Instant::now(); + writer.on_error().await.unwrap(); + assert!(started.elapsed() >= Duration::from_millis(70)); + assert!(!fault.failure_pending.load(Ordering::Acquire)); + let old_id = fault.failed_segment.lock().unwrap().unwrap().owner_chunk; + let chunk = stack + .query_chunk(&Location { + chunk_id: old_id, + ..Location::default() + }) + .await; + assert_eq!(chunk.state, ChunkState::Deleted as i32); +} + +#[tokio::test] +async fn large_mirror_repair_exhaustion_deletes_unsealed_chunk() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start(small_policy()).await; + let (allocator, disk_writer) = real_parts(&stack).await; + let fault = Arc::new(FailWriteCall { + inner: disk_writer, + calls: AtomicUsize::new(0), + fail_on: 1, + persistent: true, + failed_segment: Mutex::new(None), + segments: Mutex::new(Vec::new()), + failure_delay: Duration::ZERO, + failure_pending: AtomicBool::new(false), + successes_during_failure: AtomicUsize::new(0), + }); + let client = + ChunkIoClient::from_parts_with_small_policy(allocator, fault.clone(), small_policy()).unwrap(); + let mut configured = policy(16 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().large_mirror_copies = Some(1); + let result = client + .prepare_large_write(Some(MIB as u64), configured) + .write_stream(make_test_data(MIB).as_slice()) + .await; + assert!(matches!(result, Err(IoError::ReplicaRepairExhausted(_)))); + let repair = client.large_write_repair_metrics(); + assert!(repair.exhausted >= 1); + assert_eq!(repair.chunk_rotations, 1); + assert_eq!(repair.rotated_chunks, 0); + let failed = fault.failed_segment.lock().unwrap().expect("injected segment"); + let chunk = stack + .query_chunk(&Location { + chunk_id: failed.owner_chunk, + ..Location::default() + }) + .await; + assert_eq!(chunk.state, ChunkState::Deleted as i32); +} + #[tokio::test] async fn large_write_rotates_chunks_without_losing_data() { if !all_binaries_available() { @@ -233,9 +682,12 @@ async fn large_write_rotates_chunks_without_losing_data() { } let stack = E2eStack::start(small_policy()).await; let data = make_test_data(20 * MIB); + let mut configured = policy(8 * MIB as u64); + Arc::get_mut(&mut configured.client).unwrap().chunk_type = + crowdb_protocol::chunkdb::rpc::ChunkType::IcebergTable; let result = stack .client - .prepare_large_write(Some(data.len() as u64), policy(8 * MIB as u64)) + .prepare_large_write(Some(data.len() as u64), configured) .write_stream(data.as_slice()) .await .unwrap(); @@ -257,6 +709,14 @@ async fn large_write_rotates_chunks_without_losing_data() { let mut read_back = Vec::new(); for location in &result.locations { let chunk = stack.query_chunk(location).await; + assert_eq!( + chunk.chunk_type, + crowdb_protocol::chunkdb::rpc::ChunkType::IcebergTable as i32 + ); + assert_eq!( + chunk.id.unwrap().high >> 56, + crowdb_protocol::chunkdb::rpc::ChunkType::IcebergTable as u64 + ); assert_eq!(chunk.state, ChunkState::Sealed as i32); assert_eq!( chunk.sealed_length, @@ -373,6 +833,9 @@ async fn large_write_replaces_failed_data_and_parity_segments_end_to_end() { persistent: false, failed_segment: Mutex::new(None), segments: Mutex::new(Vec::new()), + failure_delay: Duration::ZERO, + failure_pending: AtomicBool::new(false), + successes_during_failure: AtomicUsize::new(0), }); let client = ChunkIoClient::from_parts_with_small_policy(allocator.clone(), fault.clone(), small_policy()) @@ -420,6 +883,9 @@ async fn large_write_repair_exhaustion_deletes_unsealed_chunk() { persistent: true, failed_segment: Mutex::new(None), segments: Mutex::new(Vec::new()), + failure_delay: Duration::ZERO, + failure_pending: AtomicBool::new(false), + successes_during_failure: AtomicUsize::new(0), }); let client = ChunkIoClient::from_parts_with_small_policy(allocator, fault.clone(), small_policy()).unwrap(); @@ -427,7 +893,7 @@ async fn large_write_repair_exhaustion_deletes_unsealed_chunk() { .prepare_large_write(Some(MIB as u64), policy(16 * MIB as u64)) .write_stream(make_test_data(MIB).as_slice()) .await; - assert!(matches!(result, Err(IoError::WriteFailed(_)))); + assert!(matches!(result, Err(IoError::ReplicaRepairExhausted(_)))); assert_eq!(client.large_write_repair_metrics().exhausted, 1); let failed = fault.failed_segment.lock().unwrap().expect("injected segment"); let location = Location { diff --git a/lib/crowdb-chunk-client/tests/mirror_strip_writer_test.rs b/lib/crowdb-chunk-client/tests/mirror_strip_writer_test.rs new file mode 100644 index 000000000..fda95a3e7 --- /dev/null +++ b/lib/crowdb-chunk-client/tests/mirror_strip_writer_test.rs @@ -0,0 +1,95 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use async_trait::async_trait; +use bytes::Bytes; +use crowdb_chunk_client::{DiskWriter, IoError, MirrorStripWriter, Result}; +use crowdb_protocol::chunkdb::rpc::{Chunk, ChunkStrip, MirrorStrip, Strip}; +use crowdb_protocol::common::{ChunkId, DiskId}; +use crowdb_protocol::diskdb::rpc::Segment; + +#[derive(Default)] +struct TestDiskWriter { + data: Mutex>>, + fail_disk: Option, +} + +#[async_trait] +impl DiskWriter for TestDiskWriter { + async fn write(&self, segment: &Segment, unit_bytes: u64, data: Bytes) -> Result<()> { + self.write_at_byte_offset(segment, unit_bytes, 0, data).await + } + + async fn write_at_byte_offset( + &self, + segment: &Segment, + _unit_bytes: u64, + offset: u64, + data: Bytes, + ) -> Result<()> { + let disk = segment.disk_id.unwrap().high; + if self.fail_disk == Some(disk) { + return Err(IoError::WriteFailed("injected mirror failure".into())); + } + let mut all = self.data.lock().unwrap(); + let target = all.entry(disk).or_default(); + let start = usize::try_from(offset).unwrap(); + target.resize(target.len().max(start + data.len()), 0); + target[start..start + data.len()].copy_from_slice(&data); + Ok(()) + } +} + +fn chunk() -> Arc { + Arc::new(Chunk { + id: Some(ChunkId { high: 1, low: 2 }), + strips: vec![ChunkStrip { + unit_kb: 4, + capacity: 8, + strip: Some(Strip::MirrorStrip(MirrorStrip { + segments: [11, 12] + .map(|high| Segment { + disk_id: Some(DiskId { high, low: 0 }), + unit_count: 2, + ..Segment::default() + }) + .to_vec(), + })), + ..ChunkStrip::default() + }], + ..Chunk::default() + }) +} + +#[tokio::test] +async fn mirror_strip_writes_unaligned_inputs_to_every_copy() { + let disk = Arc::new(TestDiskWriter::default()); + let mut writer = MirrorStripWriter::new(chunk(), 0, disk.clone()); + writer.push(Bytes::from(vec![3; 3 * 1024])).await.unwrap(); + writer.push(Bytes::from(vec![7; 5 * 1024])).await.unwrap(); + assert!(!writer.ready()); + assert_eq!(writer.finish().await.unwrap().bytes_written, 8 * 1024); + let copies = disk.data.lock().unwrap(); + for id in [11, 12] { + assert_eq!(&copies[&id][..3 * 1024], vec![3; 3 * 1024]); + assert_eq!(&copies[&id][3 * 1024..], vec![7; 5 * 1024]); + } +} + +#[tokio::test] +async fn mirror_strip_keeps_writing_surviving_copies_after_a_failure() { + let disk = Arc::new(TestDiskWriter { + fail_disk: Some(12), + ..TestDiskWriter::default() + }); + let mut writer = MirrorStripWriter::new(chunk(), 0, disk.clone()); + writer.push(Bytes::from_static(b"data")).await.unwrap(); + writer.push(Bytes::from_static(b"more")).await.unwrap(); + assert_eq!(writer.finish().await.unwrap().bytes_written, 8); + let copies = disk.data.lock().unwrap(); + assert_eq!(&copies[&11], b"datamore"); + assert!(!copies.contains_key(&12)); +} diff --git a/lib/crowdb-chunk-client/tests/small_object_test.rs b/lib/crowdb-chunk-client/tests/small_object_test.rs index d6ed116dd..c4940d90b 100644 --- a/lib/crowdb-chunk-client/tests/small_object_test.rs +++ b/lib/crowdb-chunk-client/tests/small_object_test.rs @@ -15,7 +15,7 @@ use crowdb_chunk_client::{ use crowdb_protocol::chunkdb::rpc::{ AdvanceChunkWriteRequest, AdvanceChunkWriteResponse, AllocateChunkRequest, AllocateChunkResponse, AllocateReplacementSegmentRequest, AllocateReplacementSegmentResponse, AppendChunkRequest, - AppendChunkResponse, Chunk, ChunkState, ChunkStrip, DeleteChunkRequest, DeleteChunkResponse, + AppendChunkResponse, Chunk, ChunkState, ChunkStrip, ChunkType, DeleteChunkRequest, DeleteChunkResponse, DiscardReplacementSegmentRequest, DiscardReplacementSegmentResponse, MirrorStrip, MutateStripReservationRequest, MutateStripReservationResponse, QueryChunkRequest, QueryChunkResponse, ReplaceChunkStripRangeRequest, ReplaceChunkStripRangeResponse, ReserveStripGroupRequest, @@ -100,7 +100,10 @@ impl ChunkAllocator for MockAllocator { if self.fail_on_attempt.load(Ordering::Relaxed) == low { return Err(IoError::AllocationFailed("injected allocation failure".into())); } - let chunk_id = req.chunk_id.unwrap_or(ChunkId { high: 7, low }); + let chunk_id = req.chunk_id.unwrap_or(ChunkId { + high: u64::try_from(req.chunk_type).unwrap() << 56, + low, + }); let strips: Vec<_> = (0..req.strip_count.max(1)) .map(|sequence| make_strip(chunk_id, sequence, req.copy_count.max(1))) .collect(); @@ -494,6 +497,32 @@ struct SelectiveFailureDiskWriter { writes: Mutex>, } +struct FailFirstChunkDiskWriter { + inner: Arc, +} + +#[async_trait] +impl DiskWriter for FailFirstChunkDiskWriter { + async fn write(&self, seg: &Segment, unit_bytes: u64, data: Bytes) -> Result<()> { + self.write_at_byte_offset(seg, unit_bytes, 0, data).await + } + + async fn write_at_byte_offset( + &self, + seg: &Segment, + unit_bytes: u64, + byte_offset: u64, + data: Bytes, + ) -> Result<()> { + if seg.owner_chunk.is_some_and(|chunk| chunk.low == 1) { + return Err(IoError::WriteFailed("first chunk disk failure".into())); + } + self.inner + .write_at_byte_offset(seg, unit_bytes, byte_offset, data) + .await + } +} + #[async_trait] impl DiskWriter for SelectiveFailureDiskWriter { async fn write(&self, seg: &Segment, _unit_bytes: u64, data: Bytes) -> Result<()> { @@ -523,6 +552,7 @@ impl DiskWriter for SelectiveFailureDiskWriter { fn policy() -> SmallWritePolicy { SmallWritePolicy { + chunk_type: crowdb_protocol::chunkdb::rpc::ChunkType::default(), object_limit: 1024 * 1024, memory_budget: 4 * 1024 * 1024, queue_capacity: 128, @@ -565,6 +595,37 @@ fn client(policy: SmallWritePolicy) -> (ChunkIoClient, Arc, Arc = allocator.state.lock().unwrap().chunks.values().cloned().collect(); + assert!(!chunks.is_empty()); + assert!(chunks + .iter() + .all(|chunk| chunk.chunk_type == ChunkType::IcebergTable as i32)); + assert!(chunks + .iter() + .all(|chunk| chunk.id.unwrap().high >> 56 == ChunkType::IcebergTable as u64)); + client.shutdown_small_writes().await.unwrap(); +} + #[tokio::test] async fn small_object_empty_finishes_without_starting_pool() { let (client, allocator, disk) = client(policy()); @@ -654,7 +715,7 @@ async fn small_object_mirror_failure_fails_every_object_without_cursor_commit() assert_eq!(client.small_write_metrics().completed, 0); assert_eq!(client.small_write_metrics().failed, 4); let (replacement_allocations, replacements, discards, _, _) = allocator.repair_snapshot(); - assert_eq!((replacement_allocations, replacements, discards), (3, 0, 3)); + assert_eq!((replacement_allocations, replacements, discards), (6, 0, 6)); disk.fail.store(false, Ordering::Relaxed); let recovered = tokio::time::timeout(Duration::from_secs(1), async { let mut writer = client.prepare_small_write(4096).await.unwrap(); @@ -689,8 +750,8 @@ async fn small_object_replacement_allocation_exhaustion_publishes_no_location() let metrics = client.small_write_metrics(); assert_eq!(metrics.completed, 0); assert_eq!(metrics.failed, 1); - assert_eq!(metrics.repair_attempts, 3); - assert_eq!(metrics.exhausted_repairs, 1); + assert_eq!(metrics.repair_attempts, 6); + assert_eq!(metrics.exhausted_repairs, 2); allocator .fail_replacement_allocations .store(false, Ordering::Relaxed); @@ -717,13 +778,13 @@ async fn small_object_metadata_exhaustion_publishes_no_location() { writer.on_data(Bytes::from(vec![8; 4096])).await.unwrap(); assert!(matches!(writer.on_finish().await, Err(IoError::WriteFailed(_)))); let (allocations, replacements, discards, _, chunks) = allocator.repair_snapshot(); - assert_eq!((allocations, replacements, discards), (1, 0, 0)); - assert_eq!(chunks[0].acknowledged_cursor, 0); + assert_eq!((allocations, replacements, discards), (2, 0, 0)); + assert!(chunks.iter().all(|chunk| chunk.acknowledged_cursor == 0)); let metrics = client.small_write_metrics(); assert_eq!(metrics.completed, 0); assert_eq!(metrics.failed, 1); - assert_eq!(metrics.repair_attempts, 3); - assert_eq!(metrics.exhausted_repairs, 1); + assert_eq!(metrics.repair_attempts, 6); + assert_eq!(metrics.exhausted_repairs, 2); allocator.fail_replacements.store(false, Ordering::Relaxed); tokio::time::timeout(Duration::from_secs(1), async { while client.small_write_metrics().pipeline_replacements == 0 { @@ -780,6 +841,149 @@ async fn small_object_repairs_two_failed_replicas_from_the_same_shadow() { client.shutdown_small_writes().await.unwrap(); } +#[tokio::test] +async fn two_copy_small_object_replaces_one_failed_replica_without_rotation() { + let allocator = Arc::new(MockAllocator::default()); + let disk = Arc::new(SelectiveFailureDiskWriter { + failed_initial_disks: vec![1], + writes: Mutex::new(Vec::new()), + }); + let mut configured = policy(); + configured.mirror_copies = 2; + let client = + ChunkIoClient::from_parts_with_small_policy(allocator.clone(), disk.clone(), configured).unwrap(); + let mut writer = client.prepare_small_write(12 * 1024).await.unwrap(); + writer.on_data(Bytes::from(vec![0x5a; 12 * 1024])).await.unwrap(); + let locations = writer.on_finish().await.unwrap(); + assert_eq!(locations.len(), 1); + + let (allocations, replacements, discards, exclusions, chunks) = allocator.repair_snapshot(); + assert_eq!((allocations, replacements, discards), (1, 1, 0)); + assert_eq!(chunks.len(), 1); + let Some(Strip::MirrorStrip(mirror)) = &chunks[0].strips[0].strip else { + panic!("expected mirror strip"); + }; + assert_eq!(mirror.segments.len(), 2); + assert!(mirror + .segments + .iter() + .all(|segment| segment.disk_id.unwrap().high != 1)); + assert!(exclusions[0].iter().any(|disk| disk.high == 1)); + { + let recorded = disk.writes.lock().unwrap(); + let retained = recorded.iter().find(|(disk, _)| *disk == 2).unwrap(); + let replaced = recorded.iter().find(|(disk, _)| *disk >= 100).unwrap(); + assert_eq!(retained.1, replaced.1); + let frame = parse_frame(&replaced.1, locations[0].chunk_id.unwrap()).unwrap(); + assert_eq!(frame.payload, vec![0x5a; 12 * 1024]); + } + assert_eq!(allocator.snapshot().0, 1); + assert_eq!(client.small_write_metrics().repairs_avoiding_rotation, 1); + assert_eq!(client.small_write_metrics().repaired_replicas, 1); + client.shutdown_small_writes().await.unwrap(); +} + +#[tokio::test] +async fn two_copy_small_write_crosses_mirror_block_boundary() { + let mut configured = policy(); + configured.mirror_copies = 2; + let (client, allocator, disk) = client(configured); + let mut first = client.prepare_small_write(4000).await.unwrap(); + first.on_data(Bytes::from(vec![0x31; 4000])).await.unwrap(); + let first = first.on_finish().await.unwrap().remove(0); + let mut second = client.prepare_small_write(100).await.unwrap(); + second.on_data(Bytes::from(vec![0x32; 100])).await.unwrap(); + let second = second.on_finish().await.unwrap().remove(0); + + assert_eq!(first.chunk_id, second.chunk_id); + assert_eq!(second.offset, frame_bytes(4000)); + assert!(second.offset < 4096 && second.offset + second.length > 4096); + let chunk_id = second.chunk_id.unwrap(); + let base = chunk_id.low * 4096 * 4096; + let mut copies = HashMap::>::new(); + { + let writes = disk.writes.lock().unwrap(); + for (disk_id, offset, data) in writes.iter() { + let relative = usize::try_from(offset - base).unwrap(); + let copy = copies.entry(*disk_id).or_default(); + copy.resize(copy.len().max(relative + data.len()), 0); + copy[relative..relative + data.len()].copy_from_slice(data); + } + } + assert_eq!(copies.len(), 2); + for copy in copies.values() { + let first_end = usize::try_from(second.offset).unwrap(); + let second_end = usize::try_from(second.offset + second.length).unwrap(); + assert_eq!( + parse_frame(©[..first_end], chunk_id).unwrap().payload, + vec![0x31; 4000] + ); + assert_eq!( + parse_frame(©[first_end..second_end], chunk_id) + .unwrap() + .payload, + vec![0x32; 100] + ); + } + assert_eq!(allocator.snapshot().2, 2); + client.shutdown_small_writes().await.unwrap(); +} + +#[tokio::test] +async fn two_copy_small_object_rotates_once_after_repair_exhaustion() { + let mut configured = policy(); + configured.mirror_copies = 2; + let (client, allocator, disk) = client(configured); + disk.fail.store(true, Ordering::Relaxed); + let mut writer = client.prepare_small_write(4096).await.unwrap(); + writer.on_data(Bytes::from(vec![0x5a; 4096])).await.unwrap(); + assert!(matches!(writer.on_finish().await, Err(IoError::WriteFailed(_)))); + let metrics = client.small_write_metrics(); + assert_eq!(metrics.completed, 0); + assert_eq!(metrics.failed, 1); + assert_eq!(metrics.exhausted_repairs, 2); + assert_eq!(metrics.repairs_avoiding_rotation, 0); + let snapshot = allocator.snapshot(); + assert!(snapshot.0 >= 2, "the failed write must rotate to a new chunk"); + assert!(snapshot.4 >= 2, "both failed chunks must be deleted"); + let _ = client.shutdown_small_writes().await; +} + +#[tokio::test] +async fn two_copy_small_object_succeeds_after_one_chunk_rotation() { + let mut configured = policy(); + configured.mirror_copies = 2; + let allocator = Arc::new(MockAllocator::default()); + let recorded = Arc::new(RecordingDiskWriter::default()); + let disk = Arc::new(FailFirstChunkDiskWriter { + inner: Arc::clone(&recorded), + }); + let client = ChunkIoClient::from_parts_with_small_policy(allocator.clone(), disk, configured).unwrap(); + let mut writer = client.prepare_small_write(4096).await.unwrap(); + writer.on_data(Bytes::from(vec![0x5a; 4096])).await.unwrap(); + let locations = writer.on_finish().await.unwrap(); + assert_eq!(locations.len(), 1); + assert_eq!(locations[0].chunk_id.unwrap().low, 2); + assert_eq!(client.small_write_metrics().exhausted_repairs, 1); + assert_eq!(client.small_write_metrics().completed, 1); + { + let state = allocator.state.lock().unwrap(); + let chunks = &state.chunks; + let high = locations[0].chunk_id.unwrap().high; + assert_eq!(chunks[&(high, 1)].state, ChunkState::Deleted as i32); + assert_eq!(chunks[&(high, 2)].state, ChunkState::Active as i32); + } + { + let recorded_images = recorded.writes.lock().unwrap(); + assert_eq!(recorded_images.len(), 2); + for (_, _, image) in recorded_images.iter() { + let frame = parse_frame(image, locations[0].chunk_id.unwrap()).unwrap(); + assert_eq!(frame.payload, vec![0x5a; 4096]); + } + } + client.shutdown_small_writes().await.unwrap(); +} + #[tokio::test] async fn small_object_retries_ambiguous_metadata_commit_without_reallocating() { let allocator = Arc::new(MockAllocator::default()); @@ -997,6 +1201,62 @@ async fn small_object_manager_scales_out_by_queued_bytes_then_drains_idle_pipeli client.shutdown_small_writes().await.unwrap(); } +#[tokio::test] +async fn protocol_small_write_pools_scale_independently() { + let mut s3_policy = policy(); + s3_policy.chunk_type = ChunkType::S3; + s3_policy.max_pipelines = 2; + s3_policy.max_batch_bytes = 16 * 1024; + s3_policy.max_batch_objects = 1; + s3_policy.scale_out_queue_bytes = 16 * 1024; + s3_policy.control_interval = Duration::from_millis(2); + s3_policy.cooldown = Duration::from_millis(1); + let mut iceberg_policy = s3_policy.clone(); + iceberg_policy.chunk_type = ChunkType::IcebergTable; + let (s3, _, s3_disk) = client(s3_policy); + let (iceberg, _, iceberg_disk) = client(iceberg_policy); + s3_disk.delay_ms.store(30, Ordering::Relaxed); + iceberg_disk.delay_ms.store(30, Ordering::Relaxed); + + let mut warmup = iceberg.prepare_small_write(1).await.unwrap(); + warmup.on_data(Bytes::from_static(b"x")).await.unwrap(); + warmup.on_finish().await.unwrap(); + let iceberg_before = iceberg.small_write_metrics(); + let mut pending_tasks = Vec::new(); + for _ in 0..12 { + let s3 = s3.clone(); + pending_tasks.push(tokio::spawn(async move { + let mut writer = s3.prepare_small_write(16 * 1024).await.unwrap(); + writer.on_data(Bytes::from(vec![1; 16 * 1024])).await.unwrap(); + writer.on_finish().await.unwrap(); + })); + } + for task in pending_tasks { + task.await.unwrap(); + } + assert!(s3.small_write_metrics().scale_out > 0); + assert_eq!(iceberg.small_write_metrics().submitted, iceberg_before.submitted); + assert_eq!(iceberg.small_write_metrics().scale_out, iceberg_before.scale_out); + + let s3_before = s3.small_write_metrics(); + let mut pending_tasks = Vec::new(); + for _ in 0..12 { + let iceberg = iceberg.clone(); + pending_tasks.push(tokio::spawn(async move { + let mut writer = iceberg.prepare_small_write(16 * 1024).await.unwrap(); + writer.on_data(Bytes::from(vec![2; 16 * 1024])).await.unwrap(); + writer.on_finish().await.unwrap(); + })); + } + for task in pending_tasks { + task.await.unwrap(); + } + assert!(iceberg.small_write_metrics().scale_out > 0); + assert_eq!(s3.small_write_metrics().submitted, s3_before.submitted); + s3.shutdown_small_writes().await.unwrap(); + iceberg.shutdown_small_writes().await.unwrap(); +} + #[tokio::test] async fn small_object_manager_scales_out_by_queued_object_count() { let mut elastic = policy(); @@ -1097,9 +1357,65 @@ async fn direct_mirror_chunk_writer_replicates_advances_and_seals() { (0, 6) ); assert_eq!(writer.cursor(), 6); - assert_eq!(disk.calls(), 3); + assert_eq!(disk.calls(), 2); assert_eq!(allocator.snapshot().2, 1); writer.seal().await.unwrap(); assert_eq!(allocator.snapshot().3, 1); } + +#[tokio::test] +async fn direct_mirror_chunk_writer_supports_five_copies() { + let allocator: Arc = Arc::new(MockAllocator::default()); + let disk = Arc::new(RecordingDiskWriter::default()); + let disk_writer: Arc = disk.clone(); + let mut writer = MirrorChunkWriter::allocate_with_copy_count( + allocator, + disk_writer, + crowdb_protocol::chunk_stream::StreamName { high: 1, low: 5 }, + 44, + 30_000, + 5, + ) + .await + .unwrap(); + writer.append(Bytes::from_static(b"stream")).await.unwrap(); + assert_eq!(disk.calls(), 5); +} + +#[tokio::test] +async fn direct_mirror_chunk_writer_accepts_protected_degraded_layout() { + let allocator: Arc = Arc::new(MockAllocator::default()); + let disk: Arc = Arc::new(RecordingDiskWriter::default()); + let stream = crowdb_protocol::chunk_stream::StreamName { high: 1, low: 3 }; + let healthy = MirrorChunkWriter::allocate_with_copy_count( + Arc::clone(&allocator), + Arc::clone(&disk), + stream, + 44, + 30_000, + 3, + ) + .await + .unwrap(); + let mut chunk = healthy.chunk().clone(); + let Some(Strip::MirrorStrip(mirror)) = &mut chunk.strips[0].strip else { + panic!("allocated stream chunk must use mirrors"); + }; + mirror.segments.pop(); + assert!(MirrorChunkWriter::open_with_copy_count( + Arc::clone(&allocator), + Arc::clone(&disk), + chunk.clone(), + stream, + 44, + 30_000, + 3, + ) + .is_ok()); + let Some(Strip::MirrorStrip(mirror)) = &mut chunk.strips[0].strip else { + panic!("allocated stream chunk must use mirrors"); + }; + mirror.segments.pop(); + assert!(MirrorChunkWriter::open_with_copy_count(allocator, disk, chunk, stream, 44, 30_000, 3).is_err()); +} diff --git a/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs b/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs index 7a9ef5532..f4ab4bc7e 100644 --- a/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs +++ b/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs @@ -19,7 +19,7 @@ use crowdb_chunkdb_client::{ChunkdbClient, ChunkdbRpcTransport}; use crowdb_common::ec::{decode, encode_parity_from_shards, EcScheme}; use crowdb_kv_client::{ClientConfig, CrowdbKvClient, HardwareClient, ServiceRegistryClient}; use crowdb_protocol::chunkdb::rpc::{ - Chunk, ChunkState, ConversionFilter, Location, Strip, TriggerConversionBatchRequest, + Chunk, ChunkState, ChunkType, ConversionFilter, Location, Strip, TriggerConversionBatchRequest, TriggerConversionRequest, }; use crowdb_protocol::common::DiskId; @@ -555,6 +555,7 @@ async fn manual_chunkdb_trigger_converts_closed_active_range_end_to_end() { let mut configured = policy(); configured.chunk_capacity = 16 * MIB as u64; configured.conversion_enabled = false; + configured.chunk_type = ChunkType::S3; let stack = E2eStack::start(configured).await; let mut object_groups = Vec::new(); for value in 11_u8..19 { @@ -577,6 +578,8 @@ async fn manual_chunkdb_trigger_converts_closed_active_range_end_to_end() { .await .expect("background close checkpoint did not reach strip 7"); assert_eq!(before.state, ChunkState::Active as i32); + assert_eq!(before.chunk_type, ChunkType::S3 as i32); + assert_eq!(before.id.unwrap().high >> 56, ChunkType::S3 as u64); assert_eq!(before.closed_strip_sequence, Some(7)); assert_eq!(before.strips.len(), 8); assert!(before @@ -609,6 +612,8 @@ async fn manual_chunkdb_trigger_converts_closed_active_range_end_to_end() { .await .expect("manual conversion task did not finish"); let strip = &converted.strips[0]; + assert_eq!(converted.id, before.id); + assert_eq!(converted.chunk_type, ChunkType::S3 as i32); let Some(Strip::EcStrip(ec)) = &strip.strip else { unreachable!(); }; @@ -903,6 +908,61 @@ async fn small_write_repairs_failed_replica_through_real_chunkdb_and_diskio() { assert_eq!(client.small_write_metrics().shadow_bytes, 0); } +#[tokio::test] +async fn single_copy_small_write_reports_real_diskio_write_failure() { + if !all_binaries_available() { + return; + } + let stack = E2eStack::start_with_diskio_fault_rate(policy(), 1.0).await; + let result = tokio::time::timeout(Duration::from_secs(15), async { + let data = Bytes::from(vec![0x5a; MAX_FRAME_PAYLOAD_BYTES]); + let mut writer = stack.client.prepare_small_write(data.len()).await?; + writer.on_data(data).await?; + writer.on_finish().await.map(|_| ()) + }) + .await + .expect("real DiskIO write failure did not return within 15 seconds"); + assert!( + matches!(result, Err(IoError::WriteFailed(_))), + "unexpected write result: {result:?}" + ); + let metrics = stack.client.small_write_metrics(); + assert_eq!(metrics.failed, 1); + assert_eq!(metrics.repairs_avoiding_rotation, 0); +} + +#[tokio::test] +async fn single_copy_read_reports_diskio_process_failure() { + if !all_binaries_available() { + return; + } + let mut stack = E2eStack::start(policy()).await; + let data = Bytes::from_static(b"diskio-read-failure"); + let location = write_object(&stack.client, data.clone()).await; + stack.client.shutdown_small_writes().await.unwrap(); + assert_eq!( + stack + .client + .read_object(std::slice::from_ref(&location)) + .await + .unwrap() + .concat() + .as_slice(), + data.as_ref() + ); + stack.crash_diskio(); + let read = tokio::time::timeout( + Duration::from_secs(15), + stack.client.read_object(std::slice::from_ref(&location)), + ) + .await + .expect("read did not return after DiskIO exited"); + assert!( + read.is_err(), + "single-copy read succeeded after its DiskIO process exited" + ); +} + #[tokio::test] async fn small_write_repair_preserves_acknowledged_prefix_in_open_block() { if !all_binaries_available() { @@ -972,7 +1032,9 @@ async fn small_write_rotates_strips_and_chunks_without_splitting_objects() { if !all_binaries_available() { return; } - let stack = E2eStack::start(policy()).await; + let mut configured = policy(); + configured.chunk_type = ChunkType::IcebergTable; + let stack = E2eStack::start(configured).await; let first_group = write_full_small_strip(&stack.client, 3).await; let second_group = write_full_small_strip(&stack.client, 5).await; let third_group = write_full_small_strip(&stack.client, 7).await; @@ -986,6 +1048,8 @@ async fn small_write_rotates_strips_and_chunks_without_splitting_objects() { assert_ne!(second.chunk_id, third.chunk_id); assert_eq!(third.offset, 0); let first_chunk = stack.query_chunk(&first).await; + assert_eq!(first_chunk.chunk_type, ChunkType::IcebergTable as i32); + assert_eq!(first_chunk.id.unwrap().high >> 56, ChunkType::IcebergTable as u64); assert_eq!(first_chunk.state, ChunkState::Sealed as i32); assert_eq!(first_chunk.strips.len(), 2); assert_eq!(first_chunk.acknowledged_cursor, 2 * MIB as u64); @@ -997,6 +1061,8 @@ async fn small_write_rotates_strips_and_chunks_without_splitting_objects() { assert_mirror_data(&stack, &first_chunk, location, data).await; } let third_chunk = stack.query_chunk(&third).await; + assert_eq!(third_chunk.chunk_type, ChunkType::IcebergTable as i32); + assert_eq!(third_chunk.id.unwrap().high >> 56, ChunkType::IcebergTable as u64); for (data, location) in &third_group { assert_mirror_data(&stack, &third_chunk, location, data).await; } diff --git a/lib/crowdb-chunk-client/tests/write_stream.rs b/lib/crowdb-chunk-client/tests/write_stream.rs index e68d09db4..a52d71e3e 100644 --- a/lib/crowdb-chunk-client/tests/write_stream.rs +++ b/lib/crowdb-chunk-client/tests/write_stream.rs @@ -229,7 +229,7 @@ impl ChunkAllocator for MockChunkAllocator { let capacity = req.write_granularity; (capacity, StripOneof::MirrorStrip(MirrorStrip { segments })) } else { - let capacity = data_num as u32; + let capacity = req.data_num.saturating_mul(4); ( capacity, StripOneof::EcStrip(EcStrip { @@ -320,7 +320,7 @@ impl ChunkAllocator for MockChunkAllocator { let capacity = req.strip_size.saturating_mul(4); (capacity, StripOneof::MirrorStrip(MirrorStrip { segments })) } else { - let capacity = data_num as u32; + let capacity = req.data_num.saturating_mul(4); ( capacity, StripOneof::EcStrip(EcStrip { @@ -434,8 +434,13 @@ impl ChunkAllocator for FailingChunkAllocator { fn test_config(max_chunk_size: u64) -> Arc { Arc::new(ChunkClientConfig { + chunk_type: crowdb_protocol::chunkdb::rpc::ChunkType::default(), + large_mirror_copies: None, max_chunk_size, prefetch_strips_per_chunk: 2, + large_prefetch_max_strips_per_batch: 32, + large_parallel_strip_writes: 4, + large_held_buffers: 4, parity_depth: 2, chunk_preparation_depth: 1, large_write_repair_attempts: 3, @@ -981,8 +986,13 @@ async fn push_mode_backpressure() { let diskio = LocalFileDiskWriter::new(tmp.path()); let ec = ec_4_1(); let config = Arc::new(ChunkClientConfig { + chunk_type: crowdb_protocol::chunkdb::rpc::ChunkType::default(), + large_mirror_copies: None, max_chunk_size: 1024 * 1024, prefetch_strips_per_chunk: 2, + large_prefetch_max_strips_per_batch: 32, + large_parallel_strip_writes: 4, + large_held_buffers: 4, parity_depth: 2, chunk_preparation_depth: 1, large_write_repair_attempts: 3, @@ -1071,8 +1081,13 @@ async fn write_stream_bounded_prealloc() { let diskio = LocalFileDiskWriter::new(tmp.path()); let ec = ec_4_1(); let config = Arc::new(ChunkClientConfig { + chunk_type: crowdb_protocol::chunkdb::rpc::ChunkType::default(), + large_mirror_copies: None, max_chunk_size: 1024 * 1024 * 1024, prefetch_strips_per_chunk: 2, + large_prefetch_max_strips_per_batch: 32, + large_parallel_strip_writes: 4, + large_held_buffers: 4, parity_depth: 2, chunk_preparation_depth: 1, large_write_repair_attempts: 3, @@ -1128,8 +1143,13 @@ async fn writer_pool_budget_rejects_over_budget() { let diskio = LocalFileDiskWriter::new(tmp.path()); let ec = ec_4_1(); let config = Arc::new(ChunkClientConfig { + chunk_type: crowdb_protocol::chunkdb::rpc::ChunkType::default(), + large_mirror_copies: None, max_chunk_size: 1024 * 1024 * 1024, prefetch_strips_per_chunk: 2, + large_prefetch_max_strips_per_batch: 32, + large_parallel_strip_writes: 4, + large_held_buffers: 4, parity_depth: 2, chunk_preparation_depth: 1, large_write_repair_attempts: 3, @@ -1158,8 +1178,13 @@ async fn writer_pool_per_writer_memory() { let diskio = LocalFileDiskWriter::new(tmp.path()); let ec = ec_4_1(); let config = Arc::new(ChunkClientConfig { + chunk_type: crowdb_protocol::chunkdb::rpc::ChunkType::default(), + large_mirror_copies: None, max_chunk_size: 1024 * 1024 * 1024, prefetch_strips_per_chunk: 2, + large_prefetch_max_strips_per_batch: 32, + large_parallel_strip_writes: 4, + large_held_buffers: 4, parity_depth: 2, chunk_preparation_depth: 1, large_write_repair_attempts: 3, diff --git a/lib/crowdb-chunk-kv/tests/partition_test.rs b/lib/crowdb-chunk-kv/tests/partition_test.rs index 65493b9d6..d6f7eb6b5 100644 --- a/lib/crowdb-chunk-kv/tests/partition_test.rs +++ b/lib/crowdb-chunk-kv/tests/partition_test.rs @@ -302,6 +302,8 @@ fn chunk_page_store(tree_id: u64, owner_epoch: u64) -> Arc { pack_bytes: 4_096, iu_size: 1, max_concurrent_packs: 2, + max_chunk_bytes: 0, + mirror_copies: 0, materialization_bytes_per_pass: 4_096, }, catalog, @@ -560,6 +562,8 @@ async fn chunk_root_checkpoint_supplies_the_wal_replay_offset() { pack_bytes: 4_096, iu_size: 1, max_concurrent_packs: 2, + max_chunk_bytes: 0, + mirror_copies: 0, materialization_bytes_per_pass: 4_096, }; let page_store = Arc::new(PageStore::open_chunk(options, Arc::clone(&catalog), None).unwrap()); diff --git a/lib/crowdb-chunk-stream/src/kv.rs b/lib/crowdb-chunk-stream/src/kv.rs index 13a66e460..785307e53 100644 --- a/lib/crowdb-chunk-stream/src/kv.rs +++ b/lib/crowdb-chunk-stream/src/kv.rs @@ -153,17 +153,47 @@ impl KvStreamMetadataStore { .await { Ok(_) => Ok(()), - Err(crowdb_kv_client::Error::CasFailed { .. } | crowdb_kv_client::Error::OutcomeUnknown) => { + Err( + error @ (crowdb_kv_client::Error::CasFailed { .. } | crowdb_kv_client::Error::OutcomeUnknown), + ) => { match self .kv .get(self.store_id, self.group_id, &key, ReadMode::Linearizable, None) .await .map_err(kv_error)? { - GetOutcome::Found { value, .. } if decode::(&value)? == *page => Ok(()), - _ => Err(StreamError::Corruption( - "immutable extent-page key contains another value".into(), - )), + GetOutcome::Found { value, .. } => { + let observed: StreamExtentPage = decode(&value)?; + if observed == *page { + Ok(()) + } else { + tracing::warn!( + stream_high = page.stream_name.high, + stream_low = page.stream_name.low, + writer_epoch = page.writer_epoch, + generation = page.generation, + page_index = page.page_index, + candidate = ?page, + observed = ?observed, + "immutable extent-page key has conflicting contents" + ); + Err(StreamError::Corruption( + "immutable extent-page key contains another value".into(), + )) + } + } + GetOutcome::NotFound => { + tracing::warn!( + stream_high = page.stream_name.high, + stream_low = page.stream_name.low, + writer_epoch = page.writer_epoch, + generation = page.generation, + page_index = page.page_index, + %error, + "immutable extent-page CAS was unresolved and its key is absent" + ); + Err(StreamError::WriteStalled) + } } } Err(crowdb_kv_client::Error::CasBusy) => Err(StreamError::Backpressure), diff --git a/lib/crowdb-chunk-stream/src/production.rs b/lib/crowdb-chunk-stream/src/production.rs index 13c30050e..5ca83c1b5 100644 --- a/lib/crowdb-chunk-stream/src/production.rs +++ b/lib/crowdb-chunk-stream/src/production.rs @@ -36,7 +36,7 @@ impl ProductionStreamRuntime { read_policy: ChunkReadPolicy, config: StreamConfig, ) -> Result { - Self::new_with_mirror_copies(kv, chunk_io, writer_lease_ms, read_policy, config, 3) + Self::new_with_mirror_copies(kv, chunk_io, writer_lease_ms, read_policy, config, 2) } /// Builds the stream runtime with an explicit stream mirror count. @@ -57,12 +57,13 @@ impl ProductionStreamRuntime { .liveness_interval .min(std::time::Duration::from_millis((writer_lease_ms / 3).max(1))); let (allocator, disk_writer) = chunk_io.storage_parts(); - let chunks = Arc::new(ProductionStreamChunkStore::new_with_mirror_copies( + let chunks = Arc::new(ProductionStreamChunkStore::new_with_mirror_copies_and_capacity( allocator, disk_writer, writer_lease_ms, read_policy, mirror_copies, + config.chunk_capacity_bytes, )?); Ok(Self { registry: Arc::new(KvStreamRegistry::new(Arc::clone(&kv))), diff --git a/lib/crowdb-chunk-stream/src/production_chunk.rs b/lib/crowdb-chunk-stream/src/production_chunk.rs index 3ed2c5278..3f9b294c0 100644 --- a/lib/crowdb-chunk-stream/src/production_chunk.rs +++ b/lib/crowdb-chunk-stream/src/production_chunk.rs @@ -36,7 +36,7 @@ struct ChunkStateView { has_checksum: AtomicBool, } -/// Production `StreamChunkStore` using direct three-copy mirror writes and the +/// Production `StreamChunkStore` using direct mirrored writes and the /// unified chunk reader. Per-chunk mutable metadata is atomic; the stream's /// single-owner worker remains the only operation sequencer. pub struct ProductionStreamChunkStore { @@ -45,6 +45,7 @@ pub struct ProductionStreamChunkStore { reader: ChunkReader, writer_lease_ms: u64, mirror_copies: u32, + chunk_capacity_bytes: u64, failed_disks: Arc, chunks: SkipMap<(u64, u64), Arc>, } @@ -61,7 +62,7 @@ impl ProductionStreamChunkStore { writer_lease_ms: u64, read_policy: ChunkReadPolicy, ) -> Result { - Self::new_with_mirror_copies(allocator, disk_writer, writer_lease_ms, read_policy, 3) + Self::new_with_mirror_copies(allocator, disk_writer, writer_lease_ms, read_policy, 2) } /// Creates a production adapter with an explicit stream mirror count. @@ -76,9 +77,34 @@ impl ProductionStreamChunkStore { read_policy: ChunkReadPolicy, mirror_copies: u32, ) -> Result { - if writer_lease_ms == 0 || mirror_copies == 0 { + Self::new_with_mirror_copies_and_capacity( + allocator, + disk_writer, + writer_lease_ms, + read_policy, + mirror_copies, + crowdb_chunk_client::STREAM_CHUNK_BYTES, + ) + } + + /// Creates a stream adapter with an explicit chunk capacity and mirror count. + /// + /// # Errors + /// Returns an error for invalid lease, capacity, mirror count, or read policy. + pub fn new_with_mirror_copies_and_capacity( + allocator: Arc, + disk_writer: Arc, + writer_lease_ms: u64, + read_policy: ChunkReadPolicy, + mirror_copies: u32, + chunk_capacity_bytes: u64, + ) -> Result { + if writer_lease_ms == 0 + || !(1..=5).contains(&mirror_copies) + || !(1024 * 1024..=crowdb_chunk_client::STREAM_CHUNK_BYTES).contains(&chunk_capacity_bytes) + { return Err(StreamError::InvalidRequest( - "stream chunk writer lease must be nonzero".into(), + "stream chunk writer lease, mirror count, or capacity is invalid".into(), )); } let reader = ChunkReader::new(Arc::clone(&allocator), Arc::clone(&disk_writer), read_policy) @@ -89,6 +115,7 @@ impl ProductionStreamChunkStore { reader, writer_lease_ms, mirror_copies, + chunk_capacity_bytes, failed_disks: Arc::new(FailedDiskList::new(Duration::from_secs(60))), chunks: SkipMap::new(), }) @@ -192,7 +219,7 @@ impl StreamChunkStore for ProductionStreamChunkStore { chunk_id: ChunkId, required_capacity: u64, ) -> Result> { - if required_capacity > crowdb_chunk_client::STREAM_CHUNK_BYTES { + if required_capacity > self.chunk_capacity_bytes { return Ok(None); } let state = self.state(chunk_id).await?; @@ -332,7 +359,13 @@ impl StreamChunkStore for ProductionStreamChunkStore { "stream chunk strip is not mirrored".into(), )); }; - if mirror.segments.len() != self.mirror_copies as usize { + let actual_copies = mirror.segments.len(); + let protected = if self.mirror_copies == 1 { + actual_copies == 1 + } else { + (2..=self.mirror_copies as usize).contains(&actual_copies) + }; + if !protected { return Err(StreamError::Corruption( "stream chunk mirror count differs from configuration".into(), )); diff --git a/lib/crowdb-chunk-stream/src/stream.rs b/lib/crowdb-chunk-stream/src/stream.rs index a4ceba403..b73404232 100644 --- a/lib/crowdb-chunk-stream/src/stream.rs +++ b/lib/crowdb-chunk-stream/src/stream.rs @@ -25,6 +25,7 @@ use crate::{Result, StreamError}; #[derive(Clone, Debug)] pub struct StreamConfig { + pub chunk_capacity_bytes: u64, pub queue_requests: usize, pub queue_bytes: u64, pub batch_requests: usize, @@ -42,6 +43,7 @@ pub struct StreamConfig { impl Default for StreamConfig { fn default() -> Self { Self { + chunk_capacity_bytes: crowdb_chunk_client::STREAM_CHUNK_BYTES, queue_requests: 1_024, queue_bytes: 64 * 1024 * 1024, batch_requests: 64, @@ -61,6 +63,7 @@ impl Default for StreamConfig { impl StreamConfig { pub(crate) fn validate(&self) -> Result<()> { if self.queue_requests == 0 + || !(1024 * 1024..=crowdb_chunk_client::STREAM_CHUNK_BYTES).contains(&self.chunk_capacity_bytes) || self.queue_bytes == 0 || self.batch_requests == 0 || self.batch_bytes == 0 @@ -1236,6 +1239,7 @@ async fn write_with_recovery( requests: &[AppendRequest], bytes: usize, ) -> std::result::Result, BatchFailure> { + let mut rotated = false; loop { match write_batch_with_watchdog(state, requests, bytes).await { Err(BatchFailure::MirrorWrite( @@ -1247,6 +1251,9 @@ async fn write_with_recovery( BatchFailure::MirrorWrite(error) | BatchFailure::Other(error @ StreamError::DefinitelyNotCommitted(_)), ) => { + if rotated { + return Err(BatchFailure::Other(error)); + } tracing::warn!( stream_high = state.stream_name.high, stream_low = state.stream_name.low, @@ -1254,24 +1261,8 @@ async fn write_with_recovery( %error, "chunk-stream append will rotate after an uncommitted write" ); - loop { - match rollover(state).await { - Ok(()) => break, - Err( - error @ (StreamError::StaleWriter - | StreamError::Corruption(_) - | StreamError::InvalidRequest(_)), - ) => { - state.stalled = true; - return Err(BatchFailure::Other(error)); - } - Err(error) => { - tracing::warn!(%error, "chunk-stream rollover remains unavailable"); - tokio::time::sleep(Duration::from_millis(100)).await; - } - } - } - tokio::time::sleep(Duration::from_millis(100)).await; + rollover(state).await.map_err(BatchFailure::Other)?; + rotated = true; } Err(BatchFailure::Other(error)) => { if matches!(rotate_externally_sealed_active(state).await, Ok(true)) { @@ -1468,10 +1459,11 @@ async fn resolve_cursor_advance( "committed cursor has an unexpected checksum".into(), )); } - Ok(_) => { - return Err(StreamError::Corruption( - "durable cursor is outside the append bounds".into(), - )); + Ok(durable) => { + return Err(StreamError::Corruption(format!( + "durable cursor is outside the append bounds: chunk={chunk_id:?} epoch={} expected={expected_cursor} new={new_cursor} durable={}", + state.writer_epoch, durable.offset + ))); } Err(StreamError::StaleWriter) => return Err(StreamError::StaleWriter), Err(error) => { diff --git a/lib/crowdb-chunk-stream/tests/production_chunk_test.rs b/lib/crowdb-chunk-stream/tests/production_chunk_test.rs index b58f28a16..43f8bc963 100644 --- a/lib/crowdb-chunk-stream/tests/production_chunk_test.rs +++ b/lib/crowdb-chunk-stream/tests/production_chunk_test.rs @@ -54,9 +54,12 @@ impl ChunkAllocator for Allocator { request: AllocateChunkRequest, ) -> crowdb_chunk_client::Result { let chunk_id = ChunkId { high: 7, low: 8 }; - let segments = (1..=3) + let segments = (1..=request.copy_count) .map(|disk| Segment { - disk_id: Some(DiskId { high: disk, low: 0 }), + disk_id: Some(DiskId { + high: u64::from(disk), + low: 0, + }), owner_chunk: Some(chunk_id), unit_offset: 0, zone_index: 0, @@ -383,7 +386,7 @@ async fn production_store_writes_reads_advances_and_releases_one_mirror_chunk() ) .await .unwrap(); - assert_eq!(disks.fsyncs.load(Ordering::Relaxed), 3); + assert_eq!(disks.fsyncs.load(Ordering::Relaxed), 2); assert_eq!( store .advance_cursor(name, 9, active.chunk_id, 0, 6, 17) @@ -469,6 +472,38 @@ async fn production_store_grows_and_writes_across_mirror_strips() { ); } +#[tokio::test] +async fn configured_stream_chunk_capacity_limits_growth_without_changing_strip_size() { + let allocator = Arc::new(Allocator::new()); + let disks = Arc::new(Disks::default()); + let store = ProductionStreamChunkStore::new_with_mirror_copies_and_capacity( + allocator, + disks, + 30_000, + ChunkReadPolicy::default(), + 1, + 2 * 1024 * 1024, + ) + .unwrap(); + let name = StreamName { high: 19, low: 20 }; + let active = store.allocate_mirrored(name, 9).await.unwrap(); + assert_eq!(active.capacity, 1024 * 1024); + assert_eq!( + store + .grow_mirrored(name, 9, active.chunk_id, 2 * 1024 * 1024) + .await + .unwrap() + .unwrap() + .capacity, + 2 * 1024 * 1024 + ); + assert!(store + .grow_mirrored(name, 9, active.chunk_id, 2 * 1024 * 1024 + 1) + .await + .unwrap() + .is_none()); +} + #[tokio::test] async fn chunk_stream_runs_end_to_end_over_the_production_chunk_adapter() { let allocator: Arc = Arc::new(Allocator::new()); diff --git a/lib/crowdb-chunk-stream/tests/production_restart_e2e.rs b/lib/crowdb-chunk-stream/tests/production_restart_e2e.rs index 533c7ba00..d43186e0e 100644 --- a/lib/crowdb-chunk-stream/tests/production_restart_e2e.rs +++ b/lib/crowdb-chunk-stream/tests/production_restart_e2e.rs @@ -15,7 +15,9 @@ use crowdb_protocol::diskdb::rpc::{DiskGroupValue, DiskType, DiskValue}; use crowdb_test_harness::chunkdb::{self as chunkdb_harness, ChunkdbProcess, ChunkdbStartOptions}; use crowdb_test_harness::cluster::KvCluster; use crowdb_test_harness::diskdb::{self as diskdb_harness, DiskdbProcess}; -use crowdb_test_harness::diskio::{self as diskio_harness, DiskArg, DiskioProcess, DiskioStartOpts}; +use crowdb_test_harness::diskio::{ + self as diskio_harness, DiskArg, DiskioGroup0Identity, DiskioProcess, DiskioStartOpts, +}; use crowdb_test_harness::hardware::INSTANCE_ID; use crowdb_test_harness::test_dirs::TestDir; @@ -305,3 +307,150 @@ async fn production_stream_recovers_exact_bytes_after_service_restarts() { Bytes::from_static(b"before-restart|after-restart") ); } + +#[tokio::test] +async fn production_stream_write_returns_after_diskio_failure() { + if !all_binaries_available() { + return; + } + + let disk_root = TestDir::new("chunk-stream-diskio-failure").expect("create test disk root"); + let disks = create_disks(&disk_root); + let cluster = KvCluster::start().await; + seed_restart_hardware(&cluster.make_hardware_client()).await; + let diskdb = DiskdbProcess::start(&cluster.mgmt_endpoints, false); + diskdb.wait_for_ready().await; + let mut diskio = start_diskio(&disks); + register_diskio(&cluster, &diskio).await; + let chunkdb = start_chunkdb(&cluster); + chunkdb.wait_for_ready().await; + + let chunk_io = connect_chunk_io(&cluster).await; + let runtime = ProductionStreamRuntime::new( + kv_client(&cluster), + &chunk_io, + 30_000, + ChunkReadPolicy::default(), + StreamConfig::default(), + ) + .expect("assemble production runtime"); + let stream_name = StreamName { + high: u64::from(std::process::id()), + low: 142, + }; + runtime + .registry() + .create(StreamBinding { + stream_name, + metadata_group_id: 1, + binding_generation: 1, + state: StreamBindingState::Active, + owner_kind: Some("diskio-failure-e2e".into()), + }) + .await + .expect("publish stream binding"); + let stream = runtime + .create_registered(stream_name, 0, 1) + .await + .expect("create stream"); + stream + .append(&[Bytes::from_static(b"durable|")]) + .await + .expect("append before DiskIO failure"); + diskio.child.kill().expect("kill diskio"); + diskio.child.wait().expect("reap diskio"); + + let result = tokio::time::timeout( + Duration::from_secs(15), + stream.append(&[Bytes::from_static(b"unavailable")]), + ) + .await + .expect("journal append exceeded the 15-second fault budget"); + assert!(result.is_err(), "journal append succeeded without DiskIO"); + assert_eq!(stream.tail(), 8, "failed append advanced the journal tail"); + assert_eq!( + stream.metrics().rollovers, + 1, + "failed mirror write did not rotate once" + ); +} + +#[tokio::test] +async fn production_stream_write_returns_after_live_diskio_errors() { + if !all_binaries_available() { + return; + } + + let cluster = KvCluster::start().await; + seed_restart_hardware(&cluster.make_hardware_client()).await; + let diskdb = DiskdbProcess::start(&cluster.mgmt_endpoints, false); + diskdb.wait_for_ready().await; + let mut diskios = Vec::new(); + for index in 0..3_u64 { + let diskio = DiskioProcess::start_for_group( + &DiskioStartOpts { + dummy_disk: "mem", + kv_seeds: &cluster.mgmt_endpoints, + disks: &[], + fault_error_rate: 1.0, + fault_latency_ms: None, + no_o_direct: false, + }, + DiskioGroup0Identity { + instance_id: INSTANCE_ID + index, + rack_id: index + 1, + node_id: index + 10, + disk_group_id: index + 100, + }, + ); + diskios.push(diskio); + } + let chunkdb = start_chunkdb(&cluster); + chunkdb.wait_for_ready().await; + let chunk_io = connect_chunk_io(&cluster).await; + let runtime = ProductionStreamRuntime::new( + kv_client(&cluster), + &chunk_io, + 30_000, + ChunkReadPolicy::default(), + StreamConfig::default(), + ) + .expect("assemble production runtime"); + let stream_name = StreamName { + high: u64::from(std::process::id()), + low: 143, + }; + runtime + .registry() + .create(StreamBinding { + stream_name, + metadata_group_id: 1, + binding_generation: 1, + state: StreamBindingState::Active, + owner_kind: Some("live-diskio-error-e2e".into()), + }) + .await + .expect("publish stream binding"); + let stream = runtime + .create_registered(stream_name, 0, 1) + .await + .expect("create stream"); + let result = tokio::time::timeout( + Duration::from_secs(15), + stream.append(&[Bytes::from_static(b"record")]), + ) + .await + .expect("journal append exceeded the 15-second fault budget"); + assert!( + result.is_err(), + "journal append succeeded with all DiskIO writes failing" + ); + assert_eq!(stream.tail(), 0); + assert_eq!(stream.metrics().rollovers, 1); + for diskio in &mut diskios { + assert!( + diskio.child.try_wait().unwrap().is_none(), + "faulted DiskIO exited" + ); + } +} diff --git a/lib/crowdb-chunk-stream/tests/stream_test.rs b/lib/crowdb-chunk-stream/tests/stream_test.rs index 4b6c9e68b..5b7b26ca4 100644 --- a/lib/crowdb-chunk-stream/tests/stream_test.rs +++ b/lib/crowdb-chunk-stream/tests/stream_test.rs @@ -375,20 +375,12 @@ async fn ambiguous_cursor_is_resolved_without_resubmission() { } #[tokio::test] -async fn repeated_mirror_failures_rotate_until_the_same_append_commits() { +async fn repeated_mirror_failures_return_after_one_rotation() { let store = Arc::new(MemoryStreamStore::new(64)); let stream = create_stream(&store, 64, StreamConfig::default()).await; - store.fail_next_writes(3); - assert_eq!( - stream - .append(&[Bytes::from_static(b"record")]) - .await - .unwrap() - .begin, - 0 - ); - assert_eq!(store.chunk_write_count(), 4); - assert_eq!(stream.read_at(0, 6).await.unwrap(), Bytes::from_static(b"record")); + store.fail_next_writes(2); + assert!(stream.append(&[Bytes::from_static(b"record")]).await.is_err()); + assert_eq!(store.chunk_write_count(), 2); } #[tokio::test] @@ -623,7 +615,12 @@ async fn higher_epoch_reopens_same_bytes_and_fences_old_writer() { let store = Arc::new(MemoryStreamStore::new(32)); let old = create_stream(&store, 32, StreamConfig::default()).await; let name = StreamName { high: 1, low: 32 }; - old.append(&[Bytes::from_static(b"old")]).await.unwrap(); + let old_chunk = old + .append(&[Bytes::from_static(b"old")]) + .await + .unwrap() + .chunk_id + .unwrap(); let registry: Arc = store.clone(); let metadata: Arc = store.clone(); @@ -632,11 +629,14 @@ async fn higher_epoch_reopens_same_bytes_and_fences_old_writer() { .await .unwrap(); assert_eq!(new.tail(), 3); + assert!(store.durable_cursor(old_chunk, 10).await.unwrap().sealed); assert_eq!( old.append(&[Bytes::from_static(b"stale")]).await, Err(StreamError::StaleWriter) ); - assert_eq!(new.append(&[Bytes::from_static(b"new")]).await.unwrap().begin, 3); + let appended = new.append(&[Bytes::from_static(b"new")]).await.unwrap(); + assert_eq!(appended.begin, 3); + assert_ne!(appended.chunk_id, Some(old_chunk)); assert_eq!(new.read_at(0, 6).await.unwrap(), Bytes::from_static(b"oldnew")); } diff --git a/lib/crowdb-chunkdb-client/examples/single_node_chunk_layout.rs b/lib/crowdb-chunkdb-client/examples/single_node_chunk_layout.rs new file mode 100644 index 000000000..fbd474026 --- /dev/null +++ b/lib/crowdb-chunkdb-client/examples/single_node_chunk_layout.rs @@ -0,0 +1,55 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use crowdb_chunkdb_client::ChunkdbRpcTransport; +use crowdb_protocol::chunkdb::rpc::{ChunkType, ListChunksRequest, Strip, StripType}; + +#[tokio::main] +async fn main() -> Result<(), Box> { + crowdb_rpc_ffi::init_logging("", "warn", 30, 5, "chunk-layout-check"); + let transport = ChunkdbRpcTransport::new(); + let mut start_token = None; + let mut seen = [false; 2]; + loop { + let listed = transport + .send_list_chunks( + "127.0.0.1:12200", + &ListChunksRequest { + start_token, + max_keys: 256, + ..ListChunksRequest::default() + }, + ) + .await?; + for chunk in listed.chunks { + let index = match ChunkType::try_from(chunk.chunk_type) { + Ok(ChunkType::S3) => 0, + Ok(ChunkType::IcebergTable) => 1, + _ => continue, + }; + seen[index] = true; + let id = chunk.id.expect("protocol chunk must have an ID"); + assert_eq!(id.high >> 56, u64::try_from(chunk.chunk_type)?); + assert!(!chunk.strips.is_empty(), "protocol chunk must have a strip"); + for strip in chunk.strips { + assert_eq!(strip.strip_type, StripType::Mirror as i32); + assert_eq!(strip.capacity, 1024); + assert!(!strip.placement_repair_required); + let Some(Strip::MirrorStrip(mirror)) = strip.strip else { + panic!("single-node protocol strip must use mirror I/O"); + }; + assert_eq!(mirror.segments.len(), 1); + } + } + let Some(next_token) = listed.next_token else { + break; + }; + start_token = Some(next_token); + } + assert!( + seen.into_iter().all(|found| found), + "both protocol chunk types must exist" + ); + println!("single-node S3 and Iceberg chunk layouts verified"); + Ok(()) +} diff --git a/lib/crowdb-chunkdb-client/src/client.rs b/lib/crowdb-chunkdb-client/src/client.rs index 79170e3e5..43950b288 100644 --- a/lib/crowdb-chunkdb-client/src/client.rs +++ b/lib/crowdb-chunkdb-client/src/client.rs @@ -283,20 +283,22 @@ impl ChunkdbClient { let mut backoff = self.retry.initial_backoff; loop { let endpoints = self.endpoints_for_chunk(chunk_id.as_ref()).await?; - let mut not_my_range = None; + let mut safe_retry = None; for endpoint in endpoints { // Allocation is not idempotent at the DiskDB layer. Trying the - // transition fallback is safe only after NotMyRange, which is - // rejected before mutation. + // transition fallback is safe only when the server rejected + // the range or the connection failed before request submission. match self.rpc_transport.send_allocate_chunk(&endpoint, &req).await { Ok(response) => return Ok(response), - Err(error @ ChunkdbClientError::NotMyRange(_)) => { - not_my_range = Some(error); + Err( + error @ (ChunkdbClientError::NotMyRange(_) | ChunkdbClientError::ConnectFailed(_)), + ) => { + safe_retry = Some(error); } Err(error) => return Err(error), } } - let Some(error) = not_my_range else { + let Some(error) = safe_retry else { return Err(ChunkdbClientError::Unreachable( "range routing supplied no endpoint".into(), )); diff --git a/lib/crowdb-chunkdb-client/src/lib.rs b/lib/crowdb-chunkdb-client/src/lib.rs index 9a16385c3..c51aa665e 100644 --- a/lib/crowdb-chunkdb-client/src/lib.rs +++ b/lib/crowdb-chunkdb-client/src/lib.rs @@ -26,6 +26,8 @@ use thiserror::Error; /// Error type for chunkdb client operations. #[derive(Debug, Error)] pub enum ChunkdbClientError { + #[error("chunkdb connection failed before request submission: {0}")] + ConnectFailed(String), #[error("chunkdb server unreachable: {0}")] Unreachable(String), #[error("chunkdb server unavailable (transient): {0}")] @@ -57,7 +59,11 @@ impl ChunkdbClientError { pub fn is_transient(&self) -> bool { matches!( self, - Self::Unavailable(_) | Self::DeadlineExceeded(_) | Self::Unreachable(_) | Self::NotMyRange(_) + Self::ConnectFailed(_) + | Self::Unavailable(_) + | Self::DeadlineExceeded(_) + | Self::Unreachable(_) + | Self::NotMyRange(_) ) } } diff --git a/lib/crowdb-chunkdb-client/src/rpc_transport.rs b/lib/crowdb-chunkdb-client/src/rpc_transport.rs index afc116d4d..ae5edf845 100644 --- a/lib/crowdb-chunkdb-client/src/rpc_transport.rs +++ b/lib/crowdb-chunkdb-client/src/rpc_transport.rs @@ -197,7 +197,7 @@ impl ChunkdbRpcTransport { self.connections .get_or_try_install(&normalized, || { let conn = self.server.connect(&host, port).map_err(|error| { - ChunkdbClientError::Unreachable(format!("rpc connect to {host}:{port}: {error:?}")) + ChunkdbClientError::ConnectFailed(format!("rpc connect to {host}:{port}: {error:?}")) })?; self.rpc.attach(&conn); Ok(conn) @@ -1519,6 +1519,8 @@ fn chunk_type_to_fb(t: ProtoChunkType) -> FBChunkType { ProtoChunkType::BtreePage => FBChunkType::BtreePage, ProtoChunkType::PageIndex => FBChunkType::PageIndex, ProtoChunkType::Stream => FBChunkType::Stream, + ProtoChunkType::S3 => FBChunkType::S3, + ProtoChunkType::IcebergTable => FBChunkType::IcebergTable, } } @@ -1528,6 +1530,8 @@ fn fb_chunk_type_to_proto(t: FBChunkType) -> ProtoChunkType { FBChunkType::BtreePage => ProtoChunkType::BtreePage, FBChunkType::PageIndex => ProtoChunkType::PageIndex, FBChunkType::Stream => ProtoChunkType::Stream, + FBChunkType::S3 => ProtoChunkType::S3, + FBChunkType::IcebergTable => ProtoChunkType::IcebergTable, _ => ProtoChunkType::Repo, } } diff --git a/lib/crowdb-chunkdb-client/tests/client_test.rs b/lib/crowdb-chunkdb-client/tests/client_test.rs index 99e920089..0636a2531 100644 --- a/lib/crowdb-chunkdb-client/tests/client_test.rs +++ b/lib/crowdb-chunkdb-client/tests/client_test.rs @@ -41,6 +41,12 @@ fn is_transient_unreachable() { assert!(err.is_transient()); } +#[test] +fn is_transient_connection_failure_before_submission() { + let err = ChunkdbClientError::ConnectFailed("test".into()); + assert!(err.is_transient()); +} + #[test] fn is_not_transient_not_found() { let err = ChunkdbClientError::NotFound("test".into()); diff --git a/lib/crowdb-common/rust/src/ec_isal.rs b/lib/crowdb-common/rust/src/ec_isal.rs index 1bdc605f9..ae1209948 100644 --- a/lib/crowdb-common/rust/src/ec_isal.rs +++ b/lib/crowdb-common/rust/src/ec_isal.rs @@ -1,7 +1,7 @@ // Copyright 2026-present Gian // Licensed under the Apache License, Version 2.0. -//! isa-l FFI bindings for Reed-Solomon GF(2^8) erasure coding. +//! isa-l FFI bindings for Reed-Solomon GF(2^8) erasure coding and CRC32C. //! //! Wraps the isa-l `erasure_code.h` API: `gf_gen_rs_matrix`, //! `ec_init_tables`, `ec_encode_data`, `gf_invert_matrix`. The safe @@ -20,6 +20,7 @@ type UcPtr = *mut u8; extern "C" { + fn crc32_iscsi(buffer: *mut u8, length: i32, seed: u32) -> u32; fn gf_gen_rs_matrix(a: UcPtr, m: i32, k: i32); fn gf_invert_matrix(input: UcPtr, output: UcPtr, n: i32); fn ec_init_tables(k: i32, rows: i32, a: UcPtr, gftbls: UcPtr); @@ -35,6 +36,19 @@ extern "C" { ); } +/// Continue the raw seed-zero CRC32C used by chunk frames and durable pages. +#[must_use] +pub fn crc32c_update(mut crc: u32, mut data: &[u8]) -> u32 { + while !data.is_empty() { + let length = data.len().min(i32::MAX as usize); + // SAFETY: ISA-L reads but does not modify the input, and `length` + // stays within the signed length accepted by its C interface. + crc = unsafe { crc32_iscsi(data.as_ptr().cast_mut(), length as i32, crc) }; + data = &data[length..]; + } + crc +} + // ── GF(2^8) arithmetic ────────────────────────────────────────── // isa-l uses the AES polynomial 0x11d. We build log/exp tables for // multiplication so the decode-matrix construction (parity row × diff --git a/lib/crowdb-console-shared/Cargo.toml b/lib/crowdb-console-shared/Cargo.toml index 4ef09e4df..29e40223b 100644 --- a/lib/crowdb-console-shared/Cargo.toml +++ b/lib/crowdb-console-shared/Cargo.toml @@ -33,5 +33,6 @@ axum = "0.7" # `KvClient` wrapper is gone; that test now exercises the real # `crowdb-kv-client` crate end-to-end instead). crowdb-kv-client = { path = "../crowdb-kv-client" } +crowdb-chunkdb-client = { path = "../crowdb-chunkdb-client" } crowdb-rpc-ffi = { path = "../crowdb-rpc/ffi", features = ["test-util"] } crowdb-test-harness = { path = "../crowdb-test-harness", features = ["kv-client"] } diff --git a/lib/crowdb-console-shared/src/lifecycle.rs b/lib/crowdb-console-shared/src/lifecycle.rs index 4937f8896..ad0f9d620 100644 --- a/lib/crowdb-console-shared/src/lifecycle.rs +++ b/lib/crowdb-console-shared/src/lifecycle.rs @@ -1217,8 +1217,19 @@ pub async fn deploy_chunkdb_local( .map(|seed| format!("{seed:?}")) .collect::>() .join(", "); + let deployment_mode = if req.allow_unsafe_ec { + "test_unsafe_placement" + } else { + "production" + }; + let placement_mode = if req.allow_unsafe_ec { + "unsafe_colocated" + } else { + "protected" + }; let config = format!( - "[server]\nrpc_workers = {}\nhttp_listen_addr = \"{}:{}\"\nrpc_listen_addr = \"{}:{}\"\ninstance_id = \"{}\"\nkv_server_mgmt_seeds = [{}]\nkeepalive_interval_secs = 1\nkv_pool_size = {}\nkv_rpc_workers = {}\ndiskdb_pool_size = {}\ndiskdb_rpc_workers = {}\n\n[topology]\nrefresh_interval_secs = 1\n\n[range_guard]\nallow_all_when_empty = false\n\n[lifecycle]\ncache_capacity = 10000\nsweep_chunk_lock_interval_secs = 60\nlock_hold_warn_threshold_ms = 1000\n\n[placement]\nallow_unsafe_ec = {}\nallow_degraded_failure_domains = {}\n", + "[deployment]\nmode = \"{}\"\n\n[server]\nrpc_workers = {}\nhttp_listen_addr = \"{}:{}\"\nrpc_listen_addr = \"{}:{}\"\ninstance_id = \"{}\"\nkv_server_mgmt_seeds = [{}]\nkeepalive_interval_secs = 1\nkv_pool_size = {}\nkv_rpc_workers = {}\ndiskdb_pool_size = {}\ndiskdb_rpc_workers = {}\n\n[topology]\nrefresh_interval_secs = 1\n\n[range_guard]\nallow_all_when_empty = false\n\n[lifecycle]\ncache_capacity = 10000\nsweep_chunk_lock_interval_secs = 60\nlock_hold_warn_threshold_ms = 1000\n\n[placement]\nmode = \"{}\"\nallow_unsafe_ec = {}\nallow_degraded_failure_domains = {}\n", + deployment_mode, req.rpc_workers.unwrap_or(2), node.host, req.http_port, @@ -1230,6 +1241,7 @@ pub async fn deploy_chunkdb_local( req.kv_client_rpc_workers.unwrap_or(2), req.diskdb_connections.unwrap_or(1), req.diskdb_client_rpc_workers.unwrap_or(2), + placement_mode, req.allow_unsafe_ec, req.allow_unsafe_ec, ); diff --git a/lib/crowdb-console-shared/src/ops/cluster.rs b/lib/crowdb-console-shared/src/ops/cluster.rs index 9709fbed4..98f6e695b 100644 --- a/lib/crowdb-console-shared/src/ops/cluster.rs +++ b/lib/crowdb-console-shared/src/ops/cluster.rs @@ -448,7 +448,7 @@ pub async fn local_deploy_combined_after_kv( let chunkdb = local_deploy_chunkdb(ctx, workspace, chunk).await?; Ok(LocalCombinedDeploySummary { kv_nodes: 3, - racks: 1, + racks: ctx.config().racks.len(), diskdb_instances: diskdb.instance_count, chunkdb_instances: chunkdb.instance_count, diskio_instances: diskio, @@ -498,7 +498,7 @@ pub async fn local_deploy_combined_file_backed_after_kv( let chunkdb = local_deploy_chunkdb(ctx, workspace, chunk).await?; Ok(LocalCombinedDeploySummary { kv_nodes: 3, - racks: 1, + racks: ctx.config().racks.len(), diskdb_instances: diskdb.instance_count, chunkdb_instances: chunkdb.instance_count, diskio_instances: diskio, @@ -1187,6 +1187,30 @@ pub async fn prepare_local_deploy( node_count: usize, workspace_dir: Option<&std::path::Path>, tunables: Option<&KvDeployTunables>, +) -> Result<(u64, Vec)> { + prepare_local_deploy_with_layout(ctx, node_count, workspace_dir, tunables, false).await +} + +/// Start a simulated local cluster with each process assigned to a distinct rack. +/// This is for validating protected placement on one development host. +/// +/// # Errors +/// Returns a validation, binary, spawn, or readiness error. +pub async fn prepare_local_deploy_distinct_racks( + ctx: &OpContext, + node_count: usize, + workspace_dir: Option<&std::path::Path>, + tunables: Option<&KvDeployTunables>, +) -> Result<(u64, Vec)> { + prepare_local_deploy_with_layout(ctx, node_count, workspace_dir, tunables, true).await +} + +async fn prepare_local_deploy_with_layout( + ctx: &OpContext, + node_count: usize, + workspace_dir: Option<&std::path::Path>, + tunables: Option<&KvDeployTunables>, + distinct_racks: bool, ) -> Result<(u64, Vec)> { if node_count == 0 { return Err(Error::Validation { @@ -1212,8 +1236,17 @@ pub async fn prepare_local_deploy( let rack_id: u64 = 1; let node_ids: Vec = (1..=u64::try_from(node_count).unwrap_or(u64::MAX)).collect(); - write_rack_and_nodes(ctx, rack_id, &node_ids); - deploy_servers(ctx, &bin, &workspace, rack_id, &node_ids, tunables).await?; + write_rack_and_nodes(ctx, rack_id, &node_ids, distinct_racks); + deploy_servers( + ctx, + &bin, + &workspace, + rack_id, + &node_ids, + tunables, + distinct_racks, + ) + .await?; // Re-seed the group-0 leader hint to the first deployed server's // RPC endpoint so sysdata writes during `init` target the right node. @@ -1278,16 +1311,17 @@ fn alloc_workspace_ports( }) } -/// Phase 1: write rack 1 + nodes 1..=N into the config (idempotent). -fn write_rack_and_nodes(ctx: &OpContext, rack_id: u64, node_ids: &[u64]) { +/// Phase 1: write the requested rack layout and nodes into the config. +fn write_rack_and_nodes(ctx: &OpContext, rack_id: u64, node_ids: &[u64], distinct_racks: bool) { let mut cfg = ctx.config_mut(); - if cfg.racks.iter().all(|r| r.id != rack_id) { - let _ = cfg.add_rack(RackEntry { - id: rack_id, - name: format!("rack-{rack_id}"), - }); - } for nid in node_ids { + let rack_id = if distinct_racks { *nid } else { rack_id }; + if cfg.racks.iter().all(|r| r.id != rack_id) { + let _ = cfg.add_rack(RackEntry { + id: rack_id, + name: format!("rack-{rack_id}"), + }); + } if cfg.nodes.iter().all(|n| n.id != *nid) { let _ = cfg.add_node(NodeEntry { id: *nid, @@ -1315,11 +1349,13 @@ async fn deploy_servers( rack_id: u64, node_ids: &[u64], tunables: Option<&KvDeployTunables>, + distinct_racks: bool, ) -> Result<()> { let n = u16::try_from(node_ids.len()).unwrap_or(u16::MAX); let rest_ports = alloc_workspace_ports(workspace, ServicePort::KvServerMgmt, 0, n)?; let rpc_ports = alloc_workspace_ports(workspace, ServicePort::KvServerListen, 0, n)?; for (i, nid) in node_ids.iter().enumerate() { + let rack_id = if distinct_racks { *nid } else { rack_id }; let rest_port = rest_ports[i]; let rpc_port = rpc_ports[i]; diff --git a/lib/crowdb-console-shared/src/ops/s3.rs b/lib/crowdb-console-shared/src/ops/s3.rs index 424727dc8..28ee1e594 100644 --- a/lib/crowdb-console-shared/src/ops/s3.rs +++ b/lib/crowdb-console-shared/src/ops/s3.rs @@ -38,6 +38,8 @@ pub enum StorageProfile { #[derive(Debug, Clone, Serialize, serde::Deserialize)] pub struct MiniClusterRecord { pub version: u32, + #[serde(default)] + pub protected_test: bool, pub endpoint: String, #[serde(default)] pub web_endpoint: String, @@ -83,7 +85,28 @@ pub fn load(data_dir: &Path) -> Result<(ConsoleConfig, MiniClusterRecord)> { /// Returns an error for an unsafe directory, missing binary, failed service, /// or failed readiness condition. pub async fn start(data_dir: &Path) -> Result { - start_with_profile(data_dir, StorageProfile::Persistent, 16 * 1024 * 1024 * 1024).await + start_with_profile( + data_dir, + StorageProfile::Persistent, + 16 * 1024 * 1024 * 1024, + false, + ) + .await +} + +/// Start a simulated three-rack cluster for protected-placement tests. +/// +/// # Errors +/// Returns an error for an unsafe directory, missing binary, failed service, +/// or failed readiness condition. +pub async fn start_protected_test_cluster(data_dir: &Path) -> Result { + start_with_profile( + data_dir, + StorageProfile::Persistent, + 16 * 1024 * 1024 * 1024, + true, + ) + .await } /// Create a fresh memory-backed cluster for an S3 benchmark. @@ -98,13 +121,14 @@ pub async fn start_memory(data_dir: &Path, memory_budget_bytes: u64) -> Result Result { archive_incomplete_attempt(data_dir)?; validate_location(data_dir)?; @@ -123,6 +147,12 @@ async fn start_with_profile( ), }); } + if record.protected_test != protected_test { + return Err(Error::Validation { + field: "protected_test".into(), + message: "existing cluster uses a different rack layout".into(), + }); + } if storage_profile == StorageProfile::Memory { return Err(Error::Validation { field: "root".into(), @@ -141,6 +171,45 @@ async fn start_with_profile( vec!["http://127.0.0.1:10000".into()], config, ); + let (disk, chunk) = storage_configs(storage_profile, capacity_bytes, protected_test); + if let Some(status) = + resume_if_interrupted(data_dir, &disk, &chunk, storage_profile, protected_test).await? + { + return Ok(status); + } + let mut record = MiniClusterRecord { + version: 1, + protected_test, + endpoint: String::new(), + web_endpoint: String::new(), + web_pid: None, + tenant: "local".into(), + storage_profile, + }; + save_record(&data_dir.join(INITIALIZING_FILE), &record)?; + let initialized = initialize_new(&ctx, data_dir, &disk, &chunk, storage_profile, protected_test).await; + let endpoints = match initialized { + Ok(endpoints) => endpoints, + Err(error) => { + stop_config_processes(&mut ctx.config_mut()); + let _ = local_state::save(data_dir, &ctx.config()); + return Err(error); + } + }; + record.endpoint = endpoints.s3; + record.web_endpoint = endpoints.web; + record.web_pid = Some(endpoints.web_pid); + save_record(&marker_path, &record)?; + let _ = std::fs::remove_file(data_dir.join(INITIALIZING_FILE)); + let status = status_from(data_dir, true, &ctx.config(), &record); + Ok(status) +} + +fn storage_configs( + storage_profile: StorageProfile, + capacity_bytes: u64, + protected_test: bool, +) -> (LocalDiskdbDeployConfig, LocalChunkdbDeployConfig) { let logical_capacity = if storage_profile == StorageProfile::Memory { 16 * 1024 * 1024 * 1024 } else { @@ -167,9 +236,7 @@ async fn start_with_profile( }; let chunk = LocalChunkdbDeployConfig { instance_count: 3, - // This loopback fixture colocates its simulated nodes in one rack. - // Production planning still requires distinct failure domains. - allow_unsafe_ec: true, + allow_unsafe_ec: !protected_test, rpc_workers: None, diskio_rpc_workers: None, kv_connections: None, @@ -178,34 +245,7 @@ async fn start_with_profile( diskdb_client_rpc_workers: None, metrics_interval: None, }; - if let Some(status) = resume_if_interrupted(data_dir, &disk, &chunk, storage_profile).await? { - return Ok(status); - } - let mut record = MiniClusterRecord { - version: 1, - endpoint: String::new(), - web_endpoint: String::new(), - web_pid: None, - tenant: "local".into(), - storage_profile, - }; - save_record(&data_dir.join(INITIALIZING_FILE), &record)?; - let initialized = initialize_new(&ctx, data_dir, &disk, &chunk, storage_profile).await; - let endpoints = match initialized { - Ok(endpoints) => endpoints, - Err(error) => { - stop_config_processes(&mut ctx.config_mut()); - let _ = local_state::save(data_dir, &ctx.config()); - return Err(error); - } - }; - record.endpoint = endpoints.s3; - record.web_endpoint = endpoints.web; - record.web_pid = Some(endpoints.web_pid); - save_record(&marker_path, &record)?; - let _ = std::fs::remove_file(data_dir.join(INITIALIZING_FILE)); - let status = status_from(data_dir, true, &ctx.config(), &record); - Ok(status) + (disk, chunk) } async fn resume_if_interrupted( @@ -213,13 +253,17 @@ async fn resume_if_interrupted( disk: &LocalDiskdbDeployConfig, chunk: &LocalChunkdbDeployConfig, storage_profile: StorageProfile, + protected_test: bool, ) -> Result> { if !data_dir.join(INITIALIZING_FILE).exists() || !local_state::path(data_dir).exists() { return Ok(None); } let record: MiniClusterRecord = serde_json::from_slice(&std::fs::read(data_dir.join(INITIALIZING_FILE))?) .map_err(|error| Error::Config(error.to_string()))?; - if record.version != 1 || record.storage_profile != storage_profile { + if record.version != 1 + || record.storage_profile != storage_profile + || record.protected_test != protected_test + { return Err(Error::Conflict { kind: "S3 bootstrap profile".into(), id: data_dir.display().to_string(), @@ -262,6 +306,7 @@ async fn initialize_new( disk: &LocalDiskdbDeployConfig, chunk: &LocalChunkdbDeployConfig, storage_profile: StorageProfile, + protected_test: bool, ) -> Result { let tunables = KvDeployTunables { kv_backend: (storage_profile == StorageProfile::Memory).then(|| "mem-block".into()), @@ -269,10 +314,14 @@ async fn initialize_new( no_fsync: (storage_profile == StorageProfile::Memory).then_some(true), ..KvDeployTunables::default() }; - let (_, nodes) = cluster::prepare_local_deploy(ctx, 3, Some(data_dir), Some(&tunables)).await?; + let (_, nodes) = if protected_test { + cluster::prepare_local_deploy_distinct_racks(ctx, 3, Some(data_dir), Some(&tunables)).await? + } else { + cluster::prepare_local_deploy(ctx, 3, Some(data_dir), Some(&tunables)).await? + }; local_state::save(data_dir, &ctx.config())?; cluster::init_with_intent(ctx, &nodes, &data_dir.join(BOOTSTRAP_INTENT_FILE)).await?; - initialize_after_kv(ctx, data_dir, disk, chunk, storage_profile).await + initialize_after_kv(ctx, data_dir, disk, chunk, storage_profile, protected_test).await } async fn initialize_after_kv( @@ -281,6 +330,7 @@ async fn initialize_after_kv( disk: &LocalDiskdbDeployConfig, chunk: &LocalChunkdbDeployConfig, storage_profile: StorageProfile, + protected_test: bool, ) -> Result { let storage_services = ctx .config() @@ -293,7 +343,7 @@ async fn initialize_after_kv( ) }) .count(); - if storage_services == 9 { + if storage_services == 6 + chunk.instance_count { for group in &disk.data_groups { if ctx.sysmd().get_group(0, *group).await?.is_none() { return Err(Error::NotFound { @@ -325,7 +375,7 @@ async fn initialize_after_kv( } let seeds = management_seeds(&ctx.config()); local_state::save(data_dir, &ctx.config())?; - let chunk_kv = spawn_chunk_kv(data_dir, &seeds).await?; + let chunk_kv = spawn_chunk_kv(data_dir, &seeds, protected_test).await?; add_service(ctx, chunk_kv)?; local_state::save(data_dir, &ctx.config())?; let access = spawn_access(data_dir, &seeds).await?; @@ -393,7 +443,7 @@ async fn resume_incomplete( mut record: MiniClusterRecord, ) -> Result { let (mut config, seeds) = local_state::load(data_dir)?; - restore_launch_nodes(&mut config)?; + restore_launch_nodes(&mut config, record.protected_test)?; let group0 = config .servers .iter() @@ -404,15 +454,24 @@ async fn resume_incomplete( .to_owned(); let ctx = OpContext::new(group0, seeds.clone(), config); for node_id in 1..=3 { + let rack_id = if record.protected_test { node_id } else { 1 }; let server_dir = data_dir - .join("rack1") + .join(format!("rack{rack_id}")) .join(format!("node{node_id}")) .join(format!("kv-server-{node_id}")); crate::ops::kv_server::restart(&ctx, node_id, Some(&server_dir), None, &seeds).await?; } local_state::save(data_dir, &ctx.config())?; cluster::init_with_intent(&ctx, &[1, 2, 3], &data_dir.join(BOOTSTRAP_INTENT_FILE)).await?; - let endpoints = initialize_after_kv(&ctx, data_dir, disk, chunk, record.storage_profile).await?; + let endpoints = initialize_after_kv( + &ctx, + data_dir, + disk, + chunk, + record.storage_profile, + record.protected_test, + ) + .await?; record.endpoint = endpoints.s3; record.web_endpoint = endpoints.web; record.web_pid = Some(endpoints.web_pid); @@ -424,7 +483,7 @@ async fn resume_incomplete( async fn restart(data_dir: &Path) -> Result { let (mut config, mut record) = load(data_dir)?; - restore_launch_nodes(&mut config)?; + restore_launch_nodes(&mut config, record.protected_test)?; if let Some(pid) = record.web_pid.take() { let _ = lifecycle::stop_pid_with_timeout(pid, Duration::from_secs(5)); save_record(&data_dir.join(MARKER_FILE), &record)?; @@ -441,8 +500,9 @@ async fn restart(data_dir: &Path) -> Result { let ctx = OpContext::new(group0, seeds.clone(), config); let node_ids = ctx.config().nodes.iter().map(|node| node.id).collect::>(); for node_id in node_ids { + let rack_id = if record.protected_test { node_id } else { 1 }; let server_dir = data_dir - .join("rack1") + .join(format!("rack{rack_id}")) .join(format!("node{node_id}")) .join(format!("kv-server-{node_id}")); crate::ops::kv_server::restart(&ctx, node_id, Some(&server_dir), None, &seeds).await?; @@ -458,7 +518,7 @@ async fn restart(data_dir: &Path) -> Result { .cloned(); let Some(server) = server else { let spawned = match kind { - ServiceType::ChunkKv => spawn_chunk_kv(data_dir, &seeds).await?, + ServiceType::ChunkKv => spawn_chunk_kv(data_dir, &seeds, record.protected_test).await?, ServiceType::AccessServer => spawn_access(data_dir, &seeds).await?, _ => unreachable!(), }; @@ -584,11 +644,7 @@ fn management_seeds(config: &ConsoleConfig) -> Vec { .collect() } -fn restore_launch_nodes(config: &mut ConsoleConfig) -> Result<()> { - config.add_rack(RackEntry { - id: 1, - name: "rack-1".into(), - })?; +fn restore_launch_nodes(config: &mut ConsoleConfig, protected_test: bool) -> Result<()> { let node_ids: Vec<_> = config .servers .iter() @@ -600,9 +656,16 @@ fn restore_launch_nodes(config: &mut ConsoleConfig) -> Result<()> { }) .collect::>()?; for id in node_ids { + let rack_id = if protected_test { id } else { 1 }; + if config.racks.iter().all(|rack| rack.id != rack_id) { + config.add_rack(RackEntry { + id: rack_id, + name: format!("rack-{rack_id}"), + })?; + } config.add_node(NodeEntry { id, - rack_id: 1, + rack_id, host: "127.0.0.1".into(), ssh_port: 22, ssh_user: String::new(), @@ -632,7 +695,7 @@ fn add_service(ctx: &OpContext, service: SpawnedService) -> Result<()> { ctx.config_mut().add_server(service.entry) } -async fn spawn_chunk_kv(data_dir: &Path, seeds: &[String]) -> Result { +async fn spawn_chunk_kv(data_dir: &Path, seeds: &[String], protected_test: bool) -> Result { let binary = find_binary("CROWDB_CHUNK_KV_SERVER_BIN", "crowdb-chunk-kv-server")?; let rpc_port = assign_cluster_port(data_dir, ServicePort::ChunkKvRpc, "chunk-kv-1-rpc")?; let http_port = assign_cluster_port(data_dir, ServicePort::ChunkKvHttp, "chunk-kv-1-http")?; @@ -644,11 +707,12 @@ async fn spawn_chunk_kv(data_dir: &Path, seeds: &[String]) -> Result>() .join(", "); + let stream_mirror_copies = if protected_test { 2 } else { 1 }; let config_path = workdir.join("chunk-kv.toml"); std::fs::write( &config_path, format!( - "instance_id = 10000\nrpc_listen_addr = \"127.0.0.1:{rpc_port}\"\nrpc_advertise_addr = \"127.0.0.1:{rpc_port}\"\nhttp_listen_addr = \"127.0.0.1:{http_port}\"\ngroup0_mgmt_seeds = [{seed_toml}]\ncatalog_refresh_interval_ms = 200\n\n[balance]\nenabled = true\ntarget_partitions_per_owner = 1\ntarget_partition_bytes = 9223372036854775807\nminimum_weighted_improvement_percent = 100\ncooldown_ms = 9223372036854775807\nmax_owner_request_rate = 0\n\n[storage]\nmetadata_store_id = 0\nstream_mirror_copies = 1\n\n[bootstrap_partition]\npartition_id = {{ high = 1, low = 1 }}\ntree_id = 1\nstream_name = {{ high = 2, low = 1 }}\nowner_epoch = 1\nmetadata_group_id = 1\n" + "instance_id = 10000\nrpc_listen_addr = \"127.0.0.1:{rpc_port}\"\nrpc_advertise_addr = \"127.0.0.1:{rpc_port}\"\nhttp_listen_addr = \"127.0.0.1:{http_port}\"\ngroup0_mgmt_seeds = [{seed_toml}]\ncatalog_refresh_interval_ms = 200\n\n[balance]\nenabled = true\ntarget_partitions_per_owner = 1\ntarget_partition_bytes = 9223372036854775807\nminimum_weighted_improvement_percent = 100\ncooldown_ms = 9223372036854775807\nmax_owner_request_rate = 0\n\n[storage]\nmetadata_store_id = 0\nstream_mirror_copies = {stream_mirror_copies}\n\n[bootstrap_partition]\npartition_id = {{ high = 1, low = 1 }}\ntree_id = 1\nstream_name = {{ high = 2, low = 1 }}\nowner_epoch = 1\nmetadata_group_id = 1\n" ), )?; let launch = LocalLaunchSpec { @@ -686,7 +750,7 @@ async fn spawn_access(data_dir: &Path, seeds: &[String]) -> Result // Licensed under the Apache License, Version 2.0. +use crowdb_chunkdb_client::{ChunkdbClient, ChunkdbRpcTransport}; use crowdb_console_shared::ops::s3; +use crowdb_console_shared::{lifecycle, ops::OpContext}; +use crowdb_kv_client::{ClientConfig, CrowdbKvClient, RangeBindingClient, ServiceRegistryClient}; +use crowdb_protocol::chunkdb::rpc::{ChunkType, ListChunksRequest, QueryChunkRequest, Strip, StripType}; +use crowdb_protocol::common::HwStatus; use crowdb_test_harness::test_dirs::TestDir; use reqwest::Method; +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; + +struct StopClusterOnDrop<'a>(&'a Path); + +impl Drop for StopClusterOnDrop<'_> { + fn drop(&mut self) { + let _ = s3::stop(self.0); + } +} #[test] fn foreign_nonempty_directory_is_not_a_cluster() { @@ -122,3 +138,330 @@ async fn persistent_cluster_survives_stop_restart_and_range_read() { s3::delete(dir.path()).expect("delete cluster"); assert!(!dir.path().exists()); } + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +#[ignore = "starts a complete simulated three-rack process stack"] +async fn protected_cluster_starts_and_reads_after_restart() { + let dir = TestDir::new("s3-mini-protected-e2e").expect("create test directory"); + let started = s3::start_protected_test_cluster(dir.path()) + .await + .expect("start protected cluster"); + let (config, record) = s3::load(dir.path()).expect("load protected cluster"); + assert!(record.protected_test); + for node_id in 1..=3 { + assert!(dir.path().join(format!("rack{node_id}/node{node_id}")).is_dir()); + for kind in [ + crowdb_console_shared::config::ServiceType::Chunkdb, + crowdb_console_shared::config::ServiceType::Diskdb, + crowdb_console_shared::config::ServiceType::Diskio, + ] { + assert!(config + .servers + .iter() + .any(|server| { server.node_id == Some(node_id) && server.service_type == kind })); + } + } + let client = s3::S3HttpClient::from_data_dir(dir.path()).expect("S3 client"); + client + .request(Method::PUT, Some("protected-bucket"), None, &[], None, None) + .await + .expect("create bucket"); + client + .request( + Method::PUT, + Some("protected-bucket"), + Some("protected-object"), + &[], + Some(b"protected-object-bytes".to_vec()), + None, + ) + .await + .expect("put protected object"); + assert_eq!( + s3::stop(dir.path()) + .expect("stop protected cluster") + .running_services, + 0 + ); + let restarted = s3::start_protected_test_cluster(dir.path()) + .await + .expect("restart protected cluster"); + assert_eq!(restarted.endpoint, started.endpoint); + let client = s3::S3HttpClient::from_data_dir(dir.path()).expect("restarted S3 client"); + let (_, body) = client + .request( + Method::GET, + Some("protected-bucket"), + Some("protected-object"), + &[], + None, + None, + ) + .await + .expect("read protected object"); + assert_eq!(body, b"protected-object-bytes"); + s3::delete(dir.path()).expect("delete protected cluster"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +#[ignore = "stops one node in a complete simulated three-rack process stack"] +async fn protected_cluster_reads_and_writes_after_node_one_stops() { + protected_cluster_reads_and_writes_after_node_stops(1).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +#[ignore = "stops one node in a complete simulated three-rack process stack"] +async fn protected_cluster_reads_and_writes_after_node_three_stops() { + protected_cluster_reads_and_writes_after_node_stops(3).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +#[ignore = "stops one node in a complete simulated three-rack process stack"] +async fn protected_cluster_reads_and_writes_after_node_two_stops() { + protected_cluster_reads_and_writes_after_node_stops(2).await; +} + +#[allow(clippy::too_many_lines)] +async fn protected_cluster_reads_and_writes_after_node_stops(failed_node: u64) { + let dir = + TestDir::new(&format!("s3-mini-protected-outage-{failed_node}")).expect("create test directory"); + s3::start_protected_test_cluster(dir.path()) + .await + .expect("start protected cluster"); + let _cleanup = StopClusterOnDrop(dir.path()); + let client = s3::S3HttpClient::from_data_dir(dir.path()).expect("S3 client"); + client + .request(Method::PUT, Some("outage-bucket"), None, &[], None, None) + .await + .expect("create bucket"); + client + .request( + Method::PUT, + Some("outage-bucket"), + Some("before-outage"), + &[], + Some(b"before-outage-bytes".to_vec()), + None, + ) + .await + .expect("write before outage"); + + let (config, _) = s3::load(dir.path()).expect("load process identities"); + for kind in [ + crowdb_console_shared::config::ServiceType::Chunkdb, + crowdb_console_shared::config::ServiceType::Diskdb, + crowdb_console_shared::config::ServiceType::Diskio, + crowdb_console_shared::config::ServiceType::Kv, + ] { + let server = config + .servers + .iter() + .find(|server| server.node_id == Some(failed_node) && server.service_type == kind) + .expect("failed-node process"); + lifecycle::stop_pid_with_timeout(server.pid.expect("process pid"), Duration::from_secs(5)) + .expect("stop failed-node process"); + } + let seeds = config + .servers + .iter() + .filter(|server| server.service_type == crowdb_console_shared::config::ServiceType::Kv) + .map(|server| server.url.clone()) + .collect::>(); + let surviving_rpc = config + .servers + .iter() + .find(|server| { + server.service_type == crowdb_console_shared::config::ServiceType::Kv + && server.node_id != Some(failed_node) + }) + .and_then(|server| server.rpc_url.as_deref()) + .expect("surviving KV RPC") + .trim_start_matches("http://") + .to_owned(); + let ctx = OpContext::new(surviving_rpc, seeds.clone(), config); + ctx.sysmd() + .set_node_status(failed_node, failed_node, HwStatus::Offline) + .await + .expect("mark unavailable node offline"); + let kv = Arc::new(CrowdbKvClient::new(ClientConfig::new(seeds))); + let bindings = RangeBindingClient::from_shared(Arc::clone(&kv)); + let failed_instance = 20_000 + failed_node - 1; + let reassigned = tokio::time::timeout(Duration::from_secs(25), async { + loop { + if bindings.refresh().await.is_ok() + && bindings.snapshot().len() == 1_024 + && bindings + .snapshot() + .iter() + .all(|binding| binding.instance_id != failed_instance) + { + break; + } + tokio::time::sleep(Duration::from_millis(200)).await; + } + }) + .await; + if reassigned.is_err() { + let snapshot = bindings.snapshot(); + let stale = snapshot + .iter() + .filter(|binding| binding.instance_id == failed_instance) + .count(); + let instances = ServiceRegistryClient::from_shared(kv) + .read_all_instance_observations("chunkdb") + .await; + panic!( + "ChunkDB ranges did not move: bindings={}, stale={stale}, instances={instances:?}", + snapshot.len() + ); + } + + let (_, old_body) = client + .request( + Method::GET, + Some("outage-bucket"), + Some("before-outage"), + &[], + None, + None, + ) + .await + .expect("read existing object with one node stopped"); + assert_eq!(old_body, b"before-outage-bytes"); + let new_body = vec![0x5a; 2 * 1024 * 1024]; + client + .request( + Method::PUT, + Some("outage-bucket"), + Some("during-outage"), + &[], + Some(new_body.clone()), + None, + ) + .await + .expect("write new object with one node stopped"); + let (_, read_back) = client + .request( + Method::GET, + Some("outage-bucket"), + Some("during-outage"), + &[], + None, + None, + ) + .await + .expect("read new object with one node stopped"); + assert_eq!(read_back, new_body); + if failed_node == 2 { + let (outage_config, _) = s3::load(dir.path()).expect("load outage cluster"); + let surviving_chunkdb = outage_config + .servers + .iter() + .find(|server| { + server.service_type == crowdb_console_shared::config::ServiceType::Chunkdb + && server.node_id != Some(failed_node) + }) + .and_then(|server| server.rpc_url.as_deref()) + .expect("surviving ChunkDB RPC"); + let chunk_transport = Arc::new(ChunkdbRpcTransport::new()); + let chunks = chunk_transport + .send_list_chunks( + surviving_chunkdb, + &ListChunksRequest { + max_keys: 1_024, + ..ListChunksRequest::default() + }, + ) + .await + .expect("list chunks written during outage"); + let degraded_chunks = chunks + .chunks + .iter() + .filter(|chunk| chunk.chunk_type == ChunkType::S3 as i32) + .filter(|chunk| { + chunk.strips.iter().any(|strip| { + strip.strip_type == StripType::Ec as i32 + && matches!(strip.strip.as_ref(), Some(Strip::EcStrip(_))) + && strip.placement_repair_required + }) + }) + .filter_map(|chunk| chunk.id) + .collect::>(); + assert!( + !degraded_chunks.is_empty(), + "outage write did not persist degraded S3 EC placement" + ); + ctx.sysmd() + .set_node_status(failed_node, failed_node, HwStatus::Up) + .await + .expect("restore recovered node status"); + s3::stop(dir.path()).expect("stop protected cluster after outage"); + s3::start_protected_test_cluster(dir.path()) + .await + .expect("restart protected cluster after outage"); + let restarted = s3::S3HttpClient::from_data_dir(dir.path()).expect("restarted S3 client"); + let (_, recovered) = restarted + .request( + Method::GET, + Some("outage-bucket"), + Some("during-outage"), + &[], + None, + None, + ) + .await + .expect("read outage write after all processes restart"); + assert_eq!(recovered, new_body); + let (restarted_config, _) = s3::load(dir.path()).expect("load restarted cluster"); + let seeds = restarted_config + .servers + .iter() + .filter(|server| server.service_type == crowdb_console_shared::config::ServiceType::Kv) + .map(|server| server.url.clone()) + .collect(); + let kv = Arc::new(CrowdbKvClient::new(ClientConfig::new(seeds))); + let chunkdb = ChunkdbClient::new( + ServiceRegistryClient::from_shared(Arc::clone(&kv)), + Arc::new(ChunkdbRpcTransport::new()), + ) + .with_range_binding(RangeBindingClient::from_shared(kv)); + chunkdb + .refresh_routes() + .await + .expect("refresh restarted ChunkDB routes"); + tokio::time::timeout(Duration::from_secs(25), async { + loop { + let mut repaired = true; + for chunk_id in °raded_chunks { + let chunk = chunkdb + .query_chunk(QueryChunkRequest { + chunk_id: Some(*chunk_id), + }) + .await + .expect("query outage chunk after restart") + .chunk + .expect("outage chunk exists after restart"); + repaired &= chunk + .strips + .iter() + .filter(|strip| strip.strip_type == StripType::Ec as i32) + .all(|strip| { + !strip.placement_repair_required + && strip.placement_assessment.as_ref().is_some_and(|assessment| { + assessment.rack_protected + && assessment.node_protected + && assessment.disk_protected + }) + }); + } + if repaired { + break; + } + tokio::time::sleep(Duration::from_millis(200)).await; + } + }) + .await + .expect("placement repair did not complete after ChunkDB restart"); + } + s3::delete(dir.path()).expect("delete protected cluster"); +} diff --git a/lib/crowdb-kv-client/src/client/core.rs b/lib/crowdb-kv-client/src/client/core.rs index 67efabd8e..0831b25a4 100644 --- a/lib/crowdb-kv-client/src/client/core.rs +++ b/lib/crowdb-kv-client/src/client/core.rs @@ -148,16 +148,13 @@ impl CrowdbKvClient { } fn build(config: ClientConfig, transport: Option>) -> Self { - // Log client creation so accidental instance proliferation - // (each with its own topology cache + connection pool) is - // visible in logs. Standalone clients (no shared transport) - // are warned at WARN level — they should be rare; repeated - // creation is a red flag that the shared client is not being - // reused. Shared clients are logged at INFO (file only). + // A standalone client is normal at process startup. Keep the + // distinction available for diagnostics without warning on each + // test listener restart or each independently deployed process. if transport.is_none() { - tracing::warn!( + tracing::debug!( seed_count = config.mgmt_seeds.len(), - "CrowdbKvClient: new standalone instance created (no shared transport) — prefer from_shared() to reuse topology cache" + "CrowdbKvClient: new standalone instance created" ); } else { tracing::info!( diff --git a/lib/crowdb-protocol/Cargo.toml b/lib/crowdb-protocol/Cargo.toml index 54cb0ee9d..27ef70038 100644 --- a/lib/crowdb-protocol/Cargo.toml +++ b/lib/crowdb-protocol/Cargo.toml @@ -24,7 +24,7 @@ serde = { version = "1", features = ["derive"] } serde_json = "1" bincode = "1" crc32fast = "1" -crc32c = "0.6.8" +crowdb-common = { workspace = true } getrandom = "0.2" xxhash-rust = { version = "0.8", features = ["xxh64"] } sha2 = "0.10" diff --git a/lib/crowdb-protocol/src/chunk_id.rs b/lib/crowdb-protocol/src/chunk_id.rs index aa8754b2a..dd77d409f 100644 --- a/lib/crowdb-protocol/src/chunk_id.rs +++ b/lib/crowdb-protocol/src/chunk_id.rs @@ -30,6 +30,9 @@ pub const CHUNK_TYPE_REPO: u8 = 0; pub const CHUNK_TYPE_WAL: u8 = 1; pub const CHUNK_TYPE_BTREE_PAGE: u8 = 2; pub const CHUNK_TYPE_PAGE_INDEX: u8 = 3; +pub const CHUNK_TYPE_STREAM: u8 = 4; +pub const CHUNK_TYPE_S3: u8 = 5; +pub const CHUNK_TYPE_ICEBERG_TABLE: u8 = 6; /// 128-bit chunk ID parts — mirrors the proto `ChunkId` (high, low). #[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] diff --git a/lib/crowdb-protocol/src/fbs/chunkdb.fbs b/lib/crowdb-protocol/src/fbs/chunkdb.fbs index f772591da..7ca538640 100644 --- a/lib/crowdb-protocol/src/fbs/chunkdb.fbs +++ b/lib/crowdb-protocol/src/fbs/chunkdb.fbs @@ -67,6 +67,8 @@ enum FBChunkType : int16 { BtreePage = 2, PageIndex = 3, Stream = 4, + S3 = 5, + IcebergTable = 6, } // ── Strip types ────────────────────────────────────────────────── diff --git a/lib/crowdb-protocol/src/frame.rs b/lib/crowdb-protocol/src/frame.rs index 60271b299..01aaa56e2 100644 --- a/lib/crowdb-protocol/src/frame.rs +++ b/lib/crowdb-protocol/src/frame.rs @@ -121,14 +121,13 @@ pub fn parse_frame_views( for view in views { let included = footer_start.saturating_sub(cursor).min(view.len()); if included > 0 { - crc = !crc32c::crc32c_append(!crc, &view[..included]); + crc = crowdb_common::ec_isal::crc32c_update(crc, &view[..included]); } cursor = cursor.saturating_add(view.len()); if cursor >= footer_start { break; } } - crc = !crc32c::crc32c_append(!crc, &footer[4..]); if crc != expected_crc { return Err(FrameError::ChecksumMismatch); } @@ -211,6 +210,23 @@ pub fn encode_frame_regions( write_time_ms: u64, header_region: &mut [u8], footer_region: &mut [u8], +) -> Result<(), FrameError> { + prepare_frame_regions(magic, payload, write_time_ms, header_region, footer_region)?; + set_frame_chunk_id(chunk_id, footer_region) +} + +/// Prepare the frame header and CRC before its destination chunk is known. +/// The CRC covers the header and payload; the chunk ID is checked separately +/// against the expected location when the frame is read. +/// +/// # Errors +/// Returns an error for an oversized payload or incorrectly sized regions. +pub fn prepare_frame_regions( + magic: FrameMagic, + payload: &[u8], + write_time_ms: u64, + header_region: &mut [u8], + footer_region: &mut [u8], ) -> Result<(), FrameError> { if header_region.len() != FRAME_HEADER_PREFIX_BYTES || footer_region.len() != FRAME_FOOTER_BYTES { return Err(FrameError::InvalidRegionLength); @@ -224,10 +240,21 @@ pub fn encode_frame_regions( }; frame_length(header)?; write_header_region(header_region, header); + let checksum = crc32c_parts([header_region, payload]); + footer_region[..4].copy_from_slice(&checksum.to_le_bytes()); + Ok(()) +} + +/// Set the destination chunk after the frame header and CRC are prepared. +/// +/// # Errors +/// Returns an error for an incorrectly sized footer region. +pub fn set_frame_chunk_id(chunk_id: ChunkId, footer_region: &mut [u8]) -> Result<(), FrameError> { + if footer_region.len() != FRAME_FOOTER_BYTES { + return Err(FrameError::InvalidRegionLength); + } footer_region[4..12].copy_from_slice(&chunk_id.high.to_be_bytes()); footer_region[12..20].copy_from_slice(&chunk_id.low.to_be_bytes()); - let checksum = crc32c_parts([header_region, payload, &footer_region[4..]]); - footer_region[..4].copy_from_slice(&checksum.to_le_bytes()); Ok(()) } @@ -273,7 +300,7 @@ pub fn parse_frame(bytes: &[u8], expected_chunk_id: ChunkId) -> Result(parts: impl IntoIterator) -> u32 { let mut crc = 0_u32; for bytes in parts { - // The frame format stores the raw seed-zero CRC, while this API applies - // initial and final XOR. Invert around each append to preserve the wire value. - crc = !crc32c::crc32c_append(!crc, bytes); + crc = crowdb_common::ec_isal::crc32c_update(crc, bytes); } crc } - -fn crc32c_frame_parts(prefix: &[u8], chunk_id: &[u8]) -> u32 { - crc32c_parts([prefix, chunk_id]) -} diff --git a/lib/crowdb-protocol/src/lib.rs b/lib/crowdb-protocol/src/lib.rs index a61049edc..53c8cfad4 100644 --- a/lib/crowdb-protocol/src/lib.rs +++ b/lib/crowdb-protocol/src/lib.rs @@ -312,7 +312,8 @@ pub use common_type::{DiskGroupId, GroupId, InstanceId, NodeId, RackId, ReplicaI pub mod chunk_id; pub use chunk_id::{ generate as generate_chunk_id, is_zero as is_zero_chunk, ChunkIdParts, CHUNK_TYPE_BTREE_PAGE, - CHUNK_TYPE_PAGE_INDEX, CHUNK_TYPE_REPO, CHUNK_TYPE_WAL, + CHUNK_TYPE_ICEBERG_TABLE, CHUNK_TYPE_PAGE_INDEX, CHUNK_TYPE_REPO, CHUNK_TYPE_S3, CHUNK_TYPE_STREAM, + CHUNK_TYPE_WAL, }; pub mod key; diff --git a/lib/crowdb-protocol/src/types/chunkdb.rs b/lib/crowdb-protocol/src/types/chunkdb.rs index 361357be0..75b6e4a72 100644 --- a/lib/crowdb-protocol/src/types/chunkdb.rs +++ b/lib/crowdb-protocol/src/types/chunkdb.rs @@ -89,6 +89,8 @@ pub enum ChunkType { BtreePage = 2, PageIndex = 3, Stream = 4, + S3 = 5, + IcebergTable = 6, } impl_enum_conversions!( ChunkType, @@ -96,7 +98,9 @@ impl_enum_conversions!( Wal = 1, BtreePage = 2, PageIndex = 3, - Stream = 4 + Stream = 4, + S3 = 5, + IcebergTable = 6 ); #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash, Serialize, Deserialize, Default)] diff --git a/lib/crowdb-protocol/tests/chunk_id_test.rs b/lib/crowdb-protocol/tests/chunk_id_test.rs index 99ef66a10..aa5db3709 100644 --- a/lib/crowdb-protocol/tests/chunk_id_test.rs +++ b/lib/crowdb-protocol/tests/chunk_id_test.rs @@ -6,11 +6,27 @@ use std::collections::HashSet; use crowdb_protocol::chunk_id::{ - generate, is_zero, ChunkIdParts, CHUNK_TYPE_BTREE_PAGE, CHUNK_TYPE_PAGE_INDEX, CHUNK_TYPE_REPO, - CHUNK_TYPE_WAL, + generate, is_zero, ChunkIdParts, CHUNK_TYPE_BTREE_PAGE, CHUNK_TYPE_ICEBERG_TABLE, CHUNK_TYPE_PAGE_INDEX, + CHUNK_TYPE_REPO, CHUNK_TYPE_S3, CHUNK_TYPE_STREAM, CHUNK_TYPE_WAL, }; use crowdb_protocol::common::ChunkId; +#[test] +fn chunk_type_prefix_values_preserve_legacy_ids() { + assert_eq!( + [ + CHUNK_TYPE_REPO, + CHUNK_TYPE_WAL, + CHUNK_TYPE_BTREE_PAGE, + CHUNK_TYPE_PAGE_INDEX, + CHUNK_TYPE_STREAM, + CHUNK_TYPE_S3, + CHUNK_TYPE_ICEBERG_TABLE, + ], + [0, 1, 2, 3, 4, 5, 6] + ); +} + #[test] fn generate_sets_chunk_type() { for ct in [ @@ -18,6 +34,9 @@ fn generate_sets_chunk_type() { CHUNK_TYPE_WAL, CHUNK_TYPE_BTREE_PAGE, CHUNK_TYPE_PAGE_INDEX, + CHUNK_TYPE_STREAM, + CHUNK_TYPE_S3, + CHUNK_TYPE_ICEBERG_TABLE, ] { let id = generate(ct); assert_eq!(id.chunk_type(), ct, "chunk type bits must match"); diff --git a/lib/crowdb-protocol/tests/frame_test.rs b/lib/crowdb-protocol/tests/frame_test.rs index e130cbe6a..3524b62da 100644 --- a/lib/crowdb-protocol/tests/frame_test.rs +++ b/lib/crowdb-protocol/tests/frame_test.rs @@ -11,7 +11,7 @@ use crowdb_protocol::frame::{ const CHUNK: ChunkId = ChunkId { high: 7, low: 11 }; const REPO_SMALL_VECTOR: [u8; 37] = [ 0x01, 0x01, 0x0E, 0x00, 0x03, 0x00, 0x2A, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x01, 0x02, 0x03, - 0x96, 0x16, 0xD6, 0x1E, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x07, 0x00, 0x00, 0x00, 0x00, 0x00, + 0xEB, 0x45, 0x53, 0x37, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x07, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x0B, ]; @@ -55,7 +55,7 @@ fn split_frame_views_verify_without_assembling_payload() { } #[test] -fn frame_crc_matches_bitwise_reference_across_header_payload_and_chunk_id() { +fn frame_crc_matches_bitwise_reference_across_header_and_payload() { for length in [0, 1, 17, 1024, MAX_FRAME_PAYLOAD_BYTES] { let payload: Vec = (0..length) .map(|index| u8::try_from(index % 251).unwrap()) @@ -63,7 +63,7 @@ fn frame_crc_matches_bitwise_reference_across_header_payload_and_chunk_id() { let frame = encode_frame(FrameMagic::RepoLargeV1, CHUNK, &payload, 42).unwrap(); let footer = frame.len() - FRAME_FOOTER_BYTES; let mut crc = 0_u32; - for byte in frame[..footer].iter().chain(frame[footer + 4..].iter()) { + for byte in &frame[..footer] { crc ^= u32::from(*byte); for _ in 0..8 { crc = (crc >> 1) ^ (0x82F6_3B78 & (0_u32.wrapping_sub(crc & 1))); diff --git a/lib/crowdb-rpc/CMakeLists.txt b/lib/crowdb-rpc/CMakeLists.txt index f25f69ced..8bd17d220 100644 --- a/lib/crowdb-rpc/CMakeLists.txt +++ b/lib/crowdb-rpc/CMakeLists.txt @@ -112,8 +112,12 @@ endif() # against. Set before find_package(folly) so folly's internal Boost lookup # inherits these. cmake_policy(SET CMP0167 OLD) +cmake_policy(SET CMP0144 NEW) set(BOOST_ROOT $ENV{CONDA_PREFIX}) set(Boost_NO_BOOST_CMAKE ON) +# The pinned pixi Boost can be newer than FindBoost's dependency table; Folly +# links the required components explicitly, so this version warning adds noise. +set(Boost_NO_WARN_NEW_VERSIONS ON) list(APPEND CMAKE_IGNORE_PATH /opt/boost) # ── folly (ConcurrentHashMap for pending map) ───────────────────── diff --git a/lib/crowdb-test-harness/src/chunkdb.rs b/lib/crowdb-test-harness/src/chunkdb.rs index 404401ecf..46567f57d 100644 --- a/lib/crowdb-test-harness/src/chunkdb.rs +++ b/lib/crowdb-test-harness/src/chunkdb.rs @@ -86,7 +86,7 @@ fn prepare_runtime(runtime: &mut crate::test_dirs::TestRuntime) -> ChunkdbRuntim } } -#[derive(Clone, Copy, Debug, Default)] +#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)] pub enum ChunkdbPlacementMode { #[default] Protected, @@ -102,9 +102,25 @@ impl ChunkdbPlacementMode { } } +fn deployment_mode(options: ChunkdbStartOptions) -> &'static str { + if options.test_single_node { + return "test_single_node"; + } + if options.placement_mode == ChunkdbPlacementMode::UnsafeColocated + || options.allow_unsafe_ec + || options.allow_degraded_failure_domains + || options.repair_allow_unsafe_placement + { + "test_unsafe_placement" + } else { + "production" + } +} + #[derive(Clone, Copy, Debug)] #[allow(clippy::struct_excessive_bools)] pub struct ChunkdbStartOptions { + pub test_single_node: bool, pub placement_mode: ChunkdbPlacementMode, pub allow_unsafe_ec: bool, pub allow_degraded_failure_domains: bool, @@ -123,6 +139,7 @@ pub struct ChunkdbStartOptions { impl Default for ChunkdbStartOptions { fn default() -> Self { Self { + test_single_node: false, placement_mode: ChunkdbPlacementMode::Protected, allow_unsafe_ec: false, allow_degraded_failure_domains: false, @@ -193,8 +210,14 @@ impl ChunkdbProcess { let rpc_port = paths.rpc_port; let http_port = paths.http_port; + let deployment_mode = deployment_mode(options); + let max_node_failures = u32::from(!options.test_single_node); let config_content = format!( - r#"[server] + r#"[deployment] +mode = "{deployment_mode}" +max_node_failures = {max_node_failures} + +[server] rpc_workers = 2 listen_addr = "127.0.0.1:{listen_port}" rpc_listen_addr = "127.0.0.1:{rpc_port}" diff --git a/lib/crowdb-tree/ffi/Cargo.lock b/lib/crowdb-tree/ffi/Cargo.lock index 7aabe4f3a..58becbed1 100644 --- a/lib/crowdb-tree/ffi/Cargo.lock +++ b/lib/crowdb-tree/ffi/Cargo.lock @@ -26,7 +26,7 @@ checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" [[package]] name = "crowtree-ffi" -version = "0.1.0" +version = "0.2.0" dependencies = [ "cc", "tempfile", diff --git a/lib/crowdb-tree/ffi/src/chunk.rs b/lib/crowdb-tree/ffi/src/chunk.rs index 813e0941a..3be15c086 100644 --- a/lib/crowdb-tree/ffi/src/chunk.rs +++ b/lib/crowdb-tree/ffi/src/chunk.rs @@ -22,6 +22,8 @@ pub struct ChunkPageStoreOptions { pub iu_size: u32, pub max_concurrent_packs: usize, pub materialization_bytes_per_pass: u64, + pub max_chunk_bytes: u64, + pub mirror_copies: u32, } #[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] @@ -435,6 +437,7 @@ pub struct ChunkRpcTransportOptions<'a> { pub writer_lease_ms: u64, pub rpc_timeout_ms: u64, pub completion_capacity: u32, + pub mirror_copies: u32, } #[cfg(feature = "chunk-rpc")] @@ -453,6 +456,7 @@ pub struct OwnedChunkRpcTransportOptions { pub writer_lease_ms: u64, pub rpc_timeout_ms: u64, pub completion_capacity: u32, + pub mirror_copies: u32, } #[cfg(feature = "chunk-rpc")] @@ -493,6 +497,7 @@ impl ChunkTransport { writer_lease_ms: options.writer_lease_ms, rpc_timeout_ms: options.rpc_timeout_ms, completion_capacity: options.completion_capacity, + mirror_copies: options.mirror_copies, }; let mut out = std::ptr::null_mut(); check(unsafe { sys::ct_rpc_chunk_transport_open(&raw, &mut out) })?; @@ -522,6 +527,7 @@ impl ChunkTransport { writer_lease_ms: options.writer_lease_ms, rpc_timeout_ms: options.rpc_timeout_ms, completion_capacity: options.completion_capacity, + mirror_copies: options.mirror_copies, }; let mut transport = unsafe { Self::open_rpc(&raw_options) }?; transport._routes = Some(OwnedTransportRoutes { @@ -579,6 +585,8 @@ impl PageStore { iu_size: options.iu_size, max_concurrent_packs: options.max_concurrent_packs, materialization_bytes_per_pass: options.materialization_bytes_per_pass, + max_chunk_bytes: options.max_chunk_bytes, + mirror_copies: options.mirror_copies, }; let mut out = std::ptr::null_mut(); let status = match transport { diff --git a/lib/crowdb-tree/ffi/src/sys.rs b/lib/crowdb-tree/ffi/src/sys.rs index ccddf43e3..ad26aa995 100644 --- a/lib/crowdb-tree/ffi/src/sys.rs +++ b/lib/crowdb-tree/ffi/src/sys.rs @@ -63,6 +63,8 @@ pub struct ct_chunk_page_store_options { pub iu_size: u32, pub max_concurrent_packs: usize, pub materialization_bytes_per_pass: u64, + pub max_chunk_bytes: u64, + pub mirror_copies: u32, } #[repr(C)] @@ -87,6 +89,7 @@ pub struct ct_chunk_rpc_transport_options { pub writer_lease_ms: u64, pub rpc_timeout_ms: u64, pub completion_capacity: u32, + pub mirror_copies: u32, } #[repr(C)] diff --git a/lib/crowdb-tree/ffi/tests/ffi_test.rs b/lib/crowdb-tree/ffi/tests/ffi_test.rs index 66746dc63..8ea762b46 100644 --- a/lib/crowdb-tree/ffi/tests/ffi_test.rs +++ b/lib/crowdb-tree/ffi/tests/ffi_test.rs @@ -143,6 +143,7 @@ fn owned_chunk_rpc_transport_retains_route_handles() { writer_lease_ms: 30_000, rpc_timeout_ms: 1_000, completion_capacity: 32, + mirror_copies: 0, }) .unwrap(); drop(transport); @@ -322,6 +323,8 @@ fn injected_chunk_store_round_trip_and_stats() { iu_size: 1, max_concurrent_packs: 2, materialization_bytes_per_pass: 4096, + max_chunk_bytes: 0, + mirror_copies: 0, }, Arc::clone(&catalog), None, @@ -369,6 +372,8 @@ fn callback_root_catalog_reopens_published_manifest() { iu_size: 1, max_concurrent_packs: 2, materialization_bytes_per_pass: 4096, + max_chunk_bytes: 0, + mirror_copies: 0, }; let store = Arc::new(PageStore::open_chunk(options, Arc::clone(&catalog), None).unwrap()); { @@ -419,6 +424,8 @@ fn callback_root_catalog_persists_transition_generation_pin() { iu_size: 1, max_concurrent_packs: 2, materialization_bytes_per_pass: 4096, + max_chunk_bytes: 0, + mirror_copies: 0, }; let store = Arc::new(PageStore::open_chunk(options, Arc::clone(&catalog), None).unwrap()); let tree = Crowdbtree::open(&Config { @@ -456,6 +463,8 @@ fn memory_root_catalog_pin_blocks_generation_reclaim_until_unpin() { iu_size: 1, max_concurrent_packs: 2, materialization_bytes_per_pass: 4096, + max_chunk_bytes: 0, + mirror_copies: 0, }; let store = Arc::new(PageStore::open_chunk(options, Arc::clone(&catalog), None).unwrap()); let tree = Crowdbtree::open(&Config { @@ -517,6 +526,8 @@ fn callback_root_catalog_opens_exact_manifest_without_latest_fallback() { iu_size: 1, max_concurrent_packs: 2, materialization_bytes_per_pass: 4096, + max_chunk_bytes: 0, + mirror_copies: 0, }; let latest_store = Arc::new(PageStore::open_chunk(latest_options, Arc::clone(&catalog), None).unwrap()); let latest = Crowdbtree::open(&Config { @@ -576,6 +587,8 @@ fn exact_manifest_tracks_durable_snapshot_after_range_rebuild() { iu_size: 1, max_concurrent_packs: 2, materialization_bytes_per_pass: 4096, + max_chunk_bytes: 0, + mirror_copies: 0, }; let store = Arc::new(PageStore::open_chunk(options, Arc::clone(&catalog), None).unwrap()); let config = Config { @@ -634,6 +647,8 @@ fn published_manifest_is_visible_through_the_same_page_store() { iu_size: 65_536, max_concurrent_packs: 2, materialization_bytes_per_pass: 4096, + max_chunk_bytes: 0, + mirror_copies: 0, }, catalog, None, diff --git a/lib/crowdb-tree/include/crowdb-tree/c_api.h b/lib/crowdb-tree/include/crowdb-tree/c_api.h index cb45b600f..497411768 100644 --- a/lib/crowdb-tree/include/crowdb-tree/c_api.h +++ b/lib/crowdb-tree/include/crowdb-tree/c_api.h @@ -153,6 +153,8 @@ using ct_chunk_page_store_options = struct uint32_t iu_size; // 0 => 64 KiB page framing size_t max_concurrent_packs; // 0 => 8 uint64_t materialization_bytes_per_pass; // 0 => 64 MiB; clamped to one pack + uint64_t max_chunk_bytes; // 0 => 256 MiB + uint32_t mirror_copies; // 0 => 2 }; using ct_chunk_page_store_stats = struct @@ -212,6 +214,7 @@ struct ct_chunk_rpc_transport_options uint64_t writer_lease_ms; uint64_t rpc_timeout_ms; // 0 => 30 seconds uint32_t completion_capacity; + uint32_t mirror_copies; // 0 => 2 }; ct_status ct_memory_root_catalog_open(uint64_t owner_epoch, ct_root_catalog **out); diff --git a/lib/crowdb-tree/src/backend/chunk/chunk_pack_pipeline.cpp b/lib/crowdb-tree/src/backend/chunk/chunk_pack_pipeline.cpp index 04a09e1c1..4d54ceab2 100644 --- a/lib/crowdb-tree/src/backend/chunk/chunk_pack_pipeline.cpp +++ b/lib/crowdb-tree/src/backend/chunk/chunk_pack_pipeline.cpp @@ -51,6 +51,7 @@ struct MirrorWriteSource void *operation = nullptr; uint32_t attempt = 0; uint64_t started_at_ns = 0; + bool disabled = false; static void submit(void *context, CallbackComplete complete_fn, void *operation_context) { @@ -58,6 +59,10 @@ struct MirrorWriteSource self->complete = complete_fn; self->operation = operation_context; self->attempt = 0; + if (self->disabled) { + complete_fn(operation_context, CallbackSignal::kValue, Status::Ok()); + return; + } self->submit_attempt(); } @@ -131,17 +136,18 @@ struct PackReceiver }; using PackSender = decltype(stdexec::when_all(std::declval(), std::declval(), + std::declval(), std::declval(), std::declval())); using PackOperation = decltype(stdexec::connect(std::declval(), std::declval())); struct PackWrite { - ChunkPagePack pack; - uint64_t physical_length = 0; - uint64_t source_offset = 0; - std::vector framed; - std::array mirrors; - std::unique_ptr operation; + ChunkPagePack pack; + uint64_t physical_length = 0; + uint64_t source_offset = 0; + std::vector framed; + std::array mirrors; + std::unique_ptr operation; }; class ChunkPackPipelineImpl final : public ChunkPackPipeline, public std::enable_shared_from_this @@ -234,7 +240,8 @@ class ChunkPackPipelineImpl final : public ChunkPackPipeline, public std::enable (store->config_.max_chunk_bytes + store->config_.pack_bytes - 1) / store->config_.pack_bytes; const uint64_t physical_pack_bytes = crowdb::protocol::framed_physical_length(store->config_.pack_bytes); - if (packs_per_chunk > std::numeric_limits::max() / physical_pack_bytes) { + if (physical_pack_bytes == 0 || + packs_per_chunk > std::numeric_limits::max() / physical_pack_bytes) { return Status::resource_exhausted("chunk page framing exceeds address space"); } Status status = store->transport_->allocate_mirror_chunk(packs_per_chunk * physical_pack_bytes, @@ -417,11 +424,14 @@ class ChunkPackPipelineImpl final : public ChunkPackPipeline, public std::enable .stop_requested = &stop_requested, .diskio_operations = &store->diskio_operations_, .diskio_latency_ns = &store->diskio_latency_ns_, + .disabled = mirror >= store->config_.mirror_copies, }; } auto sender = stdexec::when_all(CallbackSender(job.mirrors.data(), &MirrorWriteSource::submit), CallbackSender(&job.mirrors[1], &MirrorWriteSource::submit), - CallbackSender(&job.mirrors[2], &MirrorWriteSource::submit)); + CallbackSender(&job.mirrors[2], &MirrorWriteSource::submit), + CallbackSender(&job.mirrors[3], &MirrorWriteSource::submit), + CallbackSender(&job.mirrors[4], &MirrorWriteSource::submit)); // PackOperation is immovable (STDEXEC_IMMOVABLE), so make_unique // cannot be used; construct directly from the connect() prvalue. // std::move(sender) is required: connect() takes Sender&&. diff --git a/lib/crowdb-tree/src/backend/chunk/chunk_page_store.cpp b/lib/crowdb-tree/src/backend/chunk/chunk_page_store.cpp index 6bfc605ea..c5c4d7034 100644 --- a/lib/crowdb-tree/src/backend/chunk/chunk_page_store.cpp +++ b/lib/crowdb-tree/src/backend/chunk/chunk_page_store.cpp @@ -833,6 +833,9 @@ ChunkPageStore::ChunkPageStore(Config config, std::shared_ptr catal if (config_.max_chunk_bytes == 0) { config_.max_chunk_bytes = 256U * 1024U * 1024U; } + if (config_.mirror_copies == 0) { + config_.mirror_copies = 2; + } if (config_.page_alignment == 0) { config_.page_alignment = 64U * 1024U; } @@ -1214,7 +1217,7 @@ Status ChunkPageStore::read_pack(const ChunkPageRef &ref, std::shared_ptr>(physical_length); bool mirror_responded = false; - for (uint32_t mirror = 0; mirror < 3; ++mirror) { + for (uint32_t mirror = 0; mirror < config_.mirror_copies; ++mirror) { if (cancellation.cancelled()) { return Status::unavailable("chunk page read cancelled"); } @@ -1566,7 +1569,7 @@ Status ChunkPageStore::build_manifest(uint64_t expected_generation, std::shared_ return Status::resource_exhausted("chunk page frame encoding failed"); } *new_pack_bytes += length; - for (uint32_t mirror = 0; mirror < 3; ++mirror) { + for (uint32_t mirror = 0; mirror < config_.mirror_copies; ++mirror) { bool written = false; for (uint32_t attempt = 0; attempt <= config_.mirror_retry_limit; ++attempt) { if (cancellation.cancelled()) { @@ -1754,7 +1757,7 @@ Status ChunkPageStore::materialize_ownership(uint64_t *bytes_written, bool *comp orphan_bytes_.fetch_add(copied_bytes, std::memory_order_relaxed); return fail(Status::resource_exhausted("chunk materialization frame encoding failed")); } - for (uint32_t mirror = 0; mirror < 3; ++mirror) { + for (uint32_t mirror = 0; mirror < config_.mirror_copies; ++mirror) { bool written = false; for (uint32_t attempt = 0; attempt <= config_.mirror_retry_limit; ++attempt) { mirror_write_attempts_.fetch_add(1, std::memory_order_relaxed); @@ -2288,7 +2291,8 @@ void ct_root_catalog_free(ct_root_catalog *catalog) ct_status ct_chunk_page_store_open(const ct_chunk_page_store_options *options, ct_root_catalog *catalog, ct_page_store **out) { - if (options == nullptr || catalog == nullptr || catalog->catalog == nullptr || out == nullptr) { + if (options == nullptr || catalog == nullptr || catalog->catalog == nullptr || out == nullptr || + options->mirror_copies > crowdb::tree::detail::kMaxMirrorCopies) { return static_cast(crowdb::tree::Code::kInvalidArgument); } auto handle = std::make_unique(); @@ -2299,6 +2303,8 @@ ct_status ct_chunk_page_store_open(const ct_chunk_page_store_options *options, c .owner_epoch = options->owner_epoch, .open_generation = options->open_generation, .pack_bytes = options->pack_bytes, + .max_chunk_bytes = options->max_chunk_bytes, + .mirror_copies = options->mirror_copies, .iu_size = options->iu_size, .max_concurrent_packs = options->max_concurrent_packs, .materialization_bytes_per_pass = options->materialization_bytes_per_pass, @@ -2318,7 +2324,8 @@ ct_status ct_chunk_page_store_open_with_transport(const ct_chunk_page_store_opti ct_chunk_transport *transport, ct_page_store **out) { if (options == nullptr || catalog == nullptr || catalog->catalog == nullptr || transport == nullptr || - transport->transport == nullptr || out == nullptr) { + transport->transport == nullptr || out == nullptr || + options->mirror_copies > crowdb::tree::detail::kMaxMirrorCopies) { return static_cast(crowdb::tree::Code::kInvalidArgument); } auto handle = std::make_unique(); @@ -2329,6 +2336,8 @@ ct_status ct_chunk_page_store_open_with_transport(const ct_chunk_page_store_opti .owner_epoch = options->owner_epoch, .open_generation = options->open_generation, .pack_bytes = options->pack_bytes, + .max_chunk_bytes = options->max_chunk_bytes, + .mirror_copies = options->mirror_copies, .iu_size = options->iu_size, .max_concurrent_packs = options->max_concurrent_packs, .materialization_bytes_per_pass = options->materialization_bytes_per_pass, diff --git a/lib/crowdb-tree/src/backend/chunk/chunk_page_store.h b/lib/crowdb-tree/src/backend/chunk/chunk_page_store.h index ea8f55f3e..bbb1cb8ff 100644 --- a/lib/crowdb-tree/src/backend/chunk/chunk_page_store.h +++ b/lib/crowdb-tree/src/backend/chunk/chunk_page_store.h @@ -217,17 +217,17 @@ class CallbackRootCatalog final : public RootCatalog uint64_t generation) override; Status unpin_generation(uint64_t tree_id, uint64_t transition_high, uint64_t transition_low) override; - [[nodiscard]] uint64_t retained_manifest_count(uint64_t) const override + [[nodiscard]] uint64_t retained_manifest_count(uint64_t /*tree_id*/) const override { return 0; } - [[nodiscard]] uint64_t pinned_bytes(uint64_t) const override + [[nodiscard]] uint64_t pinned_bytes(uint64_t /*tree_id*/) const override { return 0; } - [[nodiscard]] uint64_t oldest_pin_age_ms(uint64_t) const override + [[nodiscard]] uint64_t oldest_pin_age_ms(uint64_t /*tree_id*/) const override { return 0; } @@ -280,8 +280,9 @@ class ChunkPageStore final : public PageStore, public AsyncPageStore uint64_t tree_id = 0; uint64_t owner_epoch = 0; uint64_t open_generation = 0; - size_t pack_bytes = 64U * 1024U - 34U; + size_t pack_bytes = (64U * 1024U) - 34U; uint64_t max_chunk_bytes = 256U * 1024U * 1024U; + uint32_t mirror_copies = 2; uint32_t page_alignment = 64U * 1024U; uint32_t iu_size = 64U * 1024U; uint32_t mirror_retry_limit = 2; diff --git a/lib/crowdb-tree/src/backend/chunk/chunk_transport.cpp b/lib/crowdb-tree/src/backend/chunk/chunk_transport.cpp index 03f62f089..6c3c863b7 100644 --- a/lib/crowdb-tree/src/backend/chunk/chunk_transport.cpp +++ b/lib/crowdb-tree/src/backend/chunk/chunk_transport.cpp @@ -250,8 +250,8 @@ Status MemoryChunkTransport::allocate_mirror_chunk(uint64_t logical_capacity, ui Status MemoryChunkTransport::write_mirror(ChunkId chunk_id, uint32_t mirror_index, uint64_t offset, const uint8_t *data, size_t length) { - if (mirror_index >= 3 || (data == nullptr && length != 0) || offset > std::numeric_limits::max() || - length > std::numeric_limits::max() - offset) { + if (mirror_index >= kMaxMirrorCopies || (data == nullptr && length != 0) || + offset > std::numeric_limits::max() || length > std::numeric_limits::max() - offset) { return Status::invalid_argument("chunk mirror write arguments are invalid"); } return mutate(chunk_id, [mirror_index, offset, data, length](Chunk &chunk) { @@ -291,7 +291,7 @@ Status MemoryChunkTransport::read_mirror(ChunkId chunk_id, uint32_t mirror_index if (unavailable_.load(std::memory_order_acquire)) { return Status::unavailable("chunk transport is unavailable"); } - if (mirror_index >= 3 || (data == nullptr && length != 0)) { + if (mirror_index >= kMaxMirrorCopies || (data == nullptr && length != 0)) { return Status::invalid_argument("chunk mirror read arguments are invalid"); } auto chunks = chunks_.load(std::memory_order_acquire); @@ -345,7 +345,7 @@ Status MemoryChunkTransport::seal_chunk(ChunkId chunk_id, uint64_t owner_epoch, void MemoryChunkTransport::corrupt_mirror(ChunkId chunk_id, uint32_t mirror_index, uint64_t offset) { - if (mirror_index >= 3) { + if (mirror_index >= kMaxMirrorCopies) { return; } static_cast(mutate(chunk_id, [mirror_index, offset](Chunk &chunk) { diff --git a/lib/crowdb-tree/src/backend/chunk/chunk_transport.h b/lib/crowdb-tree/src/backend/chunk/chunk_transport.h index 0ed31d123..ee4c6cf28 100644 --- a/lib/crowdb-tree/src/backend/chunk/chunk_transport.h +++ b/lib/crowdb-tree/src/backend/chunk/chunk_transport.h @@ -16,6 +16,8 @@ namespace crowdb::tree::detail { +inline constexpr uint32_t kMaxMirrorCopies = 5; + struct ChunkId { uint64_t high = 0; @@ -84,7 +86,7 @@ class ChunkTransport }; // Lock-free immutable-snapshot transport for tests and embedded use. It models -// the same allocation, three-mirror write, acknowledged-cursor, and seal +// the same allocation, mirror write, acknowledged-cursor, and seal // boundaries as the production RPC transport. class MemoryChunkTransport final : public ChunkTransport { @@ -108,9 +110,9 @@ class MemoryChunkTransport final : public ChunkTransport private: struct Chunk { - ChunkLayout layout; - uint64_t owner_epoch = 0; - std::array, 3> mirrors; + ChunkLayout layout; + uint64_t owner_epoch = 0; + std::array, kMaxMirrorCopies> mirrors; }; using Chunks = std::vector; diff --git a/lib/crowdb-tree/src/backend/chunk/rpc_chunk_transport.cpp b/lib/crowdb-tree/src/backend/chunk/rpc_chunk_transport.cpp index 6f3f402f6..a4318950e 100644 --- a/lib/crowdb-tree/src/backend/chunk/rpc_chunk_transport.cpp +++ b/lib/crowdb-tree/src/backend/chunk/rpc_chunk_transport.cpp @@ -220,10 +220,10 @@ struct RpcChunkTransport::Impl struct Strip { - uint64_t chunk_offset = 0; - uint64_t capacity = 0; - uint32_t unit_kb = 0; - std::array mirrors; + uint64_t chunk_offset = 0; + uint64_t capacity = 0; + uint32_t unit_kb = 0; + std::vector mirrors; }; struct RemoteChunk @@ -275,8 +275,8 @@ struct RpcChunkTransport::Impl [[nodiscard]] bool valid() const { - return options.chunkdb.client != nullptr && options.chunkdb.server != nullptr && - options.chunkdb.connection != nullptr && !disk_routes.empty(); + return options.mirror_copies <= kMaxMirrorCopies && options.chunkdb.client != nullptr && + options.chunkdb.server != nullptr && options.chunkdb.connection != nullptr && !disk_routes.empty(); } uint64_t next_request_id() const @@ -314,21 +314,22 @@ struct RpcChunkTransport::Impl for (const auto *wire_strip : *chunk->strips()) { const auto *mirror = wire_strip == nullptr ? nullptr : wire_strip->strip_body_as_FBMirrorStrip(); if (wire_strip == nullptr || wire_strip->strip_type() != FBStripType_Mirror || mirror == nullptr || - mirror->segments() == nullptr || mirror->segments()->size() != 3 || wire_strip->unit_kb() == 0) { + mirror->segments() == nullptr || mirror->segments()->empty() || wire_strip->unit_kb() == 0) { return Status::corruption("ChunkDB returned a non-mirror tree chunk layout"); } Strip strip{.chunk_offset = static_cast(wire_strip->chunk_offset()) * 1024, .capacity = static_cast(wire_strip->capacity()) * 1024, .unit_kb = wire_strip->unit_kb(), .mirrors = {}}; - for (size_t index = 0; index < strip.mirrors.size(); ++index) { - const auto *segment = mirror->segments()->Get(index); - strip.mirrors[index] = { + strip.mirrors.reserve(mirror->segments()->size()); + for (size_t index = 0; index < mirror->segments()->size(); ++index) { + const auto *segment = mirror->segments()->Get(index); + strip.mirrors.push_back({ .disk_high = segment->disk_id().high(), .disk_low = segment->disk_id().low(), .unit_offset = segment->unit_offset(), .zone_index = segment->zone_index(), - }; + }); } parsed.strips.push_back(strip); } @@ -532,6 +533,10 @@ struct RpcChunkTransport::Impl::AsyncWrite finish(this, Status::resource_exhausted("tree chunk RPC write exceeds allocated strips")); return; } + if (mirror_index >= strip->mirrors.size()) { + finish(this, Status::invalid_argument("tree chunk mirror index exceeds layout")); + return; + } const size_t part = std::min(length - consumed, strip->chunk_offset + strip->capacity - cursor); const auto &segment = strip->mirrors[mirror_index]; ct_chunk_rpc_route route{}; @@ -612,12 +617,16 @@ Status RpcChunkTransport::allocate_mirror_chunk(uint64_t logical_capacity, uint6 logical_capacity > std::numeric_limits::max()) { return Status::invalid_argument("tree chunk RPC allocation arguments are invalid"); } - const uint64_t request_id = impl_->next_request_id(); - const auto granularity_kb = static_cast((logical_capacity + 1023) / 1024); + constexpr uint64_t kMiB = 1024U * 1024U; + const uint64_t request_id = impl_->next_request_id(); + const bool single_copy = impl_->options.mirror_copies == 1; + const auto granularity_kb = single_copy ? 1024U : static_cast((logical_capacity + 1023) / 1024); + const auto strip_count = single_copy ? static_cast((logical_capacity + kMiB - 1) / kMiB) : 1U; flatbuffers::FlatBufferBuilder builder; auto request = crowdb::chunkdb::proto::CreateFBAllocateChunkRequest( - builder, request_id, monotonic_nanos(), nullptr, granularity_kb, 1, FBStripType_Mirror, 0, 0, 3, - FBChunkType_BtreePage, owner_epoch, impl_->options.writer_lease_ms); + builder, request_id, monotonic_nanos(), nullptr, granularity_kb, strip_count, FBStripType_Mirror, 0, 0, + impl_->options.mirror_copies == 0 ? 2 : impl_->options.mirror_copies, FBChunkType_BtreePage, owner_epoch, + impl_->options.writer_lease_ms); builder.Finish(request); std::vector control(builder.GetBufferPointer(), builder.GetBufferPointer() + builder.GetSize()); RpcResult result; @@ -650,7 +659,7 @@ Status RpcChunkTransport::allocate_mirror_chunk(uint64_t logical_capacity, uint6 Status RpcChunkTransport::write_mirror(ChunkId chunk_id, uint32_t mirror_index, uint64_t offset, const uint8_t *data, size_t length) { - if (mirror_index >= 3 || (data == nullptr && length != 0)) { + if ((data == nullptr && length != 0)) { return Status::invalid_argument("tree chunk RPC mirror write arguments are invalid"); } Impl::RemoteChunk chunk; @@ -669,6 +678,9 @@ Status RpcChunkTransport::write_mirror(ChunkId chunk_id, uint32_t mirror_index, if (strip == chunk.strips.end()) { return Status::resource_exhausted("tree chunk RPC write exceeds allocated strips"); } + if (mirror_index >= strip->mirrors.size()) { + return Status::invalid_argument("tree chunk mirror index exceeds layout"); + } const size_t part = std::min(length - consumed, strip->chunk_offset + strip->capacity - cursor); const auto &segment = strip->mirrors[mirror_index]; ct_chunk_rpc_route route{}; @@ -708,7 +720,7 @@ Status RpcChunkTransport::write_mirror(ChunkId chunk_id, uint32_t mirror_index, void RpcChunkTransport::submit_write_mirror(ChunkId chunk_id, uint32_t mirror_index, uint64_t offset, const uint8_t *data, size_t length, ChunkTransportCompletion completion) { - if (mirror_index >= 3 || (data == nullptr && length != 0)) { + if ((data == nullptr && length != 0)) { completion.complete(Status::invalid_argument("tree chunk RPC mirror write arguments are invalid")); return; } @@ -773,7 +785,7 @@ Status RpcChunkTransport::advance_write(ChunkId chunk_id, uint64_t expected_byte Status RpcChunkTransport::read_mirror(ChunkId chunk_id, uint32_t mirror_index, uint64_t offset, uint8_t *data, size_t length) const { - if (mirror_index >= 3 || (data == nullptr && length != 0)) { + if ((data == nullptr && length != 0)) { return Status::invalid_argument("tree chunk RPC mirror read arguments are invalid"); } Impl::RemoteChunk chunk; @@ -796,6 +808,9 @@ Status RpcChunkTransport::read_mirror(ChunkId chunk_id, uint32_t mirror_index, u if (strip == chunk.strips.end()) { return Status::corruption("tree chunk RPC read has a layout gap"); } + if (mirror_index >= strip->mirrors.size()) { + return Status::invalid_argument("tree chunk mirror index exceeds layout"); + } const size_t part = std::min(length - consumed, strip->chunk_offset + strip->capacity - cursor); const auto &segment = strip->mirrors[mirror_index]; ct_chunk_rpc_route route{}; diff --git a/lib/crowdb-tree/tests/integration/chunk_page_store_test.cpp b/lib/crowdb-tree/tests/integration/chunk_page_store_test.cpp index 14bec58a5..82d046cf1 100644 --- a/lib/crowdb-tree/tests/integration/chunk_page_store_test.cpp +++ b/lib/crowdb-tree/tests/integration/chunk_page_store_test.cpp @@ -403,13 +403,13 @@ TEST(ChunkPageStore, AsyncPackPipelineFansOutMirrorWrites) transport->hold_writes(); CompletionState completed; ASSERT_TRUE(store.submit_fsync({.context = &completed, .complete_fn = &record_completion}).ok()); - transport->wait_for_concurrent_writes(3); - EXPECT_GE(transport->max_active_writes(), 3U); + transport->wait_for_concurrent_writes(2); + EXPECT_GE(transport->max_active_writes(), 2U); EXPECT_FALSE(completed.done.load(std::memory_order_acquire)); transport->release_writes(); completed.done.wait(false, std::memory_order_acquire); EXPECT_EQ(completed.code.load(std::memory_order_relaxed), static_cast(Code::kOk)); - EXPECT_LE(transport->max_active_writes(), 3U); + EXPECT_LE(transport->max_active_writes(), 2U); ASSERT_NE(catalog->load(29), nullptr); } @@ -540,7 +540,7 @@ TEST(ChunkPageStore, ShutdownCancelsAndDrainsBlockedPackWrites) transport->hold_writes(); CompletionState completed; ASSERT_TRUE(store->submit_fsync({.context = &completed, .complete_fn = &record_completion}).ok()); - transport->wait_for_concurrent_writes(3); + transport->wait_for_concurrent_writes(2); std::thread closer([&store] { store.reset(); }); std::this_thread::sleep_for(std::chrono::milliseconds(10)); transport->release_writes(); @@ -554,7 +554,8 @@ TEST(ChunkPageStore, SnapshotPublishesBoundedChecksummedPacksAndReopens) { auto catalog = std::make_shared(7); auto transport = std::make_shared(); - ChunkPageStore::Config config{.tree_id = 42, .owner_epoch = 7, .pack_bytes = 4096, .iu_size = 1}; + ChunkPageStore::Config config{ + .tree_id = 42, .owner_epoch = 7, .pack_bytes = 4096, .mirror_copies = 3, .iu_size = 1}; { ChunkPageStore store(config, catalog, transport); Config options; @@ -1288,7 +1289,7 @@ TEST(ChunkPageStore, AvailabilityAndCorruptionRemainDistinct) transport->inject_unavailable(false); auto manifest = catalog->load(5); ASSERT_NE(manifest, nullptr); - for (uint32_t mirror = 0; mirror < 3; ++mirror) { + for (uint32_t mirror = 0; mirror < 2; ++mirror) { transport->corrupt_mirror(manifest->packs[2].ref.chunk_id, mirror, manifest->packs[2].ref.offset); } ChunkPageStore corrupted({.tree_id = 5, .owner_epoch = 1, .pack_bytes = 4096, .iu_size = 1}, catalog, transport); @@ -1777,6 +1778,8 @@ TEST(ChunkPageStore, CApiFactoryInjectsBackendWithoutChangingOpen) .iu_size = 1, .max_concurrent_packs = 2, .materialization_bytes_per_pass = 4096, + .max_chunk_bytes = 256U * 1024U * 1024U, + .mirror_copies = 5, }; ct_page_store *store = nullptr; ASSERT_EQ(ct_chunk_page_store_open(&store_options, catalog, &store), 0); @@ -1820,7 +1823,7 @@ TEST(ChunkPageStore, CApiFactoryInjectsBackendWithoutChangingOpen) ASSERT_EQ(ct_chunk_page_store_get_stats(store, &stats), 0); EXPECT_EQ(stats.generations_published, 1U); EXPECT_GT(stats.packs_written, 0U); - EXPECT_EQ(stats.mirror_write_attempts, stats.packs_written * 3U); + EXPECT_EQ(stats.mirror_write_attempts, stats.packs_written * 5U); ct_page_store_free(store); ct_close(tree); ct_root_catalog_free(catalog); diff --git a/lib/crowdb-tree/tests/integration/rpc_chunk_transport_test.cpp b/lib/crowdb-tree/tests/integration/rpc_chunk_transport_test.cpp index dbf62f433..3b73f53ae 100644 --- a/lib/crowdb-tree/tests/integration/rpc_chunk_transport_test.cpp +++ b/lib/crowdb-tree/tests/integration/rpc_chunk_transport_test.cpp @@ -194,6 +194,7 @@ TEST(RpcChunkTransport, AllocatesOneFullMirrorStripAndPreserves128BitChunkId) .writer_lease_ms = 5000, .rpc_timeout_ms = 1000, .completion_capacity = 16, + .mirror_copies = 3, }; RpcChunkTransport transport(options); ChunkId allocated; diff --git a/pixi.toml b/pixi.toml index 79ff29c08..40430fc46 100644 --- a/pixi.toml +++ b/pixi.toml @@ -1,6 +1,6 @@ [workspace] name = "crowdb-kv" -version = "0.1.0" +version = "0.2.0" description = "CROWDB — distributed key-value store with multi-paxos groups" channels = ["conda-forge"] # Keep glibc requirement low (2.17 = CentOS 7 / Ubuntu 16.04 era) so that diff --git a/tools/release.py b/tools/release.py index 7a7b4a0d1..025e09f78 100644 --- a/tools/release.py +++ b/tools/release.py @@ -66,6 +66,7 @@ def main() -> None: branch = release_branch() print(f"Dispatch release-container.yml on {branch}", flush=True) print(f"Image tag: crowdb/crowdb-iceberg:{branch.removeprefix('release/')}", flush=True) + print("Also updates: crowdb/crowdb-iceberg:latest", flush=True) if args.dry_run: return