diff --git a/.agents/skills/doc-design/SKILL.md b/.agents/skills/doc-design/SKILL.md index a2b4f0438..0f3162d31 100644 --- a/.agents/skills/doc-design/SKILL.md +++ b/.agents/skills/doc-design/SKILL.md @@ -21,7 +21,7 @@ stable IDs such as `I1`. Describe current state only. Omit requirement numbers, change history, before/after prose, file paths, and line numbers. Refer to searchable symbols. Keep architecture in root docs, detail in sub-designs, and operations in the -user guide; link rather than repeat. +relevant design or component README; link rather than repeat. When explicitly promoting a working draft, remove temporary scaffolding and requirement references, rewrite as current state, update `doc/doc_index.md`, diff --git a/.agents/skills/doc/SKILL.md b/.agents/skills/doc/SKILL.md index 0da06acb2..1f2f0ace0 100644 --- a/.agents/skills/doc/SKILL.md +++ b/.agents/skills/doc/SKILL.md @@ -15,7 +15,7 @@ Start at `doc/doc_index.md`; open only the row and section matching the task. - `doc/design//design-crowdb-.md`: architecture and rationale. - `doc/design//design-crowdb--.md`: permanent topic detail. - `doc/design/kv/kv-*-flow-analysis.md`: permanent KV path analysis. -- `doc/user-manual/user-guide.md`: operations; generated HTML is not hand-edited. +- `doc/design/`: permanent architecture and operational behavior. - `doc/backlog/`: requirement index and analysis. - `doc/working/`: implementation plans and explicitly requested design drafts. @@ -29,9 +29,6 @@ Start at `doc/doc_index.md`; open only the row and section matching the task. - Split independent topics; delete working files when their work completes. - Prefer concrete, tight prose and raw-readable bullets. Remove filler, repetition, rhetorical openings, adjective lists, and excessive em dashes. -- Rebuild `user-guide.html` with - `pixi run -- python doc/user-manual/build_html.py` whenever its Markdown - source changes. When the target is a design, backlog requirement, design draft, or working plan, use only its matching `/doc-*` guide instead of this router. diff --git a/AGENTS.md b/AGENTS.md index a72e8e9ea..4a16f3f23 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -45,4 +45,4 @@ and transport exposed through FFI. the backlog index only for selection, ordering, or status. - Pre-push or explicitly requested code review: `/review`. - Design questions: one section selected through `doc/doc_index.md`. -- Operations/user behavior: `doc/user-manual/user-guide.md`. +- Operations/user behavior: the relevant `doc/design/` or component README. diff --git a/CHANGELOG.md b/CHANGELOG.md index 3ff11d724..78fabdb2b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -35,5 +35,5 @@ The intended image is `crowdb/crowdb-iceberg:v0.1.0-dev`; no published digest is recorded yet. The GUI is not ready for this container. Multi-node deployment, production hardening and data-format upgrades are outside this release. -See the [container guide](doc/user-manual/docker-single-node-user-guide.md) for -supported startup, persistence, credentials and recovery behavior. +See the container deployment files for supported startup, persistence, +credentials and recovery behavior. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 06131b708..403f07876 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -76,9 +76,10 @@ pixi run test-single-node-container - Add dependencies through the owning package manager and avoid newly published versions until they have had time for ecosystem review. -Start documentation work at `doc/doc_index.md`. Permanent architecture belongs -under `doc/design/`, user behavior in `doc/user-manual/user-guide.md`, future -contracts in `doc/backlog/`, and temporary execution plans in `doc/working/`. +Start documentation work at `doc/doc_index.md`. Permanent architecture and +operational behavior belong under `doc/design/` or the relevant component +README, future contracts in `doc/backlog/`, and temporary execution plans in +`doc/working/`. ## Code and tests diff --git a/Cargo.lock b/Cargo.lock index bd65d2d3d..6e4c7ca91 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -510,6 +510,15 @@ version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "217698eaf96b4a3f0bc4f3662aaa55bdf913cd54d7204591faa790070c6d0853" +[[package]] +name = "crc32c" +version = "0.6.8" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a47af21622d091a8f0fb295b88bc886ac74efcc613efc19f5d0b21de5c89e47" +dependencies = [ + "rustc_version", +] + [[package]] name = "crc32fast" version = "1.5.0" @@ -614,6 +623,7 @@ dependencies = [ "arc-swap", "async-trait", "base64", + "bincode", "bytes", "chrono", "crc32fast", @@ -627,6 +637,7 @@ dependencies = [ "flate2", "hmac", "lz4_flex", + "md-5", "serde", "serde_json", "sha2", @@ -1132,6 +1143,7 @@ version = "0.1.0-dev" dependencies = [ "bincode", "bytes", + "crc32c", "crc32fast", "flatbuffers", "fs2", diff --git a/README.md b/README.md index 79d5ec707..093732d29 100644 --- a/README.md +++ b/README.md @@ -6,161 +6,55 @@ [![CI](https://github.com/buzzcrow/crowdb/actions/workflows/ci.yml/badge.svg)](https://github.com/buzzcrow/crowdb/actions/workflows/ci.yml) [![License](https://img.shields.io/badge/license-Apache--2.0-blue.svg)](LICENSE) -CROWDB is a distributed storage platform for objects, tables, and AI datasets. -It owns the data path from S3, Iceberg, and native Dataset access through -distributed metadata and chunk storage to disk—and eventually GPU memory. - -Version `0.1.0-dev` is the first development release being prepared for public -evaluation. Use disposable data; production use and on-disk upgrade compatibility -are not supported. Dataset and direct GPU delivery remain planned work. - -- Use **S3** for familiar object access. -- Use **Iceberg** for native catalogs, tables, snapshots, and immutable files. -- **Dataset** is planned for samples, shards, tensors, batches, and direct data access. - -## Three Layers, One Data Path - -```text - Applications - S3 tools Table engines AI runtimes - | | | | - | HTTP | HTTP | | native - v v v v -+--------------------------------------------------------------------------+ -| LAYER 3 — ACCESS | -| | -| S3 Iceberg Dataset | -| [implemented] [implemented] [design] | -| HTTP objects HTTP tables HTTP + native client | -| | -| Access Server serves HTTP. Dataset native access can bypass it. | -+------------------------------------+-------------------------------------+ - | - v -+--------------------------------------------------------------------------+ -| LAYER 2 — CHUNK | -| | -| Distributed structures: Chunk Stream Chunk-KV | -| | | | -| Data path: chunk client -> Chunk I/O -> ChunkDB -> DiskIO -> DiskDB | -| | | -| Accelerated path: DiskIO buffer -- RDMA / GDS -------> GPU memory | -+------------------------------------+-------------------------------------+ - | - v -+--------------------------------------------------------------------------+ -| LAYER 1 — REUSABLE KV | -| | -| crowdb-kv multi-group Paxos WAL Group 0 crowdb-tree RPC | -| | -| A standalone distributed layer—not the product-level data model. | -+--------------------------------------------------------------------------+ -``` +CROWDB is a distributed storage platform for Iceberg tables, AI datasets, and +S3 objects. Each access model keeps its own semantics while sharing one storage +core for distributed state, placement, protection, streaming, and recovery. -## Why CROWDB? - -Storage systems are usually assembled by stacking one system on another: table -metadata over object storage, dataset libraries over table or object APIs, and -new accelerators behind paths designed for disks and CPUs. Every boundary adds -another namespace, lifecycle, RPC, copy, and recovery model. When that boundary -becomes the bottleneck, the layers above it can only work around it. - -CROWDB exists to own the complete data path. S3 objects, Iceberg tables, and AI -datasets are intended as native access models over the same distributed storage core. They -share durability, placement, protection, and reclamation without pretending -that one model is merely a convention inside another. - -That control matters because both hardware and workloads keep changing. NVMe, -RDMA, GPUDirect Storage, and accelerator offload reshape more than one isolated -module. AI training and inference also need data to reach GPU memory without an -HTTP gateway or client CPU becoming the permanent middleman. Supporting those -changes cleanly requires control from protocol semantics down to buffers and -disk layout. - -The goal is a storage foundation that can evolve as one system: simple enough -to reason about, fast enough to justify owning the stack, and composed of -layers that remain useful independently. - -## Technical Foundation - -- **Parallel consensus:** crowdb-kv runs multiple independent Multi-Paxos slots - concurrently, with WAL durability, lease reads, and pluggable engines. -- **One protected chunk layer:** bounded streaming, mirrored small data, - strip-level erasure coding, shard repair, placement, and reclamation serve - every access model. -- **Chunk-based distributed structures:** Chunk Stream provides durable ordered - append. Chunk-KV uses range partitions that split and rebalance online while - reads and writes continue. -- **Native access models:** S3, Iceberg, and Dataset share the core without - being wrappers around one another. The planned Dataset model targets direct - topology access, RDMA and GPU delivery. - -## Where It Stands - -| Access model | Status | What it means | -| ------------ | ----------- | ---------------------------------------------------- | -| S3 | Implemented | Core HTTP object operations and bounded streaming | -| Iceberg | Implemented | Native catalog, FileIO and core v1/v2/v3 semantics | -| Dataset | Design | HTTP, topology-aware native client, and GPU delivery | +- **Iceberg:** native catalog and FileIO, implemented. +- **S3:** core HTTP object operations, implemented. +- **Dataset:** native access and direct GPU delivery, in design. -The KV, tree, DiskDB, ChunkDB, chunk I/O, Chunk Stream, Chunk-KV, RPC, -operations console, and core S3 foundation have working implementations. See -the [backlog](doc/backlog/backlog.md) for current delivery scope. - -## Quick Start - -The first Linux amd64 image is being prepared for manual publication. Once -`v0.1.0-dev` is published, start the Iceberg catalog and storage with Docker: - -```sh -docker run -d --name crowdb-iceberg \ - -p 127.0.0.1:80:80 \ - crowdb/crowdb-iceberg:v0.1.0-dev -``` - -Follow the [single-node Docker guide](doc/user-manual/docker-single-node-user-guide.md) -for startup checks, client credentials, persistent volumes and recovery. -For source builds and development with Pixi, see -[CONTRIBUTING.md](CONTRIBUTING.md). - -## Explore - -- [Access architecture](doc/design/access-server/design-crowdb-access-server.md) - — S3, Iceberg, Dataset, native access, and GPU delivery. -- [Documentation index](doc/doc_index.md) — every permanent architecture and - subsystem design. -- [User guide](doc/user-manual/user-guide.md) — setup, console, CLI, and - supported operations. -- [Single-node Docker guide](doc/user-manual/docker-single-node-user-guide.md) - — preview image, volume, credentials, clients, and recovery. -- [Backlog](doc/backlog/backlog.md) — what is implemented, in progress, and - planned. - -
-Current cluster demos - -### Cluster lifecycle +## Why CROWDB? -Bootstrap a cluster, register physical topology, create stores and Paxos groups, -and watch replicas elect a leader. +Storage bottlenecks move—from disks to CPUs, networks, and data movement—but +system boundaries tend to stay. A change that crosses a metadata service, an +object gateway, and a separate storage engine can become a negotiation between +systems rather than an improvement to one data path. - +CROWDB owns enough of that path to change it when workloads and hardware change. +S3 objects, Iceberg tables, and planned AI datasets are native access models +built over shared infrastructure, not conventions layered on top of one +another. The goal is not to claim novelty for Paxos, WALs, trees, or erasure +coding; it is to make their contracts agree on durability, placement, bounded +buffers, and recovery. -### KV operations +Read the full motivation in +[Why we’re building CROWDB](https://buzzcrow.github.io/blog/why-we-are-building-crowdb/). -Put, get, scan, and delete through a selected distributed group. +## Development preview - +Version `0.1.0-dev` is being prepared for public evaluation on Linux amd64. +Use disposable data. Production use and on-disk upgrade compatibility are not +supported, and Dataset and direct GPU delivery are not available yet. -### Failover and replica management +- [Project homepage](https://crowdb.dev/) +- [Quick start](https://crowdb.dev/docs/quickstart/) +- [Documentation](https://crowdb.dev/docs/) +- [Architecture](https://crowdb.dev/docs/architecture/) +- [Demos](https://crowdb.dev/demo/) -Expand a group, remove its leader, and continue operations after re-election. +## Repository - +The repository contains the storage implementation, tests, and source design +documents. Start with: -
+- [Contributing](CONTRIBUTING.md) for the development environment and workflow. +- [Documentation index](doc/doc_index.md) for subsystem designs. +- [Backlog](doc/backlog/backlog.md) for current delivery scope. +- [Single-node container](container/single-node-container/README.md) for image + development and packaging. -## Notes on AI-Assisted Development +## AI-assisted development The code in this project was written with AI assistance. The architecture, naming, module boundaries, and trade-offs remain human choices. AI is the @@ -168,4 +62,4 @@ compiler. The intent is mine. ## License -See [LICENSE](LICENSE). +Licensed under the [Apache License 2.0](LICENSE). diff --git a/app/crowdb-access-server/Cargo.toml b/app/crowdb-access-server/Cargo.toml index 9d38a1e8a..ccdbe5109 100644 --- a/app/crowdb-access-server/Cargo.toml +++ b/app/crowdb-access-server/Cargo.toml @@ -16,7 +16,6 @@ test-util = [] s3-e2e = ["s3"] iceberg-e2e = [] s3 = [ - "dep:crowdb-common", "dep:futures", ] @@ -27,7 +26,7 @@ crowdb-access-iceberg = { path = "../../lib/crowdb-access-iceberg" } crowdb-access-s3 = { path = "../../lib/crowdb-access-s3" } crowdb-chunk-client = { path = "../../lib/crowdb-chunk-client" } crowdb-chunk-kv-client = { path = "../../lib/crowdb-chunk-kv-client" } -crowdb-common = { path = "../../lib/crowdb-common/rust", optional = true } +crowdb-common = { path = "../../lib/crowdb-common/rust" } crowdb-kv-client = { path = "../../lib/crowdb-kv-client" } crowdb-protocol = { path = "../../lib/crowdb-protocol" } futures = { version = "0.3", optional = true } diff --git a/app/crowdb-access-server/conf/crowdb_access_server_config.toml b/app/crowdb-access-server/conf/crowdb_access_server_config.toml new file mode 100644 index 000000000..c9204d095 --- /dev/null +++ b/app/crowdb-access-server/conf/crowdb_access_server_config.toml @@ -0,0 +1,62 @@ +# Canonical configuration for the S3 and Iceberg access processes. +# Credentials and bearer tokens are supplied through environment variables. + +[common] +management_seeds = ["http://127.0.0.1:10000"] +diskio_connections_per_endpoint = 2 +diskio_rpc_workers = 2 + +[read] +stream_window_bytes = 1048576 +stream_slots = 3 +global_stream_bytes = 268435456 +recovery_memory_bytes = 268435456 + +[small_write] +threshold_ratio = 0.9 +disk_block_bytes = 1048576 +conversion_enabled = true +ec_data = 8 +ec_code = 4 +memory_budget_bytes = 1342177280 +queue_capacity = 1024 +min_pipelines = 1 +max_pipelines = 32 +max_batch_bytes = 1048576 + +[s3] +listen = "127.0.0.1:8081" +tenant = "preview" +region = "us-east-1" +trusted_network = false +small_object_limit = 8388608 +list_scan_items = 1001 +list_scan_bytes = 4194304 +continuation_ttl_seconds = 900 +native_budget_bytes = 268435456 +cleanup_backlog_limit = 10000 +ec_data = 8 +ec_code = 4 +max_chunk_size = 1073741824 + +[iceberg] +listen = "127.0.0.1:8181" +native_budget_bytes = 268435456 + +[iceberg.gc] +enabled = false +interval_ms = 1000 +step_bytes = 8388608 +step_ms = 1000 +page_items = 64 +page_bytes = 49152 +concurrency = 1 +kv_bytes = 67108864 +kv_requests = 128 +chunk_bytes = 8388608 +chunk_requests = 128 +retry_base_ms = 1000 +retry_max_ms = 60000 +corruption_attempts = 3 +minimum_retention_ms = 604800000 +catalogs = [] diff --git a/app/crowdb-access-server/src/config.rs b/app/crowdb-access-server/src/config.rs new file mode 100644 index 000000000..b607fea83 --- /dev/null +++ b/app/crowdb-access-server/src/config.rs @@ -0,0 +1,270 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Startup configuration shared by the S3 and Iceberg access processes. + +use std::path::Path; + +use crowdb_chunk_client::{ChunkReadPolicy, SmallWritePolicy}; +use crowdb_common::config::{load_from_file, BaseConfig}; +use serde::{Deserialize, Serialize}; + +#[derive(Clone, Debug, Default, Deserialize, Serialize)] +#[serde(default)] +pub struct AccessConfig { + pub common: CommonConfig, + pub read: ReadConfig, + pub small_write: SmallWriteConfig, + pub s3: S3Config, + pub iceberg: IcebergConfig, +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +#[serde(default)] +pub struct CommonConfig { + pub management_seeds: Vec, + pub diskio_connections_per_endpoint: usize, + pub diskio_rpc_workers: u32, +} + +impl Default for CommonConfig { + fn default() -> Self { + Self { + management_seeds: Vec::new(), + diskio_connections_per_endpoint: 2, + diskio_rpc_workers: 2, + } + } +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +#[serde(default)] +pub struct ReadConfig { + pub stream_window_bytes: usize, + pub stream_slots: usize, + pub global_stream_bytes: usize, + pub recovery_memory_bytes: usize, +} + +impl Default for ReadConfig { + fn default() -> Self { + Self { + stream_window_bytes: 1024 * 1024, + stream_slots: 3, + global_stream_bytes: 256 * 1024 * 1024, + recovery_memory_bytes: 256 * 1024 * 1024, + } + } +} + +impl ReadConfig { + #[must_use] + pub fn policy(&self) -> ChunkReadPolicy { + ChunkReadPolicy { + stream_window_bytes: self.stream_window_bytes, + stream_slots: self.stream_slots, + global_stream_bytes: self.global_stream_bytes, + recovery_memory_bytes: self.recovery_memory_bytes, + ..ChunkReadPolicy::default() + } + } +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +#[serde(default)] +pub struct SmallWriteConfig { + pub threshold_ratio: f64, + /// Data bytes per strip block used for the small-object routing boundary. + pub disk_block_bytes: usize, + pub conversion_enabled: bool, + pub ec_data: usize, + pub ec_code: usize, + pub memory_budget_bytes: usize, + pub queue_capacity: usize, + pub min_pipelines: usize, + pub max_pipelines: usize, + pub max_batch_bytes: usize, +} + +impl Default for SmallWriteConfig { + fn default() -> Self { + let policy = SmallWritePolicy::default(); + Self { + threshold_ratio: 0.9, + disk_block_bytes: 1024 * 1024, + conversion_enabled: policy.conversion_enabled, + ec_data: policy.conversion_data_num, + ec_code: policy.conversion_code_num, + memory_budget_bytes: policy.memory_budget, + queue_capacity: policy.queue_capacity, + min_pipelines: policy.min_pipelines, + max_pipelines: policy.max_pipelines, + max_batch_bytes: policy.max_batch_bytes, + } + } +} + +impl SmallWriteConfig { + #[must_use] + #[allow( + clippy::cast_possible_truncation, + clippy::cast_sign_loss, + clippy::cast_precision_loss + )] + pub fn threshold_exclusive(&self) -> usize { + let data_shards = if self.conversion_enabled { self.ec_data } else { 1 }; + (self.threshold_ratio * data_shards.saturating_mul(self.disk_block_bytes) as f64).ceil() as usize + } + + #[must_use] + pub fn policy(&self) -> SmallWritePolicy { + SmallWritePolicy { + conversion_enabled: self.conversion_enabled, + conversion_data_num: self.ec_data, + conversion_code_num: self.ec_code, + memory_budget: self.memory_budget_bytes, + queue_capacity: self.queue_capacity, + min_pipelines: self.min_pipelines, + max_pipelines: self.max_pipelines, + max_batch_bytes: self.max_batch_bytes, + ..SmallWritePolicy::default() + } + } +} + +#[derive(Clone, Debug, Default, Deserialize, Serialize)] +#[serde(default)] +pub struct S3Config { + pub listen: Option, + pub tenant: Option, + pub region: Option, + pub trusted_network: Option, + pub small_object_limit: Option, + pub list_scan_items: Option, + pub list_scan_bytes: Option, + pub continuation_ttl_seconds: Option, + pub native_budget_bytes: Option, + pub cleanup_backlog_limit: Option, + pub ec_data: Option, + pub ec_code: Option, + pub max_chunk_size: Option, +} + +#[derive(Clone, Debug, Default, Deserialize, Serialize)] +#[serde(default)] +pub struct IcebergConfig { + pub listen: Option, + pub native_budget_bytes: Option, + pub gc: IcebergGcConfig, +} + +#[derive(Clone, Debug, Default, Deserialize, Serialize)] +#[serde(default)] +pub struct IcebergGcConfig { + pub enabled: Option, + pub interval_ms: Option, + pub catalogs: Option>, + pub step_bytes: Option, + pub step_ms: Option, + pub page_items: Option, + pub page_bytes: Option, + pub concurrency: Option, + pub retry_base_ms: Option, + pub retry_max_ms: Option, + pub corruption_attempts: Option, + pub minimum_retention_ms: Option, + pub kv_bytes: Option, + pub kv_requests: Option, + pub chunk_bytes: Option, + pub chunk_requests: Option, +} + +impl BaseConfig for AccessConfig { + fn validate(&self) -> Result<(), String> { + if self.common.diskio_connections_per_endpoint == 0 || self.common.diskio_rpc_workers == 0 { + return Err("common DiskIO connections and RPC workers must be nonzero".into()); + } + if self.read.stream_slots == 0 || self.read.stream_slots > 64 { + return Err("read.stream_slots must be between 1 and 64".into()); + } + if self.read.stream_window_bytes < 64 * 1024 || self.read.stream_window_bytes > 1024 * 1024 { + return Err("read.stream_window_bytes must be between 64 KiB and 1 MiB".into()); + } + if self.read.global_stream_bytes < 1024 * 1024 || self.read.global_stream_bytes > u32::MAX as usize { + return Err("read.global_stream_bytes must be at least 1 MiB and below 4 GiB".into()); + } + if self.read.recovery_memory_bytes < 1024 * 1024 + || self.read.recovery_memory_bytes > u32::MAX as usize + { + return Err("read.recovery_memory_bytes must be between 1 MiB and 4 GiB".into()); + } + self.small_write + .policy() + .validate() + .map_err(|error| format!("invalid small_write config: {error}"))?; + if !self.small_write.threshold_ratio.is_finite() + || self.small_write.threshold_ratio <= 0.0 + || self.small_write.threshold_ratio > 1.0 + || self.small_write.disk_block_bytes < 128 * 1024 + || self.small_write.disk_block_bytes > 1024 * 1024 + || !self.small_write.disk_block_bytes.is_power_of_two() + || self.small_write.ec_data == 0 + || self.small_write.ec_data > 32 + || self.small_write.threshold_exclusive() > self.small_write.policy().object_limit + { + return Err("small_write strip capacity or threshold is invalid".into()); + } + if self.s3.ec_data == Some(0) || self.s3.ec_code == Some(0) { + return Err("S3 EC data and code counts must be nonzero".into()); + } + if self.s3.max_chunk_size == Some(0) || self.s3.native_budget_bytes == Some(0) { + return Err("S3 chunk and native budgets must be nonzero".into()); + } + if self.iceberg.native_budget_bytes == Some(0) { + return Err("Iceberg native budget must be nonzero".into()); + } + if self.s3.small_object_limit == Some(0) + || self.s3.list_scan_items == Some(0) + || self.s3.list_scan_bytes == Some(0) + || self.s3.continuation_ttl_seconds == Some(0) + { + return Err("S3 object and listing limits must be nonzero".into()); + } + for listen in [self.s3.listen.as_deref(), self.iceberg.listen.as_deref()] + .into_iter() + .flatten() + { + listen + .parse::() + .map_err(|error| error.to_string())?; + } + Ok(()) + } +} + +/// Remove a single `--config ` pair and load the named TOML file. +/// +/// # Errors +/// Rejects duplicate or missing paths and invalid configuration files. +pub fn load_args(mut args: Vec) -> Result<(AccessConfig, Vec), String> { + let positions: Vec<_> = args + .iter() + .enumerate() + .filter(|(_, value)| *value == "--config") + .map(|(index, _)| index) + .collect(); + if positions.len() > 1 { + return Err("--config may be specified only once".into()); + } + let Some(index) = positions.first().copied() else { + return Ok((AccessConfig::default(), args)); + }; + if index + 1 >= args.len() { + return Err("--config requires a path".into()); + } + let path = args.remove(index + 1); + args.remove(index); + let config = + load_from_file(Path::new(&path)).map_err(|error| format!("failed to load {path}: {error}"))?; + Ok((config, args)) +} diff --git a/app/crowdb-access-server/src/http_receive.rs b/app/crowdb-access-server/src/http_receive.rs new file mode 100644 index 000000000..6cc38b67c --- /dev/null +++ b/app/crowdb-access-server/src/http_receive.rs @@ -0,0 +1,53 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! HTTP body receive ownership shared by the S3 and Iceberg listeners. + +use std::sync::Arc; + +use crowdb_access_s3::native_buffer::{NativeBodyAllocator, NativeBodyReceiver}; +use hyper::body::{Http1BodyReceiveProvider, Incoming}; +use hyper::Request; + +#[derive(Clone)] +pub(crate) struct DeferredBodyReceiveProvider { + provider: Arc, + native: Option>, +} + +impl DeferredBodyReceiveProvider { + pub(crate) fn generic(provider: Arc) -> Self { + Self { + provider, + native: None, + } + } + + pub(crate) fn native(receiver: Arc) -> Self { + Self { + provider: receiver.clone(), + native: Some(receiver), + } + } +} + +/// Installs a provider only after the request has passed authentication and admission. +pub fn install_body_receive_provider(request: &mut Request) -> Option> { + let deferred = request.extensions_mut().remove::()?; + request + .body_mut() + .set_http1_body_receive_provider(deferred.provider); + deferred.native +} + +/// Selects a bounded native owner for one admitted upload. +pub(crate) fn install_native_body_receive_provider( + request: &mut Request, + allocator: &NativeBodyAllocator, +) -> Arc { + let receiver = Arc::new(allocator.object_receiver()); + request + .body_mut() + .set_http1_body_receive_provider(receiver.clone()); + receiver +} diff --git a/app/crowdb-access-server/src/iceberg/file_admission.rs b/app/crowdb-access-server/src/iceberg/file_admission.rs index ae9e48ca1..6a4f67a45 100644 --- a/app/crowdb-access-server/src/iceberg/file_admission.rs +++ b/app/crowdb-access-server/src/iceberg/file_admission.rs @@ -84,9 +84,11 @@ impl FileTransferAdmission { if self.operation != FileOperation::CreateMultipart { return Err(FileAdmissionError::Scope); } + let max_part_bytes = self.request_bytes.min(self.file_bytes); + let max_parts = (self.staged_bytes / max_part_bytes).min(10_000); let limits = MultipartLimits { - max_parts: 10_000, - max_part_bytes: self.request_bytes.min(self.file_bytes), + max_parts: u16::try_from(max_parts).map_err(|_| FileAdmissionError::Bounds)?, + max_part_bytes, max_file_bytes: self.file_bytes, max_staged_bytes: self.staged_bytes, ttl_ms: 24 * 60 * 60 * 1000, diff --git a/app/crowdb-access-server/src/iceberg/file_body.rs b/app/crowdb-access-server/src/iceberg/file_body.rs index d5e16329c..fb98fe282 100644 --- a/app/crowdb-access-server/src/iceberg/file_body.rs +++ b/app/crowdb-access-server/src/iceberg/file_body.rs @@ -7,6 +7,7 @@ use std::sync::{ use std::task::{Context, Poll}; use crowdb_access_iceberg::file::{ByteRange, FileBlockStore, FileIoError, FileReader, FileRecord}; +use crowdb_chunk_client::{ChunkReadStream, ReadError}; use hyper::body::{Body, Bytes, Frame, SizeHint}; #[derive(Debug, thiserror::Error)] @@ -56,10 +57,35 @@ impl FileResponseBudget { .map_err(|_| FileBodyError::Busy)?; let permit = Permit(self.active.clone()); let length = range.map_or(record.length, |range| range.end.saturating_sub(range.start)); - let reader = FileReader::new(store, record, range, 16 * 1024)?; + let stream = if let Some(locations) = record + .content + .locations(record.length) + .map_err(FileIoError::from)? + { + let interval = range.unwrap_or(ByteRange { + start: 0, + end: record.length, + }); + Some( + store + .stream_client() + .ok_or(FileIoError::Bounds)? + .read_range_stream(&locations, interval.start, interval.end) + .map_err(FileIoError::from)?, + ) + } else { + None + }; + let reader = if stream.is_none() { + Some(FileReader::new(store, record, range, 16 * 1024)?) + } else { + None + }; Ok(FileReadBody { - reader: (length > 0).then_some(reader), + reader, pending: None, + stream, + stream_pending: None, remaining: length, permit: (length > 0).then_some(permit), }) @@ -74,10 +100,13 @@ impl Drop for Permit { } type ReadFuture = Pin>, FileIoError>)> + Send>>; +type StreamFuture = Pin>)> + Send>>; pub struct FileReadBody { reader: Option, pending: Option, + stream: Option, + stream_pending: Option, remaining: u64, permit: Option, } @@ -86,6 +115,8 @@ impl FileReadBody { fn finish(&mut self) { self.reader = None; self.pending = None; + self.stream = None; + self.stream_pending = None; self.remaining = 0; self.permit = None; } @@ -103,6 +134,45 @@ impl Body for FileReadBody { if body.remaining == 0 { return Poll::Ready(None); } + if body.stream.is_some() || body.stream_pending.is_some() { + if body.stream_pending.is_none() { + let mut stream = body.stream.take().expect("stream exists"); + body.stream_pending = Some(Box::pin(async move { + let next = stream.next_chunk().await; + (stream, next) + })); + } + let (stream, next) = match body + .stream_pending + .as_mut() + .expect("stream read exists") + .as_mut() + .poll(context) + { + Poll::Pending => return Poll::Pending, + Poll::Ready(value) => value, + }; + body.stream_pending = None; + return match next { + Some(Ok(bytes)) if !bytes.is_empty() && bytes.len() as u64 <= body.remaining => { + body.remaining -= bytes.len() as u64; + if body.remaining == 0 { + body.finish(); + } else { + body.stream = Some(stream); + } + Poll::Ready(Some(Ok(Frame::data(bytes)))) + } + Some(Err(error)) => { + body.finish(); + Poll::Ready(Some(Err(FileIoError::Read(error)))) + } + _ => { + body.finish(); + Poll::Ready(Some(Err(FileIoError::Bounds))) + } + }; + } if body.pending.is_none() { let Some(mut reader) = body.reader.take() else { body.finish(); diff --git a/app/crowdb-access-server/src/iceberg/file_encoding.rs b/app/crowdb-access-server/src/iceberg/file_encoding.rs index 3b44a86a4..2a61b9692 100644 --- a/app/crowdb-access-server/src/iceberg/file_encoding.rs +++ b/app/crowdb-access-server/src/iceberg/file_encoding.rs @@ -9,6 +9,8 @@ mod checksum; mod chunks; mod content_md5; +const MAX_RECEIVE_FRAME_BYTES: usize = 1024 * 1024; + #[derive(Clone, Copy, Debug, thiserror::Error)] pub enum FileEncodingError { #[error("invalid upload framing")] @@ -28,11 +30,12 @@ pub struct FileUploadBody { buffered: Bytes, chunks: Option, checksum: Option, - content_md5: Option, + content_md5: content_md5::ContentMd5, length: Option, wire_length: Option, wire_bytes: u64, max_wire_bytes: u64, + has_integrity: bool, done: bool, failure: Option, } @@ -52,7 +55,7 @@ impl FileUploadBody { if wire_length.is_some_and(|length| length > max_wire_bytes) { return Err(FileEncodingError::Length); } - let (length, chunks, checksum) = if let Some(verifier) = verifier { + let (length, chunks, checksum, streaming_integrity) = if let Some(verifier) = verifier { if header(headers, "content-encoding")? != Some("aws-chunked") { return Err(FileEncodingError::Framing); } @@ -65,10 +68,12 @@ impl FileUploadBody { return Err(FileEncodingError::Framing); } let checksum = checksum::Checksum::from_headers(headers, verifier.has_trailer())?; + let streaming_integrity = verifier.is_signed() || checksum.is_some(); ( Some(length), Some(chunks::Chunks::new(verifier, checksum, length)), None, + streaming_integrity, ) } else { if headers.contains_key("x-amz-decoded-content-length") @@ -82,18 +87,22 @@ impl FileUploadBody { wire_length, None, checksum::Checksum::from_headers(headers, false)?, + false, ) }; + let content_md5 = content_md5::ContentMd5::from_headers(headers)?; + let has_integrity = streaming_integrity || checksum.is_some() || content_md5.is_declared(); Ok(Self { input, buffered: Bytes::new(), chunks, checksum, - content_md5: content_md5::ContentMd5::from_headers(headers)?, + content_md5, length, wire_length, wire_bytes: 0, max_wire_bytes, + has_integrity, done: false, failure: None, }) @@ -104,6 +113,21 @@ impl FileUploadBody { self.length } + #[must_use] + pub const fn has_integrity(&self) -> bool { + self.has_integrity + } + + #[must_use] + pub const fn native_handoff_eligible(&self) -> bool { + self.chunks.is_none() && self.length.is_some() + } + + #[must_use] + pub fn md5(&self) -> [u8; 16] { + self.content_md5.digest() + } + pub(super) const fn failure(&self) -> Option { self.failure } @@ -118,9 +142,7 @@ impl FileUploadBody { if let Some(checksum) = &self.checksum { checksum.verify()?; } - if let Some(checksum) = &self.content_md5 { - checksum.verify()?; - } + self.content_md5.verify()?; Ok(()) } } @@ -134,7 +156,9 @@ impl + Unpin> FileUploadBody { return Poll::Ready(Ok(Some(bytes))); } } else { - let bytes = self.buffered.split_to(self.buffered.len().min(64 * 1024)); + let bytes = self + .buffered + .split_to(self.buffered.len().min(MAX_RECEIVE_FRAME_BYTES)); if let Some(checksum) = &mut self.checksum { checksum.update(&bytes); } @@ -179,9 +203,7 @@ impl + Unpin> Body for FileUploadBody { } match std::task::ready!(body.poll_data(context)) { Ok(Some(bytes)) => { - if let Some(checksum) = &mut body.content_md5 { - checksum.update(&bytes); - } + body.content_md5.update(&bytes); Poll::Ready(Some(Ok(Frame::data(bytes)))) } Ok(None) => { diff --git a/app/crowdb-access-server/src/iceberg/file_encoding/chunks.rs b/app/crowdb-access-server/src/iceberg/file_encoding/chunks.rs index 9ea22be47..3580dddc1 100644 --- a/app/crowdb-access-server/src/iceberg/file_encoding/chunks.rs +++ b/app/crowdb-access-server/src/iceberg/file_encoding/chunks.rs @@ -2,7 +2,7 @@ use crowdb_access_s3::auth::StreamingPayloadVerifier; use hyper::body::Bytes; use sha2::{Digest, Sha256}; -use super::{checksum::Checksum, FileEncodingError}; +use super::{checksum::Checksum, FileEncodingError, MAX_RECEIVE_FRAME_BYTES}; enum State { Header, @@ -44,7 +44,7 @@ impl Chunks { if let State::Data(remaining) = self.state { let length = input .len() - .min(64 * 1024) + .min(MAX_RECEIVE_FRAME_BYTES) .min(usize::try_from(remaining).unwrap_or(usize::MAX)); let bytes = input.split_to(length); self.hash.update(&bytes); diff --git a/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs b/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs index ec8465249..767696863 100644 --- a/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs +++ b/app/crowdb-access-server/src/iceberg/file_encoding/content_md5.rs @@ -5,24 +5,33 @@ use md5::{Digest, Md5}; use super::FileEncodingError; pub(super) struct ContentMd5 { - expected: [u8; 16], + expected: Option<[u8; 16]>, digest: Md5, } impl ContentMd5 { - pub(super) fn from_headers(headers: &HeaderMap) -> Result, FileEncodingError> { - let Some(value) = super::header(headers, "content-md5")? else { - return Ok(None); - }; - let expected = STANDARD - .decode(value) - .map_err(|_| FileEncodingError::Framing)? - .try_into() - .map_err(|_| FileEncodingError::Framing)?; - Ok(Some(Self { + pub(super) fn from_headers(headers: &HeaderMap) -> Result { + let expected = super::header(headers, "content-md5")? + .map(|value| { + STANDARD + .decode(value) + .map_err(|_| FileEncodingError::Framing)? + .try_into() + .map_err(|_| FileEncodingError::Framing) + }) + .transpose()?; + Ok(Self { expected, digest: Md5::new(), - })) + }) + } + + pub(super) const fn is_declared(&self) -> bool { + self.expected.is_some() + } + + pub(super) fn digest(&self) -> [u8; 16] { + self.digest.clone().finalize().into() } pub(super) fn update(&mut self, bytes: &[u8]) { @@ -30,7 +39,7 @@ impl ContentMd5 { } pub(super) fn verify(&self) -> Result<(), FileEncodingError> { - if <[u8; 16]>::from(self.digest.clone().finalize()) != self.expected { + if self.expected.is_some_and(|expected| self.digest() != expected) { return Err(FileEncodingError::Checksum); } Ok(()) diff --git a/app/crowdb-access-server/src/iceberg/file_http.rs b/app/crowdb-access-server/src/iceberg/file_http.rs index f1de58464..02c4d608a 100644 --- a/app/crowdb-access-server/src/iceberg/file_http.rs +++ b/app/crowdb-access-server/src/iceberg/file_http.rs @@ -10,6 +10,9 @@ use crowdb_access_iceberg::file::{ }; use crowdb_access_iceberg::key::OperationId; use crowdb_access_s3::auth::{RawAuthRequest, StreamingPayloadVerifier}; +use crowdb_access_s3::native_buffer::{NativeBodyAllocator, NativeBodyReceiver}; +use crowdb_chunk_client::{ChunkClientConfig, LargeWritePolicy}; +use crowdb_common::ec::EcScheme; use hyper::body::Incoming; use hyper::http::header::{ACCEPT_RANGES, CONTENT_LENGTH, CONTENT_RANGE, ETAG, RANGE}; use hyper::{Method, Request, Response, StatusCode}; @@ -24,9 +27,9 @@ use super::file_response::{FileS3ErrorCode, MultipartResponses}; use super::file_upload::FileUploadBudget; mod multipart; +mod stream; pub(super) struct FileHttp { - pins: crowdb_access_iceberg::gc::ReaderPins, repository: FileRepository, multipart: MultipartRepository, admission: MultipartAdmission, @@ -35,16 +38,31 @@ pub(super) struct FileHttp { issuer: FileGrantIssuer, responses: FileResponseBudget, uploads: FileUploadBudget, + small_threshold_exclusive: usize, + large_write: LargeWritePolicy, + native_allocator: Option>, region: String, limits: FileServiceLimits, } impl FileHttp { + pub(super) fn chunk_metrics( + &self, + ) -> Option<( + crowdb_chunk_client::ReadFlowMetricsSnapshot, + crowdb_chunk_client::SmallWriteMetricsSnapshot, + )> { + self.blocks + .stream_client() + .map(|client| (client.read_flow_metrics(), client.small_write_metrics())) + } + pub(super) fn new( store: Arc, blocks: Arc, secret: [u8; 32], region: String, + native_allocator: Option>, ) -> Result { if region.is_empty() || region.len() > 64 @@ -55,7 +73,6 @@ impl FileHttp { return Err(FileGrantError::Invalid); } Ok(Self { - pins: crowdb_access_iceberg::gc::ReaderPins::new(store.clone()), repository: FileRepository::new(store.clone()), multipart: MultipartRepository::new(store.clone()), admission: MultipartAdmission::new(store.clone()), @@ -64,6 +81,12 @@ impl FileHttp { issuer: FileGrantIssuer::new(secret, 15 * 60 * 1000)?, responses: FileResponseBudget::new(64).map_err(|_| FileGrantError::Invalid)?, uploads: FileUploadBudget::new(64).map_err(|_| FileGrantError::Invalid)?, + small_threshold_exclusive: crate::config::SmallWriteConfig::default().threshold_exclusive(), + large_write: LargeWritePolicy { + ec_scheme: EcScheme::new(8, 4), + client: Arc::new(ChunkClientConfig::default()), + }, + native_allocator, region, limits: FileServiceLimits { max_request_bytes: 1024 * 1024 * 1024, @@ -74,6 +97,25 @@ impl FileHttp { }) } + pub(super) fn set_small_threshold(&mut self, threshold_exclusive: usize) -> Result<(), FileGrantError> { + if threshold_exclusive == 0 || threshold_exclusive > 32 * 1024 * 1024 { + return Err(FileGrantError::Invalid); + } + self.small_threshold_exclusive = threshold_exclusive; + Ok(()) + } + + pub(super) fn set_large_write(&mut self, policy: LargeWritePolicy) -> Result<(), FileGrantError> { + if policy.ec_scheme.data_num == 0 + || policy.ec_scheme.code_num == 0 + || policy.client.read_buffer_size == 0 + { + return Err(FileGrantError::Invalid); + } + self.large_write = policy; + Ok(()) + } + pub(super) async fn dispatch( self: &Arc, catalog: &CatalogRepository, @@ -93,7 +135,7 @@ impl FileHttp { async fn execute( self: &Arc, catalog: &CatalogRepository, - request: Request, + mut request: Request, request_timeout: Duration, ) -> Result, FileS3ErrorCode> { let file_request = FileRequest::parse(request.method(), request.uri()).map_err(request_error)?; @@ -117,45 +159,41 @@ impl FileHttp { grant .authorize(file_request.operation, &file_request.location, 0, 0) .map_err(|_| FileS3ErrorCode::AccessDenied)?; - let expires_ms = self - .pins - .request_expiry(root.context, now_ms) - .await - .map_err(catalog_error)?; - if matches!(file_request.operation, FileOperation::Head | FileOperation::Get) { - self.pins - .protect_file_reads( - root.context, - file_request.location.table().table, - "file-request", - expires_ms, - now_ms, - ) - .await - .map_err(catalog_error)?; - } else { - self.pins - .protect_files( - root.context, - file_request.location.table().table, - "file-request", - expires_ms, - now_ms, - ) - .await - .map_err(catalog_error)?; - } let session = self.load_session(root.context, &file_request).await?; let admission = FileTransferAdmission::authorize(&grant, &file_request, self.limits, session.as_ref(), now_ms) .map_err(admission_error)?; + let declared_receive = request + .headers() + .get("x-amz-decoded-content-length") + .or_else(|| request.headers().get(CONTENT_LENGTH)) + .and_then(|value| value.to_str().ok()) + .and_then(|value| value.parse::().ok()); + let native_receiver = if matches!( + file_request.operation, + FileOperation::Put | FileOperation::UploadPart + ) && declared_receive.map_or(true, |length| length >= 1024 * 1024) + { + self.native_allocator.as_ref().map(|allocator| { + crate::http_receive::install_native_body_receive_provider(&mut request, allocator) + }) + } else { + None + }; match file_request.operation { FileOperation::Head | FileOperation::Get => { self.read(&file_request, &request, root.context, &admission).await } FileOperation::Put => { - self.put(&file_request, request, root.context, &admission, streaming) - .await + self.put( + &file_request, + request, + root.context, + &admission, + streaming, + native_receiver.as_deref(), + ) + .await } _ => { Box::pin(self.multipart_request( @@ -165,6 +203,7 @@ impl FileHttp { &admission, now_ms, streaming, + native_receiver.as_deref(), )) .await } @@ -178,6 +217,7 @@ impl FileHttp { context: crowdb_access_iceberg::catalog::CatalogContext, admission: &FileTransferAdmission, streaming: Option, + native_receiver: Option<&NativeBodyReceiver>, ) -> Result, FileS3ErrorCode> { let digest = if streaming.is_some() { None @@ -187,11 +227,38 @@ impl FileHttp { let (parts, body) = request.into_parts(); let mut body = FileUploadBody::new(body, &parts.headers, streaming, admission.request_byte_limit()) .map_err(multipart::encoding_error)?; + if digest.is_none() && !body.has_integrity() { + return Err(FileS3ErrorCode::InvalidRequest); + } let length = body.decoded_length(); let owner = crowdb_access_iceberg::file::FileIdentity { table: file_request.location.table(), file: crowdb_access_iceberg::key::FileId::random(), }; + if let Some(client) = self.blocks.stream_client() { + let sealed = stream::upload( + client, + &self.uploads, + admission, + &mut body, + owner, + file_request.location.clone(), + length, + digest, + native_receiver, + self.small_threshold_exclusive, + &self.large_write, + ) + .await?; + let published = self + .repository + .publish(context, &sealed) + .await + .map_err(catalog_error)?; + let mut response = Response::new(IcebergBody::new(Vec::new())); + set_header(&mut response, ETAG, &etag(&published))?; + return Ok(response); + } let tree = admission .receive( &self.uploads, @@ -292,6 +359,9 @@ fn set_header( } fn etag(record: &FileRecord) -> String { + if let Some(etag) = record.content.etag() { + return format!("\"{etag}\""); + } let mut value = String::from("\""); for byte in record.digest { write!(value, "{byte:02x}").expect("string writes do not fail"); diff --git a/app/crowdb-access-server/src/iceberg/file_http/multipart.rs b/app/crowdb-access-server/src/iceberg/file_http/multipart.rs index 962003acf..e9a9de6a6 100644 --- a/app/crowdb-access-server/src/iceberg/file_http/multipart.rs +++ b/app/crowdb-access-server/src/iceberg/file_http/multipart.rs @@ -3,10 +3,11 @@ use std::sync::Arc; use crowdb_access_iceberg::catalog::{CatalogContext, CatalogError}; use crowdb_access_iceberg::file::{ FileIdentity, FileOperation, FileSealError, FileSealer, MultipartAdmissionLimits, MultipartPart, - MultipartPhase, MultipartSession, MultipartWorkError, + MultipartPhase, MultipartSession, MultipartStreamPart, MultipartWorkError, }; use crowdb_access_iceberg::key::{FileId, OperationId}; use crowdb_access_s3::auth::StreamingPayloadVerifier; +use crowdb_access_s3::native_buffer::NativeBodyReceiver; use http_body_util::BodyExt; use hyper::body::Incoming; use hyper::http::header::HeaderValue; @@ -45,6 +46,7 @@ impl FileHttp { .map(Some) } + #[allow(clippy::too_many_arguments)] pub(super) async fn multipart_request( self: &Arc, file: &FileRequest, @@ -53,6 +55,7 @@ impl FileHttp { admission: &FileTransferAdmission, now_ms: u64, streaming: Option, + native_receiver: Option<&NativeBodyReceiver>, ) -> Result, FileS3ErrorCode> { match (&file.multipart, file.operation) { (Some(MultipartRequest::Create), FileOperation::CreateMultipart) => { @@ -66,6 +69,7 @@ impl FileHttp { admission, now_ms, streaming, + native_receiver, ) .await } @@ -159,6 +163,7 @@ impl FileHttp { .map_err(|_| FileS3ErrorCode::InternalError) } + #[allow(clippy::too_many_arguments, clippy::too_many_lines)] async fn upload( &self, session: MultipartSession, @@ -167,6 +172,7 @@ impl FileHttp { admission: &FileTransferAdmission, now_ms: u64, streaming: Option, + native_receiver: Option<&NativeBodyReceiver>, ) -> Result, FileS3ErrorCode> { let digest = if streaming.is_some() { None @@ -176,72 +182,91 @@ impl FileHttp { let (parts, body) = request.into_parts(); let mut body = FileUploadBody::new(body, &parts.headers, streaming, admission.request_byte_limit()) .map_err(encoding_error)?; + if digest.is_none() && !body.has_integrity() { + return Err(FileS3ErrorCode::InvalidRequest); + } let length = body.decoded_length(); let owner = FileIdentity { table: session.owner.table, file: FileId::random(), }; - let tree = admission - .receive( + let (tree, stream) = if let Some(client) = self.blocks.stream_client() { + let record = super::stream::upload( + client, &self.uploads, + admission, &mut body, - self.blocks.clone(), owner, + session.location.clone(), length, digest, + native_receiver, + self.small_threshold_exclusive, + &self.large_write, ) - .await - .map_err(|error| { - body.failure() - .map_or_else(|| admission_error(error), encoding_error) - })?; - let before = self - .multipart - .part(&session, part_number) - .await - .map_err(catalog_error)?; - let part = MultipartPart { + .await?; + ( + None, + Some(MultipartStreamPart { + length: record.length, + content: record.content, + }), + ) + } else { + let tree = admission + .receive( + &self.uploads, + &mut body, + self.blocks.clone(), + owner, + length, + digest, + ) + .await + .map_err(|error| { + body.failure() + .map_or_else(|| admission_error(error), encoding_error) + })?; + (Some(tree), None) + }; + let mut part = MultipartPart { upload: session.upload, number: part_number, - revision: before.map_or(1, |part| part.revision.checked_add(1).unwrap_or(0)), + revision: 1, modified_ms: now_ms, owner, tree, + stream, }; - if !self - .multipart - .reserve_part(&session, &part, now_ms) - .await - .map_err(catalog_error)? - { - return Err(FileS3ErrorCode::SlowDown); - } - let pending = self - .multipart - .load(session.context, session.upload) - .await - .map_err(catalog_error)? - .ok_or(FileS3ErrorCode::InternalError)?; - if !self - .multipart - .settle_part(&pending) - .await - .map_err(catalog_error)? - { - return Err(FileS3ErrorCode::SlowDown); + if part.stream.is_some() { + part = self + .multipart + .put_stream_part(&session, &part, now_ms) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::SlowDown)?; + } else { + let before = self + .multipart + .part_for_upload(&session, part_number) + .await + .map_err(catalog_error)?; + part.revision = before.map_or(1, |part| part.revision.checked_add(1).unwrap_or(0)); + let pending = self + .multipart + .reserve_part_state(&session, &part, now_ms) + .await + .map_err(catalog_error)? + .ok_or(FileS3ErrorCode::SlowDown)?; + if !self + .multipart + .settle_part(&pending) + .await + .map_err(catalog_error)? + { + return Err(FileS3ErrorCode::SlowDown); + } } - let settled = self - .multipart - .load(session.context, session.upload) - .await - .map_err(catalog_error)? - .ok_or(FileS3ErrorCode::InternalError)?; - let part = self - .multipart - .part(&settled, part_number) - .await - .map_err(catalog_error)? - .ok_or(FileS3ErrorCode::InternalError)?; MultipartResponses::upload_part(&part) .map(|response| response.map(IcebergBody::new)) .map_err(|_| FileS3ErrorCode::InternalError) @@ -358,6 +383,17 @@ impl FileHttp { .completion .as_ref() .ok_or(FileS3ErrorCode::InternalError)?; + if completion.progress.next_part == 0 + && self + .multipart + .prepare_stream_publication(&session, now_ms) + .await + .map_err(catalog_error)? + .is_some() + { + session = self.current(&session).await?; + continue; + } if completion.progress.next_part < completion.selected_parts { self.multipart .advance_completion( diff --git a/app/crowdb-access-server/src/iceberg/file_http/stream.rs b/app/crowdb-access-server/src/iceberg/file_http/stream.rs new file mode 100644 index 000000000..c509c61b5 --- /dev/null +++ b/app/crowdb-access-server/src/iceberg/file_http/stream.rs @@ -0,0 +1,189 @@ +use std::fmt::Write; + +use crowdb_access_iceberg::file::{ + ContentFormat, FileContent, FileIdentity, FileKind, FileLocation, FileRecord, +}; +use crowdb_access_s3::native_buffer::NativeBodyReceiver; +use crowdb_chunk_client::{ChunkIoClient, ChunkIoWriter, LargeWritePolicy}; +use crowdb_protocol::frame::MAX_FRAME_PAYLOAD_BYTES; +use http_body_util::BodyExt; +use hyper::body::Bytes; +use hyper::body::Incoming; +use sha2::{Digest, Sha256}; + +use super::{ + admission_error, multipart, FileS3ErrorCode, FileTransferAdmission, FileUploadBody, FileUploadBudget, +}; + +#[allow(clippy::too_many_arguments, clippy::too_many_lines)] +pub(super) async fn upload( + client: &ChunkIoClient, + budget: &FileUploadBudget, + admission: &FileTransferAdmission, + body: &mut FileUploadBody, + owner: FileIdentity, + location: FileLocation, + declared_length: Option, + expected_sha256: Option<[u8; 32]>, + native_receiver: Option<&NativeBodyReceiver>, + small_threshold_exclusive: usize, + large_write: &LargeWritePolicy, +) -> Result { + let _permit = budget.acquire().map_err(|_| FileS3ErrorCode::SlowDown)?; + let small = declared_length + .and_then(|length| usize::try_from(length).ok()) + .filter(|length| *length < small_threshold_exclusive); + let handoff = native_receiver.filter(|_| small.is_none() && body.native_handoff_eligible()); + let mut writer: Box = if let Some(length) = small { + let key = location.to_string(); + if length <= MAX_FRAME_PAYLOAD_BYTES { + let mut small = client + .prepare_small_write_for_key(length, key.as_bytes()) + .await + .map_err(|_| FileS3ErrorCode::SlowDown)?; + small.require_durable_completion(); + Box::new(small) + } else { + Box::new( + client + .prepare_shared_object_write_for_key(length, key.as_bytes()) + .await + .map_err(|_| FileS3ErrorCode::SlowDown)?, + ) + } + } else { + let mut large = client.prepare_large_write(declared_length, large_write.clone()); + large + .wait_until_prepared() + .await + .map_err(|_| FileS3ErrorCode::SlowDown)?; + Box::new(large) + }; + if let Some(receiver) = handoff { + receiver.enable_owner_handoff(); + } + let mut length = 0u64; + let mut sha256 = expected_sha256.map(|_| Sha256::new()); + let target_buffer = usize::try_from(declared_length.unwrap_or(1024 * 1024)) + .unwrap_or(1024 * 1024) + .clamp(1, 1024 * 1024); + let mut pending = Vec::with_capacity(if handoff.is_some() { 0 } else { target_buffer }); + let transfer = async { + while let Some(frame) = body.frame().await { + let mut bytes = frame + .map_err(multipart::encoding_error)? + .into_data() + .map_err(|_| FileS3ErrorCode::InvalidRequest)?; + if bytes.is_empty() { + continue; + } + length = length + .checked_add(bytes.len() as u64) + .ok_or(FileS3ErrorCode::EntityTooLarge)?; + admission.check_bytes(length, length).map_err(admission_error)?; + if declared_length.is_some_and(|declared| length > declared) { + return Err(FileS3ErrorCode::InvalidRequest); + } + if let Some(hash) = &mut sha256 { + hash.update(&bytes); + } + while !writer.require_data() && !writer.input_complete() { + writer.wait_for_capacity().await; + } + if let Some(receiver) = handoff { + if let Some(owner) = receiver.take_ready_owner() { + writer + .on_framed_data(Box::new(owner)) + .await + .map_err(|_| FileS3ErrorCode::SlowDown)?; + } + } else { + while !bytes.is_empty() { + let count = (target_buffer - pending.len()).min(bytes.len()); + pending.extend_from_slice(&bytes.split_to(count)); + if pending.len() == target_buffer { + writer + .on_data(Bytes::from(std::mem::replace( + &mut pending, + Vec::with_capacity(target_buffer), + ))) + .await + .map_err(|_| FileS3ErrorCode::SlowDown)?; + } + } + } + } + if let Some(receiver) = handoff { + if let Some(owner) = receiver + .finish_owner_when_ready() + .await + .map_err(|_| FileS3ErrorCode::InvalidRequest)? + { + writer + .on_framed_data(Box::new(owner)) + .await + .map_err(|_| FileS3ErrorCode::SlowDown)?; + } + } + if !pending.is_empty() { + writer + .on_data(Bytes::from(pending)) + .await + .map_err(|_| FileS3ErrorCode::SlowDown)?; + } + if declared_length.is_some_and(|declared| declared != length) + || expected_sha256 + .zip(sha256) + .is_some_and(|(expected, hash)| <[u8; 32]>::from(hash.finalize()) != expected) + { + return Err(FileS3ErrorCode::InvalidRequest); + } + writer.on_finish().await.map_err(|_| FileS3ErrorCode::SlowDown) + } + .await; + let locations = match transfer { + Ok(locations) => locations, + Err(error) => { + let _ = writer.on_error().await; + return Err(error); + } + }; + let mut etag = String::with_capacity(32); + for byte in body.md5() { + write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); + } + let content = + FileContent::from_locations(&locations, length, etag).map_err(|_| FileS3ErrorCode::InternalError)?; + let (kind, format) = format_for_location(&location); + let record = FileRecord { + file: owner.file, + location, + kind, + format, + length, + digest: [0; 32], + content, + hint: None, + }; + record.validate().map_err(|_| FileS3ErrorCode::InternalError)?; + Ok(record) +} + +fn format_for_location(location: &FileLocation) -> (FileKind, ContentFormat) { + let path = location.relative_key(); + let extension = std::path::Path::new(path).extension(); + let has_extension = |wanted: &str| extension.is_some_and(|value| value.eq_ignore_ascii_case(wanted)); + if has_extension("json") { + (FileKind::Metadata, ContentFormat::Json) + } else if has_extension("avro") { + (FileKind::Unbound, ContentFormat::Avro) + } else if has_extension("parquet") { + (FileKind::Unbound, ContentFormat::Parquet) + } else if has_extension("orc") { + (FileKind::Unbound, ContentFormat::Orc) + } else if has_extension("puffin") { + (FileKind::Unbound, ContentFormat::Puffin) + } else { + (FileKind::Unbound, ContentFormat::Opaque) + } +} diff --git a/app/crowdb-access-server/src/iceberg/file_response.rs b/app/crowdb-access-server/src/iceberg/file_response.rs index 9fbf3bed3..b4d858990 100644 --- a/app/crowdb-access-server/src/iceberg/file_response.rs +++ b/app/crowdb-access-server/src/iceberg/file_response.rs @@ -110,7 +110,7 @@ impl MultipartResponses { let mut response = Response::new(Vec::new()); response.headers_mut().insert( ETAG, - HeaderValue::from_str(&etag(part.tree.digest)).map_err(|_| FileResponseError::Invalid)?, + HeaderValue::from_str(&format!("\"{}\"", part.etag())).map_err(|_| FileResponseError::Invalid)?, ); Ok(response) } @@ -169,8 +169,8 @@ impl MultipartResponses { "LastModified", ×tamp.to_rfc3339_opts(SecondsFormat::Millis, true), ); - element(&mut body, "ETag", &etag(part.tree.digest)); - element(&mut body, "Size", &part.tree.length.to_string()); + element(&mut body, "ETag", &format!("\"{}\"", part.etag())); + element(&mut body, "Size", &part.length().to_string()); body.push_str(""); } end(&mut body, "ListPartsResult"); @@ -191,20 +191,24 @@ impl MultipartResponses { || session.location != record.location || response_url.len() > 2048 || !(response_url.starts_with("https://") || response_url.starts_with("http://")) - || session - .completion - .as_ref() - .and_then(|completion| completion.candidate.as_ref()) - .map_or(true, |candidate| { + || session.completion.as_ref().map_or(true, |completion| { + if let Some(candidate) = &completion.candidate { candidate.length != record.length || candidate.digest != record.digest - }) + } else { + completion.publication.is_none() || completion.progress.completed_bytes != record.length + } + }) { return Err(FileResponseError::Invalid); } let mut body = start("CompleteMultipartUploadResult"); element(&mut body, "Location", response_url); location_fields(&mut body, session); - element(&mut body, "ETag", &etag(record.digest)); + let tag = record + .content + .etag() + .map_or_else(|| etag(record.digest), |value| format!("\"{value}\"")); + element(&mut body, "ETag", &tag); end(&mut body, "CompleteMultipartUploadResult"); Ok(xml(StatusCode::OK, body)) } diff --git a/app/crowdb-access-server/src/iceberg/file_selection.rs b/app/crowdb-access-server/src/iceberg/file_selection.rs index 88a1dc5b2..2bb6c4cf5 100644 --- a/app/crowdb-access-server/src/iceberg/file_selection.rs +++ b/app/crowdb-access-server/src/iceberg/file_selection.rs @@ -1,5 +1,7 @@ use crowdb_access_iceberg::catalog::CatalogError; -use crowdb_access_iceberg::file::{MultipartRepository, MultipartSelection, MultipartSession, SelectedPart}; +use crowdb_access_iceberg::file::{ + MultipartPhase, MultipartRepository, MultipartSelection, MultipartSession, SelectedPart, +}; use quick_xml::events::Event; use quick_xml::Reader; @@ -7,10 +9,10 @@ const MAX_COMPLETE_XML_BYTES: usize = 2 * 1024 * 1024; const MAX_COMPLETE_PARTS: usize = 10_000; const S3_NAMESPACE: &[u8] = b"http://s3.amazonaws.com/doc/2006-03-01/"; -#[derive(Clone, Copy, Debug, Eq, PartialEq)] +#[derive(Clone, Debug, Eq, PartialEq)] pub struct CompletePart { pub number: u16, - pub digest: [u8; 32], + pub etag: String, } #[derive(Clone, Debug, Eq, PartialEq)] @@ -96,7 +98,7 @@ impl CompleteSelection { } (State::Part, b"Part") => { let number = number.take().ok_or(CompleteRequestError)?; - let digest = digest.take().ok_or(CompleteRequestError)?; + let etag = digest.take().ok_or(CompleteRequestError)?; if number == 0 || number > 10_000 || parts @@ -105,7 +107,7 @@ impl CompleteSelection { { return Err(CompleteRequestError); } - parts.push(CompletePart { number, digest }); + parts.push(CompletePart { number, etag }); State::Root } (State::Root, b"CompleteMultipartUpload") if !parts.is_empty() => State::Done, @@ -134,25 +136,48 @@ impl CompleteSelection { if self.parts.len() > usize::from(session.limits.max_parts) { return Err(CompleteResolveError::InvalidPart); } + if session.completion.is_some() { + let frozen = repository.load_selection(session).await?; + if let Some(snapshots) = frozen.snapshots() { + if self.parts.len() != snapshots.len() + || self.parts.iter().zip(frozen.parts().iter().zip(snapshots)).any( + |(requested, (selected, snapshot))| { + requested.number != selected.number || requested.etag != snapshot.etag + }, + ) + { + return Err(CompleteResolveError::InvalidPart); + } + return Ok(frozen); + } + } let mut selected = Vec::with_capacity(self.parts.len()); + let mut parts = Vec::with_capacity(self.parts.len()); for (index, requested) in self.parts.iter().enumerate() { - let part = repository - .part(session, requested.number) - .await? - .ok_or(CompleteResolveError::InvalidPart)?; - if part.tree.digest != requested.digest { + let part = if session.phase == MultipartPhase::Open { + repository.part_for_upload(session, requested.number).await? + } else { + repository.part(session, requested.number).await? + } + .ok_or(CompleteResolveError::InvalidPart)?; + if part.etag() != requested.etag { return Err(CompleteResolveError::InvalidPart); } - if index + 1 < self.parts.len() && part.tree.length < 5 * 1024 * 1024 { + if index + 1 < self.parts.len() && part.length() < 5 * 1024 * 1024 { return Err(CompleteResolveError::EntityTooSmall); } selected.push(SelectedPart { number: part.number, revision: part.revision, - digest: part.tree.digest, + digest: part.selection_digest(), }); + parts.push(part); + } + if parts.iter().all(|part| part.stream.is_some()) { + MultipartSelection::with_stream_parts(&parts).map_err(|_| CompleteResolveError::InvalidPart) + } else { + MultipartSelection::new(selected).map_err(|_| CompleteResolveError::InvalidPart) } - MultipartSelection::new(selected).map_err(|_| CompleteResolveError::InvalidPart) } } @@ -166,19 +191,18 @@ enum State { Done, } -fn parse_etag(bytes: &[u8]) -> Result<[u8; 32], CompleteRequestError> { +fn parse_etag(bytes: &[u8]) -> Result { let hex = bytes .strip_prefix(b"\"") .and_then(|bytes| bytes.strip_suffix(b"\"")) .ok_or(CompleteRequestError)?; - if hex.len() != 64 { + if hex.len() != 32 && hex.len() != 64 { return Err(CompleteRequestError); } - let mut digest = [0; 32]; - for (target, pair) in digest.iter_mut().zip(hex.chunks_exact(2)) { - *target = (hex_digit(pair[0])? << 4) | hex_digit(pair[1])?; + for pair in hex.chunks_exact(2) { + let _ = (hex_digit(pair[0])? << 4) | hex_digit(pair[1])?; } - Ok(digest) + String::from_utf8(hex.to_vec()).map_err(|_| CompleteRequestError) } fn append_etag(etag: &mut Vec, bytes: &[u8]) -> Result<(), CompleteRequestError> { diff --git a/app/crowdb-access-server/src/iceberg/file_upload.rs b/app/crowdb-access-server/src/iceberg/file_upload.rs index 6a7992505..4888e6159 100644 --- a/app/crowdb-access-server/src/iceberg/file_upload.rs +++ b/app/crowdb-access-server/src/iceberg/file_upload.rs @@ -69,6 +69,15 @@ impl FileUploadBudget { self.active.load(Ordering::Acquire) } + pub(crate) fn acquire(&self) -> Result { + self.active + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |active| { + (active < self.limit).then_some(active + 1) + }) + .map_err(|_| FileUploadError::Busy)?; + Ok(Permit(self.active.clone())) + } + /// Stages bytes only; the caller must authorize intersected limits and seal before publication. /// Writes each received frame in bounded slices and never polls ahead of a pending storage write. /// # Errors @@ -82,12 +91,7 @@ impl FileUploadBudget { constraints: FileUploadConstraints, ) -> Result { constraints.validate()?; - self.active - .fetch_update(Ordering::AcqRel, Ordering::Acquire, |active| { - (active < self.limit).then_some(active + 1) - }) - .map_err(|_| FileUploadError::Busy)?; - let _permit = Permit(self.active.clone()); + let _permit = self.acquire()?; let mut writer = FileTreeWriter::new(store, owner, NATIVE_FILE_BLOCK_BYTES)?; while let Some(frame) = body.frame().await { let bytes = frame @@ -125,7 +129,7 @@ impl FileUploadBudget { } } -struct Permit(Arc); +pub(crate) struct Permit(Arc); impl Drop for Permit { fn drop(&mut self) { diff --git a/app/crowdb-access-server/src/iceberg/gc_control.rs b/app/crowdb-access-server/src/iceberg/gc_control.rs index 3b8aeac3a..3a784ebc5 100644 --- a/app/crowdb-access-server/src/iceberg/gc_control.rs +++ b/app/crowdb-access-server/src/iceberg/gc_control.rs @@ -4,7 +4,7 @@ use crowdb_access_iceberg::{ catalog::{ CatalogContext, CatalogRepository, CatalogStore, ManagementPrivilege, RootState, RoutedCatalogStore, }, - gc::{GcLimits, GcPin, GcRepository, GcTask, ReaderPins}, + gc::{GcLimits, GcRepository, GcTask}, key::{CatalogId, OperationId, TableId}, operation::{ManagementAction, ManagementPhase}, record::StorageRecord, @@ -13,6 +13,7 @@ use crowdb_access_iceberg::{ }; use super::gc_runtime::GcRuntimeConfig; +use crate::config::IcebergGcConfig; type BoxError = Box; @@ -21,6 +22,7 @@ pub(super) async fn manage( store: Arc, authentication: &BearerAuthenticator, arguments: &[String], + gc_settings: &IcebergGcConfig, ) -> Result<(), BoxError> { let token = std::env::var("CROWDB_ICEBERG_TOKEN")?; let principal = authentication @@ -29,10 +31,9 @@ pub(super) async fn manage( if principal.management == ManagementPrivilege::None { return Err("management privilege is required".into()); } - let config = GcRuntimeConfig::from_env()?; + let config = GcRuntimeConfig::from_config(gc_settings)?; let limits = config.limits; let repository = GcRepository::new(store.clone()); - let pins = ReaderPins::new(store.clone()); match arguments.iter().map(String::as_str).collect::>().as_slice() { ["limits"] => { println!("{}", serde_json::json!({ @@ -74,13 +75,7 @@ pub(super) async fn manage( }; show(&task); } - ["pin", identity, table] => { - pin_table(catalog, store.as_ref(), &pins, principal.name, identity, table).await?; - } - ["unpin", catalog_id, table, identity] => { - unpin_table(&pins, principal.name, catalog_id, table, identity).await?; - } - _ => return Err("usage: crowdb-iceberg gc limits | start-table UUID TABLE_ID | start-retired UUID CATALOG_ID EPOCH | inspect|pause|resume|retry CATALOG_ID TASK_ID | pin UUID TABLE_ID | unpin CATALOG_ID TABLE_ID PIN_ID".into()), + _ => return Err("usage: crowdb-iceberg gc limits | start-table UUID TABLE_ID | start-retired UUID CATALOG_ID EPOCH | inspect|pause|resume|retry CATALOG_ID TASK_ID".into()), } Ok(()) } @@ -154,59 +149,6 @@ async fn start_retired( Ok(()) } -async fn pin_table( - catalog: &CatalogRepository, - store: &RoutedCatalogStore, - pins: &ReaderPins, - principal: &str, - identity: &str, - table: &str, -) -> Result<(), BoxError> { - let (root, _) = catalog.status().await?; - if root.state != RootState::Ready { - return Err("catalog is not ready".into()); - } - let table: TableId = table.parse()?; - let pin = GcPin { - context: root.context, - identity: identity.parse()?, - head: load_head(store, root.context.catalog, table).await?, - principal: principal.to_owned(), - expires_ms: 0, - released: false, - operator: true, - protects_uploads: true, - }; - pins.acquire(&pin).await?; - println!( - "{}", - serde_json::json!({"pin": pin.identity.to_string(), "table": table.to_string()}) - ); - Ok(()) -} - -async fn unpin_table( - pins: &ReaderPins, - principal: &str, - catalog_id: &str, - table: &str, - identity: &str, -) -> Result<(), BoxError> { - let catalog_id: CatalogId = catalog_id.parse()?; - let table: TableId = table.parse()?; - let identity: OperationId = identity.parse()?; - let pin = pins - .get(catalog_id, table, identity) - .await? - .ok_or("GC pin is missing")?; - if !pin.operator || pin.principal != principal { - return Err("operator pin is owned by another principal".into()); - } - pins.release(&pin).await?; - println!("{}", serde_json::json!({"released": identity.to_string()})); - Ok(()) -} - async fn load_head( store: &RoutedCatalogStore, catalog: CatalogId, diff --git a/app/crowdb-access-server/src/iceberg/gc_runtime.rs b/app/crowdb-access-server/src/iceberg/gc_runtime.rs index abafaa9ea..e84dc73bd 100644 --- a/app/crowdb-access-server/src/iceberg/gc_runtime.rs +++ b/app/crowdb-access-server/src/iceberg/gc_runtime.rs @@ -1,5 +1,6 @@ use std::{sync::Arc, time::Duration}; +use crate::config::IcebergGcConfig; use crowdb_access_iceberg::{ catalog::{CatalogContext, CatalogRepository, RootState, RoutedCatalogStore}, file::FileBlockStore, @@ -32,21 +33,35 @@ pub(super) struct GcRuntimeConfig { } impl GcRuntimeConfig { - pub fn from_env() -> Result> { + pub fn from_config(file: &IcebergGcConfig) -> Result> { let mut limits = GcLimits::default(); - limits.step_bytes = setting("CROWDB_ICEBERG_GC_STEP_BYTES", limits.step_bytes)?; - limits.step_ms = setting("CROWDB_ICEBERG_GC_STEP_MS", limits.step_ms)?; - limits.page_items = setting("CROWDB_ICEBERG_GC_PAGE_ITEMS", limits.page_items)?; - limits.page_bytes = setting("CROWDB_ICEBERG_GC_PAGE_BYTES", limits.page_bytes)?; - limits.concurrency = setting("CROWDB_ICEBERG_GC_CONCURRENCY", limits.concurrency)?; - limits.retry_base_ms = setting("CROWDB_ICEBERG_GC_RETRY_BASE_MS", limits.retry_base_ms)?; - limits.retry_max_ms = setting("CROWDB_ICEBERG_GC_RETRY_MAX_MS", limits.retry_max_ms)?; - limits.corruption_attempts = setting( + limits.step_bytes = setting_or("CROWDB_ICEBERG_GC_STEP_BYTES", file.step_bytes, limits.step_bytes)?; + limits.step_ms = setting_or("CROWDB_ICEBERG_GC_STEP_MS", file.step_ms, limits.step_ms)?; + limits.page_items = setting_or("CROWDB_ICEBERG_GC_PAGE_ITEMS", file.page_items, limits.page_items)?; + limits.page_bytes = setting_or("CROWDB_ICEBERG_GC_PAGE_BYTES", file.page_bytes, limits.page_bytes)?; + limits.concurrency = setting_or( + "CROWDB_ICEBERG_GC_CONCURRENCY", + file.concurrency, + limits.concurrency, + )?; + limits.retry_base_ms = setting_or( + "CROWDB_ICEBERG_GC_RETRY_BASE_MS", + file.retry_base_ms, + limits.retry_base_ms, + )?; + limits.retry_max_ms = setting_or( + "CROWDB_ICEBERG_GC_RETRY_MAX_MS", + file.retry_max_ms, + limits.retry_max_ms, + )?; + limits.corruption_attempts = setting_or( "CROWDB_ICEBERG_GC_CORRUPTION_ATTEMPTS", + file.corruption_attempts, limits.corruption_attempts, )?; - limits.minimum_retention_ms = setting( + limits.minimum_retention_ms = setting_or( "CROWDB_ICEBERG_GC_MINIMUM_RETENTION_MS", + file.minimum_retention_ms, limits.minimum_retention_ms, )?; limits.validate()?; @@ -56,35 +71,47 @@ impl GcRuntimeConfig { if limits.minimum_retention_ms < GcLimits::default().minimum_retention_ms { return Err("GC retention must be at least seven days".into()); } - let interval_ms = setting("CROWDB_ICEBERG_GC_INTERVAL_MS", 1000_u64)?; + let interval_ms = setting_or("CROWDB_ICEBERG_GC_INTERVAL_MS", file.interval_ms, 1000_u64)?; if !(100..=60_000).contains(&interval_ms) { return Err("GC interval must be between 100 and 60000 milliseconds".into()); } - let catalogs = match std::env::var("CROWDB_ICEBERG_GC_CATALOGS") { - Ok(value) => value, - Err(std::env::VarError::NotPresent) => String::new(), - Err(error) => return Err(error.into()), - } - .split(',') - .map(str::trim) - .filter(|value| !value.is_empty()) - .map(str::parse) - .collect::, _>>()?; + let names = match &file.catalogs { + Some(names) => names.clone(), + None => match std::env::var("CROWDB_ICEBERG_GC_CATALOGS") { + Ok(value) => value.split(',').map(str::to_owned).collect(), + Err(std::env::VarError::NotPresent) => Vec::new(), + Err(error) => return Err(error.into()), + }, + }; + let catalogs = names + .iter() + .map(String::as_str) + .map(str::trim) + .filter(|value| !value.is_empty()) + .map(str::parse) + .collect::, _>>()?; if catalogs.len() > 64 { return Err("too many GC catalog scopes".into()); } let mut catalogs: Vec = catalogs; catalogs.sort_unstable(); catalogs.dedup(); - let enabled = match std::env::var("CROWDB_ICEBERG_GC_ENABLED").as_deref() { - Ok("1") => true, - Ok("0") | Err(std::env::VarError::NotPresent) => false, - _ => return Err("CROWDB_ICEBERG_GC_ENABLED must be 0 or 1".into()), + let enabled = match file.enabled { + Some(enabled) => enabled, + None => match std::env::var("CROWDB_ICEBERG_GC_ENABLED").as_deref() { + Ok("1") => true, + Ok("0") | Err(std::env::VarError::NotPresent) => false, + _ => return Err("CROWDB_ICEBERG_GC_ENABLED must be 0 or 1".into()), + }, }; - let kv_bytes = setting("CROWDB_ICEBERG_GC_KV_BYTES", 64 * 1024 * 1024_u64)?; - let kv_requests = setting("CROWDB_ICEBERG_GC_KV_REQUESTS", 128_u32)?; - let chunk_bytes = setting("CROWDB_ICEBERG_GC_CHUNK_BYTES", 8 * 1024 * 1024_u64)?; - let chunk_requests = setting("CROWDB_ICEBERG_GC_CHUNK_REQUESTS", 128_u32)?; + let kv_bytes = setting_or("CROWDB_ICEBERG_GC_KV_BYTES", file.kv_bytes, 64 * 1024 * 1024_u64)?; + let kv_requests = setting_or("CROWDB_ICEBERG_GC_KV_REQUESTS", file.kv_requests, 128_u32)?; + let chunk_bytes = setting_or( + "CROWDB_ICEBERG_GC_CHUNK_BYTES", + file.chunk_bytes, + 8 * 1024 * 1024_u64, + )?; + let chunk_requests = setting_or("CROWDB_ICEBERG_GC_CHUNK_REQUESTS", file.chunk_requests, 128_u32)?; if !(4 * 1024 * 1024..=256 * 1024 * 1024).contains(&kv_bytes) || !(8..=4096).contains(&kv_requests) || !(256 * 1024..=64 * 1024 * 1024).contains(&chunk_bytes) @@ -117,6 +144,21 @@ where } } +fn setting_or( + name: &str, + configured: Option, + default: T, +) -> Result> +where + T: std::str::FromStr, + T::Err: std::error::Error + Send + Sync + 'static, +{ + match configured { + Some(value) => Ok(value), + None => setting(name, default), + } +} + pub(super) async fn run( catalog: Arc, store: Arc, diff --git a/app/crowdb-access-server/src/iceberg/http.rs b/app/crowdb-access-server/src/iceberg/http.rs index a823f1631..fcb2f55a1 100644 --- a/app/crowdb-access-server/src/iceberg/http.rs +++ b/app/crowdb-access-server/src/iceberg/http.rs @@ -13,6 +13,8 @@ use crowdb_access_iceberg::catalog::{ Capabilities, CatalogError, CatalogLifecycle, CatalogRepository, ManagementPrivilege, RootState, }; use crowdb_access_iceberg::wire::{BearerAuthenticator, CatalogConfig, IcebergErrorResponse}; +use crowdb_access_s3::native_buffer::NativeBodyAllocator; +use crowdb_chunk_client::LargeWritePolicy; use hyper::body::Incoming; use hyper::server::conn::http1; use hyper::service::service_fn; @@ -55,7 +57,13 @@ impl IcebergHttpService { #[must_use] pub fn metrics_snapshot(&self) -> IcebergMetricsSnapshot { - self.metrics.snapshot() + let mut snapshot = self.metrics.snapshot(); + snapshot.catalog = self.repository.store_operation_counts(); + if let Some((read, write)) = self.files.as_ref().and_then(|files| files.chunk_metrics()) { + snapshot.chunk_read = Some(read); + snapshot.chunk_small_write = Some(write); + } + snapshot } /// # Errors @@ -65,16 +73,65 @@ impl IcebergHttpService { store: Arc, blocks: Arc, region: String, + ) -> Result { + self = self.with_fileio_native(store, blocks, region, None)?; + Ok(self) + } + + /// Installs native receive owners for authenticated file uploads. + /// # Errors + /// Rejects invalid native file listener limits or signing configuration. + pub fn with_fileio_native( + mut self, + store: Arc, + blocks: Arc, + region: String, + native_allocator: Option>, ) -> Result { self.files = Some(Arc::new(FileHttp::new( store, blocks, self.authentication.namespace_token_key(), region, + native_allocator, )?)); Ok(self) } + /// Applies the configured shared-chunk routing threshold to native file uploads. + /// # Errors + /// Rejects an unavailable file service or an invalid threshold. + pub fn with_small_object_threshold( + mut self, + threshold_exclusive: usize, + ) -> Result { + Arc::get_mut( + self.files + .as_mut() + .ok_or(crowdb_access_iceberg::file::FileGrantError::Invalid)?, + ) + .ok_or(crowdb_access_iceberg::file::FileGrantError::Invalid)? + .set_small_threshold(threshold_exclusive)?; + Ok(self) + } + + /// Applies the configured EC layout and data-block size to native large writes. + /// # Errors + /// Rejects an unavailable file service or invalid write policy. + pub fn with_large_write_policy( + mut self, + policy: LargeWritePolicy, + ) -> Result { + Arc::get_mut( + self.files + .as_mut() + .ok_or(crowdb_access_iceberg::file::FileGrantError::Invalid)?, + ) + .ok_or(crowdb_access_iceberg::file::FileGrantError::Invalid)? + .set_large_write(policy)?; + Ok(self) + } + /// # Errors /// Rejects invalid namespace token signing configuration. pub fn with_namespaces( @@ -217,7 +274,7 @@ impl IcebergHttpService { } return Ok(response( 200, - serde_json::to_vec(&self.metrics.snapshot()).map_err(|_| service_unavailable())?, + serde_json::to_vec(&self.metrics_snapshot()).map_err(|_| service_unavailable())?, )); } let (root, authority) = self diff --git a/app/crowdb-access-server/src/iceberg/metrics.rs b/app/crowdb-access-server/src/iceberg/metrics.rs index 02722c428..f538e846e 100644 --- a/app/crowdb-access-server/src/iceberg/metrics.rs +++ b/app/crowdb-access-server/src/iceberg/metrics.rs @@ -49,6 +49,9 @@ pub struct IcebergMetricsSnapshot { pub retry_resume: u64, pub retry_replay: u64, pub selected_versions: [u64; 3], + pub chunk_read: Option, + pub chunk_small_write: Option, + pub catalog: Option, } struct Counters { @@ -105,6 +108,9 @@ impl IcebergMetrics { retry_resume: self.retry[1].load(Ordering::Relaxed), retry_replay: self.retry[2].load(Ordering::Relaxed), selected_versions: array::from_fn(|index| self.selected_versions[index].load(Ordering::Relaxed)), + chunk_read: None, + chunk_small_write: None, + catalog: None, } } } diff --git a/app/crowdb-access-server/src/iceberg/runtime.rs b/app/crowdb-access-server/src/iceberg/runtime.rs index cc2c78e4d..4c143c2fa 100644 --- a/app/crowdb-access-server/src/iceberg/runtime.rs +++ b/app/crowdb-access-server/src/iceberg/runtime.rs @@ -7,14 +7,20 @@ use crowdb_access_iceberg::catalog::{ }; use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; use crowdb_access_iceberg::wire::BearerAuthenticator; -use crowdb_chunk_client::{ChunkIoClient, ChunkIoClientConfig, SmallWritePolicy}; +use crowdb_access_s3::native_buffer::NativeBodyAllocator; +use crowdb_chunk_client::{ + ChunkClientConfig, ChunkIoClient, ChunkIoClientConfig, ChunkReadPolicy, LargeWritePolicy, + SmallWritePolicy, +}; use crowdb_chunk_kv_client::{ ChunkKvClient, ChunkKvRpcTransport, ClientConfig, Group0ChunkKvRangeCatalogSource, }; +use crowdb_common::ec::EcScheme; use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; use tokio::net::TcpListener; use super::{serve, IcebergHttpService}; +use crate::config::{load_args, AccessConfig}; type BoxError = Box; @@ -28,16 +34,26 @@ impl IcebergRuntimeConfig { /// # Errors /// Rejects missing/invalid credentials, seeds or listener configuration. pub fn from_env() -> Result { - let seeds = std::env::var("CROWDB_MANAGEMENT_SEEDS")?; - if seeds.len() > 8192 { - return Err("management seed configuration is oversized".into()); - } - let management_seeds: Vec<_> = seeds - .split(',') - .map(str::trim) - .filter(|value| !value.is_empty()) - .map(str::to_owned) - .collect(); + Self::from_config(&AccessConfig::default()) + } + + /// # Errors + /// Rejects missing/invalid credentials, seeds or listener configuration. + pub fn from_config(access: &AccessConfig) -> Result { + let management_seeds: Vec<_> = if access.common.management_seeds.is_empty() { + let seeds = std::env::var("CROWDB_MANAGEMENT_SEEDS")?; + if seeds.len() > 8192 { + return Err("management seed configuration is oversized".into()); + } + seeds + .split(',') + .map(str::trim) + .filter(|value| !value.is_empty()) + .map(str::to_owned) + .collect() + } else { + access.common.management_seeds.clone() + }; if management_seeds.is_empty() || management_seeds.len() > 16 { return Err("one to sixteen management seeds are required".into()); } @@ -49,7 +65,12 @@ impl IcebergRuntimeConfig { &std::env::var("CROWDB_ICEBERG_CLEAR_TOKEN")?, ) .map_err(|error| format!("invalid Iceberg bearer credential configuration: {error}"))?; - let listen = std::env::var("CROWDB_ICEBERG_LISTEN").unwrap_or_else(|_| "127.0.0.1:8181".into()); + let listen = access + .iceberg + .listen + .clone() + .or_else(|| std::env::var("CROWDB_ICEBERG_LISTEN").ok()) + .unwrap_or_else(|| "127.0.0.1:8181".into()); let _: std::net::SocketAddr = listen.parse()?; Ok(Self { listen, @@ -62,12 +83,19 @@ impl IcebergRuntimeConfig { /// # Errors /// Returns configuration, authentication, storage, management or listener failures. pub async fn run() -> Result<(), BoxError> { - let config = IcebergRuntimeConfig::from_env()?; - let arguments: Vec<_> = std::env::args().skip(1).collect(); + let (access_config, arguments) = load_args(std::env::args().skip(1).collect())?; + let config = IcebergRuntimeConfig::from_config(&access_config)?; if arguments.len() > 7 { return Err("too many Iceberg command arguments".into()); } - let (repository, store, chunks) = connect(config.management_seeds.clone()).await?; + let (repository, store, chunks) = connect( + config.management_seeds.clone(), + access_config.read.policy(), + access_config.small_write.policy(), + access_config.common.diskio_connections_per_endpoint, + access_config.common.diskio_rpc_workers, + ) + .await?; let result = if arguments.is_empty() || arguments == ["serve"] { Box::pin(start_listener( &config.listen, @@ -76,6 +104,7 @@ pub async fn run() -> Result<(), BoxError> { config.authentication, chunks.clone(), config.management_seeds, + access_config, )) .await } else if arguments.first().is_some_and(|argument| argument == "gc") { @@ -84,6 +113,7 @@ pub async fn run() -> Result<(), BoxError> { store.clone(), &config.authentication, &arguments[1..], + &access_config.iceberg.gc, ) .await } else { @@ -97,6 +127,10 @@ pub async fn run() -> Result<(), BoxError> { async fn connect( seeds: Vec, + read_policy: ChunkReadPolicy, + small_write: SmallWritePolicy, + diskio_connections_per_endpoint: usize, + diskio_rpc_workers: u32, ) -> Result<(Arc, Arc, ChunkIoClient), BoxError> { let control = Arc::new(CrowdbKvClient::new(KvConfig::new(seeds.clone()))); let client_config = ClientConfig::default(); @@ -108,14 +142,15 @@ async fn connect( )); let client = Arc::new(ChunkKvClient::new(client_config, source, transport)?); client.refresh_catalog().await?; - let chunks = ChunkIoClient::connect_with_kv( + let chunks = ChunkIoClient::connect_with_kv_read_policy( ChunkIoClientConfig { management_seeds: seeds, - diskio_connections_per_endpoint: 2, - diskio_rpc_workers: 2, - small_write: SmallWritePolicy::default(), + diskio_connections_per_endpoint, + diskio_rpc_workers, + small_write, }, control, + read_policy, ) .await?; let store = Arc::new(RoutedCatalogStore::new(client)); @@ -137,8 +172,9 @@ async fn start_listener( authentication: BearerAuthenticator, chunks: ChunkIoClient, management_seeds: Vec, + access_config: AccessConfig, ) -> Result<(), BoxError> { - let gc_config = super::gc_runtime::GcRuntimeConfig::from_env()?; + let gc_config = super::gc_runtime::GcRuntimeConfig::from_config(&access_config.iceberg.gc)?; for _ in 0..600 { match repository.recover(now_ms()?).await { Ok(()) => break, @@ -157,9 +193,31 @@ async fn start_listener( let blocks: Arc = Arc::new( crowdb_access_iceberg::file::NativeFileBlocks::new(chunks.clone(), store.clone()), ); + let native_budget = access_config + .iceberg + .native_budget_bytes + .unwrap_or(256 * 1024 * 1024); + let native_allocator = Arc::new(NativeBodyAllocator::new(native_budget, 1024 * 1024)?); + let large_write = LargeWritePolicy { + ec_scheme: EcScheme::new( + access_config.small_write.ec_data, + access_config.small_write.ec_code, + ), + client: Arc::new(ChunkClientConfig { + read_buffer_size: access_config.small_write.disk_block_bytes, + ..ChunkClientConfig::default() + }), + }; let mut service = IcebergHttpService::new(repository.clone(), authentication, timeout) .with_namespaces(store.clone())? - .with_fileio(store.clone(), blocks.clone(), "us-east-1".into())?; + .with_fileio_native( + store.clone(), + blocks.clone(), + "us-east-1".into(), + Some(native_allocator), + )? + .with_small_object_threshold(access_config.small_write.threshold_exclusive())? + .with_large_write_policy(large_write)?; if authority.admission_bounds.delegated_access_ms >= 900_000 { let endpoint = std::env::var("CROWDB_ICEBERG_PUBLIC_URI").unwrap_or_else(|_| format!("http://{address}")); @@ -182,7 +240,14 @@ async fn start_listener( )); let tables = super::table_recovery::run(repository.clone(), store.clone(), blocks.clone()); let (gc_store, gc_chunks) = if gc_config.enabled { - let (_, gc_store, gc_chunks) = connect(management_seeds).await?; + let (_, gc_store, gc_chunks) = connect( + management_seeds, + access_config.read.policy(), + access_config.small_write.policy(), + access_config.common.diskio_connections_per_endpoint, + access_config.common.diskio_rpc_workers, + ) + .await?; (gc_store, Some(gc_chunks)) } else { (store.clone(), None) diff --git a/app/crowdb-access-server/src/iceberg/table_credentials.rs b/app/crowdb-access-server/src/iceberg/table_credentials.rs index 9d40b2dd1..155f2b4da 100644 --- a/app/crowdb-access-server/src/iceberg/table_credentials.rs +++ b/app/crowdb-access-server/src/iceberg/table_credentials.rs @@ -227,14 +227,6 @@ impl TableCredentials { } .issue(&self.issuer, principal, context, &authority, target, now) .map_err(|_| service_unavailable())?; - let pins = crowdb_access_iceberg::gc::ReaderPins::new(self.store.clone()); - let expires_ms = pins - .request_expiry(context, credentials.grant().expires_ms) - .await - .map_err(|_| service_unavailable())?; - pins.protect_files(context, target.table, principal.name, expires_ms, now) - .await - .map_err(|_| service_unavailable())?; Ok(response( 200, serde_json::to_vec(&LoadCredentialsResponse::from(credentials)) diff --git a/app/crowdb-access-server/src/iceberg/table_read.rs b/app/crowdb-access-server/src/iceberg/table_read.rs index d8d80e422..1b047f157 100644 --- a/app/crowdb-access-server/src/iceberg/table_read.rs +++ b/app/crowdb-access-server/src/iceberg/table_read.rs @@ -45,8 +45,7 @@ impl TableHttp { string_bytes: 1024 * 1024, collection_entries: 10_000, }, - ) - .with_catalog_reader_pins(), + ), lister: TableLister::new(store, secret)?, spools: Arc::new(AtomicUsize::new(0)), file_config: None, diff --git a/app/crowdb-access-server/src/lib.rs b/app/crowdb-access-server/src/lib.rs index 306c81425..d7b637f7c 100644 --- a/app/crowdb-access-server/src/lib.rs +++ b/app/crowdb-access-server/src/lib.rs @@ -3,6 +3,8 @@ //! Independent listener lifecycle for external access protocols. +pub mod config; +mod http_receive; pub mod iceberg; #[cfg(feature = "s3")] diff --git a/app/crowdb-access-server/src/main.rs b/app/crowdb-access-server/src/main.rs index 1294b36d6..dfe8eb8f4 100644 --- a/app/crowdb-access-server/src/main.rs +++ b/app/crowdb-access-server/src/main.rs @@ -18,6 +18,8 @@ use crowdb_access_s3::metrics::{DependencyHealth, S3Health, S3Metrics}; #[cfg(feature = "s3")] use crowdb_access_s3::native_buffer::NativeBodyAllocator; #[cfg(feature = "s3")] +use crowdb_access_server::config::{load_args, AccessConfig}; +#[cfg(feature = "s3")] use crowdb_access_server::credentials::CredentialAuthority; #[cfg(feature = "s3")] use crowdb_access_server::s3::{serve, ProductionS3Operations, S3Dispatcher, S3ServiceConfig}; @@ -26,6 +28,7 @@ use crowdb_access_server::storage::S3StorageClients; #[cfg(feature = "s3")] use crowdb_chunk_client::SmallWritePolicy; #[cfg(feature = "s3")] +#[cfg(feature = "s3")] use crowdb_common::ec::EcScheme; #[cfg(feature = "s3")] use crowdb_kv_client::{ClientConfig as KvConfig, CrowdbKvClient}; @@ -36,58 +39,73 @@ use tokio::net::TcpListener; async fn main() -> Result<(), Box> { tracing_subscriber::fmt().with_writer(std::io::stderr).init(); #[cfg(feature = "s3")] + let (access_config, remaining_args) = load_args(std::env::args().skip(1).collect())?; + #[cfg(feature = "s3")] if matches!( - std::env::args().nth(1).as_deref(), + remaining_args.first().map(String::as_str), Some("issue-user" | "ensure-user" | "lookup-user") ) { - return issue_user().await; + return issue_user(&remaining_args, &access_config).await; } #[cfg(feature = "s3")] - if let Ok(address) = std::env::var("CROWDB_S3_LISTEN") { - let management_seeds = management_seeds()?; - let tenant = TenantId::new(required_env("CROWDB_S3_TENANT")?.into_bytes())?; + if !remaining_args.is_empty() { + return Err("unexpected S3 server arguments".into()); + } + #[cfg(feature = "s3")] + run_s3(&access_config).await?; + Ok(()) +} + +#[cfg(feature = "s3")] +async fn run_s3(access_config: &AccessConfig) -> Result<(), Box> { + if let Some(address) = access_config + .s3 + .listen + .clone() + .or_else(|| std::env::var("CROWDB_S3_LISTEN").ok()) + { + let management_seeds = if access_config.common.management_seeds.is_empty() { + management_seeds()? + } else { + access_config.common.management_seeds.clone() + }; + let tenant_name = match &access_config.s3.tenant { + Some(name) => name.clone(), + None => required_env("CROWDB_S3_TENANT")?, + }; + let tenant = TenantId::new(tenant_name.into_bytes())?; let master_key = MasterKey::from_hex(&required_env("CROWDB_S3_MASTER_KEY")?)?; let credential_cipher = Arc::new(CredentialCipher::new(&master_key)); let continuation_key = credential_cipher.continuation_key().to_vec(); - let small_write = SmallWritePolicy::default(); - let storage = S3StorageClients::connect(management_seeds, 2, 2, small_write.clone()).await?; + let (ec_scheme, small_write, small_threshold) = s3_write_routing(access_config)?; + let storage = S3StorageClients::connect_with_read_policy( + management_seeds, + access_config.common.diskio_connections_per_endpoint, + access_config.common.diskio_rpc_workers, + small_write.clone(), + access_config.read.policy(), + ) + .await?; let chunks = Arc::clone(&storage.chunks); let authority = Arc::new(CredentialAuthority::new( Arc::clone(&storage.control), Arc::clone(&credential_cipher), )); - let trusted_network = std::env::var("CROWDB_S3_TRUSTED_NETWORK").as_deref() == Ok("true"); - let (authenticator, credential_refresh): ( - Arc, - Option>, - ) = if trusted_network { - tracing::warn!(%address, "starting S3 with explicit trusted-network authentication bypass"); - (Arc::new(TrustedNetworkAuthenticator::new()), None) - } else { - let cache = Arc::new(CredentialCache::new(90)); - refresh_credentials(&authority, &cache).await?; - let refresh_authority = Arc::clone(&authority); - let refresh_cache = Arc::clone(&cache); - let task = tokio::spawn(async move { - let mut interval = tokio::time::interval(Duration::from_secs(30)); - interval.tick().await; - loop { - interval.tick().await; - if let Err(error) = refresh_credentials(&refresh_authority, &refresh_cache).await { - tracing::error!(%error, "S3 credential refresh failed; cache will fail closed when stale"); - } - } - }); - let region = std::env::var("CROWDB_S3_REGION").unwrap_or_else(|_| "us-east-1".into()); - (Arc::new(SigV4Verifier::new(cache, region, 900)), Some(task)) - }; - let mut service_config = S3ServiceConfig::basic(tenant, continuation_key, small_write.object_limit); - service_config.small_object_limit = - optional_usize("CROWDB_S3_SMALL_OBJECT_LIMIT")?.unwrap_or(service_config.small_object_limit); - configure_large_write(&mut service_config)?; + let (authenticator, credential_refresh, trusted_network) = + authenticate_s3(access_config, &address, &authority).await?; + let service_config = s3_service_config( + access_config, + tenant, + continuation_key, + small_threshold, + ec_scheme, + )?; let metrics = Arc::new(S3Metrics::default()); - let cleanup_backlog_limit = optional_usize("CROWDB_S3_CLEANUP_BACKLOG_LIMIT")? - .map_or(10_000, |value| u64::try_from(value).unwrap_or(u64::MAX)); + let cleanup_backlog_limit = configured_u64( + access_config.s3.cleanup_backlog_limit, + "CROWDB_S3_CLEANUP_BACKLOG_LIMIT", + )? + .unwrap_or(10_000); let health = Arc::new(S3Health::starting(cleanup_backlog_limit)); health.set_metadata(DependencyHealth::Ready); health.set_chunks(DependencyHealth::Ready); @@ -98,7 +116,11 @@ async fn main() -> Result<(), Box> { .with_metrics(Arc::clone(&metrics)) .with_health(Arc::clone(&health)), ); - let native_budget = optional_usize("CROWDB_S3_NATIVE_BUDGET_BYTES")?.unwrap_or(256 * 1024 * 1024); + let native_budget = configured_usize( + access_config.s3.native_budget_bytes, + "CROWDB_S3_NATIVE_BUDGET_BYTES", + )? + .unwrap_or(256 * 1024 * 1024); let body_allocator = Arc::new(NativeBodyAllocator::new(native_budget, 1024 * 1024)?); let handler = Arc::new( S3Dispatcher::new( @@ -130,36 +152,142 @@ async fn main() -> Result<(), Box> { } #[cfg(feature = "s3")] -fn configure_large_write(config: &mut S3ServiceConfig) -> Result<(), Box> { - let ec_data = optional_usize("CROWDB_S3_EC_DATA")?.unwrap_or(config.large_write.ec_scheme.data_num); - let ec_code = optional_usize("CROWDB_S3_EC_CODE")?.unwrap_or(config.large_write.ec_scheme.code_num); - if ec_data == 0 || ec_code == 0 { - return Err("CROWDB S3 EC data and code counts must be nonzero".into()); +async fn authenticate_s3( + access: &AccessConfig, + address: &str, + authority: &Arc, +) -> Result< + ( + Arc, + Option>, + bool, + ), + Box, +> { + let trusted_network = access + .s3 + .trusted_network + .unwrap_or(std::env::var("CROWDB_S3_TRUSTED_NETWORK").as_deref() == Ok("true")); + if trusted_network { + tracing::warn!(%address, "starting S3 with explicit trusted-network authentication bypass"); + return Ok((Arc::new(TrustedNetworkAuthenticator::new()), None, true)); } - config.large_write.ec_scheme = EcScheme::new(ec_data, ec_code); - if let Some(max_chunk_size) = optional_usize("CROWDB_S3_MAX_CHUNK_SIZE")? { + let cache = Arc::new(CredentialCache::new(90)); + refresh_credentials(authority, &cache).await?; + let refresh_authority = Arc::clone(authority); + let refresh_cache = Arc::clone(&cache); + let task = tokio::spawn(async move { + let mut interval = tokio::time::interval(Duration::from_secs(30)); + interval.tick().await; + loop { + interval.tick().await; + if let Err(error) = refresh_credentials(&refresh_authority, &refresh_cache).await { + tracing::error!(%error, "S3 credential refresh failed; cache will fail closed when stale"); + } + } + }); + let region = access + .s3 + .region + .clone() + .or_else(|| std::env::var("CROWDB_S3_REGION").ok()) + .unwrap_or_else(|| "us-east-1".into()); + Ok(( + Arc::new(SigV4Verifier::new(cache, region, 900)), + Some(task), + false, + )) +} + +#[cfg(feature = "s3")] +fn s3_service_config( + access: &AccessConfig, + tenant: TenantId, + continuation_key: Vec, + small_write_limit: usize, + ec_scheme: EcScheme, +) -> Result> { + let mut config = S3ServiceConfig::basic(tenant, continuation_key, small_write_limit); + config.large_write.ec_scheme = ec_scheme; + if let Some(limit) = access.s3.list_scan_items { + config.list_scan_items = limit; + } + if let Some(limit) = access.s3.list_scan_bytes { + config.list_scan_bytes = limit; + } + if let Some(ttl) = access.s3.continuation_ttl_seconds { + config.continuation_ttl_seconds = ttl; + } + config.small_object_limit = + configured_usize(access.s3.small_object_limit, "CROWDB_S3_SMALL_OBJECT_LIMIT")? + .unwrap_or(config.small_object_limit) + .min(small_write_limit.saturating_sub(1)); + configure_large_write(&mut config, access)?; + Ok(config) +} + +#[cfg(feature = "s3")] +fn configure_large_write( + config: &mut S3ServiceConfig, + access: &AccessConfig, +) -> Result<(), Box> { + Arc::make_mut(&mut config.large_write.client).read_buffer_size = access.small_write.disk_block_bytes; + if let Some(max_chunk_size) = configured_u64(access.s3.max_chunk_size, "CROWDB_S3_MAX_CHUNK_SIZE")? { if max_chunk_size == 0 { return Err("CROWDB S3 max chunk size must be nonzero".into()); } - Arc::make_mut(&mut config.large_write.client).max_chunk_size = - u64::try_from(max_chunk_size).unwrap_or(u64::MAX); + Arc::make_mut(&mut config.large_write.client).max_chunk_size = max_chunk_size; } Ok(()) } #[cfg(feature = "s3")] -async fn issue_user() -> Result<(), Box> { - let command = std::env::args().nth(1).ok_or("missing S3 user command")?; - let user = std::env::args() - .nth(2) +fn s3_ec_scheme(access: &AccessConfig) -> Result> { + let ec_data = + configured_usize(access.s3.ec_data, "CROWDB_S3_EC_DATA")?.unwrap_or(access.small_write.ec_data); + let ec_code = + configured_usize(access.s3.ec_code, "CROWDB_S3_EC_CODE")?.unwrap_or(access.small_write.ec_code); + if ec_data == 0 || ec_data > 32 || ec_code == 0 { + return Err("CROWDB S3 EC data and code counts are invalid".into()); + } + Ok(EcScheme::new(ec_data, ec_code)) +} + +#[cfg(feature = "s3")] +fn s3_write_routing( + access: &AccessConfig, +) -> Result<(EcScheme, SmallWritePolicy, usize), Box> { + let ec_scheme = s3_ec_scheme(access)?; + let mut config = access.small_write.clone(); + config.ec_data = ec_scheme.data_num; + config.ec_code = ec_scheme.code_num; + let policy = config.policy(); + policy.validate()?; + let threshold = config.threshold_exclusive(); + if threshold > policy.object_limit { + return Err("S3 small-object threshold exceeds the shared writer limit".into()); + } + Ok((ec_scheme, policy, threshold)) +} + +#[cfg(feature = "s3")] +async fn issue_user(args: &[String], access: &AccessConfig) -> Result<(), Box> { + let command = args.first().ok_or("missing S3 user command")?; + let user = args + .get(1) .filter(|value| !value.is_empty()) .ok_or("usage: crowdb-access-server issue-user|ensure-user|lookup-user USER")?; - if std::env::args().nth(3).is_some() { + if args.len() != 2 { return Err("usage: crowdb-access-server issue-user|ensure-user|lookup-user USER".into()); } let master_key = MasterKey::from_hex(&required_env("CROWDB_S3_MASTER_KEY")?)?; let cipher = Arc::new(CredentialCipher::new(&master_key)); - let control = Arc::new(CrowdbKvClient::new(KvConfig::new(management_seeds()?))); + let seeds = if access.common.management_seeds.is_empty() { + management_seeds()? + } else { + access.common.management_seeds.clone() + }; + let control = Arc::new(CrowdbKvClient::new(KvConfig::new(seeds))); let authority = CredentialAuthority::new(control, cipher); let token = match command.as_str() { "ensure-user" => authority.ensure_user(user.as_bytes()).await?, @@ -211,6 +339,30 @@ fn optional_usize(name: &str) -> Result, Box, + env_name: &str, +) -> Result, Box> { + match configured { + Some(value) => Ok(Some(value)), + None => optional_usize(env_name), + } +} + +#[cfg(feature = "s3")] +fn configured_u64( + configured: Option, + env_name: &str, +) -> Result, Box> { + match configured { + Some(value) => Ok(Some(value)), + None => { + optional_usize(env_name).map(|value| value.map(|value| u64::try_from(value).unwrap_or(u64::MAX))) + } + } +} + #[cfg(feature = "s3")] fn required_env(name: &str) -> Result> { std::env::var(name).map_err(|_| format!("{name} is required when S3 is enabled").into()) diff --git a/app/crowdb-access-server/src/s3.rs b/app/crowdb-access-server/src/s3.rs index d60d2bf5c..8c8201029 100644 --- a/app/crowdb-access-server/src/s3.rs +++ b/app/crowdb-access-server/src/s3.rs @@ -10,10 +10,9 @@ use std::time::Instant; use crowdb_access_s3::error::S3Error; use crowdb_access_s3::metrics::{OutcomeClass, S3Metrics}; -use crowdb_access_s3::native_buffer::NativeBodyReceiver; use crowdb_access_s3::route::S3Operation; use http_body_util::{BodyExt, Full}; -use hyper::body::{Body, Bytes, Frame, Http1BodyReceiveProvider, Incoming, SizeHint}; +use hyper::body::{Body, Bytes, Frame, Incoming, SizeHint}; use hyper::server::conn::http1; use hyper::service::service_fn; use hyper::{Request, Response}; @@ -23,6 +22,7 @@ use tokio::net::TcpListener; mod dispatcher; mod operations; +pub use crate::http_receive::install_body_receive_provider; pub use dispatcher::S3Dispatcher; pub use operations::{ProductionS3Operations, S3Operations, S3OperationsFuture, S3ServiceConfig}; @@ -35,39 +35,6 @@ pub trait S3HttpHandler: Send + Sync + 'static { fn handle(&self, request: Request) -> HandlerFuture; } -#[derive(Clone)] -pub(crate) struct DeferredBodyReceiveProvider { - provider: Arc, - native: Option>, -} - -impl DeferredBodyReceiveProvider { - fn generic(provider: Arc) -> Self { - Self { - provider, - native: None, - } - } - - fn native(receiver: Arc) -> Self { - Self { - provider: receiver.clone(), - native: Some(receiver), - } - } -} - -/// Installs the admitted request's provider immediately before body polling. -pub fn install_body_receive_provider(request: &mut Request) -> Option> { - if let Some(deferred) = request.extensions_mut().remove::() { - request - .body_mut() - .set_http1_body_receive_provider(deferred.provider); - return deferred.native; - } - None -} - /// Runs one independent HTTP/1 S3 listener until shutdown. /// /// # Errors diff --git a/app/crowdb-access-server/src/s3/dispatcher.rs b/app/crowdb-access-server/src/s3/dispatcher.rs index 318fb88fb..90cd14586 100644 --- a/app/crowdb-access-server/src/s3/dispatcher.rs +++ b/app/crowdb-access-server/src/s3/dispatcher.rs @@ -16,9 +16,9 @@ use hyper::{Method, Request, Response, StatusCode}; use tracing::Instrument; use super::{ - error_response, full_body, measured_body, DeferredBodyReceiveProvider, HandlerFuture, ResponseBody, - S3HttpHandler, S3Operations, + error_response, full_body, measured_body, HandlerFuture, ResponseBody, S3HttpHandler, S3Operations, }; +use crate::http_receive::DeferredBodyReceiveProvider; pub struct S3Dispatcher { authenticator: Arc, diff --git a/app/crowdb-access-server/src/s3/operations.rs b/app/crowdb-access-server/src/s3/operations.rs index 7b44287d2..8e6616621 100644 --- a/app/crowdb-access-server/src/s3/operations.rs +++ b/app/crowdb-access-server/src/s3/operations.rs @@ -17,8 +17,8 @@ use crowdb_access_s3::publication::PublicationRequest; use crowdb_access_s3::retrieval::{self, ObjectHeaders, RetrievalError}; use crowdb_access_s3::route::{S3Operation, S3Route}; use crowdb_access_s3::streaming::{ - publish_completed_locations, write_body_with_checksums_metered, write_native_body_with_checksums_metered, - PutErrorCode, PutOutcome, + publish_completed_locations, write_body_with_checksums_buffered, + write_native_body_with_checksums_metered, PutErrorCode, PutOutcome, }; use crowdb_chunk_client::{ ChunkClientConfig, ChunkIoWriter, IoError, LargeWritePolicy, PreparedLargeWrite, SharedObjectWriter, @@ -245,8 +245,11 @@ impl ProductionS3Operations { route_key.extend_from_slice(bucket_id.as_bytes()); route_key.extend_from_slice(&key); let mut writer = self.prepare_writer(content_length, &route_key).await?; - let native_receiver = install_body_receive_provider(&mut request); - let native_receiver = native_receiver.filter(|_| writer.is_large() && content_length.is_some()); + let native_receiver = if writer.is_large() && content_length.is_some() { + install_body_receive_provider(&mut request) + } else { + None + }; if let Some(receiver) = &native_receiver { receiver.enable_owner_handoff(); } @@ -262,11 +265,16 @@ impl ProductionS3Operations { ) .await } else { - write_body_with_checksums_metered( + let receive_bytes = content_length + .and_then(|length| usize::try_from(length).ok()) + .unwrap_or(1024 * 1024) + .clamp(1, 1024 * 1024); + write_body_with_checksums_buffered( &mut body, &mut writer, content_md5.as_deref(), payload_sha256.as_deref(), + receive_bytes, self.metrics.as_deref(), ) .await @@ -455,6 +463,15 @@ impl ProductionS3Operations { .and_then(|length| usize::try_from(length).ok()) .filter(|length| *length <= self.config.small_object_limit) { + if length > crowdb_protocol::frame::MAX_FRAME_PAYLOAD_BYTES { + return self + .storage + .chunks + .prepare_shared_object_write_for_key(length, key) + .await + .map(ObjectWriter::Small) + .map_err(|_| S3ErrorCode::SlowDown); + } match self.storage.chunks.prepare_small_write_for_key(length, key).await { Ok(writer) => return Ok(ObjectWriter::Small(writer)), Err(IoError::ObjectTooLarge { .. }) => {} @@ -523,7 +540,7 @@ impl ChunkIoWriter for ObjectWriter { async fn on_finish(&mut self) -> crowdb_chunk_client::Result> { match self { - Self::Small(writer) => writer.on_finish().await, + Self::Small(writer) => writer.finish_durable().await, Self::Large(writer) => writer.on_finish().await, } } diff --git a/app/crowdb-access-server/src/storage.rs b/app/crowdb-access-server/src/storage.rs index 3983c63db..ff9d1eb45 100644 --- a/app/crowdb-access-server/src/storage.rs +++ b/app/crowdb-access-server/src/storage.rs @@ -6,7 +6,7 @@ use std::sync::Arc; use crowdb_access_s3::metadata::ChunkKvMetadataStore; -use crowdb_chunk_client::{ChunkIoClient, ChunkIoClientConfig, SmallWritePolicy}; +use crowdb_chunk_client::{ChunkIoClient, ChunkIoClientConfig, ChunkReadPolicy, SmallWritePolicy}; use crowdb_chunk_kv_client::{ ChunkKvClient, ChunkKvRpcTransport, ClientConfig as ChunkKvConfig, Group0ChunkKvRangeCatalogSource, }; @@ -38,6 +38,25 @@ impl S3StorageClients { diskio_connections_per_endpoint: usize, diskio_rpc_workers: u32, small_write: SmallWritePolicy, + ) -> Result { + Self::connect_with_read_policy( + management_seeds, + diskio_connections_per_endpoint, + diskio_rpc_workers, + small_write, + ChunkReadPolicy::default(), + ) + .await + } + + /// # Errors + /// Returns an error when the management or `DiskIO` connection cannot be established. + pub async fn connect_with_read_policy( + management_seeds: Vec, + diskio_connections_per_endpoint: usize, + diskio_rpc_workers: u32, + small_write: SmallWritePolicy, + read_policy: ChunkReadPolicy, ) -> Result { let kv = Arc::new(CrowdbKvClient::new(KvConfig::new(management_seeds.clone()))); let config = ChunkKvConfig::default(); @@ -49,7 +68,7 @@ impl S3StorageClients { .refresh_catalog() .await .map_err(|error| StorageConnectError::ChunkKv(error.to_string()))?; - let chunks = ChunkIoClient::connect_with_kv( + let chunks = ChunkIoClient::connect_with_kv_read_policy( ChunkIoClientConfig { management_seeds, diskio_connections_per_endpoint, @@ -57,6 +76,7 @@ impl S3StorageClients { small_write, }, Arc::clone(&kv), + read_policy, ) .await .map_err(|error| StorageConnectError::ChunkIo(error.to_string()))?; diff --git a/app/crowdb-access-server/tests/common/iceberg_commit_fault.rs b/app/crowdb-access-server/tests/common/iceberg_commit_fault.rs index 60c8a01c4..3fe5b66e4 100644 --- a/app/crowdb-access-server/tests/common/iceberg_commit_fault.rs +++ b/app/crowdb-access-server/tests/common/iceberg_commit_fault.rs @@ -12,6 +12,7 @@ use crowdb_access_iceberg::{ namespace::{ChildScan, NamespaceStore}, record::StorageRecord, }; +use crowdb_chunk_client::ChunkIoClient; use crowdb_chunk_kv_client::MultiScanPage; use crowdb_protocol::chunk_kv::ClientRequestId; @@ -115,6 +116,10 @@ pub struct TestCommitBlocks { #[async_trait] impl FileBlockStore for TestCommitBlocks { + fn stream_client(&self) -> Option<&ChunkIoClient> { + self.inner.stream_client() + } + async fn put(&self, owner: FileIdentity, height: u8, bytes: &[u8]) -> Result { let index = self.boundary.before("file-block").await; let result = self.inner.put(owner, height, bytes).await; diff --git a/app/crowdb-access-server/tests/common/iceberg_file_recovery.rs b/app/crowdb-access-server/tests/common/iceberg_file_recovery.rs index b73c94e07..82f320257 100644 --- a/app/crowdb-access-server/tests/common/iceberg_file_recovery.rs +++ b/app/crowdb-access-server/tests/common/iceberg_file_recovery.rs @@ -57,8 +57,6 @@ pub async fn run() { assert!(labels.contains("multipart-Completing")); assert!(labels.contains("multipart-Publishing")); assert!(labels.contains("multipart-Published")); - } else { - assert!(labels.contains("file-block")); } } } diff --git a/app/crowdb-access-server/tests/common/iceberg_file_worker.rs b/app/crowdb-access-server/tests/common/iceberg_file_worker.rs index ffdd4b0be..2f0b35194 100644 --- a/app/crowdb-access-server/tests/common/iceberg_file_worker.rs +++ b/app/crowdb-access-server/tests/common/iceberg_file_worker.rs @@ -64,11 +64,12 @@ pub async fn verify(stack: &TestIcebergStack, context: CatalogContext, table: Ta table, file: FileId::random(), }, - tree: FileTree { + tree: Some(FileTree { root: None, length: 0, digest: Sha256::digest([]).into(), - }, + }), + stream: None, }; assert!(repository.reserve_part(&admitted, &part, 2).await.unwrap()); let mut worker = TestWorker::start(&stack.cluster.mgmt_endpoints); diff --git a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileIO.java b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileIO.java index 13125128d..54a09e5e8 100644 --- a/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileIO.java +++ b/app/crowdb-access-server/tests/common/iceberg_java/src/main/java/TestIcebergFileIO.java @@ -15,7 +15,6 @@ import software.amazon.awssdk.core.sync.RequestBody; import software.amazon.awssdk.services.s3.S3Client; import software.amazon.awssdk.services.s3.model.CompletedPart; -import software.amazon.awssdk.services.s3.model.S3Exception; public class TestIcebergFileIO { public static void main(String[] args) throws Exception { @@ -51,6 +50,7 @@ public static void main(String[] args) throws Exception { files.initialize(properties); String prefix = configuration.getProperty("location"); byte[] small = "{\"client\":\"iceberg-java-1.11.0\"}".getBytes(StandardCharsets.UTF_8); + System.out.println("S3FileIO: small PUT/GET"); verify(files, prefix + "metadata/sdk-small.json", small); byte[] large = new byte[6 * 1024 * 1024]; Arrays.fill(large, (byte) 'x'); @@ -58,8 +58,11 @@ public static void main(String[] args) throws Exception { System.arraycopy(start, 0, large, 0, start.length); large[large.length - 2] = '"'; large[large.length - 1] = '}'; + System.out.println("S3FileIO: multipart PUT/GET"); verify(files, prefix + "metadata/sdk-multipart.json", large); - verifyLateError(files.client(), prefix + "metadata/sdk-invalid.json"); + System.out.println("S3FileIO: opaque multipart Complete"); + verifyOpaqueMultipart(files.client(), prefix + "metadata/sdk-opaque.json"); + System.out.println("S3FileIO: file operations"); TestIcebergFileOperations.run(files.client(), prefix); if (credentialRequests.get() != 1) { throw new AssertionError("SDK did not fetch and cache the delegated credential response"); @@ -67,27 +70,23 @@ public static void main(String[] args) throws Exception { } finally { credentials.stop(0); } - System.out.println("Apache Iceberg 1.11.0 S3FileIO PUT, multipart, HEAD, GET, seek and embedded error passed"); + System.out.println("Apache Iceberg 1.11.0 S3FileIO PUT, multipart, HEAD, GET, seek and opaque file passed"); } - private static void verifyLateError(S3Client client, String location) { + private static void verifyOpaqueMultipart(S3Client client, String location) { URI uri = URI.create(location); String bucket = uri.getHost(); String key = uri.getPath().substring(1); String upload = client.createMultipartUpload(request -> request.bucket(bucket).key(key)).uploadId(); + byte[] payload = "{invalid-json".getBytes(StandardCharsets.UTF_8); String etag = client.uploadPart( request -> request.bucket(bucket).key(key).uploadId(upload).partNumber(1), - RequestBody.fromString("{invalid-json")).eTag(); - try { - client.completeMultipartUpload(request -> request.bucket(bucket).key(key).uploadId(upload) - .multipartUpload(parts -> parts.parts(CompletedPart.builder().partNumber(1).eTag(etag).build()))); - throw new AssertionError("SDK accepted an embedded Complete error as success"); - } catch (S3Exception error) { - if (!"InvalidRequest".equals(error.awsErrorDetails().errorCode())) { - throw error; - } - } finally { - client.abortMultipartUpload(request -> request.bucket(bucket).key(key).uploadId(upload)); + RequestBody.fromBytes(payload)).eTag(); + client.completeMultipartUpload(request -> request.bucket(bucket).key(key).uploadId(upload) + .multipartUpload(parts -> parts.parts(CompletedPart.builder().partNumber(1).eTag(etag).build()))); + byte[] stored = client.getObjectAsBytes(request -> request.bucket(bucket).key(key)).asByteArray(); + if (!Arrays.equals(stored, payload)) { + throw new AssertionError("opaque multipart payload changed"); } } diff --git a/app/crowdb-access-server/tests/common/iceberg_multipart.rs b/app/crowdb-access-server/tests/common/iceberg_multipart.rs index ca149a34f..cbb9d55a8 100644 --- a/app/crowdb-access-server/tests/common/iceberg_multipart.rs +++ b/app/crowdb-access-server/tests/common/iceberg_multipart.rs @@ -54,7 +54,8 @@ pub async fn verify_restart(stack: &mut TestIcebergStack, context: CatalogContex revision: 1, modified_ms: 101, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, }; assert!(repository.reserve_part(&initial, &part, 101).await.unwrap()); let recovery = MultipartRecovery::new(stack.store().await, blocks.clone(), 7, 8).unwrap(); @@ -66,7 +67,7 @@ pub async fn verify_restart(stack: &mut TestIcebergStack, context: CatalogContex let selection = MultipartSelection::new(vec![SelectedPart { number: 1, revision: 1, - digest: part.tree.digest, + digest: part.selection_digest(), }]) .unwrap(); assert!(repository diff --git a/app/crowdb-access-server/tests/common/iceberg_process.rs b/app/crowdb-access-server/tests/common/iceberg_process.rs index 54951fdcb..27a1d103c 100644 --- a/app/crowdb-access-server/tests/common/iceberg_process.rs +++ b/app/crowdb-access-server/tests/common/iceberg_process.rs @@ -41,10 +41,9 @@ impl TestIcebergProcess { let mut process = Self { child, address }; tokio::time::timeout(Duration::from_secs(30), async { loop { - assert!( - process.child.try_wait().unwrap().is_none(), - "Iceberg listener exited" - ); + if let Some(status) = process.child.try_wait().unwrap() { + panic!("Iceberg listener exited before readiness: {status}"); + } if tokio::net::TcpStream::connect(address).await.is_ok() { break; } diff --git a/app/crowdb-access-server/tests/config_test.rs b/app/crowdb-access-server/tests/config_test.rs new file mode 100644 index 000000000..028f46251 --- /dev/null +++ b/app/crowdb-access-server/tests/config_test.rs @@ -0,0 +1,95 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +use std::path::Path; + +use crowdb_access_server::config::{load_args, AccessConfig, SmallWriteConfig}; +use crowdb_common::config::{load_from_file, BaseConfig}; + +#[test] +fn tracked_access_configs_load_and_set_bounded_read_resources() { + let root = Path::new(env!("CARGO_MANIFEST_DIR")); + let canonical: AccessConfig = + load_from_file(&root.join("conf/crowdb_access_server_config.toml")).unwrap(); + let container: AccessConfig = + load_from_file(&root.join("../../container/single-node-container/templates/access.toml")).unwrap(); + for (config, expected_ec, expected_threshold) in + [(canonical, (8, 4), 7_549_748), (container, (2, 1), 1_887_437)] + { + assert_eq!(config.read.stream_slots, 3); + assert_eq!(config.read.stream_window_bytes, 1024 * 1024); + assert_eq!(config.read.global_stream_bytes, 256 * 1024 * 1024); + assert_eq!(config.read.recovery_memory_bytes, 256 * 1024 * 1024); + assert_eq!(config.read.policy().stream_slots, 3); + assert_eq!(config.small_write.memory_budget_bytes, 1280 * 1024 * 1024); + assert_eq!(config.small_write.disk_block_bytes, 1024 * 1024); + assert_eq!( + (config.small_write.ec_data, config.small_write.ec_code), + expected_ec + ); + assert_eq!( + (config.s3.ec_data, config.s3.ec_code), + (Some(expected_ec.0), Some(expected_ec.1)) + ); + assert_eq!(config.small_write.threshold_exclusive(), expected_threshold); + assert_eq!(config.small_write.policy().max_pipelines, 32); + assert!(config.s3.listen.is_some()); + assert!(config.iceberg.listen.is_some()); + assert_eq!(config.iceberg.native_budget_bytes, Some(256 * 1024 * 1024)); + assert_eq!(config.s3.list_scan_bytes, Some(4 * 1024 * 1024)); + assert_eq!(config.iceberg.gc.kv_bytes, Some(64 * 1024 * 1024)); + } +} + +#[test] +fn config_argument_is_removed_from_service_commands() { + let root = Path::new(env!("CARGO_MANIFEST_DIR")); + let path = root.join("conf/crowdb_access_server_config.toml"); + let (_, remaining) = load_args(vec![ + "serve".into(), + "--config".into(), + path.display().to_string(), + ]) + .unwrap(); + assert_eq!(remaining, ["serve"]); + assert!(load_args(vec!["--config".into()]).is_err()); +} + +#[test] +fn invalid_read_budget_is_rejected() { + let mut config = AccessConfig::default(); + config.read.stream_slots = 0; + assert!(config.validate().is_err()); + config.read.stream_slots = 3; + config.small_write.memory_budget_bytes = 1; + assert!(config.validate().is_err()); +} + +#[test] +fn small_threshold_uses_strip_data_capacity_for_ec_and_mirror() { + let mut config = SmallWriteConfig { + ec_data: 8, + ec_code: 2, + ..SmallWriteConfig::default() + }; + assert_eq!(config.threshold_exclusive(), 7_549_748); + config.disk_block_bytes = 512 * 1024; + assert_eq!(config.threshold_exclusive(), 3_774_874); + config.conversion_enabled = false; + assert_eq!(config.threshold_exclusive(), 471_860); + config.disk_block_bytes = 1024 * 1024; + assert_eq!(config.threshold_exclusive(), 943_719); +} + +#[test] +fn unsupported_disk_block_size_is_rejected_before_routing() { + let mut config = AccessConfig::default(); + config.small_write.disk_block_bytes = 0; + assert!(config.validate().is_err()); + config.small_write.disk_block_bytes = 2 * 1024 * 1024; + assert!(config.validate().is_err()); + config.small_write.disk_block_bytes = 3 * 1024; + assert!(config.validate().is_err()); + config.small_write.disk_block_bytes = 768 * 1024; + assert!(config.validate().is_err()); +} diff --git a/app/crowdb-access-server/tests/iceberg_file_admission_test.rs b/app/crowdb-access-server/tests/iceberg_file_admission_test.rs index 5b8da7acc..39c3754e9 100644 --- a/app/crowdb-access-server/tests/iceberg_file_admission_test.rs +++ b/app/crowdb-access-server/tests/iceberg_file_admission_test.rs @@ -226,6 +226,10 @@ fn multipart_session_and_global_credit_intersections_fail_closed() { ]); let create = request(&grant, &Method::POST, "?uploads"); let admitted = FileTransferAdmission::authorize(&grant, &create, service(), None, 101).unwrap(); + let limits = admitted.multipart_limits().unwrap(); + assert_eq!(limits.max_part_bytes, 4); + assert_eq!(limits.max_parts, 7); + assert!(u64::from(limits.max_parts) * limits.max_part_bytes <= limits.max_staged_bytes); let mut session = session(&grant); let mut policy = policy(&grant); assert!(admitted.check_create(&session, &policy).is_ok()); diff --git a/app/crowdb-access-server/tests/iceberg_file_encoding_test.rs b/app/crowdb-access-server/tests/iceberg_file_encoding_test.rs index 20411ebc0..766a1abd1 100644 --- a/app/crowdb-access-server/tests/iceberg_file_encoding_test.rs +++ b/app/crowdb-access-server/tests/iceberg_file_encoding_test.rs @@ -14,6 +14,33 @@ use hyper::{header::HeaderValue, HeaderMap}; struct TestFrames(VecDeque); +#[test] +fn upload_integrity_is_present_only_with_a_declared_checksum_or_signed_chunks() { + let empty = HeaderMap::new(); + assert!(!FileUploadBody::new(Full::new(Bytes::new()), &empty, None, 100) + .unwrap() + .has_integrity()); + let mut md5 = HeaderMap::new(); + md5.insert( + "content-md5", + HeaderValue::from_static("1B2M2Y8AsgTpgAmY7PhCfg=="), + ); + assert!(FileUploadBody::new(Full::new(Bytes::new()), &md5, None, 100) + .unwrap() + .has_integrity()); + let mut crc = HeaderMap::new(); + crc.insert("x-amz-checksum-crc32c", HeaderValue::from_static("AAAAAA==")); + assert!(FileUploadBody::new(Full::new(Bytes::new()), &crc, None, 100) + .unwrap() + .has_integrity()); + let (headers, verifier, wire) = signed::fixture(); + assert!( + FileUploadBody::new(Full::new(Bytes::from(wire)), &headers, Some(verifier), 100_000) + .unwrap() + .has_integrity() + ); +} + #[tokio::test] async fn content_md5_checks_decoded_bytes_before_successful_eof() { use base64::{engine::general_purpose::STANDARD, Engine}; @@ -85,7 +112,7 @@ async fn aws_published_signed_trailer_vector_survives_arbitrary_http_boundaries( let mut output = Vec::new(); while let Some(frame) = body.frame().await { let bytes = frame.unwrap().into_data().unwrap(); - assert!(bytes.len() <= 64 * 1024); + assert!(bytes.len() <= 1024 * 1024); output.extend_from_slice(&bytes); } assert!(body.is_end_stream()); diff --git a/app/crowdb-access-server/tests/iceberg_file_http_test.rs b/app/crowdb-access-server/tests/iceberg_file_http_test.rs index 36e79e384..e53b4d4c0 100644 --- a/app/crowdb-access-server/tests/iceberg_file_http_test.rs +++ b/app/crowdb-access-server/tests/iceberg_file_http_test.rs @@ -22,7 +22,13 @@ use crowdb_access_iceberg::file::{ use crowdb_access_iceberg::key::OperationId; use crowdb_access_iceberg::operation::{ManagementAction, ManagementRequest, RequestIdentity}; use crowdb_access_iceberg::wire::BearerAuthenticator; +use crowdb_access_server::config::SmallWriteConfig; +use crowdb_protocol::chunkdb::rpc::{QueryChunkRequest, Strip}; +use crowdb_test_harness::chunkdb::make_client as make_chunkdb_client; +use md5::{Digest, Md5}; use reqwest::{Client, Method}; +use std::fmt::Write as _; +use std::time::Instant; #[path = "common/iceberg_signed_file.rs"] mod signed; @@ -66,6 +72,19 @@ async fn setup_with_bounds( process::TestIcebergProcess, TestFileClient, TableLocation, +) { + setup_with_bounds_and_file_limit(bounds, 16 * 1024 * 1024, 64 * 1024 * 1024).await +} + +async fn setup_with_bounds_and_file_limit( + bounds: ClearBounds, + max_request_bytes: u64, + max_file_bytes: u64, +) -> ( + TestIcebergStack, + process::TestIcebergProcess, + TestFileClient, + TableLocation, ) { let stack = TestIcebergStack::start().await; let repository = CatalogRepository::new(stack.store().await, bounds).unwrap(); @@ -137,8 +156,8 @@ async fn setup_with_bounds( FileOperation::AbortMultipart, ]) .unwrap(), - max_request_bytes: 16 * 1024 * 1024, - max_file_bytes: 64 * 1024 * 1024, + max_request_bytes, + max_file_bytes, }) .unwrap(); let client = TestFileClient { @@ -153,6 +172,107 @@ fn path(table: TableLocation, key: &str) -> String { format!("/{}/{}", table.bucket(), table.file(key).unwrap().object_key()) } +async fn catalog_counts(client: &TestFileClient) -> (u64, u64, u64, u64) { + let response: serde_json::Value = client + .client + .get(format!("http://{}/_crowdb/metrics", client.address)) + .bearer_auth("m".repeat(32)) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + let catalog = &response["catalog"]; + ( + catalog["get"].as_u64().unwrap(), + catalog["compare_exchange"].as_u64().unwrap(), + catalog["scan"].as_u64().unwrap(), + catalog["conditional_delete"].as_u64().unwrap(), + ) +} + +fn catalog_delta(before: (u64, u64, u64, u64), after: (u64, u64, u64, u64)) -> String { + format!( + "get={} cas={} scan={} delete={}", + after.0 - before.0, + after.1 - before.1, + after.2 - before.2, + after.3 - before.3 + ) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +#[ignore = "native null-DiskIO release performance fixture"] +async fn native_file_5_mib_profile() { + let (_stack, _process, client, table) = setup().await; + let bytes = vec![0x5a; 5 * 1024 * 1024]; + let object = path(table, "data/profile-put.bin"); + let before = catalog_counts(&client).await; + let started = Instant::now(); + let put = client.send(Method::PUT, &object, "", &bytes, true).await; + let put_ms = started.elapsed().as_millis(); + assert_eq!(put.status(), 200, "{}", put.text().await.unwrap()); + println!( + "iceberg 5MiB PUT: {put_ms}ms {}", + catalog_delta(before, catalog_counts(&client).await) + ); + + let multipart = path(table, "data/profile-mpu.bin"); + let created = client + .send(Method::POST, &multipart, "uploads=", b"", false) + .await; + assert_eq!(created.status(), 200); + let created = created.text().await.unwrap(); + let upload = created + .split_once("") + .unwrap() + .1 + .split_once("") + .unwrap() + .0; + let query = format!("partNumber=1&uploadId={upload}"); + let before = catalog_counts(&client).await; + let started = Instant::now(); + let part = client.send(Method::PUT, &multipart, &query, &bytes, true).await; + let part_ms = started.elapsed().as_millis(); + assert_eq!(part.status(), 200, "{}", part.text().await.unwrap()); + let etag = part.headers()["etag"].to_str().unwrap(); + println!( + "iceberg 5MiB UploadPart: {part_ms}ms {}", + catalog_delta(before, catalog_counts(&client).await) + ); + let manifest = format!( + "{etag}1" + ); + let complete = client + .send( + Method::POST, + &multipart, + &format!("uploadId={upload}"), + manifest.as_bytes(), + false, + ) + .await; + assert_eq!(complete.status(), 200); + assert!(complete + .text() + .await + .unwrap() + .contains("")); + + let before = catalog_counts(&client).await; + let started = Instant::now(); + let get = client.send(Method::GET, &object, "", b"", false).await; + assert_eq!(get.status(), 200); + assert_eq!(get.bytes().await.unwrap().as_ref(), bytes); + let get_ms = started.elapsed().as_millis(); + println!( + "iceberg 5MiB GET: {get_ms}ms {}", + catalog_delta(before, catalog_counts(&client).await) + ); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn signed_standard_put_get_and_multipart_publish_unbound_files() { let (stack, _process, client, table) = setup().await; @@ -184,10 +304,52 @@ async fn signed_standard_put_get_and_multipart_publish_unbound_files() { assert_eq!(record.kind, FileKind::Unbound); assert!(record.bind_kind(FileKind::EqualityDelete).is_ok()); let conflict = client - .send(Method::PUT, &object, "", b"PAR1difffoot\x04\0\0\0PAR1", false) + .send(Method::PUT, &object, "", b"PAR1difffoot\x04\0\0\0PAR1", true) .await; assert_eq!(conflict.status(), 409); + let medium = path(table, "data/medium.parquet"); + let medium_bytes = (0..1_200_000) + .map(|index| (index % 251) as u8) + .collect::>(); + let response = client.send(Method::PUT, &medium, "", &medium_bytes, true).await; + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + let stored = repository + .load( + client.credentials.grant().context, + &table.file("data/medium.parquet").unwrap(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(stored.content.locations(stored.length).unwrap().unwrap().len(), 1); + let range = client + .send_range( + Method::GET, + &medium, + "", + b"", + false, + Some("bytes=1048550-1048600"), + ) + .await; + assert_eq!(range.status(), 206); + assert_eq!( + range.bytes().await.unwrap().as_ref(), + &medium_bytes[1048550..1048601] + ); + let metrics: serde_json::Value = Client::new() + .get(format!("http://{}/_crowdb/metrics", client.address)) + .bearer_auth("m".repeat(32)) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + assert!(metrics["chunk_read"]["stream_windows"].as_u64().unwrap() > 0); + assert!(metrics["chunk_small_write"]["completed"].as_u64().unwrap() > 0); + let metadata = path(table, "metadata/b.json"); let create = client.send(Method::POST, &metadata, "uploads=", b"", false).await; assert_eq!(create.status(), 200); @@ -236,6 +398,33 @@ async fn signed_standard_put_get_and_multipart_publish_unbound_files() { .await .unwrap() .ends_with("")); + let mut composite = Md5::new(); + composite.update(Md5::digest(first)); + composite.update(Md5::digest(second)); + let expected_etag = composite + .finalize() + .iter() + .map(|byte| format!("{byte:02x}")) + .collect::() + + "-2"; + let published = repository + .load( + client.credentials.grant().context, + &table.file("metadata/b.json").unwrap(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(published.content.etag(), Some(expected_etag.as_str())); + assert_eq!( + published + .content + .locations(published.length) + .unwrap() + .unwrap() + .len(), + 2 + ); let replay = client .send( Method::POST, @@ -256,6 +445,214 @@ async fn signed_standard_put_get_and_multipart_publish_unbound_files() { assert_eq!(get.bytes().await.unwrap().as_ref(), document); } +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn ordinary_put_size_matrix_streams_and_reads_ranges() { + let (stack, _process, client, table) = setup_with_bounds_and_file_limit( + ClearBounds { + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + 128 * 1024 * 1024, + 128 * 1024 * 1024, + ) + .await; + let repository = FileRepository::new(stack.store().await); + for size in [10 * 1024, 1024 * 1024, 12 * 1024 * 1024, 100 * 1024 * 1024] { + let key = format!("data/size-{size}.parquet"); + let object = path(table, &key); + let bytes = (0..size).map(|offset| (offset % 256) as u8).collect::>(); + let expected = Md5::digest(&bytes); + let started = Instant::now(); + let put = client.send(Method::PUT, &object, "", &bytes, true).await; + assert_eq!(put.status(), 200, "{}", put.text().await.unwrap()); + let put_elapsed = started.elapsed(); + let head = client.send(Method::HEAD, &object, "", b"", false).await; + assert_eq!(head.status(), 200); + assert_eq!(head.headers()["content-length"], size.to_string()); + let record = repository + .load(client.credentials.grant().context, &table.file(&key).unwrap()) + .await + .unwrap() + .unwrap(); + assert_eq!(record.length, size as u64); + let locations = record.content.locations(record.length).unwrap().unwrap(); + assert!(!locations.is_empty()); + if size == 12 * 1024 * 1024 { + let chunkdb = make_chunkdb_client(stack.cluster.make_service_registry_client()); + let chunk = chunkdb + .query_chunk(QueryChunkRequest { + chunk_id: locations[0].chunk_id, + }) + .await + .unwrap() + .chunk + .unwrap(); + let ec = chunk.strips.iter().find_map(|strip| match strip.strip.as_ref() { + Some(Strip::EcStrip(ec)) => Some(ec), + _ => None, + }); + let ec = ec.expect("large Iceberg write has an EC strip"); + assert_eq!((ec.data_num, ec.code_num), (8, 4)); + } + let get_started = Instant::now(); + let mut response = client.send(Method::GET, &object, "", b"", false).await; + assert_eq!(response.status(), 200); + let mut received = Md5::new(); + let mut received_size = 0; + while let Some(chunk) = response.chunk().await.unwrap() { + received.update(&chunk); + received_size += chunk.len(); + } + assert_eq!(received_size, size); + assert_eq!(received.finalize().as_slice(), expected.as_slice()); + let get_elapsed = get_started.elapsed(); + for start in [0, usize::min(size - 1, 65500), size - 1] { + let end = usize::min(size - 1, start + 127); + let range = format!("bytes={start}-{end}"); + let response = client + .send_range(Method::GET, &object, "", b"", false, Some(&range)) + .await; + assert_eq!(response.status(), 206); + assert_eq!(response.bytes().await.unwrap().as_ref(), &bytes[start..=end]); + } + println!("iceberg ordinary size={size} PUT={put_elapsed:?} GET={get_elapsed:?}"); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn small_routing_is_strict_at_the_strip_threshold() { + let (_stack, _process, client, table) = setup().await; + let threshold = SmallWriteConfig::default().threshold_exclusive(); + let completed = async || { + let metrics: serde_json::Value = Client::new() + .get(format!("http://{}/_crowdb/metrics", client.address)) + .bearer_auth("m".repeat(32)) + .send() + .await + .unwrap() + .json() + .await + .unwrap(); + metrics["chunk_small_write"]["completed"].as_u64().unwrap() + }; + let mut previous = completed().await; + for size in [threshold - 1, threshold, threshold + 1] { + let object = path(table, &format!("data/threshold-{size}.parquet")); + let bytes = vec![u8::try_from(size % 251).unwrap(); size]; + let response = client.send(Method::PUT, &object, "", &bytes, true).await; + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + let current = completed().await; + assert_eq!(current - previous, u64::from(size < threshold), "size={size}"); + previous = current; + let last = format!("bytes={}-{}", size - 2, size - 1); + let response = client + .send_range(Method::GET, &object, "", b"", false, Some(&last)) + .await; + assert_eq!(response.status(), 206); + assert_eq!(response.bytes().await.unwrap().as_ref(), &bytes[size - 2..]); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn multipart_100_mib_survives_restart_and_complete_replay() { + const PART_BYTES: usize = 5 * 1024 * 1024; + const PART_COUNT: usize = 20; + let (stack, process, mut client, table) = setup_with_bounds_and_file_limit( + ClearBounds { + delegated_access_ms: 900_000, + ..ClearBounds::default() + }, + 128 * 1024 * 1024, + 128 * 1024 * 1024, + ) + .await; + let object = path(table, "data/large-multipart.parquet"); + let created = client.send(Method::POST, &object, "uploads=", b"", false).await; + assert_eq!(created.status(), 200, "{}", created.text().await.unwrap()); + let created = created.text().await.unwrap(); + let upload = created + .split_once("") + .unwrap() + .1 + .split_once("") + .unwrap() + .0; + let mut manifest = String::from( + "", + ); + let mut composite = Md5::new(); + for first_number in (1..=PART_COUNT).step_by(2) { + let second_number = first_number + 1; + let first_bytes = vec![u8::try_from(first_number).unwrap(); PART_BYTES]; + let second_bytes = vec![u8::try_from(second_number).unwrap(); PART_BYTES]; + let first_query = format!("partNumber={first_number}&uploadId={upload}"); + let second_query = format!("partNumber={second_number}&uploadId={upload}"); + let (first_response, second_response) = tokio::join!( + client.send(Method::PUT, &object, &first_query, &first_bytes, true), + client.send(Method::PUT, &object, &second_query, &second_bytes, true), + ); + for (number, bytes, response) in [ + (first_number, first_bytes, first_response), + (second_number, second_bytes, second_response), + ] { + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + let etag = response.headers()["etag"].to_str().unwrap(); + composite.update(Md5::digest(&bytes)); + write!( + manifest, + "{etag}{number}" + ) + .unwrap(); + } + } + manifest.push_str(""); + let mut expected_etag = String::new(); + for byte in composite.finalize() { + write!(expected_etag, "{byte:02x}").unwrap(); + } + expected_etag.push_str("-20"); + let query = format!("uploadId={upload}"); + let completed = client + .send(Method::POST, &object, &query, manifest.as_bytes(), false) + .await; + assert_eq!(completed.status(), 200); + let completed = completed.text().await.unwrap(); + assert!( + completed.contains(""), + "{completed}" + ); + let repository = FileRepository::new(stack.store().await); + let record = repository + .load( + client.credentials.grant().context, + &table.file("data/large-multipart.parquet").unwrap(), + ) + .await + .unwrap() + .unwrap(); + assert_eq!(record.length, (PART_BYTES * PART_COUNT) as u64); + assert_eq!(record.content.etag(), Some(expected_etag.as_str())); + assert_eq!( + record.content.locations(record.length).unwrap().unwrap().len(), + PART_COUNT + ); + + drop(process); + let restarted = process::TestIcebergProcess::start(&stack.cluster.mgmt_endpoints).await; + client.address = restarted.address; + let replay = client + .send(Method::POST, &object, &query, manifest.as_bytes(), false) + .await; + assert_eq!(replay.status(), 200, "{}", replay.text().await.unwrap()); + let response = client.send(Method::GET, &object, "", b"", false).await; + assert_eq!(response.status(), 200, "{}", response.text().await.unwrap()); + let bytes = response.bytes().await.unwrap(); + assert_eq!(bytes.len(), PART_BYTES * PART_COUNT); + for (index, part) in bytes.chunks_exact(PART_BYTES).enumerate() { + assert!(part.iter().all(|byte| *byte == u8::try_from(index + 1).unwrap())); + } +} + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] #[ignore = "requires Maven and the pinned Apache Iceberg Java dependencies"] async fn official_java_s3_fileio_uploads_and_reads_native_files() { diff --git a/app/crowdb-access-server/tests/iceberg_file_response_test.rs b/app/crowdb-access-server/tests/iceberg_file_response_test.rs index d50c9fe03..59d36f71e 100644 --- a/app/crowdb-access-server/tests/iceberg_file_response_test.rs +++ b/app/crowdb-access-server/tests/iceberg_file_response_test.rs @@ -60,11 +60,12 @@ fn part(session: &MultipartSession, number: u16, modified_ms: u64) -> MultipartP file: FileId::random(), ..session.owner }, - tree: FileTree { + tree: Some(FileTree { root: None, length: 0, digest: Sha256::digest([]).into(), - }, + }), + stream: None, } } @@ -161,7 +162,7 @@ fn complete_uses_only_a_published_matching_record() { let candidate = FileTree { root: match &record.content { FileContent::Chunks { root } => root.clone(), - FileContent::Inline { .. } => None, + FileContent::Inline { .. } | FileContent::Locations { .. } => None, }, length: record.length, digest: record.digest, diff --git a/app/crowdb-access-server/tests/iceberg_file_selection_test.rs b/app/crowdb-access-server/tests/iceberg_file_selection_test.rs index bb0eea7c2..c21181761 100644 --- a/app/crowdb-access-server/tests/iceberg_file_selection_test.rs +++ b/app/crowdb-access-server/tests/iceberg_file_selection_test.rs @@ -1,26 +1,26 @@ use crowdb_access_server::iceberg::CompleteSelection; #[test] -fn complete_xml_accepts_only_ordered_sha256_parts() { +fn complete_xml_accepts_ordered_part_etags() { let first = "01".repeat(32); let second = "ab".repeat(32); let xml = format!("1\"{first}\"10000\"{second}\""); let selection = CompleteSelection::parse(xml.as_bytes()).unwrap(); assert_eq!(selection.parts().len(), 2); assert_eq!(selection.parts()[0].number, 1); - assert_eq!(selection.parts()[0].digest, [1; 32]); + assert_eq!(selection.parts()[0].etag, first); assert_eq!(selection.parts()[1].number, 10_000); - assert_eq!(selection.parts()[1].digest, [0xab; 32]); + assert_eq!(selection.parts()[1].etag, second); let sdk_xml = format!("\"{first}\"1"); assert_eq!( - CompleteSelection::parse(sdk_xml.as_bytes()).unwrap().parts()[0].digest, - [1; 32] + CompleteSelection::parse(sdk_xml.as_bytes()).unwrap().parts()[0].etag, + first ); for quote in [""", """, """] { let escaped = sdk_xml.replace(&format!("\"{first}\""), &format!("{quote}{first}{quote}")); assert_eq!( - CompleteSelection::parse(escaped.as_bytes()).unwrap().parts()[0].digest, - [1; 32] + CompleteSelection::parse(escaped.as_bytes()).unwrap().parts()[0].etag, + first ); } } diff --git a/app/crowdb-access-server/tests/iceberg_file_storage_test.rs b/app/crowdb-access-server/tests/iceberg_file_storage_test.rs index 02893266f..32069fc44 100644 --- a/app/crowdb-access-server/tests/iceberg_file_storage_test.rs +++ b/app/crowdb-access-server/tests/iceberg_file_storage_test.rs @@ -23,6 +23,7 @@ async fn chunks(stack: &TestIcebergStack) -> ChunkIoClient { diskio_connections_per_endpoint: 2, diskio_rpc_workers: 1, small_write: SmallWritePolicy { + object_limit: 1024 * 1024, min_pipelines: 1, max_pipelines: 1, memory_budget: 8 * 1024 * 1024, diff --git a/app/crowdb-access-server/tests/iceberg_gc_capacity_test.rs b/app/crowdb-access-server/tests/iceberg_gc_capacity_test.rs index 3a169b60f..aa836cdef 100644 --- a/app/crowdb-access-server/tests/iceberg_gc_capacity_test.rs +++ b/app/crowdb-access-server/tests/iceberg_gc_capacity_test.rs @@ -60,6 +60,7 @@ async fn chunks(stack: &TestIcebergStack) -> ChunkIoClient { diskio_connections_per_endpoint: 1, diskio_rpc_workers: 1, small_write: SmallWritePolicy { + object_limit: 1024 * 1024, min_pipelines: 1, max_pipelines: 1, memory_budget: 8 * 1024 * 1024, diff --git a/app/crowdb-access-server/tests/iceberg_gc_control_test.rs b/app/crowdb-access-server/tests/iceberg_gc_control_test.rs index d9c6b6732..e3c0204b1 100644 --- a/app/crowdb-access-server/tests/iceberg_gc_control_test.rs +++ b/app/crowdb-access-server/tests/iceberg_gc_control_test.rs @@ -10,7 +10,7 @@ use crowdb_access_iceberg::{ CatalogContext, CatalogRepository, CatalogStore, ClearBounds, ManagementPrivilege, RoutedCatalogStore, }, file::{file_key, ContentFormat, FileContent, FileKind, FileRecord, TableLocation}, - gc::{GcLimits, GcRepository, GcStalledReason, ReaderPins}, + gc::{GcLimits, GcRepository, GcStalledReason}, key::{FileId, NamespaceId, OperationId, TableId}, operation::{mutation_identity, ManagementAction, ManagementRequest, RequestIdentity}, record::StorageRecord, @@ -110,33 +110,6 @@ async fn seed_head(store: &RoutedCatalogStore, context: CatalogContext) -> Table table } -async fn check_operator_pin( - stack: &common::TestIcebergStack, - store: Arc, - context: CatalogContext, - table: TableId, -) { - let pin_id = OperationId::random().to_string(); - let table_id = table.to_string(); - let catalog_id = context.catalog.to_string(); - response(command(stack, 'm', &["pin", &pin_id, &table_id])); - let pin_identity = pin_id.parse().unwrap(); - let pins = ReaderPins::new(store); - assert!(pins - .get(context.catalog, table, pin_identity) - .await - .unwrap() - .unwrap() - .protects(common::now_ms())); - response(command(stack, 'm', &["unpin", &catalog_id, &table_id, &pin_id])); - assert!(!pins - .get(context.catalog, table, pin_identity) - .await - .unwrap() - .unwrap() - .protects(common::now_ms())); -} - async fn tombstone_head(store: &RoutedCatalogStore, context: CatalogContext, table: TableId) { let key = head_key(context.catalog, table); let previous = store.get(&key.encode().unwrap()).await.unwrap().unwrap(); @@ -218,7 +191,6 @@ async fn authenticated_gc_controls_survive_separate_processes() { assert!(!live.status.success()); assert!(String::from_utf8_lossy(&live.stderr).contains("live-table GC is disabled")); - check_operator_pin(&stack, store.clone(), context, table).await; tombstone_head(store.as_ref(), context, table).await; let created = response(command(&stack, 'm', &["start-table", &identity, &table_id])); assert_eq!(created["phase"], "Discover"); diff --git a/app/crowdb-access-server/tests/iceberg_table_credentials_test.rs b/app/crowdb-access-server/tests/iceberg_table_credentials_test.rs index b0dfd6690..21613ac96 100644 --- a/app/crowdb-access-server/tests/iceberg_table_credentials_test.rs +++ b/app/crowdb-access-server/tests/iceberg_table_credentials_test.rs @@ -140,7 +140,6 @@ async fn same_name_drafts_refresh_only_the_exact_original_writer_scope() { .parse() .unwrap(); assert_eq!(grant.grant().table, table.table); - assert_credential_pin(&fixture, grant.grant()).await; assert!(grant.grant().operations.allows(FileOperation::Put)); let wrong = path.replace("/events/", "/other/"); assert_eq!( @@ -176,19 +175,6 @@ async fn same_name_drafts_refresh_only_the_exact_original_writer_scope() { fixture.finish().await; } -async fn assert_credential_pin(fixture: &TestTableHttp, grant: &crowdb_access_iceberg::file::FileGrant) { - use crowdb_access_iceberg::{key::IcebergKey, record::StorageRecord}; - let repository = CatalogRepository::new(fixture.store.clone(), ClearBounds::default()).unwrap(); - let (_, authority) = repository.status().await.unwrap(); - let expiry = - grant.expires_ms + authority.admission_bounds.request_ms + authority.admission_bounds.clock_skew_ms; - assert!(fixture.store.values.load().iter().any(|(key, value)| { - matches!(IcebergKey::decode(key).and_then(|key| StorageRecord::decode(&key, &value.bytes)), - Ok(StorageRecord::GcPin(pin)) if pin.head.table == grant.table && pin.expires_ms == expiry - && pin.protects_uploads && !pin.released) - })); -} - async fn expire_first(fixture: &TestTableHttp, first: &Value, second: &Value) { let first_table: TableLocation = format!("{}/", first["metadata"]["location"].as_str().unwrap()) .parse() diff --git a/app/crowdb-access-server/tests/s3_e2e/basic.py b/app/crowdb-access-server/tests/s3_e2e/basic.py index 050502cb2..0b22c586c 100644 --- a/app/crowdb-access-server/tests/s3_e2e/basic.py +++ b/app/crowdb-access-server/tests/s3_e2e/basic.py @@ -240,7 +240,7 @@ def test_slow_signed_upload_releases_native_buffers(self): bucket = f"{self.bucket}-slow" path = f"/{bucket}/slow.bin" second_path = f"/{bucket}/slow-second.bin" - payload = bytes(range(256)) * (4096 + 1) + payload = bytes(range(256)) * (32768 + 1) status, _, _ = self.signed_http("PUT", f"/{bucket}") self.assertEqual(status, 200) barrier = Barrier(2) @@ -259,12 +259,12 @@ def upload(target): status, _, metrics = self.signed_http("GET", "/_crowdb/metrics") self.assertEqual(status, 200) self.assertIn(b"crowdb_s3_native_retained_bytes 0\n", metrics) - backpressure = next( + direct_bytes = next( int(line.split()[-1]) for line in metrics.splitlines() - if line.startswith(b"crowdb_s3_native_backpressure_events_total ") + if line.startswith(b"crowdb_s3_native_direct_bytes_total ") ) - self.assertGreater(backpressure, 0) + self.assertGreater(direct_bytes, 0) self.client.delete_object(Bucket=bucket, Key="slow.bin") self.client.delete_object(Bucket=bucket, Key="slow-second.bin") self.client.delete_bucket(Bucket=bucket) @@ -478,6 +478,43 @@ def test_fragmentation_and_storage_boundaries(self): client.delete_object(Bucket=bucket, Key=key) client.delete_bucket(Bucket=bucket) + def test_ordinary_put_size_matrix(self): + client = self.client + bucket = f"{self.bucket}-sizes" + client.create_bucket(Bucket=bucket) + sizes = [10 * 1024, 1024 * 1024, 12 * 1024 * 1024, 100 * 1024 * 1024] + for size in sizes: + key = f"size-{size}.bin" + payload = bytes(range(256)) * (size // 256) + digest = md5(payload) + started = time.monotonic() + result = client.put_object( + Bucket=bucket, + Key=key, + Body=BytesIO(payload), + ContentLength=size, + ContentMD5=b64encode(digest.digest()).decode("ascii"), + ) + self.assertEqual(result["ETag"], f'"{digest.hexdigest()}"') + put_s = time.monotonic() - started + self.assertEqual(client.head_object(Bucket=bucket, Key=key)["ContentLength"], size) + get_started = time.monotonic() + fetched = client.get_object(Bucket=bucket, Key=key)["Body"] + downloaded = md5() + while block := fetched.read(1024 * 1024): + downloaded.update(block) + self.assertEqual(downloaded.digest(), digest.digest()) + get_s = time.monotonic() - get_started + for start in [0, min(size - 1, 65500), size - 1]: + end = min(size - 1, start + 127) + ranged = client.get_object( + Bucket=bucket, Key=key, Range=f"bytes={start}-{end}" + )["Body"].read() + self.assertEqual(ranged, payload[start : end + 1]) + print(f"S3 ordinary size={size} PUT={put_s:.3f}s GET={get_s:.3f}s") + client.delete_object(Bucket=bucket, Key=key) + client.delete_bucket(Bucket=bucket) + if __name__ == "__main__": unittest.main() diff --git a/app/crowdb-access-server/tests/s3_full_stack_test.rs b/app/crowdb-access-server/tests/s3_full_stack_test.rs index b1091722c..77fdbf564 100644 --- a/app/crowdb-access-server/tests/s3_full_stack_test.rs +++ b/app/crowdb-access-server/tests/s3_full_stack_test.rs @@ -32,7 +32,7 @@ use hyper::body::Bytes; use serde_json::json; const MASTER_KEY: &str = "1111111111111111111111111111111111111111111111111111111111111111"; -const TEST_COUNT: usize = 17; +const TEST_COUNT: usize = 18; const BOTO3_CASES: &[&str] = &[ "test_signed_raw_http_wire_contract", "test_independent_frontends_share_one_namespace", @@ -42,6 +42,7 @@ const BOTO3_CASES: &[&str] = &[ "test_slow_response_reader_keeps_full_object_consistent", "test_basic_bucket_object_matrix", "test_fragmentation_and_storage_boundaries", + "test_ordinary_put_size_matrix", ]; struct AccessServerProcess { @@ -454,10 +455,19 @@ fn assert_native_write_metrics(listen: &str) { assert!(native_body_bytes > 0); assert!(metric_value(&exported, "crowdb_s3_large_write_framed_owners_total") > 0); assert!(metric_value(&exported, "crowdb_s3_large_write_framed_views_total") > 0); + assert!(metric_value(&exported, "crowdb_s3_small_write_completed_total") > 0); assert_eq!( metric_value(&exported, "crowdb_s3_large_write_payload_copy_operations_total"), 0 ); + assert!(metric_value(&exported, "crowdb_s3_chunk_read_location_normalizations_total") > 0); + assert!(metric_value(&exported, "crowdb_s3_chunk_read_range_locations_examined_total") > 0); + assert!(metric_value(&exported, "crowdb_s3_chunk_read_stream_windows_total") > 0); + assert!(metric_value(&exported, "crowdb_s3_chunk_read_layout_queries_total") > 0); + assert!(metric_value(&exported, "crowdb_s3_chunk_read_layout_query_wait_ns_total") > 0); + assert!(metric_value(&exported, "crowdb_s3_chunk_read_strip_read_wait_ns_total") > 0); + assert!(metric_value(&exported, "crowdb_s3_chunk_range_read_wait_ns_total") > 0); + assert!(metric_value(&exported, "crowdb_s3_chunk_read_frame_decode_wait_ns_total") > 0); } fn issue_credentials(access_binary: &Path, seeds: &str) -> (String, String) { @@ -525,7 +535,6 @@ fn start_access_server( .env("CROWDB_S3_TENANT", "boto3-e2e") .env("CROWDB_S3_MASTER_KEY", MASTER_KEY) .env("CROWDB_S3_REGION", "us-east-1") - .env("CROWDB_S3_SMALL_OBJECT_LIMIT", "0") .env("CROWDB_S3_EC_DATA", "2") .env("CROWDB_S3_EC_CODE", "1") .env("CROWDB_S3_NATIVE_BUDGET_BYTES", (1024 * 1024).to_string()) @@ -563,6 +572,9 @@ fn run_boto3_case(method: &str, context: &Boto3CaseContext<'_>) { context.access_server.log_content(), context.chunk_kv.log_content(), ); + if method == "test_ordinary_put_size_matrix" { + print!("{}", String::from_utf8_lossy(&python.stdout)); + } } fn run_restart_phase(phase: &str, listen: &str, access_key: &str, secret_key: &str) { @@ -669,14 +681,14 @@ async fn run_direct_chunk_benchmark(seeds: &[String], artifacts_dir: &Path) { .await .expect("direct chunk read"); let get_ns = started.elapsed().as_nanos(); - assert_eq!(read, payload); + assert_eq!(read.concat(), payload); let started = Instant::now(); let range = chunks .read_range(&result.locations, 0, 4096) .await .expect("direct chunk range read"); let range_ns = started.elapsed().as_nanos(); - assert_eq!(range, payload.slice(..4096)); + assert_eq!(range.concat(), payload.slice(..4096)); json!({ "size": size, "concurrency": concurrency, diff --git a/app/crowdb-cli/src/commands/bench/io.rs b/app/crowdb-cli/src/commands/bench/io.rs index e176a1916..57b20f1fc 100644 --- a/app/crowdb-cli/src/commands/bench/io.rs +++ b/app/crowdb-cli/src/commands/bench/io.rs @@ -294,13 +294,19 @@ fn print_read(workload: ReadBenchmarkWorkload, args: &ChunkioReadArgs, result: & ReadBenchmarkWorkload::Mixed => "read-mix", }; println!( - "chunkio {name}: requested={} dataset_objects={} prepare_s={:.3} reads={} small_reads={} large_reads={} errors={} incomplete={} stop={} reads_s={:.2} logical_mib_s={:.1} logical_bytes={} avg_us={} p50_us={} p99_us={} dram_read_mib_s={} dram_write_mib_s={} dram_total_mib_s={}", + "chunkio {name}: requested={} dataset_objects={} prepare_s={:.3} reads={} small_reads={} large_reads={} errors={} incomplete={} stop={} reads_s={:.2} logical_mib_s={:.1} logical_bytes={} avg_us={} p50_us={} p99_us={} dram_read_mib_s={} dram_write_mib_s={} dram_total_mib_s={} location_normalizations={} locations_examined={} range_locations_examined={} stream_windows={} layout_queries={} layout_query_wait_ns={} strip_read_wait_ns={} chunk_read_wait_ns={} frame_decode_wait_ns={} frame_parse_wait_ns={}", result.requested_reads, args.dataset_objects, result.preparation_secs, result.reads, result.small_reads, result.large_reads, result.errors, result.incomplete_reads, result.stop_reason, result.reads_per_sec, result.logical_mib_per_sec, result.logical_bytes, result.latency_avg_us, result.latency_p50_us, result.latency_p99_us, dram(result.dram_read_mib_s), dram(result.dram_write_mib_s), dram(result.dram_total_mib_s), + result.read_flow.location_normalizations, result.read_flow.locations_examined, + result.read_flow.range_locations_examined, result.read_flow.stream_windows, + result.read_flow.layout_queries, + result.read_flow.layout_query_wait_ns, result.read_flow.strip_read_wait_ns, + result.read_flow.chunk_read_wait_ns, result.read_flow.frame_decode_wait_ns, + result.read_flow.frame_parse_wait_ns, ); print_errors(name, &result.error_messages); } diff --git a/container/single-node-container/Dockerfile b/container/single-node-container/Dockerfile index 682b73374..db18caefb 100644 --- a/container/single-node-container/Dockerfile +++ b/container/single-node-container/Dockerfile @@ -29,7 +29,6 @@ LABEL org.opencontainers.image.title="CROWDB Iceberg" \ org.opencontainers.image.description="Apache Iceberg REST catalog with native storage, in one container for development and testing." \ org.opencontainers.image.url="https://crowdb.dev/" \ org.opencontainers.image.source="https://github.com/buzzcrow/crowdb" \ - org.opencontainers.image.documentation="https://github.com/buzzcrow/crowdb/blob/v${PREVIEW_VERSION}/doc/user-manual/docker-single-node-user-guide.md" \ org.opencontainers.image.authors="Gian " \ org.opencontainers.image.licenses="Apache-2.0" \ org.opencontainers.image.revision="$SOURCE_REVISION" \ diff --git a/container/single-node-container/README.md b/container/single-node-container/README.md index facd8f9d2..e8b5b3dbd 100644 --- a/container/single-node-container/README.md +++ b/container/single-node-container/README.md @@ -3,8 +3,8 @@ # Single-node container development -For running CROWDB, see the [Docker user guide](../../doc/user-manual/docker-single-node-user-guide.md). -This page describes building from source on a Linux amd64 development or CI host. +This page describes building and running CROWDB from source on a Linux amd64 +development or CI host. ```sh pixi run build-single-node-container diff --git a/container/single-node-container/profile.toml b/container/single-node-container/profile.toml index 92ebcda5e..b3bf07dd4 100644 --- a/container/single-node-container/profile.toml +++ b/container/single-node-container/profile.toml @@ -173,10 +173,11 @@ backoff_max_ms = 5000 [[services]] id = "s3" program = "/opt/crowdb/bin/crowdb-access-server" -args = [] -env = { CROWDB_S3_LISTEN = "0.0.0.0:81", CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000", CROWDB_S3_TENANT = "preview", CROWDB_S3_REGION = "us-east-1", CROWDB_S3_EC_DATA = "2", CROWDB_S3_EC_CODE = "1" } +args = ["--config", "/opt/crowdb/run/config/s3.toml"] +env = { CROWDB_S3_PUBLIC_URI = "http://localhost:81", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000" } dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] fence_listeners = ["127.0.0.1:81"] +config_template = "/opt/crowdb/etc/templates/access.toml" [services.probe] kind = "http" target = "http://127.0.0.1:81/_crowdb/health/ready" @@ -190,10 +191,11 @@ backoff_max_ms = 5000 [[services]] id = "iceberg" program = "/opt/crowdb/bin/crowdb-iceberg" -args = ["serve"] -env = { CROWDB_ICEBERG_LISTEN = "0.0.0.0:80", CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_MANAGEMENT_SEEDS = "http://127.0.0.1:10000", CROWDB_ICEBERG_GC_ENABLED = "0" } +args = ["serve", "--config", "/opt/crowdb/run/config/iceberg.toml"] +env = { CROWDB_ICEBERG_PUBLIC_URI = "http://localhost", CROWDB_ICEBERG_GC_ENABLED = "0" } dependencies = ["kv", "chunk-kv", "chunkdb", "diskio"] fence_listeners = ["127.0.0.1:80"] +config_template = "/opt/crowdb/etc/templates/access.toml" [services.probe] kind = "http" target = "http://127.0.0.1:80/v1/config" diff --git a/container/single-node-container/templates/access.toml b/container/single-node-container/templates/access.toml new file mode 100644 index 000000000..4eb6885f9 --- /dev/null +++ b/container/single-node-container/templates/access.toml @@ -0,0 +1,61 @@ +# Access configuration shared by S3 and Iceberg processes. +# Secrets stay in /opt/crowdb/data/secrets/server.env. + +[common] +management_seeds = ["http://127.0.0.1:10000"] +diskio_connections_per_endpoint = 2 +diskio_rpc_workers = 2 + +[read] +stream_window_bytes = 1048576 +stream_slots = 3 +global_stream_bytes = 268435456 +recovery_memory_bytes = 268435456 + +[small_write] +threshold_ratio = 0.9 +disk_block_bytes = 1048576 +conversion_enabled = true +ec_data = 2 +ec_code = 1 +memory_budget_bytes = 1342177280 +queue_capacity = 1024 +min_pipelines = 1 +max_pipelines = 32 +max_batch_bytes = 1048576 + +[s3] +listen = "0.0.0.0:81" +tenant = "preview" +region = "us-east-1" +small_object_limit = 8388608 +list_scan_items = 1001 +list_scan_bytes = 4194304 +continuation_ttl_seconds = 900 +native_budget_bytes = 268435456 +cleanup_backlog_limit = 10000 +ec_data = 2 +ec_code = 1 +max_chunk_size = 1073741824 + +[iceberg] +listen = "0.0.0.0:80" +native_budget_bytes = 268435456 + +[iceberg.gc] +enabled = false +interval_ms = 1000 +step_bytes = 8388608 +step_ms = 1000 +page_items = 64 +page_bytes = 49152 +concurrency = 1 +kv_bytes = 67108864 +kv_requests = 128 +chunk_bytes = 8388608 +chunk_requests = 128 +retry_base_ms = 1000 +retry_max_ms = 60000 +corruption_attempts = 3 +minimum_retention_ms = 604800000 +catalogs = [] diff --git a/doc/assets/demo-cluster.mp4 b/doc/assets/demo-cluster.mp4 deleted file mode 100644 index 86ef09f76..000000000 Binary files a/doc/assets/demo-cluster.mp4 and /dev/null differ diff --git a/doc/assets/demo-failover.mp4 b/doc/assets/demo-failover.mp4 deleted file mode 100644 index 38bb32ed8..000000000 Binary files a/doc/assets/demo-failover.mp4 and /dev/null differ diff --git a/doc/assets/demo-kv.mp4 b/doc/assets/demo-kv.mp4 deleted file mode 100644 index ff3c1db16..000000000 Binary files a/doc/assets/demo-kv.mp4 and /dev/null differ diff --git a/doc/backlog/R167-s3-multipart-upload.md b/doc/backlog/R167-s3-multipart-upload.md index 0ba860157..c7927bd96 100644 --- a/doc/backlog/R167-s3-multipart-upload.md +++ b/doc/backlog/R167-s3-multipart-upload.md @@ -22,17 +22,20 @@ The scope boundary is ## Solution 1. Add create, upload-part, list-parts, complete, abort, and required upload - listing operations as a separate S3-owned state machine. + listing operations through S3-owned API and namespace adapters over the + protocol-neutral multipart session, part, and completion core from R190. 2. Store immutable part identities and integrity records durably; a retried part number replaces only that part's selected generation and schedules old private data for cleanup. 3. Complete with one fenced metadata transaction that validates ordered part identities, sizes, checksums, and expected upload state before publishing - one immutable object generation. No concatenation through access-server - memory is allowed. -4. Define multipart-specific ETag/checksum behavior without changing the basic - single-part generation rules. Abort and expiry create bounded, - idempotent cleanup records. + one immutable object generation. Compose the selected parts' chunk-location + arrays with adjusted logical offsets. Complete does not read part data or + concatenate it through access-server memory. +4. Persist each uploaded part's raw 16-byte MD5. The multipart ETag is the + lowercase hexadecimal MD5 of the selected parts' raw MD5 bytes in order, + followed by `-`. Keep the basic single-part ETag rule unchanged. + Abort and expiry create bounded, idempotent cleanup records. 5. Preserve the basic admission bounds for parallel part traffic and wire compatibility for all retry and conflict outcomes. @@ -41,6 +44,8 @@ The scope boundary is - Depends on R152–R166. - Reuses the basic milestone's publication/recovery, streaming input, logical deletion, and integrity contracts. +- Reuses R190's protocol-neutral multipart core; S3 retains its own + authorization, namespace, wire errors, ETag response, and object publication. - R170 owns any accelerated multipart transfer and additionally depends on this requirement before enabling that operation. @@ -48,8 +53,12 @@ The scope boundary is - Given parts uploaded out of order with part retries, when completion names a valid order, assert exact concatenated bytes become visible through one - generation without gateway concatenation. Invariant: completion is atomic - and storage-backed. E2E test. + generation without gateway concatenation or part reads. Invariant: completion + is metadata-only, atomic and storage-backed. E2E test. +- Given completed parts with known MD5 values, when completion selects and + reorders them, assert the ETag uses only the selected raw part MD5 values in + completion order and the part count suffix. Invariant: multipart ETag matches + the S3-compatible composite algorithm. Unit test. - Given missing, duplicated, undersized, checksum-mismatched, or concurrently replaced parts, when completion runs, assert no object publishes and exact errors are stable. Invariant: only the validated part set can publish. diff --git a/doc/backlog/R188-console-group0-authority.md b/doc/backlog/R188-console-group0-authority.md index 6b6650239..ee422080d 100644 --- a/doc/backlog/R188-console-group0-authority.md +++ b/doc/backlog/R188-console-group0-authority.md @@ -68,12 +68,10 @@ not block R187 completion. 7. Audit the S3 mini-cluster's local `console.toml` and restart path under the same authority boundary. Retain only launch inputs and bootstrap seeds locally after Group 0 cutover; do not replay a local topology copy. -8. Migrate the verified bare-metal deployment and operations material from the - old `doc/user-manual/user-guide.md` into - `doc/user-manual/bare-metal-user-guide.md`, organized by KV cluster, chunk - layer, and data access servers. State that bare-metal is not yet - production-ready. Remove the old combined guide only after its supported - material and links are migrated; Docker documentation remains independent. +8. Migrate the verified bare-metal deployment and operations material into + dedicated bare-metal deployment documentation, organized by KV cluster, + chunk layer, and data access servers. State that bare-metal is not yet + production-ready. Keep Docker deployment documentation independent. 9. Complete container crash diagnostics without changing host-wide collector policy. Respect file-based core patterns, Ubuntu Apport, systemd-coredump and Docker Desktop's Linux VM; document where dumps actually go or why collection diff --git a/doc/backlog/R189-access-iceberg-container-ecosystem.md b/doc/backlog/R189-access-iceberg-container-ecosystem.md index 2cb721da9..53e52530c 100644 --- a/doc/backlog/R189-access-iceberg-container-ecosystem.md +++ b/doc/backlog/R189-access-iceberg-container-ecosystem.md @@ -63,8 +63,8 @@ is a normal S3 object. separately; Kafka infrastructure is not required for the Python/SQL/engine acceptance above. Exercise a simple BI query through a tested SQL engine; do not claim direct BI-tool or catalog compatibility without its own fixture. -6. Add only passing, reproducible recipes to - `doc/user-manual/docker-single-node-user-guide.md`. Maintain a client +6. Add only passing, reproducible recipes to the container deployment + documentation. Maintain a client capability matrix with tested versions, read/write scope, known exclusions, and links to executable fixtures. Label the image and all examples as development/test, not production data storage or upgrade-stable service. diff --git a/doc/backlog/R190-access-iceberg-shared-streaming-io.md b/doc/backlog/R190-access-iceberg-shared-streaming-io.md deleted file mode 100644 index 10b976b79..000000000 --- a/doc/backlog/R190-access-iceberg-shared-streaming-io.md +++ /dev/null @@ -1,113 +0,0 @@ - - - -### R190: access — Shared S3 and Iceberg streaming data path - -Status: Ready after R187 completion, at the user's request. Begin with a -complete read/write/delete/GC flow review before implementation. - -## Problem - -Iceberg FileIO treats each roughly 64 KiB leaf as a separately durable small -object. Each leaf registers physical ownership through six catalog reads and -one conditional write, then waits for the Chunk readable cursor. Receiving the -next leaf waits for that entire chain. Reads fetch individual leaves and copy -returned bytes. S3 already receives into 1 MiB owners, frames at 64 KiB, and uses -whole-object Chunk writers and lazy read streams. - -A measured 5 MiB release-build upload on the native null-DiskIO stack took -2.68–2.79 seconds for ordinary PUT and 2.09–2.18 seconds for UploadPart, excluding -multipart completion. These are API measurements, not NVMe throughput. - -Root designs: [Iceberg](../design/access-server/iceberge/design-crowdb-iceberg.md), -[S3](../design/access-server/s3/design-crowdb-access-s3.md), and -[Chunk I/O](../design/chunkio/design-crowdb-chunkio.md). - -## Solution - -The user-selected data path is shared streaming infrastructure with independent -S3 and Iceberg metadata semantics: - - HTTP receive owner (1 MiB) -> Chunk writer (64 KiB frames) -> DiskIO - durable complete locations -> one atomic file/part publication point - published locations -> Chunk read stream -> owner-backed HTTP response - -1. Share deferred HTTP receive-provider installation and bounded native owner - allocation. Authenticate and admit before reading bodies. Preserve signed - AWS-chunked decoding, checksums and unknown-length bounded streaming. -2. Write a whole file or multipart part through the Chunk writer selected by - object size. Do not register catalog intents or await a durable cursor per - frame. A frame is transport/integrity granularity, not a catalog transaction. -3. Publish complete immutable file or part metadata only after data completion - and validation. Keep fencing, conflicting-path rejection and ambiguous-result - resolution. Readers cannot observe partial data. -4. Stream GET and Range through the same Chunk read machinery as S3, retaining - owner-backed buffers and bounded backpressure. Keep Iceberg credentials, - generation checks, format validation, full-file integrity and GC protection. -5. Make multipart completion consume complete part references without restoring - the per-leaf write/commit path. Preserve ordering, replay, format validation - and atomic final-file visibility. -6. Retain crash-safe allocation ownership and reclamation below the per-frame - catalog path. Use durable Chunk allocation/lifecycle ownership rather than - deleting protection and assuming S3 already implements all orphan GC. - Drain submitted writes before reclaim; never free published or pinned data. -7. Review delete and GC end to end alongside reads and writes: logical - invisibility, reader pins, owner discovery, grace periods, cancelled writes, - shared ranges, compaction and physical reuse must form one coherent model. -8. Keep existing stored file descriptors readable, or implement an explicit - migration within this work; do not silently invalidate persisted volumes. - -## Dependencies - -- Existing native receive owners, prepared Chunk writers and Chunk read streams. -- Existing Iceberg file, multipart and GC contracts remain acceptance obligations. -- R168/R169/R147 contain deferred shared-storage reclamation work. Do not claim - those are implemented or weaken Iceberg recovery to bypass them; implement - any ownership support required for this path within this requirement. -- R188 remains a separate console-authority follow-up. - -## Acceptance - -- Given ordinary PUT bodies of 10 KiB, 1 MiB, 12 MiB and 100 MiB, upload - through S3 and Iceberg -> both use bounded - 1 MiB owners and 64 KiB frames; no catalog operation is issued per frame. - **Bounded shared ingress. Integration test.** -- Given signed chunks, corrupted signatures/checksums, short bodies and cancelled - requests, upload -> reject without publishing metadata; release owner credits - and drain in-flight writes. **No partial visibility. Integration test.** -- Given completed bytes, publish with a conflicting path or a lost reply -> keep - one complete authoritative outcome without overwriting different content. - **Atomic publication. Integration test.** -- Given full, cross-frame and cross-chunk ranges, read -> exact bytes, bounded - retained buffers, integrity checks and cancellation propagation. - **Shared bounded reads. Integration test.** -- Given a 100 MiB multipart upload (twenty 5 MiB parts), complete/replay/restart - -> one correct immutable file, - no per-leaf rewrite/commit loop and no premature part reclamation. - **Multipart correctness. E2E test.** -- Given process loss before/after data completion and metadata publication, - recover -> unpublished allocations remain discoverable and eventually reclaim; - published/pinned data remain readable. **Crash-safe ownership. E2E test.** -- Given published, pinned and unpublished data, delete and run GC across restart - -> logical deletion precedes physical reclamation; retained reads remain valid; - ownership is discoverable and freed ranges are not reused before writes drain. - **Delete/GC consistency. E2E test.** -- Given existing file descriptors, restart and read -> preserve bytes and ranges. - **Persisted-data readability. Integration test.** -- Given the same 5 MiB fixtures and build/storage profile, measure PUT, UploadPart - and GET -> record elapsed time and dependency-operation counts against the - baseline without raising deadlines or weakening assertions. - **Measured operation reduction. E2E test.** - -Commands: - -```sh -pixi run test-access-iceberg -pixi run test-access-s3 -pixi run test-access-server -pixi run -e s3-e2e test-boto3-e2e -pixi run -e iceberg-e2e test-iceberg-native -pixi run -e iceberg-e2e test-iceberg-sdk -pixi run rs-fmt-check -pixi run rs-lint -``` diff --git a/doc/backlog/R92-chunkdb-in-chunk-gc.md b/doc/backlog/R92-chunkdb-in-chunk-gc.md index 194f0e299..933ff05b3 100644 --- a/doc/backlog/R92-chunkdb-in-chunk-gc.md +++ b/doc/backlog/R92-chunkdb-in-chunk-gc.md @@ -10,6 +10,19 @@ to avoid global merge overhead. **Solution**: Implement in-chunk GC operations (ReclaimStrip, CollapseStrip, MergeStrips) for shared chunks. Add logical-to-physical offset mapping -to support GC while keeping chunk IDs stable. +to support GC while keeping chunk IDs stable. Add a ChunkDB orphan scanner for +chunks and shared ranges allocated by access uploads that crash or fail before +their complete file/object descriptor is published. R190 intentionally does not +write per-chunk catalog intents or upload-owner records on its write hot path. +The scanner must compare candidates with authoritative published S3 and Iceberg +references and reader protection before reclaiming, and must not infer orphan +status merely from age or a missing intermediate upload record. Account for +in-flight writers and delayed publication so physical ranges are never reused +while a writer or reader can still own them. Report candidate and reclaimed +bytes separately. Include Iceberg MPU Complete's frozen selection payload as +an authoritative reference while completion is in progress: it stores the +selected parts' exact chunk locations, which remain live even if a concurrent +UploadPart replaces the same part number before publication. After publication, +the immutable file descriptor is the authoritative reference. -**Scope**: Placeholder - detailed design to be refined before implementation. \ No newline at end of file +**Scope**: Placeholder - detailed design to be refined before implementation. diff --git a/doc/backlog/backlog.md b/doc/backlog/backlog.md index 6b28d7658..4ff88de49 100644 --- a/doc/backlog/backlog.md +++ b/doc/backlog/backlog.md @@ -13,13 +13,6 @@ complexity, and dependency. Before implementation, follow the **Next R number: R191** — Bump this line in the same commit when adding a new item. -### Planned — Shared access streaming - -- **[R190](R190-access-iceberg-shared-streaming-io.md)** — align Iceberg PUT, - multipart and GET/Range with S3's bounded native receive and Chunk data path; - remove per-frame catalog transactions while preserving publication and recovery. - Ready for the full read, write, delete and GC review before implementation. - ### Next Milestone — Chunk-backed range KV R144 is a deferred merge follow-up to the completed chunk-KV server and routed diff --git a/doc/design/access-server/iceberge/design-crowdb-iceberg.md b/doc/design/access-server/iceberge/design-crowdb-iceberg.md index 9095bbe0a..425980573 100644 --- a/doc/design/access-server/iceberge/design-crowdb-iceberg.md +++ b/doc/design/access-server/iceberge/design-crowdb-iceberg.md @@ -239,7 +239,7 @@ namespace-write capability and no catalog management or clear privilege; reader, manager and clearer do not inherit namespace-write rights. All four can read the configuration endpoint and namespaces. Only writer may invoke namespace or table mutations. Management commands are separate from the Iceberg REST listener. Operational -configuration is in the [user guide](../../../user-manual/user-guide.md#9-iceberg-catalog-foundation). +configuration follows the selected deployment profile and its startup inputs. Iceberg FileIO uses reserved S3-shaped locations so existing Iceberg clients can address immutable metadata and data files. The shape is a compatibility @@ -252,22 +252,27 @@ bytes. Dot traversal, leading slash, backslash, controls, query and fragment delimiters are rejected rather than normalized. HTTP percent decoding belongs only at the transport boundary, not in stored S3-shaped locations. -Native file records bind FileId to exact location, kind, format, canonical length -and SHA-256 digest. Eligible metadata stores at most 16 KiB inline; bounded LZ4 +Native file records bind FileId to exact location, kind, format and canonical length. +Legacy tree and inline records also bind a whole-file SHA-256 digest; streamed +records bind a bounded array of complete Chunk locations and an HTTP ETag, and +rely on verified 64-KiB storage frames rather than a whole-file digest. Eligible +metadata stores at most 16 KiB inline; bounded LZ4 compression considers at most 64 KiB original input, and decoding verifies the -canonical length and digest. Other file kinds retain a fixed-size chunk root, -never a growing location vector. Hints are non-authoritative and out-of-bounds +canonical length and digest. Location vectors are validated for exact logical +coverage and bounded by the record limit. Hints are non-authoritative and out-of-bounds hints are ignored. The publication primitive stages an immutable authority before the exact-location CAS; equal-content retries return the selected FileId, while conflicts retain losing candidates without overwriting or physical deletion. An SDK upload supplies a path and bytes, not the eventual Iceberg data/delete -use. Sealing validates physical container bytes and records ambiguous Avro, -Parquet, ORC and Puffin uses as unbound. Selected metadata and manifests must -validate declared uses against these canonical records before table publication. +use. FileIO treats client-supplied bytes as opaque. It verifies the declared +transport checksum and durable frame writes before publishing the descriptor; +format interpretation belongs to the client or to a CROWDB component that +constructs those bytes. Selected metadata and manifests validate their declared +uses during table publication. The isolated native HTTP surface exposes signed immutable object reads/writes and multipart operations, but no general S3 bucket authority or file DELETE. -Chunk-backed files use bounded leaf blocks and immutable chunk-resident directory +Legacy chunk-backed files use bounded leaf blocks and immutable chunk-resident directory pages, with at most 256 children per page and eight directory levels. Each page binds its catalog, table and file identity, child heights and covered byte count. The writer retains only one partial leaf and bounded per-level frontiers. Native @@ -289,7 +294,7 @@ coverage. SHA-256 compression uses RustCrypto; versioned digest checkpoints reta only chaining state, byte length and a partial block. They are trusted-storage recovery records, not client authentication assertions. Failed checkpoint writes poison the current writer without invalidating earlier durable checkpoints. -Native block writes persist an exact physical-range ownership intent in the +Legacy native block writes persist an exact physical-range ownership intent in the catalog before DiskIO. A shared-writer callback receives the assigned location; uncertain catalog writes are read back before the physical batch proceeds. This ledger also covers process loss before file publication and checkpoints @@ -318,19 +323,29 @@ immutable before/after session references; policy-bound sequence receipts make create and terminal release recoverable without double accounting. Released receipts remain in terminal sessions. These logical credits are not physical disk reclamation or accounting for retained orphan bytes. -The native multipart repository reserves one part mutation in the session before +The legacy tree multipart repository reserves one part mutation in the session before changing its part authority. A bounded before/after snapshot and monotonically increasing revisions make the write and fence release recoverable across servers. Counts and current staged bytes are reserved once at the session CAS. Abort cannot bypass an unresolved mutation; stale helpers cannot restore an older part. Abort retains parts and completion evidence rather than deleting physical storage. -Completion freezes an ordered part-number/revision/digest selection in immutable +Legacy completion freezes an ordered part-number/revision/digest selection in immutable payload pages, then changes the session phase by CAS to fence part replacement. Selections are independently bounded to 10,000 entries and 420,007 encoded bytes. Each completion step verifies that bounded selection and one selected part before copying a bounded byte window and publishing its checkpoint by session CAS. Lost replies reload progress without appending selected bytes twice. Assembled bytes remain unexposed until semantic sealing and immutable location publication. +Streamed UploadPart writes one part key by CAS without changing the session for +each part. Create caps part count by the reserved staged-byte ceiling divided +by the per-part byte ceiling. Complete freezes an ordered selection with each +part's exact location bytes, length and MD5 ETag in bounded immutable payload +pages. A later replacement cannot change that selection. Completion composes +logical offsets across the selected locations and derives the multipart ETag +from the ordered raw part MD5 values; it does not read part data or assemble a +new chunk. Session phase CAS freezes publication against later selections. The +final file descriptor becomes visible through the immutable location publication +protocol. Foreground and recovery drivers use the same native-block-aligned byte window below the one-MiB assembly ceiling. Equal windows prevent systematic CAS losses to a smaller competing recovery step; alignment avoids checkpoint-only tiny leaves. @@ -369,12 +384,15 @@ number. Current-session checks bracket each scan; concurrent mutations invalidat the page rather than mixing pending counters with old part records. Expired or terminal sessions and malformed storage pages are not reported as successful lists. -Native HTTP upload staging holds an independent concurrency -credit, slices each received frame into bounded writes and awaits storage before pulling more -input. Declared/actual byte limits, exact content length and optional signed SHA-256 -are checked before returning a tree. Failed or cancelled uploads retain orphan -blocks without publishing file authority. This transport adapter does not infer -semantic file kind, authorize grants or accept unchecked checksum trailers. +Native HTTP upload staging holds an independent concurrency credit and bounded +1-MiB receive owners. A body below 1 MiB enters its write pipeline in one push; +larger bodies push each filled owner while the next owner receives. The Chunk +writer frames data at 64 KiB and chooses shared chunks only below the configured +fraction of one strip's data capacity. Unknown-length bodies choose a dedicated +chunk. PUT and UploadPart require a verified request checksum or signed payload; +SHA-256 is calculated only when declared. Durable completion and exact length +precede metadata publication. Failed or cancelled uploads leave unpublished +allocations for orphan scanning without exposing partial file authority. Metadata JSON structural validation uses a bounded pull-reader bridge and an ignored-value parser rather than retaining the metadata graph. A separate scanner @@ -542,7 +560,7 @@ Physical reclamation follows a proof that no live metadata, snapshot, reference, lease, or retained operation can reach the file. General S3 deletion and lifecycle rules cannot reclaim Iceberg-owned data. -The reclamation proof binds current and pinned historical metadata to their +The reclamation proof binds current and retained historical metadata to their captured heads. Its immutable traversal stack and compressed binary file-ID index use content-addressed payload pages. A task CAS publishes the pending stack and mark root together; a missing page is an error, including during a nonmembership @@ -564,19 +582,19 @@ authorizes deletion. Committed files remain readable when new chunk allocation fails. Progress resumes after capacity is restored through the normal storage flow. Shared-chunk ranges remain pending while range deletion is unsupported. -Metadata readers, direct FileIO, file publication and both published and staged -credentials persist pins before rechecking their authority. Pin expiry includes -the applicable persisted request and clock-skew bounds. Once a file's canonical +Metadata readers, direct FileIO, file publication and credentials recheck their +authority without writing request-level pins. Physical reclamation observes a +minimum retention interval that covers admitted request lifetimes. Once a file's canonical deletion intent has started, ordinary resolution and publication reject it even if physical range reclamation is deferred. Legacy live tasks are retired without further deletion, releasing an owned table fence. Retained and deferred candidates remain durable work for later inactive passes. -An already authorized FileIO GET or HEAD can pin a tombstoned table while its -credential remains valid. The pin is persisted before the exact head is -rechecked; a concurrent transition to `Reclaiming` rejects admission. Uploads -and new table credentials still require a Ready table. Logical drop therefore -does not invalidate retained file reads or bypass the physical deletion fence. +An already authorized FileIO GET or HEAD can continue within the configured +retention interval after a logical deletion. A concurrent transition to +`Reclaiming` rejects new admission. Uploads and new table credentials require +a Ready table. Logical drop preserves admitted file reads until their bounded +response lifetime ends. Retired catalog recovery scans system retry and management ledgers before file deletion and after the final file rescan. Pending or retained bindings stop the @@ -588,6 +606,12 @@ candidates. Assembly checkpoints have separate claims and a durable frontier-roo index. Abandoned frontiers are authenticated before traversal; a conflicted final tree is traversed once instead of revisiting its shared frontier. Published sessions reclaim only the checkpoint block, preserving the assembled data tree. +Streamed file and part candidates instead reclaim their exact Chunk location +ranges after retention. A published selected part remains owned by the immutable +file descriptor; part cleanup does not reclaim it a second time. Allocations +abandoned before a file or part descriptor is published have no per-chunk catalog +intent and remain for ChunkDB's orphan scanner to discover after checking +published descriptors, frozen selections and active writers. Each physical step rechecks the terminal session and retention. The checkpoint block is deleted after its children, and session cleanup requires its completed claim. Block intents are swept after tree candidates, preserving reachable owners diff --git a/doc/design/config/design-crowdb-config.md b/doc/design/config/design-crowdb-config.md index bc910ef2f..58d339210 100644 --- a/doc/design/config/design-crowdb-config.md +++ b/doc/design/config/design-crowdb-config.md @@ -22,10 +22,10 @@ validated, and activated. ## 1. Scope -The `crowdb-kv-server`, `crowdb-diskdb`, `crowdb-chunkdb`, and -`crowdb-diskio` processes accept typed TOML startup configuration. A service -may require a file or make it optional, but a supplied file follows the same -resolution and failure rules in every process. +The `crowdb-kv-server`, `crowdb-diskdb`, `crowdb-chunkdb`, `crowdb-diskio`, +`crowdb-access-server`, and `crowdb-iceberg` processes accept typed TOML startup +configuration. A service may require a file or make it optional, but a supplied +file follows the same resolution and failure rules in every process. This contract covers process configuration. Durable cluster topology, membership, allocation metadata, and application data keep their existing @@ -79,12 +79,29 @@ Common section names describe process roles: The service schemas retain their different domain sections: -| Service | File policy | `server.rpc_workers` default | Principal domain sections | -|---------|-------------|------------------------------|-------------------------------------------------| -| KV | Optional | 2 | Paxos, WAL, engine, metrics | -| diskdb | Required | 2 | Storage, heartbeat, persistence, scanner, sync | -| chunkdb | Required | 2 | Storage, placement, lifecycle, topology, clients| -| diskio | Optional | 4 | Engine, group-0 discovery, metrics, disk entries| +| Service | File policy | `server.rpc_workers` default | Principal domain sections | +| ------- | ----------- | ---------------------------- | ------------------------------------------------ | +| KV | Optional | 2 | Paxos, WAL, engine, metrics | +| diskdb | Required | 2 | Storage, heartbeat, persistence, scanner, sync | +| chunkdb | Required | 2 | Storage, placement, lifecycle, topology, clients | +| diskio | Optional | 4 | Engine, group-0 discovery, metrics, disk entries | +| Access | Optional | N/A | HTTP listeners, shared read, S3, Iceberg GC | + +The S3 and Iceberg access processes accept the same `AccessConfig` schema. +Each process activates only its own protocol section plus `[common]`, `[read]`, +and `[small_write]`. The file is selected with `--config`; a supplied file is +validated before connecting to dependencies or binding listeners. Read slots, +the per-read window, the global retained-byte budget, the separate EC recovery +budget, and small-write memory limits are static startup settings. Credentials +and bearer tokens remain in the environment. For access settings that predate the +file, an explicitly populated TOML value takes precedence over the legacy +environment variable; an omitted TOML value keeps the legacy behavior. + +The small-object routing bound is `threshold_ratio × disk_block_bytes × EC data +shards`; mirrored strips use one data shard. `[small_write]` supplies the default +EC layout for both access writers, and S3's EC override changes its small and +large writers together. The default is 8+4 with 1 MiB data blocks and ratio +0.9. The single-node container profile uses 2+1. All schemas tolerate unknown keys so a newer file can be staged before a binary upgrade. A known key with the wrong type or invalid value rejects the @@ -135,9 +152,8 @@ not produce an empty or cached topology. Hardware/process mutation is disabled, while authenticated logical operations use the existing operation paths. Only the container profile overrides public listeners to Iceberg 80, S3 81 and -Web 8080. Bare-metal defaults remain independent. See the -[single-node container guide](../../user-manual/docker-single-node-user-guide.md) -for publication, volume and endpoint usage. +Web 8080. Bare-metal defaults remain independent. Container publication, +volume and endpoint usage are defined by the container deployment files. ## 7. Failure Handling diff --git a/doc/doc_index.md b/doc/doc_index.md index 1fb23e13c..8c54e702c 100644 --- a/doc/doc_index.md +++ b/doc/doc_index.md @@ -21,8 +21,6 @@ listed document or section needed by the task. | `doc/design/console/design-crowdb-console.md` | Console architecture and service lifecycle. | | `doc/design/config/design-crowdb-config.md` | Configuration ownership, precedence, validation, reload. | | `doc/design/access-server/design-crowdb-access-server.md` | S3, Iceberg, native Dataset access, and GPU delivery. | -| `doc/user-manual/user-guide.md` | Web UI, CLI, REST API, setup, operations, upgrade. | -| `doc/user-manual/docker-single-node-user-guide.md` | Single-node Docker preview, credentials, clients, recovery. | ## Backlog (`doc/backlog/`) diff --git a/doc/user-manual/build_html.py b/doc/user-manual/build_html.py deleted file mode 100644 index 8bc48d3d9..000000000 --- a/doc/user-manual/build_html.py +++ /dev/null @@ -1,585 +0,0 @@ -#!/usr/bin/env python3 -"""Convert user-guide.md to a standalone HTML page with tabbed code sections. - -Usage: - python3 doc/user-manual/build_html.py [source.md] - -Output: - doc/user-manual/user-guide.html - -The script is self-contained (no third-party dependencies). It handles the -markdown subset used in the user guide: headings, fenced code blocks, bold -text, bullet lists, tables, horizontal rules, and inline code. - -Sections labelled "**CLI:**" and "**curl:**" followed by a fenced code block -are automatically wrapped in a tabbed interface so the reader can switch -between CLI and curl examples with a single click. -""" - -from __future__ import annotations - -import html -import re -import sys -from pathlib import Path - -SCRIPT_DIR = Path(__file__).resolve().parent -SOURCE = SCRIPT_DIR / "user-guide.md" - -CSS = """ -:root { - /* ── Hugging Face inspired palette ──────────────────────────────── */ - --bg: #ffffff; - --bg-subtle: #fafafa; - --fg: #111827; - --fg-default: #374151; - --fg-muted: #6b7280; - --sidebar-bg: #fafafa; - --sidebar-border: #e5e7eb; - --code-bg: #f3f4f6; - --code-border: #e5e7eb; - --border: #e5e7eb; - --border-strong: #d1d5db; - --accent: #ffd21e; - --accent-hover: #f59e0b; - --accent-soft: #fff4cc; - --accent-on: #111827; - --link: #155dfc; - --link-hover: #0b4ed6; - --tab-active-bg: #ffffff; - --tab-inactive-bg: #f3f4f6; - --tab-active-fg: #111827; - --tab-inactive-fg: #6b7280; -} -@media (prefers-color-scheme: dark) { - :root { - --bg: #0d1117; - --bg-subtle: #161b22; - --fg: #e6edf3; - --fg-default: #c9d1d9; - --fg-muted: #8b949e; - --sidebar-bg: #161b22; - --sidebar-border: #30363d; - --code-bg: #161b22; - --code-border: #30363d; - --border: #30363d; - --border-strong: #484f58; - --accent: #ffd21e; - --accent-hover: #f59e0b; - --accent-soft: #2a2410; - --accent-on: #0d1117; - --link: #58a6ff; - --link-hover: #79b8ff; - --tab-active-bg: #21262d; - --tab-inactive-bg: #161b22; - --tab-active-fg: #e6edf3; - --tab-inactive-fg: #8b949e; - } -} -* { box-sizing: border-box; margin: 0; padding: 0; } -body { - font-family: "Source Sans Pro", -apple-system, BlinkMacSystemFont, "Segoe UI", "Noto Sans", Roboto, sans-serif; - background: var(--bg-subtle); - color: var(--fg-default); - line-height: 1.65; - font-size: 15px; -} - -/* ── Layout ─────────────────────────────────────────────────────── */ -.layout { - display: flex; - min-height: 100vh; -} -.sidebar { - width: 280px; - flex-shrink: 0; - background: var(--sidebar-bg); - border-right: 1px solid var(--sidebar-border); - position: fixed; - top: 0; - left: 0; - bottom: 0; - overflow-y: auto; - padding: 1.5rem 0; -} -.content { - margin-left: 280px; - flex: 1; - padding: 2rem 3rem; - max-width: calc(100vw - 280px); - background: var(--bg); -} - -/* ── Sidebar ────────────────────────────────────────────────────── */ -.sidebar-brand { - padding: 0 1.5rem 1rem; - font-size: 1.05rem; - font-weight: 700; - color: var(--fg); - border-bottom: 1px solid var(--sidebar-border); - margin-bottom: 0.75rem; - letter-spacing: -0.01em; -} -.sidebar-brand .brand-mark { - display: inline-block; - width: 20px; - height: 20px; - background: var(--accent); - border-radius: 4px; - vertical-align: middle; - margin-right: 0.5rem; -} -.sidebar nav { padding: 0 0.5rem; } -.sidebar nav a { - display: block; - padding: 0.35rem 1rem; - color: var(--fg-muted); - text-decoration: none; - font-size: 0.875rem; - border-radius: 6px; - transition: all 0.12s; - line-height: 1.4; -} -.sidebar nav a:hover { - color: var(--link); - background: var(--code-bg); -} -.sidebar nav a.active { - color: var(--link); - font-weight: 600; - background: var(--code-bg); -} -.sidebar nav a.sub { padding-left: 2rem; font-size: 0.82rem; } - -/* ── Content ────────────────────────────────────────────────────── */ -.content h1 { - font-size: 1.9rem; - font-weight: 700; - color: var(--fg); - margin-bottom: 0.5rem; - padding-bottom: 0.5rem; - border-bottom: 1px solid var(--border); - letter-spacing: -0.02em; -} -.content h2 { - font-size: 1.4rem; - font-weight: 600; - color: var(--fg); - margin-top: 2.5rem; - margin-bottom: 0.75rem; - padding-bottom: 0.3rem; - border-bottom: 1px solid var(--border); - scroll-margin-top: 1rem; - letter-spacing: -0.01em; -} -.content h3 { - font-size: 1.15rem; - font-weight: 600; - color: var(--fg); - margin-top: 1.75rem; - margin-bottom: 0.5rem; - scroll-margin-top: 1rem; -} -.content p { margin: 0.6rem 0; } -.content a { color: var(--link); text-decoration: none; } -.content a:hover { color: var(--link-hover); text-decoration: underline; } -.content ul { padding-left: 1.5rem; margin: 0.6rem 0; } -.content li { margin: 0.3rem 0; } -.content hr { border: none; border-top: 1px solid var(--border); margin: 2rem 0; } - -/* ── Inline code ────────────────────────────────────────────────── */ -code { - font-family: "IBM Plex Mono", "SF Mono", "Fira Code", "Consolas", monospace; - font-size: 0.85em; - background: var(--code-bg); - padding: 0.15em 0.4em; - border-radius: 4px; - border: 1px solid var(--code-border); -} - -/* ── Code blocks ────────────────────────────────────────────────── */ -pre { - background: var(--code-bg); - border: 1px solid var(--code-border); - border-radius: 6px; - padding: 0.85rem 1rem; - overflow-x: auto; - font-size: 0.84rem; - line-height: 1.5; -} -pre code { background: none; border: none; padding: 0; font-size: inherit; } - -/* ── Tables ─────────────────────────────────────────────────────── */ -table { - border-collapse: collapse; - width: 100%; - margin: 1rem 0; - font-size: 0.88rem; -} -th, td { - border: 1px solid var(--border); - padding: 0.5rem 0.75rem; - text-align: left; -} -th { - background: var(--code-bg); - font-weight: 600; - color: var(--fg); -} -tr:hover td { background: var(--code-bg); } - -/* ── Tabbed code ────────────────────────────────────────────────── */ -.tab-group { - margin: 1rem 0; - border-radius: 6px; - overflow: hidden; - border: 1px solid var(--border); -} -.tab-bar { - display: flex; - background: var(--tab-inactive-bg); - border-bottom: 1px solid var(--border); -} -.tab-btn { - padding: 0.45rem 1.1rem; - border: none; - background: transparent; - cursor: pointer; - font-size: 0.8rem; - font-weight: 500; - color: var(--tab-inactive-fg); - border-bottom: 2px solid transparent; - transition: all 0.12s; - font-family: inherit; -} -.tab-btn:hover { color: var(--fg); background: var(--tab-active-bg); } -.tab-btn.active { - color: var(--tab-active-fg); - background: var(--tab-active-bg); - border-bottom-color: var(--accent); - font-weight: 600; -} -.tab-panel { display: none; } -.tab-panel.active { display: block; } -.tab-panel-content { padding: 1rem; } -.tab-panel-content table { margin: 0.75rem 0; } -.tab-panel-content h4 { margin-top: 1.25rem; } -.tab-panel-content h4:first-child { margin-top: 0; } -.tab-panel pre { - border-radius: 0 0 6px 6px; - margin: 0; - border: none; - border-top: none; -} - -/* ── Mobile ─────────────────────────────────────────────────────── */ -@media (max-width: 768px) { - .sidebar { - position: static; - width: 100%; - border-right: none; - border-bottom: 1px solid var(--sidebar-border); - max-height: 300px; - } - .content { - margin-left: 0; - max-width: 100%; - padding: 1.5rem; - } - .layout { flex-direction: column; } -} -""" - -JS = """ -// Tab switching -document.querySelectorAll('.tab-group').forEach(function(group) { - var btns = group.querySelectorAll('.tab-btn'); - btns.forEach(function(btn) { - btn.addEventListener('click', function() { - var idx = btn.getAttribute('data-tab'); - btns.forEach(function(b) { b.classList.remove('active'); }); - group.querySelectorAll('.tab-panel').forEach(function(p) { - p.classList.toggle('active', p.getAttribute('data-panel') === idx); - }); - btn.classList.add('active'); - }); - }); -}); - -// Sidebar active section tracking -var navLinks = document.querySelectorAll('.sidebar nav a'); -var sections = []; -navLinks.forEach(function(link) { - var id = link.getAttribute('href'); - if (id && id.startsWith('#')) { - var el = document.getElementById(id.slice(1)); - if (el) sections.push({ id: id, el: el, link: link }); - } -}); - -function updateActive() { - var scrollY = window.scrollY + 100; - var current = sections[0]; - for (var i = 0; i < sections.length; i++) { - if (sections[i].el.offsetTop <= scrollY) current = sections[i]; - } - navLinks.forEach(function(l) { l.classList.remove('active'); }); - if (current) current.link.classList.add('active'); -} -window.addEventListener('scroll', updateActive, { passive: true }); -updateActive(); -""" - - -def escape(text: str) -> str: - return html.escape(text, quote=False) - - -def inline_format(text: str) -> str: - text = escape(text) - text = re.sub(r"`([^`]+)`", r"\1", text) - text = re.sub(r"\*\*([^*]+)\*\*", r"\1", text) - text = re.sub(r'\[([^\]]+)\]\((https?://[^\s)"<>]+)\)', r'\1', text) - return text - - -def convert_table(lines: list[str]) -> str: - rows = [l.strip().strip("|").split("|") for l in lines if l.strip().startswith("|")] - if len(rows) < 2: - return "".join(f"

{inline_format(l)}

" for l in lines) - header = [c.strip() for c in rows[0]] - body = [[c.strip() for c in r] for r in rows[2:]] if len(rows) > 2 else [] - out = "\n\n" - for h in header: - out += f"" - out += "\n\n\n" - for row in body: - out += "" - for cell in row: - out += f"" - out += "\n" - out += "\n
{inline_format(h)}
{inline_format(cell)}
\n" - return out - - -def slugify(text: str) -> str: - s = re.sub(r"[^\w\s-]", "", text.lower()) - return re.sub(r"[\s_-]+", "-", s).strip("-") - - -def convert_markdown(md: str) -> tuple[str, list[dict]]: - """Convert markdown to HTML body, return (html, toc_entries).""" - lines = md.split("\n") - i = 0 - out: list[str] = [] - toc: list[dict] = [] - in_list = False - - def close_list(): - nonlocal in_list - if in_list: - out.append("\n") - in_list = False - - while i < len(lines): - line = lines[i] - - # Skip HTML comments - if line.strip().startswith("" not in lines[i]: - i += 1 - i += 1 - continue - - # Fenced code block (standalone, not part of a tab group) - if line.strip().startswith("```"): - lang = line.strip().strip("`").strip() - code_lines: list[str] = [] - i += 1 - while i < len(lines) and not lines[i].strip().startswith("```"): - code_lines.append(lines[i]) - i += 1 - i += 1 - close_list() - code = escape("\n".join(code_lines)) - out.append(f'
{code}
\n') - continue - - # Tab pattern: **CLI:** / **curl:** / **Web UI:** followed by code block or other content - if re.match(r"^\*\*(CLI|curl|Web UI):\*\*\s*$", line.strip()): - tabs: list[tuple[str, str, bool]] = [] # (label, content, is_code) - while i < len(lines): - while i < len(lines) and lines[i].strip() == "": - i += 1 - if i >= len(lines): - break - m = re.match(r"^\*\*(CLI|curl|Web UI):\*\*\s*$", lines[i].strip()) - if not m: - break - label = m.group(1) - if label == "CLI": - label = "cli" - i += 1 - while i < len(lines) and lines[i].strip() == "": - i += 1 - if i < len(lines) and lines[i].strip().startswith("```"): - # Code-block tab (existing behaviour) - i += 1 - cl: list[str] = [] - while i < len(lines) and not lines[i].strip().startswith("```"): - cl.append(lines[i]) - i += 1 - i += 1 - tabs.append((label, escape("\n".join(cl)), True)) - else: - # Raw-markdown tab — capture lines until next tab marker, - # a horizontal rule, a new ## heading, or EOF. - raw_lines: list[str] = [] - while i < len(lines): - if re.match(r"^\*\*(CLI|curl|Web UI):\*\*\s*$", lines[i].strip()): - break - if lines[i].strip() == "---": - break - if re.match(r"^##\s+", lines[i]): - break - raw_lines.append(lines[i]) - i += 1 - tabs.append((label, "\n".join(raw_lines), False)) - - if tabs: - close_list() - out.append('
\n
\n') - for idx, (label, _, _) in enumerate(tabs): - active = " active" if idx == 0 else "" - out.append(f'\n') - out.append("
\n") - for idx, (_, content, is_code) in enumerate(tabs): - active = " active" if idx == 0 else "" - if is_code: - out.append(f'
{content}
\n') - else: - raw_html, _ = convert_markdown(content) - out.append(f'
{raw_html}
\n') - out.append("
\n") - continue - - # Headings - m = re.match(r"^(#{1,4})\s+(.*)", line) - if m: - close_list() - level = len(m.group(1)) - text_raw = m.group(2) - text = inline_format(text_raw) - slug = slugify(text_raw) - if level <= 2: - toc.append({"level": level, "text": text_raw, "slug": slug}) - out.append(f'{text}\n') - i += 1 - continue - - # Horizontal rule - if line.strip() == "---": - close_list() - out.append("
\n") - i += 1 - continue - - # Table - if line.strip().startswith("|"): - close_list() - table_lines: list[str] = [] - while i < len(lines) and lines[i].strip().startswith("|"): - table_lines.append(lines[i]) - i += 1 - out.append(convert_table(table_lines)) - continue - - # Bullet list - if re.match(r"^\s*[-*]\s+", line): - if not in_list: - out.append("
    \n") - in_list = True - text = inline_format(re.sub(r"^\s*[-*]\s+", "", line)) - out.append(f"
  • {text}
  • \n") - i += 1 - continue - - # Numbered list - if re.match(r"^\s*\d+\.\s+", line): - close_list() - text = inline_format(re.sub(r"^\s*\d+\.\s+", "", line)) - out.append(f"
  • {text}
  • \n") - i += 1 - continue - - # Blank line - if line.strip() == "": - close_list() - i += 1 - continue - - # Regular paragraph - close_list() - text = inline_format(line) - out.append(f"

    {text}

    \n") - i += 1 - - close_list() - return "".join(out), toc - - -def build_sidebar(toc: list[dict]) -> str: - lines = ['', "") - return "\n".join(lines) - - -def build(source: Path = SOURCE) -> None: - if not source.exists(): - print(f"error: {source} not found", file=sys.stderr) - sys.exit(1) - - md = source.read_text(encoding="utf-8") - output = source.with_suffix(".html") - title_match = re.search(r"^# (.+)$", md, re.MULTILINE) - title = escape(title_match.group(1) if title_match else "CROWDB User Guide") - body, toc = convert_markdown(md) - sidebar = build_sidebar(toc) - - doc = f""" - - - - -{title} - - - -
    - -
    -{body} -
    -
    - - - -""" - - output.write_text(doc, encoding="utf-8") - print(f"wrote {output} ({len(doc)} bytes)") - - -if __name__ == "__main__": - build(Path(sys.argv[1]) if len(sys.argv) > 1 else SOURCE) diff --git a/doc/user-manual/docker-single-node-user-guide.html b/doc/user-manual/docker-single-node-user-guide.html deleted file mode 100644 index 82888597d..000000000 --- a/doc/user-manual/docker-single-node-user-guide.html +++ /dev/null @@ -1,486 +0,0 @@ - - - - - -CROWDB Single-Node Container Guide - - - -
    - -
    -

    CROWDB Single-Node Container Guide

    -

    Run an Iceberg REST catalog and its storage on one Linux amd64 host.

    -

    For evaluation only; no production or upgrade guarantee.

    -

    Quick start

    -
    docker run -d --name crowdb-iceberg \
    -  -p 127.0.0.1:80:80 \
    -  crowdb/crowdb-iceberg:v0.1.0-dev
    -

    The image repository is crowdb/crowdb-iceberg; examples use the version tag

    -

    v0.1.0-dev. Docker pulls the image if it is not present locally. Docker creates

    -

    an anonymous volume for the data.

    -

    Check startup, then retrieve your client credentials:

    -
    docker inspect --format '{{.State.Health.Status}}' crowdb-iceberg
    -docker exec crowdb-iceberg crowdb-monitor credentials show --format env
    -

    Wait for healthy. Connect your Iceberg client to http://localhost using

    -

    ICEBERG_TOKEN from the credential output. Port 80 serves both the REST catalog

    -

    and Iceberg FileIO; no separate S3 port is needed for Iceberg.

    -

    Stop and start the same container without losing its data:

    -
    docker stop --time 120 crowdb-iceberg
    -docker start crowdb-iceberg
    -

    The quick start exposes only Iceberg to the host. The GUI is not ready for use

    -

    and is not published. For data you want to reuse after deleting and recreating

    -

    the container, use a named volume as shown below.

    -

    Common options

    -
      -
    • -d: run in the background.
    • -
    • --name crowdb-iceberg: give the container a convenient name for later commands.
    • -
    • -p 127.0.0.1:80:80: publish Iceberg on host loopback. Fixed host ports require
    • -
    -

    explicit mapping; the image cannot publish them automatically.

    -
      -
    • -v crowdb-data:/opt/crowdb/data: keep data in a named volume. Docker creates
    • -
    -

    it if needed; reuse it with the same image when recreating the container.

    -
      -
    • --restart unless-stopped: automatically restart after failure or Docker
    • -
    -

    daemon restart, unless you explicitly stopped the container.

    -
      -
    • --stop-timeout 120: allow 120 seconds for shutdown before Docker sends SIGKILL.
    • -
    • --log-driver json-file --log-opt max-size=30m --log-opt max-file=5:
    • -
    -

    rotate Docker's captured logs separately from CROWDB's internal logs.

    -

    Example: persistent data and automatic restart

    -

    Use this instead of the quick-start command:

    -
    docker run -d --name crowdb-iceberg \
    -  -p 127.0.0.1:80:80 \
    -  -v crowdb-data:/opt/crowdb/data \
    -  --restart unless-stopped --stop-timeout 120 \
    -  --log-driver json-file --log-opt max-size=30m --log-opt max-file=5 \
    -  crowdb/crowdb-iceberg:v0.1.0-dev
    -

    Do not attach the same data volume to two running containers. A bind mount can

    -

    replace the named volume, but its directory must be writable by UID/GID 10001.

    -

    Connect clients

    -

    The credential command prints ICEBERG_URI and ICEBERG_TOKEN, plus credentials

    -

    for the optional independent S3 service. Keep this output private. Credentials

    -

    remain the same when the data volume is reused.

    -

    Iceberg with PyIceberg

    -

    Set ICEBERG_URI and ICEBERG_TOKEN in your client environment using the values

    -

    from the credential command. With PyIceberg installed:

    -
    import os
    -from pyiceberg.catalog import load_catalog
    -
    -catalog = load_catalog(
    -    "crowdb",
    -    type="rest",
    -    uri=os.environ["ICEBERG_URI"],
    -    token=os.environ["ICEBERG_TOKEN"],
    -)
    -catalog.create_namespace_if_not_exists("demo")
    -print(catalog.list_namespaces())
    -

    Optional independent S3 access

    -

    Add -p 127.0.0.1:81:81 when creating the container if you also need the general

    -

    S3 API. Connect to http://localhost:81 with path-style addressing, region

    -

    us-east-1, and the printed AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY.

    -

    Uploading an S3 object does not register an Iceberg table.

    -

    The examples provide local HTTP access. Changing host ports or using a remote

    -

    client also requires reachable Iceberg FileIO URLs; changing only the client's

    -

    catalog URI is insufficient. The container's ports are independent of bare-metal

    -

    service defaults.

    -

    Troubleshooting

    -
    docker logs --tail 100 crowdb-iceberg
    -docker exec crowdb-iceberg crowdb-monitor liveness
    -docker exec crowdb-iceberg crowdb-monitor readiness
    -
      -
    • starting: initialization or recovery is still running; inspect the logs.
    • -
    • unhealthy: a required service is unavailable. Liveness checks the monitor;
    • -
    -

    readiness checks the complete deployment. Successful probes exit zero.

    -
      -
    • exited: inspect logs before restarting. Repeated child failures exhaust the
    • -
    -

    restart budget. Docker's restart policy handles container exits; an unhealthy

    -

    health check alone does not restart the container.

    -
      -
    • Port already allocated: another process uses host port 80. Free the port or
    • -
    -

    configure a different host mapping and reachable client/FileIO endpoints.

    -
      -
    • Existing volume rejected: use the original image and credentials. Do not edit
    • -
    -

    bootstrap manifests or secrets to bypass compatibility checks.

    -

    CROWDB logs are under /opt/crowdb/data/log. Monitor events are in

    -

    monitor/monitor.log; child logs have service-specific directories. Log rotation

    -

    uses a 30 MiB target and five retained files per channel across child restarts.

    -

    Docker's own log retention is configured separately, as in the extended example.

    -

    Data, backup and limitations

    -
      -
    • The data volume contains storage, metadata, credentials and logs. Anonymous
    • -
    -

    volumes are not automatically reused by a newly created container; record the

    -

    volume identity before deleting the original container.

    -
      -
    • Stop the container before copying the entire volume. Preserve ownership,
    • -
    -

    private permissions and sparse files. Restore with the exact image version;

    -

    copying only disk images is insufficient.

    -
      -
    • All services and four sparse 16 GiB disk images share one host. This provides
    • -
    -

    no host fault tolerance. Monitor actual filesystem space; configured disk

    -

    capacity is not a usable-capacity guarantee.

    -
      -
    • Deleted Iceberg content can continue occupying space. Physical reclamation
    • -
    -

    is disabled in this profile. Cross-version volume migration is not promised.

    -
      -
    • PyIceberg namespace/table operations and boto3 object operations have container
    • -
    -

    acceptance coverage. Spark, Flink, Trino, dataframe workflows and selected ORC

    -

    data are not certified by those checks.

    -

    Advanced crash diagnostics

    -

    Core dumps can contain credentials and user data. Linux host kernel.core_pattern

    -

    controls their destination; the image does not change it. A Docker core ulimit

    -

    alone does not guarantee a dump.

    -
      -
    • File patterns use the process's filesystem namespace; relative paths use its
    • -
    -

    working directory. The destination must be writable.

    -
      -
    • Ubuntu Apport can reject container crashes when its container forwarding
    • -
    -

    support is absent. This image does not include an Apport agent; do not assume

    -

    a report will appear under the host's /var/crash.

    -
      -
    • systemd-coredump uses the host journal and usually /var/lib/systemd/coredump;
    • -
    -

    inspect with coredumpctl on the host.

    -
      -
    • Docker Desktop uses its Linux VM's collector policy.
    • -
    -

    The image retains function symbols but currently has no DWARF source-line

    -

    information. Do not assume that the data volume contains usable core dumps.

    -

    See the Linux core manual

    -

    for host collection rules.

    - -
    -
    - - - diff --git a/doc/user-manual/docker-single-node-user-guide.md b/doc/user-manual/docker-single-node-user-guide.md deleted file mode 100644 index e49c09e7f..000000000 --- a/doc/user-manual/docker-single-node-user-guide.md +++ /dev/null @@ -1,169 +0,0 @@ - - - -# CROWDB Single-Node Container Guide - -Run an Iceberg REST catalog and its storage on one Linux amd64 host. -For evaluation only; no production or upgrade guarantee. - -## Quick start - -```sh -docker run -d --name crowdb-iceberg \ - -p 127.0.0.1:80:80 \ - crowdb/crowdb-iceberg:v0.1.0-dev -``` - -The image repository is `crowdb/crowdb-iceberg`; examples use the version tag -`v0.1.0-dev`. Docker pulls the image if it is not present locally. Docker creates -an anonymous volume for the data. - -Check startup, then retrieve your client credentials: - -```sh -docker inspect --format '{{.State.Health.Status}}' crowdb-iceberg -docker exec crowdb-iceberg crowdb-monitor credentials show --format env -``` - -Wait for `healthy`. Connect your Iceberg client to `http://localhost` using -`ICEBERG_TOKEN` from the credential output. Port 80 serves both the REST catalog -and Iceberg FileIO; no separate S3 port is needed for Iceberg. - -Stop and start the same container without losing its data: - -```sh -docker stop --time 120 crowdb-iceberg -docker start crowdb-iceberg -``` - -The quick start exposes only Iceberg to the host. The GUI is not ready for use -and is not published. For data you want to reuse after deleting and recreating -the container, use a named volume as shown below. - -## Common options - -- `-d`: run in the background. -- `--name crowdb-iceberg`: give the container a convenient name for later commands. -- `-p 127.0.0.1:80:80`: publish Iceberg on host loopback. Fixed host ports require - explicit mapping; the image cannot publish them automatically. -- `-v crowdb-data:/opt/crowdb/data`: keep data in a named volume. Docker creates - it if needed; reuse it with the same image when recreating the container. -- `--restart unless-stopped`: automatically restart after failure or Docker - daemon restart, unless you explicitly stopped the container. -- `--stop-timeout 120`: allow 120 seconds for shutdown before Docker sends SIGKILL. -- `--log-driver json-file --log-opt max-size=30m --log-opt max-file=5`: - rotate Docker's captured logs separately from CROWDB's internal logs. - -### Example: persistent data and automatic restart - -Use this instead of the quick-start command: - -```sh -docker run -d --name crowdb-iceberg \ - -p 127.0.0.1:80:80 \ - -v crowdb-data:/opt/crowdb/data \ - --restart unless-stopped --stop-timeout 120 \ - --log-driver json-file --log-opt max-size=30m --log-opt max-file=5 \ - crowdb/crowdb-iceberg:v0.1.0-dev -``` - -Do not attach the same data volume to two running containers. A bind mount can -replace the named volume, but its directory must be writable by UID/GID 10001. - -## Connect clients - -The credential command prints `ICEBERG_URI` and `ICEBERG_TOKEN`, plus credentials -for the optional independent S3 service. Keep this output private. Credentials -remain the same when the data volume is reused. - -### Iceberg with PyIceberg - -Set `ICEBERG_URI` and `ICEBERG_TOKEN` in your client environment using the values -from the credential command. With PyIceberg installed: - -```python -import os -from pyiceberg.catalog import load_catalog - -catalog = load_catalog( - "crowdb", - type="rest", - uri=os.environ["ICEBERG_URI"], - token=os.environ["ICEBERG_TOKEN"], -) -catalog.create_namespace_if_not_exists("demo") -print(catalog.list_namespaces()) -``` - -### Optional independent S3 access - -Add `-p 127.0.0.1:81:81` when creating the container if you also need the general -S3 API. Connect to `http://localhost:81` with path-style addressing, region -`us-east-1`, and the printed `AWS_ACCESS_KEY_ID` / `AWS_SECRET_ACCESS_KEY`. -Uploading an S3 object does not register an Iceberg table. - -The examples provide local HTTP access. Changing host ports or using a remote -client also requires reachable Iceberg FileIO URLs; changing only the client's -catalog URI is insufficient. The container's ports are independent of bare-metal -service defaults. - -## Troubleshooting - -```sh -docker logs --tail 100 crowdb-iceberg -docker exec crowdb-iceberg crowdb-monitor liveness -docker exec crowdb-iceberg crowdb-monitor readiness -``` - -- `starting`: initialization or recovery is still running; inspect the logs. -- `unhealthy`: a required service is unavailable. Liveness checks the monitor; - readiness checks the complete deployment. Successful probes exit zero. -- `exited`: inspect logs before restarting. Repeated child failures exhaust the - restart budget. Docker's restart policy handles container exits; an unhealthy - health check alone does not restart the container. -- Port already allocated: another process uses host port 80. Free the port or - configure a different host mapping and reachable client/FileIO endpoints. -- Existing volume rejected: use the original image and credentials. Do not edit - bootstrap manifests or secrets to bypass compatibility checks. - -CROWDB logs are under `/opt/crowdb/data/log`. Monitor events are in -`monitor/monitor.log`; child logs have service-specific directories. Log rotation -uses a 30 MiB target and five retained files per channel across child restarts. -Docker's own log retention is configured separately, as in the extended example. - -## Data, backup and limitations - -- The data volume contains storage, metadata, credentials and logs. Anonymous - volumes are not automatically reused by a newly created container; record the - volume identity before deleting the original container. -- Stop the container before copying the entire volume. Preserve ownership, - private permissions and sparse files. Restore with the exact image version; - copying only disk images is insufficient. -- All services and four sparse 16 GiB disk images share one host. This provides - no host fault tolerance. Monitor actual filesystem space; configured disk - capacity is not a usable-capacity guarantee. -- Deleted Iceberg content can continue occupying space. Physical reclamation - is disabled in this profile. Cross-version volume migration is not promised. -- PyIceberg namespace/table operations and boto3 object operations have container - acceptance coverage. Spark, Flink, Trino, dataframe workflows and selected ORC - data are not certified by those checks. - -### Advanced crash diagnostics - -Core dumps can contain credentials and user data. Linux host `kernel.core_pattern` -controls their destination; the image does not change it. A Docker core ulimit -alone does not guarantee a dump. - -- File patterns use the process's filesystem namespace; relative paths use its - working directory. The destination must be writable. -- Ubuntu Apport can reject container crashes when its container forwarding - support is absent. This image does not include an Apport agent; do not assume - a report will appear under the host's `/var/crash`. -- systemd-coredump uses the host journal and usually `/var/lib/systemd/coredump`; - inspect with `coredumpctl` on the host. -- Docker Desktop uses its Linux VM's collector policy. - -The image retains function symbols but currently has no DWARF source-line -information. Do not assume that the data volume contains usable core dumps. -See the [Linux core manual](https://www.man7.org/linux/man-pages/man5/core.5.html) -for host collection rules. diff --git a/doc/user-manual/user-guide.html b/doc/user-manual/user-guide.html deleted file mode 100644 index 8e018cae5..000000000 --- a/doc/user-manual/user-guide.html +++ /dev/null @@ -1,1124 +0,0 @@ - - - - - -CrowdbKV User Guide - - - -
    - -
    -

    CROWDB User Guide

    -

    CROWDB is a distributed storage platform with an S3-compatible object data

    -

    plane and a multi-group Multi-Paxos key-value foundation. This guide starts

    -

    with the shortest usable path: create a persistent local S3 cluster and use

    -

    bucket and object operations. The later sections cover the underlying KV

    -

    cluster, physical topology, individual servers, upgrades, and recovery.

    -

    CROWDB provides three user-facing interfaces:

    -
      -
    • Web UI — the crowdb-web service provides a visual dashboard
    • -
    -

    with cluster topology, group health, a KV Operator panel (store/group

    -

    selector, paginated scan, inline CRUD, demo data injection), and

    -

    Swagger UI for browsing the OpenAPI spec of any registered

    -

    crowdb-kv-server instance.

    -
      -
    • CLI — crowdb-cli s3 owns the local S3 cluster lifecycle and sends
    • -
    -

    bucket/object requests directly to the cluster recorded by --root.

    -

    Lower-level management commands discover services through group 0 and call

    -

    them directly. Output is a human-readable console transcript.

    -
      -
    • HTTP APIs — the local access server exposes the S3 HTTP API. The
    • -
    -

    console service exposes the lower-level cluster management API documented

    -

    in §8.

    -

    S3 examples use the loopback access endpoint. The first local cluster normally

    -

    receives 127.0.0.1:16000; when that port is already assigned, use the

    -

    endpoint printed by s3 cluster start or `s3 cluster status.

    -
    S3_ENDPOINT=http://127.0.0.1:16000
    -

    Prerequisites

    -

    Before following the steps below:

    -
      -
    • Run pixi run build from the repository root.
    • -
    • Add target/release to PATH, or invoke the binaries by their full paths.
    • -
    • Choose a dedicated cluster directory. The directory is the persistent
    • -
    -

    cluster identity and contains configuration, logs, and data files.

    -
    -

    1. Quick Start: Persistent Local S3

    -

    1.1 Start or restart the cluster

    -

    Choose a directory and start the cluster:

    -
    S3_ROOT="$PWD/.crowdb-runtime/persistent/s3-local"
    -crowdb-cli s3 cluster start --root "$S3_ROOT"
    -

    An absent or empty directory creates a new file-backed cluster. A recognized

    -

    cluster directory restarts the same cluster with its existing data and port

    -

    assignments. A non-empty directory that is not a CROWDB cluster is rejected

    -

    without modification.

    -

    The command prints the S3 endpoint, Web management URL, running service count,

    -

    and cluster root. The default local endpoints are:

    -
    S3_ENDPOINT=http://127.0.0.1:16000
    -WEB_URL=http://127.0.0.1:14000
    -

    If the command prints different endpoints because a default port is already

    -

    assigned, use the printed values. Open WEB_URL for the topology, service, and

    -

    cluster management console.

    -

    Inspect the recorded processes without changing them:

    -
    crowdb-cli s3 cluster status --root "$S3_ROOT"
    -

    The CLI owns S3 mini-cluster lifecycle. The bundled Web service loads that

    -

    mini-cluster's console registry and presents its running services.

    -

    1.2 Bucket operations

    -

    Create a bucket:

    -
    crowdb-cli s3 bucket put --root "$S3_ROOT" photos
    -curl -X PUT "$S3_ENDPOINT/photos"
    -

    List buckets:

    -
    crowdb-cli s3 bucket list --root "$S3_ROOT"
    -curl "$S3_ENDPOINT/"
    -

    GET one bucket and show its XML result:

    -
    crowdb-cli s3 bucket get --root "$S3_ROOT" photos
    -curl "$S3_ENDPOINT/photos"
    -

    Remove an empty bucket:

    -
    crowdb-cli s3 bucket delete --root "$S3_ROOT" photos
    -curl -X DELETE "$S3_ENDPOINT/photos"
    -

    Removing a non-empty bucket returns the S3 error and leaves its objects intact.

    -

    1.3 Object CRUD

    -

    Create the bucket used by the following examples, then upload an object from a

    -

    file, literal text, generated random bytes, or standard input:

    -
    crowdb-cli s3 bucket put --root "$S3_ROOT" documents
    -
    -crowdb-cli s3 object put --root "$S3_ROOT" \
    -  documents reports/hello.txt --file ./hello.txt
    -
    -crowdb-cli s3 object put --root "$S3_ROOT" \
    -  documents reports/text.txt --text 'object content'
    -
    -crowdb-cli s3 object put --root "$S3_ROOT" \
    -  documents reports/random.bin --random-size 1048576
    -
    -printf 'hello from CROWDB\n' | crowdb-cli s3 object put \
    -  --root "$S3_ROOT" documents reports/stdin.txt
    -
    -curl -X PUT --data-binary @hello.txt \
    -  "$S3_ENDPOINT/documents/reports/hello.txt"
    -

    put creates a new object or replaces the bytes of an existing key.

    -

    Read an object to standard output or a file:

    -
    crowdb-cli s3 object get --root "$S3_ROOT" \
    -  documents reports/hello.txt
    -
    -crowdb-cli s3 object get --root "$S3_ROOT" \
    -  documents reports/hello.txt --output ./downloaded.txt
    -
    -curl "$S3_ENDPOINT/documents/reports/hello.txt" \
    -  --output ./downloaded-with-curl.txt
    -

    Check that an object exists:

    -
    crowdb-cli s3 object head --root "$S3_ROOT" \
    -  documents reports/hello.txt
    -
    -curl -I "$S3_ENDPOINT/documents/reports/hello.txt"
    -

    Delete an object:

    -
    crowdb-cli s3 object delete --root "$S3_ROOT" \
    -  documents reports/hello.txt
    -
    -curl -X DELETE "$S3_ENDPOINT/documents/reports/hello.txt"
    -

    1.4 List objects

    -

    List the first 100 keys below a prefix:

    -
    crowdb-cli s3 object list --root "$S3_ROOT" documents \
    -  --prefix reports/ --limit 100
    -
    -curl --get "$S3_ENDPOINT/documents" \
    -  --data-urlencode 'list-type=2' \
    -  --data-urlencode 'prefix=reports/' \
    -  --data-urlencode 'max-keys=100'
    -

    When a response is truncated, pass its opaque continuation token unchanged:

    -
    crowdb-cli s3 object list --root "$S3_ROOT" documents \
    -  --prefix reports/ --limit 100 --continuation "$TOKEN"
    -
    -curl --get "$S3_ENDPOINT/documents" \
    -  --data-urlencode 'list-type=2' \
    -  --data-urlencode 'prefix=reports/' \
    -  --data-urlencode 'max-keys=100' \
    -  --data-urlencode "continuation-token=$TOKEN"
    -

    1.5 Read a byte range

    -

    Ranges are inclusive. --range 3-9 returns seven bytes:

    -
    crowdb-cli s3 object get --root "$S3_ROOT" \
    -  documents reports/hello.txt --range 3-9
    -
    -curl -H 'Range: bytes=3-9' \
    -  "$S3_ENDPOINT/documents/reports/hello.txt"
    -

    1.6 Stop, restart, and delete

    -

    Stop every process while preserving the cluster directory and stored objects:

    -
    crowdb-cli s3 cluster stop --root "$S3_ROOT"
    -

    Restart from the same directory and read the same data:

    -
    crowdb-cli s3 cluster start --root "$S3_ROOT"
    -crowdb-cli s3 object get --root "$S3_ROOT" \
    -  documents reports/hello.txt
    -

    Permanently stop the cluster, release its port assignments, and remove its

    -

    directory:

    -
    crowdb-cli s3 cluster delete --root "$S3_ROOT"
    -

    delete is destructive. Use stop when the cluster must be started again.

    -

    Cluster lifecycle does not currently have an HTTP management endpoint.

    -
    -

    2. Advanced: Bootstrap a KV Cluster

    -

    The remaining sections describe lower-level cluster and server administration.

    -

    They are not required for the local S3 workflow above.

    -

    Management CLI commands omit --system-ip and --system-port for brevity.

    -

    They default to the system-group discovery endpoint 127.0.0.1:10000; either

    -

    flag may point to any system-group node because leader discovery is automatic.

    -

    The CROWDB_SYSTEM_IP/CROWDB_SYSTEM_PORT environment variables provide the

    -

    same overrides. The following console HTTP curl examples assume:

    -
    IP=127.0.0.1
    -PORT=14000
    -

    An S3 mini-cluster already starts crowdb-web with its persisted registry.

    -

    For a separately managed cluster, before using these commands:

    -
      -
    • Start crowdb-web with crowdb-web --port 14000. Add --test-mode for an
    • -
    -

    in-memory console configuration that is lost on restart.

    -
      -
    • Set CROWDB_KV_SERVER_BIN when crowdb-kv-server is not next to
    • -
    -

    crowdb-web and not available through PATH.

    -
      -
    • Use each machine's reachable hostname or IP instead of 127.0.0.1 for a
    • -
    -

    multi-machine deployment.

    -

    Server configuration files

    -

    KV server, diskdb, chunkdb, and diskio use TOML startup configuration. Valid

    -

    templates are shipped in each server's conf/ directory. Values resolve in

    -

    this order: compiled defaults, then file values, then CLI options that were

    -

    explicitly supplied. A CLI option you omit does not erase its file value.

    -

    KV server and diskio make --config optional; diskdb and chunkdb require a

    -

    config path. A malformed, unreadable, or invalid named file stops startup

    -

    instead of silently falling back to defaults. server.rpc_workers controls

    -

    the inbound RPC worker count (default 2 for the Rust servers and 4 for diskio),

    -

    must be positive, and takes effect only after restart. File watchers may report

    -

    static changes before restart, but the active listener does not change.

    -

    Console local deployment generates node-specific config files and reuses the

    -

    same paths when restarting services. Keep manually managed files with the

    -

    server's data and deployment records; group 0 currently stores topology, not

    -

    process configuration.

    -

    2.1 Register the physical topology

    -

    Create a rack, add nodes, and deploy a server on each node. The

    -

    deploy command starts crowdb-kv-server on the target node (via SSH

    -

    if ssh_user is set, or as a local subprocess otherwise). No manual

    -

    start needed.

    -
    -
    - - -
    -
    # Create a rack
    -crowdb-cli rack add --id r1 --name "rack-one"
    -
    -# Register each node (repeat for n2, n3)
    -crowdb-cli node add --id n1 --rack r1 --host 127.0.0.1
    -
    -# Deploy a crowdb-kv-server process on each node (repeat for n2, n3)
    -crowdb-cli server deploy --node n1 --rest-port 2001 --rpc-port 20001
    -
    # Create a rack
    -curl -X POST "http://$IP:$PORT/api/racks" -H 'Content-Type: application/json' \
    -  -d '{"id":"r1"}'
    -
    -# Register each node (repeat for n2, n3)
    -curl -X POST "http://$IP:$PORT/api/nodes" -H 'Content-Type: application/json' \
    -  -d '{"id":"n1","rack_id":"r1","host":"127.0.0.1","ssh_port":22,"ssh_user":""}'
    -
    -# Deploy a crowdb-kv-server process on each node (repeat for n2, n3)
    -curl -X POST "http://$IP:$PORT/api/nodes/n1/server/deploy" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"rest_port":2001,"rpc_port":20001}'
    -
    -

    2.2 Initialize the cluster

    -

    Before creating data stores or groups, the cluster must be

    -

    initialized. This creates the system group (store 0, group 0) which

    -

    stores cluster topology metadata as KV entries, providing HA for

    -

    the topology itself.

    -
    -
    - - -
    -
    # Initialize with all deployed nodes
    -crowdb-cli cluster init --nodes n1,n2,n3
    -
    curl -X POST "http://$IP:$PORT/api/cluster/init" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"nodes":["n1","n2","n3"]}'
    -
    -

    This creates store 0 and group 0 on each selected node, wires remotes

    -

    for multi-node, persists topology in console config, and writes

    -

    hardware hierarchy + KV-cluster topology into group 0 via

    -

    HardwareClient + KVClusterMetaClient (text-path keys, JSON

    -

    values). After initialization, data store/group creation is unblocked.

    -

    For a single-node dev cluster, pass one node:

    -
    crowdb-cli cluster init --nodes n1
    -

    2.3 Create a store and group

    -

    A store is the logical container that owns one or more groups.

    -
    -
    - - -
    -
    # Create a store on n1
    -crowdb-cli store add --store-id 3 --nodes n1
    -
    -# Create a group with an initial replica on n1
    -crowdb-cli paxos add \
    -  --store-id 3 --group-id 3 --replica-id 1 --nodes n1
    -
    curl -X POST "http://$IP:$PORT/api/stores" -H 'Content-Type: application/json' \
    -  -d '{"store_id":3,"nodes":["n1"]}'
    -
    -curl -X POST "http://$IP:$PORT/api/stores/3/groups" -H 'Content-Type: application/json' \
    -  -d '{"group_id":3,"replica_id":1,"nodes":["n1"]}'
    -
    -

    If the cluster has not been initialized, store/group creation returns

    -

    409 Conflict with a message directing you to run cluster init first.

    -

    2.4 Add the remaining replicas

    -
    -
    - - -
    -
    crowdb-cli replica add \
    -  --store-id 3 --group-id 3 --node n2 --replica-id 2
    -
    -crowdb-cli replica add \
    -  --store-id 3 --group-id 3 --node n3 --replica-id 3
    -
    curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/replicas" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"node_id":"n2","replica_id":2}'
    -
    -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/replicas" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"node_id":"n3","replica_id":3}'
    -
    -

    The service orchestrates the full add-replica flow: creates the local

    -

    group on the target node, wires remotes bidirectionally, and the new

    -

    replica catches up via snapshot streaming before joining the voting set.

    -

    2.5 Verify and smoke test

    -
    -
    - - -
    -
    # Check group health
    -crowdb-cli paxos inspect --store-id 3 --group-id 3
    -# Look for "leader=" and replica states
    -
    -# Put / Get
    -crowdb-cli kv put --store-id 3 --group-id 3 \
    -  --key hello --value world
    -
    -crowdb-cli kv get --store-id 3 --group-id 3 --key hello
    -
    curl "http://$IP:$PORT/api/stores/3/groups/3"
    -
    -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/kv/put" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"key":"hello","value":"world"}'
    -
    -curl "http://$IP:$PORT/api/stores/3/groups/3/kv/get?key=hello"
    -
    -
    -

    3. KV Operations

    -

    All KV operations target a specific (store_id, group_id).

    -
    -
    - - -
    -
    # Put
    -crowdb-cli kv put --store-id 3 --group-id 3 --key user:1 --value alice
    -
    -# Get
    -crowdb-cli kv get --store-id 3 --group-id 3 --key user:1
    -
    -# Delete
    -crowdb-cli kv delete --store-id 3 --group-id 3 --key user:1
    -
    -# Prefix scan (list mode — fast, latest values, S3-list semantics)
    -crowdb-cli kv scan --store-id 3 --group-id 3 --prefix user: --limit 100
    -
    curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/kv/put" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"key":"user:1","value":"alice"}'
    -
    -curl "http://$IP:$PORT/api/stores/3/groups/3/kv/get?key=user:1"
    -
    -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/kv/delete" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"key":"user:1"}'
    -
    -curl "http://$IP:$PORT/api/stores/3/groups/3/kv/scan?prefix=user:&limit=100"
    -
    -

    The Web UI KV Operator panel provides the same operations with a

    -

    store/group selector, paginated scan, and inline editing.

    -

    3.1 Scan modes

    -

    CROWDB provides two range-read modes for different use cases:

    -
      -
    • List scan (kv scan) — the default scan. Fast, always returns the
    • -
    -

    latest value per key at each page's read point. S3-list semantics:

    -

    each page is independently consistent, but a key can vanish (deleted

    -

    between pages) or a value can drift (overwritten between pages) within

    -

    a single logical scan. No server-side state beyond the per-page read

    -

    barrier. Use for interactive listing, key discovery, and the KV

    -

    Operator UI.

    -
      -
    • Snapshot scan (snapshot create + snapshot scan) —
    • -
    -

    point-in-time-consistent. Pins a frozen view of the keyspace at a

    -

    specific slot; every page is served from the same frozen view. No key

    -

    vanishes, no value drifts, no phantom keys appear. Use for backup,

    -

    analytics, and any consumer that needs a consistent point-in-time

    -

    view. See §3.2 below.

    -

    3.2 Snapshot versioning

    -

    A snapshot scan pins a point-in-time view of the keyspace. Creating a

    -

    snapshot flushes the in-memory write buffer (L0) into the durable tree

    -

    (L1), then pins L1 at the current applied slot. The snapshot is a frozen,

    -

    immutable view. Iterating it is pure array traversal with no concurrency

    -

    concerns. Each snapshot has a server-side handle with a lease (default 5

    -

    minutes); the handle is reaped if the client disconnects, preventing

    -

    unbounded pin retention.

    -

    Create a snapshot:

    -
    crowdb-cli snapshot create --store-id 3 --group-id 3
    -# Returns: snapshot_handle=42, at_slot=12345
    -

    List active snapshots:

    -
    crowdb-cli snapshot list --store-id 3 --group-id 3
    -# Returns: handle, at_slot, lease_remaining for each active snapshot
    -

    Scan a snapshot (paginated, same prefix/start_after/limit as list scan):

    -
    # First page
    -crowdb-cli snapshot scan --store-id 3 --group-id 3 \
    -  --handle 42 --prefix user: --limit 100
    -
    -# Next page (start_after = last key from previous page)
    -crowdb-cli snapshot scan --store-id 3 --group-id 3 \
    -  --handle 42 --prefix user: --limit 100 \
    -  --start-after user:50
    -

    Release a snapshot (free the pinned pages):

    -
    crowdb-cli snapshot release --store-id 3 --group-id 3 --handle 42
    -
    -
    - -
    -
    # Create
    -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/snapshots"
    -
    -# List
    -curl "http://$IP:$PORT/api/stores/3/groups/3/snapshots"
    -
    -# Scan
    -curl "http://$IP:$PORT/api/stores/3/groups/3/snapshots/42/scan?prefix=user:&limit=100"
    -
    -# Release
    -curl -X DELETE "http://$IP:$PORT/api/stores/3/groups/3/snapshots/42"
    -
    -

    GC and snapshots: the engine's garbage collector reclaims tombstones

    -

    and stale versions with slot <= gc_watermark. Active snapshots protect

    -

    their pinned pages via refcount — GC never frees a page a live snapshot

    -

    still references. Once a snapshot is released (or its lease expires), the

    -

    next GC sweep can reclaim those pages. The GC watermark can be advanced

    -

    explicitly via the management API to control retention:

    -
    # Advance GC watermark (data with slot <= watermark becomes reclaimable)
    -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/gc-watermark" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"slot":12000}'
    -
    -

    4. Cluster Management

    -

    4.1 Check cluster health

    -
    -
    - - -
    -
    # High-level summary (servers + store/group counts)
    -crowdb-cli cluster status
    -
    -# Full topology (logical stores/groups/replicas + physical nodes/servers)
    -crowdb-cli cluster topology
    -
    -# Inspect a specific store, group, or node
    -crowdb-cli cluster inspect s3          # store 3
    -crowdb-cli cluster inspect s3/g3       # group 3 in store 3
    -crowdb-cli cluster inspect n1          # node n1
    -
    # All nodes
    -curl "http://$IP:$PORT/api/nodes"
    -
    -# All deployed servers
    -curl "http://$IP:$PORT/api/servers"
    -
    -# A specific group
    -curl "http://$IP:$PORT/api/stores/3/groups/3"
    -# healthy: all replicas up, leader known
    -# degraded: some replicas down, quorum + leader available
    -# unavailable: quorum lost
    -
    -

    4.2 Add a read replica

    -
    -
    - - -
    -
    crowdb-cli replica add --store-id 3 --group-id 3 --node n4 --replica-id 4
    -
    curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/replicas" \
    -  -H 'Content-Type: application/json' \
    -  -d '{"node_id":"n4","replica_id":4}'
    -
    -

    The new replica streams a snapshot from the leader, catches up, then

    -

    joins the voting set automatically.

    -

    4.3 Remove a replica

    -
    -
    - - -
    -
    crowdb-cli replica remove --store-id 3 --group-id 3 --replica-id 3
    -
    curl -X DELETE "http://$IP:$PORT/api/stores/3/groups/3/replicas/3"
    -
    -

    If the target is the leader, the service asks it to step down first,

    -

    waits for a new leader, then removes the replica.

    -

    4.4 Replace a failed node

    -
  • Provision the new machine with the same node ID, management port,
  • -

    and RPC port.

    -
  • Deploy the server via the service. The server auto-loads its
  • -

    store/group configuration from conf/node-config.json on startup.

    -

    No --stores/--groups CLI args needed for normal restart:

    -
    -
    - - -
    -
       crowdb-cli server deploy --node n1 --rest-port 2001 --rpc-port 20001
    -
       curl -X POST "http://$IP:$PORT/api/nodes/n1/server/deploy" \
    -     -H 'Content-Type: application/json' \
    -     -d '{"rest_port":2001,"rpc_port":20001}'
    -
    -

    If node-config.json is lost, fall back to explicit bootstrap args

    -

    by starting crowdb-kv-server manually with --stores/--groups/

    -

    --replica:

    -
       crowdb-kv-server \
    -     --management-addr 0.0.0.0 --management-port 2001 \
    -     --ports 20001 --election-profile default \
    -     --stores 3 --groups 3 --replica 1
    -
  • Verify group health.
  • -

    If the WAL and config directory were also lost, add the replacement as

    -

    a new replica with a new replica ID instead of reusing the old one.

    -
    -

    5. Rolling Upgrade

    -

    Upgrade one node at a time. Wait for each node to rejoin and catch up

    -

    before moving to the next.

    -

    For each node:

    -
  • Stop:
  • -
    -
    - - -
    -
       crowdb-cli server stop --node n1
    -
       curl -X POST "http://$IP:$PORT/api/nodes/n1/server/stop"
    -
    -
  • Install the new binary on the node.
  • -
  • Restart the server. The server auto-loads its store/group
  • -

    configuration from conf/node-config.json on startup:

    -
    -
    - - -
    -
       crowdb-cli server restart --node n1
    -
       curl -X POST "http://$IP:$PORT/api/nodes/n1/server/restart"
    -
    -

    If node-config.json is missing, start crowdb-kv-server manually

    -

    with explicit args:

    -
       crowdb-kv-server \
    -     --management-addr 0.0.0.0 --management-port 2001 \
    -     --ports 20001 --election-profile default \
    -     --stores 3 --groups 3 --replica 1
    -

    --stores/--groups tells the server to reopen the WAL and rejoin

    -

    as a full member. --replica must match the assigned replica ID.

    -
  • Wait for healthy:
  • -
       crowdb-cli cluster status
    -   crowdb-cli paxos inspect --store-id 3 --group-id 3
    -
  • Smoke test:
  • -
       crowdb-cli kv get --store-id 3 --group-id 3 --key hello
    -
  • Move to the next node.
  • -

    What to watch: after stopping a node, the remaining nodes elect a

    -

    new leader. Wait for the group view to show a leader before proceeding.

    -

    A brief latency spike during leader transition is normal.

    -
    -

    6. Emergency: Loss of Quorum

    -

    If two of three nodes fail, the remaining node cannot elect itself

    -

    leader. Writes and linearizable reads block.

    -
      -
    • Restore the failed nodes from backups and restart. The server
    • -
    -

    auto-loads from conf/node-config.json; if the config is lost, fall

    -

    back to --stores/--groups/--replica args. This is always the

    -

    safest path.

    -
      -
    • Recover with data loss (last resort): force the surviving node to
    • -
    -

    become leader by manually truncating the log. Only safe when the

    -

    other nodes are permanently lost.

    -

    Do not add a new node to a quorum-less group without first recovering

    -

    leadership.

    -
    -

    7. Backup

    -

    CROWDB durability comes from the per-store WAL (--wal-root), the

    -

    per-node config cache (--config-root), and the durable KV engine

    -

    (--data-root). For disaster recovery, back up:

    -
      -
    • {wal-root}/store{store_id}/ for each store
    • -
    • {config-root}/node-config.json — per-node store/group config cache
    • -
    • {data-root}/store{store_id}/group{group_id}/ if using crowdb-tree
    • -
    -

    durable KV engine

    -

    Restore by placing these on the replacement node and starting the

    -

    server. With node-config.json present, no --stores/--groups

    -

    bootstrap args are needed. If the config is lost, use explicit

    -

    --stores/--groups/--replica args to recover from WAL.

    -
    -

    8. API Reference

    -
    -
    - - -
    -

    Local S3 commands use --root to identify the cluster root and discover its access endpoint:

    -
      -
    • crowdb-cli s3 cluster start --root <path> — create or restart
    • -
    • crowdb-cli s3 cluster status --root <path> — inspect process liveness
    • -
    • crowdb-cli s3 cluster stop --root <path> — stop and preserve data
    • -
    • crowdb-cli s3 cluster delete --root <path> — stop and delete permanently
    • -
    • crowdb-cli s3 bucket put --root <path> <bucket>
    • -
    • crowdb-cli s3 bucket delete --root <path> <bucket>
    • -
    • crowdb-cli s3 bucket list --root <path>
    • -
    • crowdb-cli s3 bucket get --root <path> <bucket>
    • -
    • crowdb-cli s3 object put --root <path> <bucket> <key> [--file <path> | --text <content> | --random-size <bytes>]
    • -
    • crowdb-cli s3 object get --root <path> <bucket> <key> [--output <file>] [--range <start-end>]
    • -
    • crowdb-cli s3 object delete --root <path> <bucket> <key>
    • -
    • crowdb-cli s3 object head --root <path> <bucket> <key>
    • -
    • crowdb-cli s3 object list --root <path> <bucket> [--prefix <prefix>] [--limit <n>] [--continuation <token>]
    • -
    -

    Lower-level management commands accept --system-ip <addr> (default

    -

    127.0.0.1) and --system-port <port> (default 10000).

    -
      -
    • crowdb-cli cluster status — servers + store/group summary
    • -
    • crowdb-cli cluster topology — full logical + physical hierarchy
    • -
    • crowdb-cli cluster inspect <id> — s<sid>, s<sid>/g<gid>,
    • -
    -

    s<sid>/g<gid>/r<rid>, or <node-id>

    -
      -
    • crowdb-cli cluster init --nodes n1,n2,... — initialize cluster (system group)
    • -
    • crowdb-cli rack add --id <id> [--name <name>]
    • -
    • crowdb-cli rack remove --id <id>
    • -
    • crowdb-cli rack list
    • -
    • crowdb-cli node add --id <id> --rack <rack> [--host <host>] [--ssh-user <user>]
    • -
    • crowdb-cli node remove --id <id>
    • -
    • crowdb-cli node list
    • -
    • crowdb-cli node ping <node>
    • -
    • crowdb-cli server deploy --node <id> --rest-port <p> --rpc-port <p>
    • -
    • crowdb-cli server restart --node <id>
    • -
    • crowdb-cli server stop --node <id>
    • -
    • crowdb-cli server list
    • -
    • crowdb-cli store add --store-id <id> [--nodes n1,n2,...]
    • -
    • crowdb-cli store remove --store-id <id>
    • -
    • crowdb-cli store list
    • -
    • crowdb-cli store inspect --store-id <id>
    • -
    • crowdb-cli paxos add --store-id <s> --group-id <g> --replica-id <r> --nodes n1,n2,...
    • -
    • crowdb-cli paxos remove --store-id <s> --group-id <g>
    • -
    • crowdb-cli paxos list --store-id <s>
    • -
    • crowdb-cli paxos inspect --store-id <s> --group-id <g>
    • -
    • crowdb-cli replica add --store-id <s> --group-id <g> --node <n> [--replica-id <r>]
    • -
    • crowdb-cli replica remove --store-id <s> --group-id <g> --replica-id <r>
    • -
    • crowdb-cli kv put --store-id <s> --group-id <g> --key <k> --value <v>
    • -
    • crowdb-cli kv get --store-id <s> --group-id <g> --key <k>
    • -
    • crowdb-cli kv delete --store-id <s> --group-id <g> --key <k>
    • -
    • crowdb-cli kv scan --store-id <s> --group-id <g> --prefix <p> [--limit <n>] — list scan (fast, latest values, S3-list semantics)
    • -
    • crowdb-cli snapshot create --store-id <s> --group-id <g> — pin a point-in-time snapshot
    • -
    • crowdb-cli snapshot list --store-id <s> --group-id <g> — list active snapshots
    • -
    • crowdb-cli snapshot scan --store-id <s> --group-id <g> --handle <h> --prefix <p> [--limit <n>] [--start-after <k>] — scan a pinned snapshot
    • -
    • crowdb-cli snapshot release --store-id <s> --group-id <g> --handle <h> — release a snapshot
    • -
    -
    -

    S3 data plane

    -

    These endpoints use S3_ENDPOINT, whose local default is

    -

    http://127.0.0.1:16000. Cluster lifecycle and benchmark operations do not

    -

    currently have HTTP endpoints.

    - - - - - - - - - - - - - - - - - -
    OperationEndpoint
    List bucketsGET /
    Create bucketPUT /{bucket}
    Inspect bucketHEAD /{bucket}
    Delete empty bucketDELETE /{bucket}
    Put or replace objectPUT /{bucket}/{key}
    Get objectGET /{bucket}/{key}
    Get inclusive rangeGET /{bucket}/{key} with Range: bytes={start}-{end}
    Inspect objectHEAD /{bucket}/{key}
    Delete objectDELETE /{bucket}/{key}
    List objectsGET /{bucket}?list-type=2&prefix=...&max-keys=...
    Continue object listGET /{bucket}?list-type=2&continuation-token=...
    -

    Cluster lifecycle

    - - - - - - - -
    OperationEndpoint
    Initialize clusterPOST /api/cluster/init
    -

    Physical topology

    - - - - - - - - - - - - - - - - - - -
    OperationEndpoint
    List racksGET /api/racks
    Create rackPOST /api/racks
    Delete rackDELETE /api/racks/{rack_id}
    List nodesGET /api/nodes
    Add nodePOST /api/nodes
    Get nodeGET /api/nodes/{id}
    Remove nodeDELETE /api/nodes/{id}
    Ping nodePOST /api/nodes/{id}/ping
    Get server infoGET /api/nodes/{id}/server
    Deploy serverPOST /api/nodes/{id}/server/deploy
    Restart serverPOST /api/nodes/{id}/server/restart
    Stop serverPOST /api/nodes/{id}/server/stop
    -

    Logical topology (stores and groups)

    - - - - - - - - - - - - - - - - - - - -
    OperationEndpoint
    List storesGET /api/stores
    Create storePOST /api/stores
    Get storeGET /api/stores/{sid}
    Remove storeDELETE /api/stores/{sid}
    List groupsGET /api/stores/{sid}/groups
    Create groupPOST /api/stores/{sid}/groups
    Get group viewGET /api/stores/{sid}/groups/{gid}
    Remove groupDELETE /api/stores/{sid}/groups/{gid}
    List replicasGET /api/stores/{sid}/groups/{gid}/replicas
    Add replicaPOST /api/stores/{sid}/groups/{gid}/replicas
    Get replicaGET /api/stores/{sid}/groups/{gid}/replicas/{rid}
    Remove replicaDELETE /api/stores/{sid}/groups/{gid}/replicas/{rid}
    Resolve leader endpointGET /api/stores/{sid}/groups/{gid}/endpoint
    -

    KV data plane

    - - - - - - - - - - - - - - - -
    OperationEndpoint
    GetGET /api/stores/{sid}/groups/{gid}/kv/get?key=...
    PutPOST /api/stores/{sid}/groups/{gid}/kv/put
    DeletePOST /api/stores/{sid}/groups/{gid}/kv/delete
    Scan (list mode)GET /api/stores/{sid}/groups/{gid}/kv/scan?prefix=...&limit=N
    Create snapshotPOST /api/stores/{sid}/groups/{gid}/snapshots
    List snapshotsGET /api/stores/{sid}/groups/{gid}/snapshots
    Snapshot scanGET /api/stores/{sid}/groups/{gid}/snapshots/{handle}/scan?prefix=...&limit=N&start_after=...
    Release snapshotDELETE /api/stores/{sid}/groups/{gid}/snapshots/{handle}
    Set GC watermarkPOST /api/stores/{sid}/groups/{gid}/gc-watermark
    -

    Server management (per-node, internal)

    - - - - - - - - - - - - - - - - -
    OperationEndpoint
    System init (bootstrap group 0)POST /system/init
    Add storePOST /stores
    Remove storeDELETE /stores/{sid}
    Add groupPOST /stores/{sid}/groups
    Remove groupDELETE /stores/{sid}/groups/{gid}
    Add remote replicasPOST /stores/{sid}/groups/{gid}/remotes
    Step down leaderPOST /stores/{sid}/groups/{gid}/step-down
    Export topologyGET /topology
    Health checkGET /health
    MetricsGET /metrics
    -

    These endpoints are on the crowdb-kv-server management API (internal,

    -

    only called by crowdb-kv-client's KVClusterAdmin). The console's

    -

    POST /api/cluster/init orchestrates

    -

    /system/init across nodes and auto-finalizes.

    -
    -
    -

    9. Iceberg Catalog Foundation

    -

    The independent crowdb-iceberg binary exposes authenticated catalog configuration

    -

    only. Namespace, table and FileIO endpoints are not enabled. It uses an existing

    -

    healthy Group 0, Chunk-KV and chunk-storage deployment; S3 credentials and buckets

    -

    do not select or authorize an Iceberg catalog.

    -
    pixi run -- cargo build -p crowdb-access-server --bin crowdb-iceberg
    -export CROWDB_MANAGEMENT_SEEDS=127.0.0.1:10000
    -export CROWDB_ICEBERG_LISTEN=127.0.0.1:8181
    -

    Supply three distinct, randomly generated 32–256-character ASCII tokens through

    -

    your secret-management environment: CROWDB_ICEBERG_READ_TOKEN,

    -

    CROWDB_ICEBERG_MANAGE_TOKEN and CROWDB_ICEBERG_CLEAR_TOKEN. Configure every

    -

    instance consistently. Management credentials can rename/initialize; only the

    -

    clear credential can replace the catalog. All three can read configuration.

    -

    The listener is plain HTTP: keep it on a trusted loopback/private hop behind a

    -

    TLS-terminating proxy. Do not transmit bearer credentials over public plain HTTP.

    -

    Set CROWDB_ICEBERG_TOKEN to the appropriate management token for CLI commands.

    -

    Each mutation takes a fresh UUIDv7 request identity. Preserve both that identity

    -

    and the exact arguments when retrying an interrupted command.

    -
    export CROWDB_ICEBERG_TOKEN="$CROWDB_ICEBERG_MANAGE_TOKEN"
    -pixi run -- target/debug/crowdb-iceberg initialize "$INIT_UUIDV7" primary
    -pixi run -- target/debug/crowdb-iceberg status
    -pixi run -- target/debug/crowdb-iceberg rename "$RENAME_UUIDV7" renamed "$ACTIVE_EPOCH"
    -pixi run -- target/debug/crowdb-iceberg serve
    -

    Status reports CatalogId, activation epoch, name and phase. Rename preserves the

    -

    CatalogId. The server validates dependencies and reconciles the root before

    -

    opening its listener. Ctrl-C stops admission and drains accepted connections.

    -
    pixi run -- curl -H "Authorization: Bearer $CROWDB_ICEBERG_READ_TOKEN" \
    -  http://127.0.0.1:8181/v1/config
    -

    Absent or empty warehouse selects the active catalog. A nonempty warehouse

    -

    returns 404 NoSuchWarehouseException. Unsupported endpoints return 406; all

    -

    table-format capabilities are false and HTTP idempotency is not advertised.

    -

    Clear makes the old domain inaccessible and selects a new empty catalog. It is

    -

    not physical erasure. Obtain the exact epoch and CatalogId from status, then

    -

    explicitly confirm both:

    -
    export CROWDB_ICEBERG_TOKEN="$CROWDB_ICEBERG_CLEAR_TOKEN"
    -pixi run -- target/debug/crowdb-iceberg clear "$CLEAR_UUIDV7" empty \
    -  "$ACTIVE_EPOCH" "$ACTIVE_CATALOG_ID"
    -

    Admission returns 503 during maintenance. Default persisted limits require an

    -

    11-second grace after the durable fence is observed. Restart cannot shorten it.

    -

    Another healthy instance resumes interrupted operations. An uncertain command

    -

    must be retried with its original identity and input, not a newly generated key.

    -

    Requests have a 24-hour retry window; expired identities are rejected. Bounded

    -

    ledger-slot collisions can reject new operations without evicting live receipts.

    -

    Run backend restart, two-instance and official-client checks with

    -

    pixi run -e iceberg-e2e test-pyiceberg-e2e. This uses a separate disposable runtime

    -

    registry and leaves persistent local cluster reservations intact.

    - -
    -
    - - - diff --git a/doc/user-manual/user-guide.md b/doc/user-manual/user-guide.md deleted file mode 100644 index 668fcfe23..000000000 --- a/doc/user-manual/user-guide.md +++ /dev/null @@ -1,1012 +0,0 @@ - - - -# CROWDB User Guide - -CROWDB is a distributed storage platform with an S3-compatible object data -plane and a multi-group Multi-Paxos key-value foundation. This guide starts -with the shortest usable path: create a persistent local S3 cluster and use -bucket and object operations. The later sections cover the underlying KV -cluster, physical topology, individual servers, upgrades, and recovery. - -CROWDB provides three user-facing interfaces: - -- **Web UI** — the `crowdb-web` service provides a visual dashboard - with cluster topology, group health, a KV Operator panel (store/group - selector, paginated scan, inline CRUD, demo data injection), and - Swagger UI for browsing the OpenAPI spec of any registered - `crowdb-kv-server` instance. -- **CLI** — `crowdb-cli s3` owns the local S3 cluster lifecycle and sends - bucket/object requests directly to the cluster recorded by `--root`. - Lower-level management commands discover services through group 0 and call - them directly. Output is a human-readable console transcript. -- **HTTP APIs** — the local access server exposes the S3 HTTP API. The - console service exposes the lower-level cluster management API documented - in §8. - -S3 examples use the loopback access endpoint. The first local cluster normally -receives `127.0.0.1:16000`; when that port is already assigned, use the -`endpoint` printed by `s3 cluster start` or `s3 cluster status. - -```bash -S3_ENDPOINT=http://127.0.0.1:16000 -``` - -### Prerequisites - -Before following the steps below: - -- Run `pixi run build` from the repository root. -- Add `target/release` to `PATH`, or invoke the binaries by their full paths. -- Choose a dedicated cluster directory. The directory is the persistent - cluster identity and contains configuration, logs, and data files. - ---- - -## 1. Quick Start: Persistent Local S3 - -### 1.1 Start or restart the cluster - -Choose a directory and start the cluster: - -```bash -S3_ROOT="$PWD/.crowdb-runtime/persistent/s3-local" -crowdb-cli s3 cluster start --root "$S3_ROOT" -``` - -An absent or empty directory creates a new file-backed cluster. A recognized -cluster directory restarts the same cluster with its existing data and port -assignments. A non-empty directory that is not a CROWDB cluster is rejected -without modification. - -The command prints the S3 endpoint, Web management URL, running service count, -and cluster root. The default local endpoints are: - -```bash -S3_ENDPOINT=http://127.0.0.1:16000 -WEB_URL=http://127.0.0.1:14000 -``` - -If the command prints different endpoints because a default port is already -assigned, use the printed values. Open `WEB_URL` for the topology, service, and -cluster management console. - -Inspect the recorded processes without changing them: - -```bash -crowdb-cli s3 cluster status --root "$S3_ROOT" -``` - -The CLI owns S3 mini-cluster lifecycle. The bundled Web service loads that -mini-cluster's console registry and presents its running services. - -### 1.2 Bucket operations - -Create a bucket: - -```bash -crowdb-cli s3 bucket put --root "$S3_ROOT" photos -curl -X PUT "$S3_ENDPOINT/photos" -``` - -List buckets: - -```bash -crowdb-cli s3 bucket list --root "$S3_ROOT" -curl "$S3_ENDPOINT/" -``` - -GET one bucket and show its XML result: - -```bash -crowdb-cli s3 bucket get --root "$S3_ROOT" photos -curl "$S3_ENDPOINT/photos" -``` - -Remove an empty bucket: - -```bash -crowdb-cli s3 bucket delete --root "$S3_ROOT" photos -curl -X DELETE "$S3_ENDPOINT/photos" -``` - -Removing a non-empty bucket returns the S3 error and leaves its objects intact. - -### 1.3 Object CRUD - -Create the bucket used by the following examples, then upload an object from a -file, literal text, generated random bytes, or standard input: - -```bash -crowdb-cli s3 bucket put --root "$S3_ROOT" documents - -crowdb-cli s3 object put --root "$S3_ROOT" \ - documents reports/hello.txt --file ./hello.txt - -crowdb-cli s3 object put --root "$S3_ROOT" \ - documents reports/text.txt --text 'object content' - -crowdb-cli s3 object put --root "$S3_ROOT" \ - documents reports/random.bin --random-size 1048576 - -printf 'hello from CROWDB\n' | crowdb-cli s3 object put \ - --root "$S3_ROOT" documents reports/stdin.txt - -curl -X PUT --data-binary @hello.txt \ - "$S3_ENDPOINT/documents/reports/hello.txt" -``` - -`put` creates a new object or replaces the bytes of an existing key. - -Read an object to standard output or a file: - -```bash -crowdb-cli s3 object get --root "$S3_ROOT" \ - documents reports/hello.txt - -crowdb-cli s3 object get --root "$S3_ROOT" \ - documents reports/hello.txt --output ./downloaded.txt - -curl "$S3_ENDPOINT/documents/reports/hello.txt" \ - --output ./downloaded-with-curl.txt -``` - -Check that an object exists: - -```bash -crowdb-cli s3 object head --root "$S3_ROOT" \ - documents reports/hello.txt - -curl -I "$S3_ENDPOINT/documents/reports/hello.txt" -``` - -Delete an object: - -```bash -crowdb-cli s3 object delete --root "$S3_ROOT" \ - documents reports/hello.txt - -curl -X DELETE "$S3_ENDPOINT/documents/reports/hello.txt" -``` - -### 1.4 List objects - -List the first 100 keys below a prefix: - -```bash -crowdb-cli s3 object list --root "$S3_ROOT" documents \ - --prefix reports/ --limit 100 - -curl --get "$S3_ENDPOINT/documents" \ - --data-urlencode 'list-type=2' \ - --data-urlencode 'prefix=reports/' \ - --data-urlencode 'max-keys=100' -``` - -When a response is truncated, pass its opaque continuation token unchanged: - -```bash -crowdb-cli s3 object list --root "$S3_ROOT" documents \ - --prefix reports/ --limit 100 --continuation "$TOKEN" - -curl --get "$S3_ENDPOINT/documents" \ - --data-urlencode 'list-type=2' \ - --data-urlencode 'prefix=reports/' \ - --data-urlencode 'max-keys=100' \ - --data-urlencode "continuation-token=$TOKEN" -``` - -### 1.5 Read a byte range - -Ranges are inclusive. `--range 3-9` returns seven bytes: - -```bash -crowdb-cli s3 object get --root "$S3_ROOT" \ - documents reports/hello.txt --range 3-9 - -curl -H 'Range: bytes=3-9' \ - "$S3_ENDPOINT/documents/reports/hello.txt" -``` - -### 1.6 Stop, restart, and delete - -Stop every process while preserving the cluster directory and stored objects: - -```bash -crowdb-cli s3 cluster stop --root "$S3_ROOT" -``` - -Restart from the same directory and read the same data: - -```bash -crowdb-cli s3 cluster start --root "$S3_ROOT" -crowdb-cli s3 object get --root "$S3_ROOT" \ - documents reports/hello.txt -``` - -Permanently stop the cluster, release its port assignments, and remove its -directory: - -```bash -crowdb-cli s3 cluster delete --root "$S3_ROOT" -``` - -`delete` is destructive. Use `stop` when the cluster must be started again. -Cluster lifecycle does not currently have an HTTP management endpoint. - ---- - -## 2. Advanced: Bootstrap a KV Cluster - -The remaining sections describe lower-level cluster and server administration. -They are not required for the local S3 workflow above. - -Management CLI commands omit `--system-ip` and `--system-port` for brevity. -They default to the system-group discovery endpoint `127.0.0.1:10000`; either -flag may point to any system-group node because leader discovery is automatic. -The `CROWDB_SYSTEM_IP`/`CROWDB_SYSTEM_PORT` environment variables provide the -same overrides. The following console HTTP `curl` examples assume: - -```bash -IP=127.0.0.1 -PORT=14000 -``` - -An S3 mini-cluster already starts `crowdb-web` with its persisted registry. -For a separately managed cluster, before using these commands: - -- Start `crowdb-web` with `crowdb-web --port 14000`. Add `--test-mode` for an - in-memory console configuration that is lost on restart. -- Set `CROWDB_KV_SERVER_BIN` when `crowdb-kv-server` is not next to - `crowdb-web` and not available through `PATH`. -- Use each machine's reachable hostname or IP instead of `127.0.0.1` for a - multi-machine deployment. - -### Server configuration files - -KV server, diskdb, chunkdb, and diskio use TOML startup configuration. Valid -templates are shipped in each server's `conf/` directory. Values resolve in -this order: compiled defaults, then file values, then CLI options that were -explicitly supplied. A CLI option you omit does not erase its file value. - -KV server and diskio make `--config` optional; diskdb and chunkdb require a -config path. A malformed, unreadable, or invalid named file stops startup -instead of silently falling back to defaults. `server.rpc_workers` controls -the inbound RPC worker count (default 2 for the Rust servers and 4 for diskio), -must be positive, and takes effect only after restart. File watchers may report -static changes before restart, but the active listener does not change. - -Console local deployment generates node-specific config files and reuses the -same paths when restarting services. Keep manually managed files with the -server's data and deployment records; group 0 currently stores topology, not -process configuration. - -### 2.1 Register the physical topology - -Create a rack, add nodes, and deploy a server on each node. The -`deploy` command starts `crowdb-kv-server` on the target node (via SSH -if `ssh_user` is set, or as a local subprocess otherwise). No manual -start needed. - -**CLI:** - -```bash -# Create a rack -crowdb-cli rack add --id r1 --name "rack-one" - -# Register each node (repeat for n2, n3) -crowdb-cli node add --id n1 --rack r1 --host 127.0.0.1 - -# Deploy a crowdb-kv-server process on each node (repeat for n2, n3) -crowdb-cli server deploy --node n1 --rest-port 2001 --rpc-port 20001 -``` - -**curl:** - -```bash -# Create a rack -curl -X POST "http://$IP:$PORT/api/racks" -H 'Content-Type: application/json' \ - -d '{"id":"r1"}' - -# Register each node (repeat for n2, n3) -curl -X POST "http://$IP:$PORT/api/nodes" -H 'Content-Type: application/json' \ - -d '{"id":"n1","rack_id":"r1","host":"127.0.0.1","ssh_port":22,"ssh_user":""}' - -# Deploy a crowdb-kv-server process on each node (repeat for n2, n3) -curl -X POST "http://$IP:$PORT/api/nodes/n1/server/deploy" \ - -H 'Content-Type: application/json' \ - -d '{"rest_port":2001,"rpc_port":20001}' -``` - -### 2.2 Initialize the cluster - -Before creating data stores or groups, the cluster must be -initialized. This creates the system group (store 0, group 0) which -stores cluster topology metadata as KV entries, providing HA for -the topology itself. - -**CLI:** - -```bash -# Initialize with all deployed nodes -crowdb-cli cluster init --nodes n1,n2,n3 -``` - -**curl:** - -```bash -curl -X POST "http://$IP:$PORT/api/cluster/init" \ - -H 'Content-Type: application/json' \ - -d '{"nodes":["n1","n2","n3"]}' -``` - -This creates store 0 and group 0 on each selected node, wires remotes -for multi-node, persists topology in console config, and writes -hardware hierarchy + KV-cluster topology into group 0 via -`HardwareClient` + `KVClusterMetaClient` (text-path keys, JSON -values). After initialization, data store/group creation is unblocked. - -For a single-node dev cluster, pass one node: - -```bash -crowdb-cli cluster init --nodes n1 -``` - -### 2.3 Create a store and group - -A store is the logical container that owns one or more groups. - -**CLI:** - -```bash -# Create a store on n1 -crowdb-cli store add --store-id 3 --nodes n1 - -# Create a group with an initial replica on n1 -crowdb-cli paxos add \ - --store-id 3 --group-id 3 --replica-id 1 --nodes n1 -``` - -**curl:** - -```bash -curl -X POST "http://$IP:$PORT/api/stores" -H 'Content-Type: application/json' \ - -d '{"store_id":3,"nodes":["n1"]}' - -curl -X POST "http://$IP:$PORT/api/stores/3/groups" -H 'Content-Type: application/json' \ - -d '{"group_id":3,"replica_id":1,"nodes":["n1"]}' -``` - -If the cluster has not been initialized, store/group creation returns -`409 Conflict` with a message directing you to run `cluster init` first. - -### 2.4 Add the remaining replicas - -**CLI:** - -```bash -crowdb-cli replica add \ - --store-id 3 --group-id 3 --node n2 --replica-id 2 - -crowdb-cli replica add \ - --store-id 3 --group-id 3 --node n3 --replica-id 3 -``` - -**curl:** - -```bash -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/replicas" \ - -H 'Content-Type: application/json' \ - -d '{"node_id":"n2","replica_id":2}' - -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/replicas" \ - -H 'Content-Type: application/json' \ - -d '{"node_id":"n3","replica_id":3}' -``` - -The service orchestrates the full add-replica flow: creates the local -group on the target node, wires remotes bidirectionally, and the new -replica catches up via snapshot streaming before joining the voting set. - -### 2.5 Verify and smoke test - -**CLI:** - -```bash -# Check group health -crowdb-cli paxos inspect --store-id 3 --group-id 3 -# Look for "leader=" and replica states - -# Put / Get -crowdb-cli kv put --store-id 3 --group-id 3 \ - --key hello --value world - -crowdb-cli kv get --store-id 3 --group-id 3 --key hello -``` - -**curl:** - -```bash -curl "http://$IP:$PORT/api/stores/3/groups/3" - -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/kv/put" \ - -H 'Content-Type: application/json' \ - -d '{"key":"hello","value":"world"}' - -curl "http://$IP:$PORT/api/stores/3/groups/3/kv/get?key=hello" -``` - ---- - -## 3. KV Operations - -All KV operations target a specific `(store_id, group_id)`. - -**CLI:** - -```bash -# Put -crowdb-cli kv put --store-id 3 --group-id 3 --key user:1 --value alice - -# Get -crowdb-cli kv get --store-id 3 --group-id 3 --key user:1 - -# Delete -crowdb-cli kv delete --store-id 3 --group-id 3 --key user:1 - -# Prefix scan (list mode — fast, latest values, S3-list semantics) -crowdb-cli kv scan --store-id 3 --group-id 3 --prefix user: --limit 100 -``` - -**curl:** - -```bash -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/kv/put" \ - -H 'Content-Type: application/json' \ - -d '{"key":"user:1","value":"alice"}' - -curl "http://$IP:$PORT/api/stores/3/groups/3/kv/get?key=user:1" - -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/kv/delete" \ - -H 'Content-Type: application/json' \ - -d '{"key":"user:1"}' - -curl "http://$IP:$PORT/api/stores/3/groups/3/kv/scan?prefix=user:&limit=100" -``` - -The Web UI KV Operator panel provides the same operations with a -store/group selector, paginated scan, and inline editing. - -### 3.1 Scan modes - -CROWDB provides two range-read modes for different use cases: - -- **List scan** (`kv scan`) — the default scan. Fast, always returns the - latest value per key at each page's read point. S3-list semantics: - each page is independently consistent, but a key can vanish (deleted - between pages) or a value can drift (overwritten between pages) within - a single logical scan. No server-side state beyond the per-page read - barrier. Use for interactive listing, key discovery, and the KV - Operator UI. -- **Snapshot scan** (`snapshot create` + `snapshot scan`) — - point-in-time-consistent. Pins a frozen view of the keyspace at a - specific slot; every page is served from the same frozen view. No key - vanishes, no value drifts, no phantom keys appear. Use for backup, - analytics, and any consumer that needs a consistent point-in-time - view. See §3.2 below. - -### 3.2 Snapshot versioning - -A snapshot scan pins a point-in-time view of the keyspace. Creating a -snapshot flushes the in-memory write buffer (L0) into the durable tree -(L1), then pins L1 at the current applied slot. The snapshot is a frozen, -immutable view. Iterating it is pure array traversal with no concurrency -concerns. Each snapshot has a server-side handle with a lease (default 5 -minutes); the handle is reaped if the client disconnects, preventing -unbounded pin retention. - -**Create a snapshot:** - -```bash -crowdb-cli snapshot create --store-id 3 --group-id 3 -# Returns: snapshot_handle=42, at_slot=12345 -``` - -**List active snapshots:** - -```bash -crowdb-cli snapshot list --store-id 3 --group-id 3 -# Returns: handle, at_slot, lease_remaining for each active snapshot -``` - -**Scan a snapshot (paginated, same prefix/start_after/limit as list scan):** - -```bash -# First page -crowdb-cli snapshot scan --store-id 3 --group-id 3 \ - --handle 42 --prefix user: --limit 100 - -# Next page (start_after = last key from previous page) -crowdb-cli snapshot scan --store-id 3 --group-id 3 \ - --handle 42 --prefix user: --limit 100 \ - --start-after user:50 -``` - -**Release a snapshot (free the pinned pages):** - -```bash -crowdb-cli snapshot release --store-id 3 --group-id 3 --handle 42 -``` - -**curl:** - -```bash -# Create -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/snapshots" - -# List -curl "http://$IP:$PORT/api/stores/3/groups/3/snapshots" - -# Scan -curl "http://$IP:$PORT/api/stores/3/groups/3/snapshots/42/scan?prefix=user:&limit=100" - -# Release -curl -X DELETE "http://$IP:$PORT/api/stores/3/groups/3/snapshots/42" -``` - -**GC and snapshots**: the engine's garbage collector reclaims tombstones -and stale versions with `slot <= gc_watermark`. Active snapshots protect -their pinned pages via refcount — GC never frees a page a live snapshot -still references. Once a snapshot is released (or its lease expires), the -next GC sweep can reclaim those pages. The GC watermark can be advanced -explicitly via the management API to control retention: - -```bash -# Advance GC watermark (data with slot <= watermark becomes reclaimable) -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/gc-watermark" \ - -H 'Content-Type: application/json' \ - -d '{"slot":12000}' -``` - ---- - -## 4. Cluster Management - -### 4.1 Check cluster health - -**CLI:** - -```bash -# High-level summary (servers + store/group counts) -crowdb-cli cluster status - -# Full topology (logical stores/groups/replicas + physical nodes/servers) -crowdb-cli cluster topology - -# Inspect a specific store, group, or node -crowdb-cli cluster inspect s3 # store 3 -crowdb-cli cluster inspect s3/g3 # group 3 in store 3 -crowdb-cli cluster inspect n1 # node n1 -``` - -**curl:** - -```bash -# All nodes -curl "http://$IP:$PORT/api/nodes" - -# All deployed servers -curl "http://$IP:$PORT/api/servers" - -# A specific group -curl "http://$IP:$PORT/api/stores/3/groups/3" -# healthy: all replicas up, leader known -# degraded: some replicas down, quorum + leader available -# unavailable: quorum lost -``` - -### 4.2 Add a read replica - -**CLI:** - -```bash -crowdb-cli replica add --store-id 3 --group-id 3 --node n4 --replica-id 4 -``` - -**curl:** - -```bash -curl -X POST "http://$IP:$PORT/api/stores/3/groups/3/replicas" \ - -H 'Content-Type: application/json' \ - -d '{"node_id":"n4","replica_id":4}' -``` - -The new replica streams a snapshot from the leader, catches up, then -joins the voting set automatically. - -### 4.3 Remove a replica - -**CLI:** - -```bash -crowdb-cli replica remove --store-id 3 --group-id 3 --replica-id 3 -``` - -**curl:** - -```bash -curl -X DELETE "http://$IP:$PORT/api/stores/3/groups/3/replicas/3" -``` - -If the target is the leader, the service asks it to step down first, -waits for a new leader, then removes the replica. - -### 4.4 Replace a failed node - -1. Provision the new machine with the same node ID, management port, - and RPC port. -2. Deploy the server via the service. The server auto-loads its - store/group configuration from `conf/node-config.json` on startup. - No `--stores`/`--groups` CLI args needed for normal restart: - - **CLI:** - - ```bash - crowdb-cli server deploy --node n1 --rest-port 2001 --rpc-port 20001 - ``` - - **curl:** - - ```bash - curl -X POST "http://$IP:$PORT/api/nodes/n1/server/deploy" \ - -H 'Content-Type: application/json' \ - -d '{"rest_port":2001,"rpc_port":20001}' - ``` - - If `node-config.json` is lost, fall back to explicit bootstrap args - by starting `crowdb-kv-server` manually with `--stores`/`--groups`/ - `--replica`: - - ```bash - crowdb-kv-server \ - --management-addr 0.0.0.0 --management-port 2001 \ - --ports 20001 --election-profile default \ - --stores 3 --groups 3 --replica 1 - ``` - -3. Verify group health. - -If the WAL and config directory were also lost, add the replacement as -a new replica with a new replica ID instead of reusing the old one. - ---- - -## 5. Rolling Upgrade - -Upgrade one node at a time. Wait for each node to rejoin and catch up -before moving to the next. - -For each node: - -1. **Stop:** - - **CLI:** - - ```bash - crowdb-cli server stop --node n1 - ``` - - **curl:** - - ```bash - curl -X POST "http://$IP:$PORT/api/nodes/n1/server/stop" - ``` - -2. **Install the new binary** on the node. - -3. **Restart the server.** The server auto-loads its store/group - configuration from `conf/node-config.json` on startup: - - **CLI:** - - ```bash - crowdb-cli server restart --node n1 - ``` - - **curl:** - - ```bash - curl -X POST "http://$IP:$PORT/api/nodes/n1/server/restart" - ``` - - If `node-config.json` is missing, start `crowdb-kv-server` manually - with explicit args: - - ```bash - crowdb-kv-server \ - --management-addr 0.0.0.0 --management-port 2001 \ - --ports 20001 --election-profile default \ - --stores 3 --groups 3 --replica 1 - ``` - - `--stores`/`--groups` tells the server to reopen the WAL and rejoin - as a full member. `--replica` must match the assigned replica ID. - -4. **Wait for healthy:** - - ```bash - crowdb-cli cluster status - crowdb-cli paxos inspect --store-id 3 --group-id 3 - ``` - -5. **Smoke test:** - - ```bash - crowdb-cli kv get --store-id 3 --group-id 3 --key hello - ``` - -6. Move to the next node. - -**What to watch:** after stopping a node, the remaining nodes elect a -new leader. Wait for the group view to show a leader before proceeding. -A brief latency spike during leader transition is normal. - ---- - -## 6. Emergency: Loss of Quorum - -If two of three nodes fail, the remaining node cannot elect itself -leader. Writes and linearizable reads block. - -- **Restore the failed nodes** from backups and restart. The server - auto-loads from `conf/node-config.json`; if the config is lost, fall - back to `--stores`/`--groups`/`--replica` args. This is always the - safest path. -- **Recover with data loss** (last resort): force the surviving node to - become leader by manually truncating the log. Only safe when the - other nodes are permanently lost. - -Do not add a new node to a quorum-less group without first recovering -leadership. - ---- - -## 7. Backup - -CROWDB durability comes from the per-store WAL (`--wal-root`), the -per-node config cache (`--config-root`), and the durable KV engine -(`--data-root`). For disaster recovery, back up: - -- `{wal-root}/store{store_id}/` for each store -- `{config-root}/node-config.json` — per-node store/group config cache -- `{data-root}/store{store_id}/group{group_id}/` if using crowdb-tree - durable KV engine - -Restore by placing these on the replacement node and starting the -server. With `node-config.json` present, no `--stores`/`--groups` -bootstrap args are needed. If the config is lost, use explicit -`--stores`/`--groups`/`--replica` args to recover from WAL. - ---- - -## 8. API Reference - -**CLI:** - -Local S3 commands use `--root` to identify the cluster root and discover its access endpoint: - -- **`crowdb-cli s3 cluster start --root `** — create or restart -- **`crowdb-cli s3 cluster status --root `** — inspect process liveness -- **`crowdb-cli s3 cluster stop --root `** — stop and preserve data -- **`crowdb-cli s3 cluster delete --root `** — stop and delete permanently -- **`crowdb-cli s3 bucket put --root `** -- **`crowdb-cli s3 bucket delete --root `** -- **`crowdb-cli s3 bucket list --root `** -- **`crowdb-cli s3 bucket get --root `** -- **`crowdb-cli s3 object put --root [--file | --text | --random-size ]`** -- **`crowdb-cli s3 object get --root [--output ] [--range ]`** -- **`crowdb-cli s3 object delete --root `** -- **`crowdb-cli s3 object head --root `** -- **`crowdb-cli s3 object list --root [--prefix ] [--limit ] [--continuation ]`** - -Lower-level management commands accept `--system-ip ` (default -`127.0.0.1`) and `--system-port ` (default `10000`). - -- **`crowdb-cli cluster status`** — servers + store/group summary -- **`crowdb-cli cluster topology`** — full logical + physical hierarchy -- **`crowdb-cli cluster inspect `** — `s`, `s/g`, - `s/g/r`, or `` -- **`crowdb-cli cluster init --nodes n1,n2,...`** — initialize cluster (system group) -- **`crowdb-cli rack add --id [--name ]`** -- **`crowdb-cli rack remove --id `** -- **`crowdb-cli rack list`** -- **`crowdb-cli node add --id --rack [--host ] [--ssh-user ]`** -- **`crowdb-cli node remove --id `** -- **`crowdb-cli node list`** -- **`crowdb-cli node ping `** -- **`crowdb-cli server deploy --node --rest-port

    --rpc-port

    `** -- **`crowdb-cli server restart --node `** -- **`crowdb-cli server stop --node `** -- **`crowdb-cli server list`** -- **`crowdb-cli store add --store-id [--nodes n1,n2,...]`** -- **`crowdb-cli store remove --store-id `** -- **`crowdb-cli store list`** -- **`crowdb-cli store inspect --store-id `** -- **`crowdb-cli paxos add --store-id --group-id --replica-id --nodes n1,n2,...`** -- **`crowdb-cli paxos remove --store-id --group-id `** -- **`crowdb-cli paxos list --store-id `** -- **`crowdb-cli paxos inspect --store-id --group-id `** -- **`crowdb-cli replica add --store-id --group-id --node [--replica-id ]`** -- **`crowdb-cli replica remove --store-id --group-id --replica-id `** -- **`crowdb-cli kv put --store-id --group-id --key --value `** -- **`crowdb-cli kv get --store-id --group-id --key `** -- **`crowdb-cli kv delete --store-id --group-id --key `** -- **`crowdb-cli kv scan --store-id --group-id --prefix

    [--limit ]`** — list scan (fast, latest values, S3-list semantics) -- **`crowdb-cli snapshot create --store-id --group-id `** — pin a point-in-time snapshot -- **`crowdb-cli snapshot list --store-id --group-id `** — list active snapshots -- **`crowdb-cli snapshot scan --store-id --group-id --handle --prefix

    [--limit ] [--start-after ]`** — scan a pinned snapshot -- **`crowdb-cli snapshot release --store-id --group-id --handle `** — release a snapshot - -**curl:** - -#### S3 data plane - -These endpoints use `S3_ENDPOINT`, whose local default is -`http://127.0.0.1:16000`. Cluster lifecycle and benchmark operations do not -currently have HTTP endpoints. - -| Operation | Endpoint | -| ---------------------- | ---------------------------------------------------------- | -| List buckets | `GET /` | -| Create bucket | `PUT /{bucket}` | -| Inspect bucket | `HEAD /{bucket}` | -| Delete empty bucket | `DELETE /{bucket}` | -| Put or replace object | `PUT /{bucket}/{key}` | -| Get object | `GET /{bucket}/{key}` | -| Get inclusive range | `GET /{bucket}/{key}` with `Range: bytes={start}-{end}` | -| Inspect object | `HEAD /{bucket}/{key}` | -| Delete object | `DELETE /{bucket}/{key}` | -| List objects | `GET /{bucket}?list-type=2&prefix=...&max-keys=...` | -| Continue object list | `GET /{bucket}?list-type=2&continuation-token=...` | - -#### Cluster lifecycle - -| Operation | Endpoint | -| --- | --- | -| Initialize cluster | `POST /api/cluster/init` | - -#### Physical topology - -| Operation | Endpoint | -| --- | --- | -| List racks | `GET /api/racks` | -| Create rack | `POST /api/racks` | -| Delete rack | `DELETE /api/racks/{rack_id}` | -| List nodes | `GET /api/nodes` | -| Add node | `POST /api/nodes` | -| Get node | `GET /api/nodes/{id}` | -| Remove node | `DELETE /api/nodes/{id}` | -| Ping node | `POST /api/nodes/{id}/ping` | -| Get server info | `GET /api/nodes/{id}/server` | -| Deploy server | `POST /api/nodes/{id}/server/deploy` | -| Restart server | `POST /api/nodes/{id}/server/restart` | -| Stop server | `POST /api/nodes/{id}/server/stop` | - -#### Logical topology (stores and groups) - -| Operation | Endpoint | -| --- | --- | -| List stores | `GET /api/stores` | -| Create store | `POST /api/stores` | -| Get store | `GET /api/stores/{sid}` | -| Remove store | `DELETE /api/stores/{sid}` | -| List groups | `GET /api/stores/{sid}/groups` | -| Create group | `POST /api/stores/{sid}/groups` | -| Get group view | `GET /api/stores/{sid}/groups/{gid}` | -| Remove group | `DELETE /api/stores/{sid}/groups/{gid}` | -| List replicas | `GET /api/stores/{sid}/groups/{gid}/replicas` | -| Add replica | `POST /api/stores/{sid}/groups/{gid}/replicas` | -| Get replica | `GET /api/stores/{sid}/groups/{gid}/replicas/{rid}` | -| Remove replica | `DELETE /api/stores/{sid}/groups/{gid}/replicas/{rid}` | -| Resolve leader endpoint | `GET /api/stores/{sid}/groups/{gid}/endpoint` | - -#### KV data plane - -| Operation | Endpoint | -| --- | --- | -| Get | `GET /api/stores/{sid}/groups/{gid}/kv/get?key=...` | -| Put | `POST /api/stores/{sid}/groups/{gid}/kv/put` | -| Delete | `POST /api/stores/{sid}/groups/{gid}/kv/delete` | -| Scan (list mode) | `GET /api/stores/{sid}/groups/{gid}/kv/scan?prefix=...&limit=N` | -| Create snapshot | `POST /api/stores/{sid}/groups/{gid}/snapshots` | -| List snapshots | `GET /api/stores/{sid}/groups/{gid}/snapshots` | -| Snapshot scan | `GET /api/stores/{sid}/groups/{gid}/snapshots/{handle}/scan?prefix=...&limit=N&start_after=...` | -| Release snapshot | `DELETE /api/stores/{sid}/groups/{gid}/snapshots/{handle}` | -| Set GC watermark | `POST /api/stores/{sid}/groups/{gid}/gc-watermark` | - -#### Server management (per-node, internal) - -| Operation | Endpoint | -| --- | --- | -| System init (bootstrap group 0) | `POST /system/init` | -| Add store | `POST /stores` | -| Remove store | `DELETE /stores/{sid}` | -| Add group | `POST /stores/{sid}/groups` | -| Remove group | `DELETE /stores/{sid}/groups/{gid}` | -| Add remote replicas | `POST /stores/{sid}/groups/{gid}/remotes` | -| Step down leader | `POST /stores/{sid}/groups/{gid}/step-down` | -| Export topology | `GET /topology` | -| Health check | `GET /health` | -| Metrics | `GET /metrics` | - -These endpoints are on the `crowdb-kv-server` management API (internal, -only called by `crowdb-kv-client`'s `KVClusterAdmin`). The console's -`POST /api/cluster/init` orchestrates -`/system/init` across nodes and auto-finalizes. - -## 9. Iceberg Catalog Foundation - -The independent `crowdb-iceberg` binary exposes authenticated catalog configuration -only. Namespace, table and FileIO endpoints are not enabled. It uses an existing -healthy Group 0, Chunk-KV and chunk-storage deployment; S3 credentials and buckets -do not select or authorize an Iceberg catalog. - -```bash -pixi run -- cargo build -p crowdb-access-server --bin crowdb-iceberg -export CROWDB_MANAGEMENT_SEEDS=127.0.0.1:10000 -export CROWDB_ICEBERG_LISTEN=127.0.0.1:8181 -``` - -Supply three distinct, randomly generated 32–256-character ASCII tokens through -your secret-management environment: `CROWDB_ICEBERG_READ_TOKEN`, -`CROWDB_ICEBERG_MANAGE_TOKEN` and `CROWDB_ICEBERG_CLEAR_TOKEN`. Configure every -instance consistently. Management credentials can rename/initialize; only the -clear credential can replace the catalog. All three can read configuration. -The listener is plain HTTP: keep it on a trusted loopback/private hop behind a -TLS-terminating proxy. Do not transmit bearer credentials over public plain HTTP. - -Set `CROWDB_ICEBERG_TOKEN` to the appropriate management token for CLI commands. -Each mutation takes a fresh UUIDv7 request identity. Preserve both that identity -and the exact arguments when retrying an interrupted command. - -```bash -export CROWDB_ICEBERG_TOKEN="$CROWDB_ICEBERG_MANAGE_TOKEN" -pixi run -- target/debug/crowdb-iceberg initialize "$INIT_UUIDV7" primary -pixi run -- target/debug/crowdb-iceberg status -pixi run -- target/debug/crowdb-iceberg rename "$RENAME_UUIDV7" renamed "$ACTIVE_EPOCH" -pixi run -- target/debug/crowdb-iceberg serve -``` - -Status reports CatalogId, activation epoch, name and phase. Rename preserves the -CatalogId. The server validates dependencies and reconciles the root before -opening its listener. Ctrl-C stops admission and drains accepted connections. - -```bash -pixi run -- curl -H "Authorization: Bearer $CROWDB_ICEBERG_READ_TOKEN" \ - http://127.0.0.1:8181/v1/config -``` - -Absent or empty `warehouse` selects the active catalog. A nonempty warehouse -returns 404 `NoSuchWarehouseException`. Unsupported endpoints return 406; all -table-format capabilities are false and HTTP idempotency is not advertised. - -Clear makes the old domain inaccessible and selects a new empty catalog. It is -not physical erasure. Obtain the exact epoch and CatalogId from `status`, then -explicitly confirm both: - -```bash -export CROWDB_ICEBERG_TOKEN="$CROWDB_ICEBERG_CLEAR_TOKEN" -pixi run -- target/debug/crowdb-iceberg clear "$CLEAR_UUIDV7" empty \ - "$ACTIVE_EPOCH" "$ACTIVE_CATALOG_ID" -``` - -Admission returns 503 during maintenance. Default persisted limits require an -11-second grace after the durable fence is observed. Restart cannot shorten it. -Another healthy instance resumes interrupted operations. An uncertain command -must be retried with its original identity and input, not a newly generated key. -Requests have a 24-hour retry window; expired identities are rejected. Bounded -ledger-slot collisions can reject new operations without evicting live receipts. - -Run backend restart, two-instance and official-client checks with -`pixi run -e iceberg-e2e test-pyiceberg-e2e`. This uses a separate disposable runtime -registry and leaves persistent local cluster reservations intact. diff --git a/doc/working/plan-access-streaming-io.md b/doc/working/plan-access-streaming-io.md deleted file mode 100644 index e6b949529..000000000 --- a/doc/working/plan-access-streaming-io.md +++ /dev/null @@ -1,65 +0,0 @@ - - - -# Shared Access Streaming I/O Plan - -Implements [R190](../backlog/R190-access-iceberg-shared-streaming-io.md). -Goal: share S3's whole-object write/read path with Iceberg, preserving authority -and crash recovery while eliminating per-frame catalog operations. - -Status: Ready after R187 completion. No production refactor has started. - -## Execution - -- [ ] **Complete flow review**: map read/write/delete/GC, ownership, buffer - lifetime, publication and physical reuse for S3 and Iceberg. Review the whole - flow before implementation; include cancellation and crash boundaries. -- [ ] **Shared receive plumbing**: extract deferred native HTTP receive-provider - installation from the S3 facade into a protocol-neutral access-server module; - keep S3 behavior covered by existing receive-provider tests. Inspect native - owner handoff and Iceberg signed-body decoding before wiring the fast path. -- [ ] **Complete stream descriptors and ownership**: define bounded durable - locations for complete files/parts and allocation ownership for unpublished - data. Preserve old descriptors; integrate GC with the new owner references. -- [ ] **Whole-object uploads**: reuse prepared Chunk writers, 1 MiB native owners - and 64 KiB frames; remove per-leaf catalog intent/durable-completion waits. - Publish only after checksum, format and storage completion. -- [ ] **Shared reads**: use Chunk read streams and owner-backed Bytes for full - GET and ranges; preserve integrity, pins and cancellation. -- [ ] **Multipart completion**: compose validated completed parts without the - old serial per-leaf rewrite/commit path; retain recovery and terminal credits. -- [ ] **Faults and measurements**: test cancellation, lost replies, crash points, - ownership reclamation, stale grants and existing records; compare identical - 5 MiB baseline plus ordinary 10 KiB/1 MiB/12 MiB/100 MiB PUTs and a - 100 MiB multipart upload (twenty 5 MiB parts), with matching build profiles. -- [ ] **Final gates and cleanup**: run affected S3/Iceberg suites, fmt/clippy, - update current architecture, close this requirement after the full reviewed flow passes. - -## Evidence - -- Existing native Iceberg writer uses 65,502-byte leaves. A 5 MiB upload has - 81 leaves plus a directory block. Each registers ownership with six reads - and one conditional write, then forces readable-cursor completion. -- Existing S3 selects the prepared large writer for this size and publishes - object metadata after `on_finish`; receive owners are configured at 1 MiB. -- Release/null-DiskIO measurements: ordinary PUT 2791/2682/2752 ms; UploadPart - 2091/2184/2159 ms. All HTTP 200. Temporary probe was removed. -- Debug multipart intermittently exceeds its existing 10 s deadline even after - streaming MD5 verification. This remains unresolved; do not hide it by only - changing test profiles or raising the timeout. - -## Files - -- Access server: shared body receive module, S3 dispatcher/operations, Iceberg - HTTP, body decoding, uploads, reads and runtime wiring. -- Access libraries: native buffers, streaming, file descriptors, multipart, - file validation, GC and record codecs. -- Chunk client/protocol: existing large writer, read streams and durable - allocation ownership; extend only where the shared path requires it. - -## Tests - -- Unit: checksums/framing, descriptor validation, bounded reads and lifecycle. -- Integration: native receive-provider tests, S3/Iceberg upload/read tests, - counted catalog calls and backwards-compatible descriptors. -- E2E: native file/multipart crash tests, official SDKs, container acceptance. diff --git a/doc/working/plan-console-authority.md b/doc/working/plan-console-authority.md index dd3c666c2..ac0987caa 100644 --- a/doc/working/plan-console-authority.md +++ b/doc/working/plan-console-authority.md @@ -145,12 +145,13 @@ is paused; it does not block the single-node image requirement. change the host-wide `core_pattern` from inside the container. Files: `container/single-node-container/{Dockerfile,entrypoint.sh,tests/**}`, `container/crowdb-monitor/src/**`, - `doc/user-manual/docker-single-node-user-guide.md`. + `container/single-node-container/README.md`. ## Documentation and completion -- [ ] **Bare-metal guide**: migrate verified KV, chunk and access setup from - the old combined guide, state the non-production boundary, then fix links and - remove the old guide. Keep the Docker end-user guide independent. +- [ ] **Bare-metal documentation**: migrate verified KV, chunk and access + setup into dedicated deployment documentation, state the non-production + boundary, then fix links and remove obsolete combined material. Keep Docker + deployment notes independent. - [ ] **Acceptance and cleanup**: run affected integration cases, full console and UI suites, Rust fmt and lint; update the relevant permanent architecture, then remove the requirement, backlog entry and this plan when complete. diff --git a/doc/working/test.md b/doc/working/test.md index f50a528d8..196374ff6 100644 --- a/doc/working/test.md +++ b/doc/working/test.md @@ -77,8 +77,9 @@ and subprocess startup/shutdown, so feature changes and cold builds affect it. Counts are runner-reported cases, not assertions; ignored cases are excluded. Native Iceberg and Java/Rust/RCK SDK acceptance use release binaries, matching the published container profile. Component suites retain their default test -profile. Debug native upload deadline failures are tracked for the R190 I/O -review; successful release acceptance does not resolve that issue. +profile. The focused debug native 100 MiB multipart upload, completion, +restart, replay, and full read passed on 2026-09-29 in 54.40 s. Its previous +10 s completion deadline failure did not recur after the streaming I/O changes. Status icons: ✅ = PASS, ⚠️ = PASS with ignored tests, ❌ = FAIL, ⏳ = measurement pending. diff --git a/lib/crowdb-access-iceberg/Cargo.toml b/lib/crowdb-access-iceberg/Cargo.toml index 39bb0faef..252c27795 100644 --- a/lib/crowdb-access-iceberg/Cargo.toml +++ b/lib/crowdb-access-iceberg/Cargo.toml @@ -16,12 +16,14 @@ test-util = [] [dependencies] async-trait = "0.1" base64 = "0.22" +bincode = "1.3" bytes = "1" chrono = { version = "0.4", default-features = false, features = ["std"] } crc32fast = "1" data-encoding = "2" hmac = "0.12" lz4_flex = { version = "0.11", default-features = false, features = ["std", "safe-encode", "safe-decode", "frame"] } +md-5 = "0.10" crowdb-chunk-kv-client = { path = "../crowdb-chunk-kv-client" } crowdb-chunk-client = { path = "../crowdb-chunk-client" } crowdb-common = { workspace = true } diff --git a/lib/crowdb-access-iceberg/src/catalog.rs b/lib/crowdb-access-iceberg/src/catalog.rs index 6234d44a9..8fef30d80 100644 --- a/lib/crowdb-access-iceberg/src/catalog.rs +++ b/lib/crowdb-access-iceberg/src/catalog.rs @@ -15,4 +15,6 @@ pub use deadline::ClearBounds; pub use repository::{CatalogError, CatalogRepository, ManagementPrivilege}; pub use root::{ActiveCatalogRecord, RootState}; pub use state::{CatalogAuthority, CatalogContext, CatalogLifecycle, ClearTransition}; -pub use storage::{CasOutcome, CatalogStore, RoutedCatalogStore, StoreError, StoredValue}; +pub use storage::{ + CasOutcome, CatalogStore, CatalogStoreOperationCounts, RoutedCatalogStore, StoreError, StoredValue, +}; diff --git a/lib/crowdb-access-iceberg/src/catalog/repository.rs b/lib/crowdb-access-iceberg/src/catalog/repository.rs index beb7fe4c7..58f177c80 100644 --- a/lib/crowdb-access-iceberg/src/catalog/repository.rs +++ b/lib/crowdb-access-iceberg/src/catalog/repository.rs @@ -10,7 +10,8 @@ use crate::operation::{ use crate::record::StorageRecord; use super::{ - ActiveCatalogRecord, CasOutcome, CatalogAuthority, CatalogStore, ClearBounds, RootState, StoreError, + ActiveCatalogRecord, CasOutcome, CatalogAuthority, CatalogStore, CatalogStoreOperationCounts, + ClearBounds, RootState, StoreError, }; #[derive(Debug, thiserror::Error)] @@ -42,6 +43,11 @@ pub struct CatalogRepository { } impl CatalogRepository { + #[must_use] + pub fn store_operation_counts(&self) -> Option { + self.store.operation_counts() + } + /// # Errors /// Rejects invalid timing limits before any storage access. pub fn new(store: Arc, bounds: ClearBounds) -> Result { diff --git a/lib/crowdb-access-iceberg/src/catalog/storage.rs b/lib/crowdb-access-iceberg/src/catalog/storage.rs index cad893f75..af3613f57 100644 --- a/lib/crowdb-access-iceberg/src/catalog/storage.rs +++ b/lib/crowdb-access-iceberg/src/catalog/storage.rs @@ -1,4 +1,7 @@ -use std::sync::Arc; +use std::sync::{ + atomic::{AtomicU64, Ordering}, + Arc, +}; use async_trait::async_trait; use crowdb_chunk_kv_client::{ChunkKvClient, ClientError, MultiScanPage, MultiScanRequest}; @@ -37,8 +40,39 @@ pub enum CasOutcome { Conflict(Option), } +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, serde::Serialize)] +pub struct CatalogStoreOperationCounts { + pub get: u64, + pub compare_exchange: u64, + pub scan: u64, + pub conditional_delete: u64, +} + +#[derive(Default)] +struct OperationCounters { + get: AtomicU64, + compare_exchange: AtomicU64, + scan: AtomicU64, + conditional_delete: AtomicU64, +} + +impl OperationCounters { + fn snapshot(&self) -> CatalogStoreOperationCounts { + CatalogStoreOperationCounts { + get: self.get.load(Ordering::Relaxed), + compare_exchange: self.compare_exchange.load(Ordering::Relaxed), + scan: self.scan.load(Ordering::Relaxed), + conditional_delete: self.conditional_delete.load(Ordering::Relaxed), + } + } +} + #[async_trait] pub trait CatalogStore: Send + Sync { + fn operation_counts(&self) -> Option { + None + } + async fn get(&self, key: &[u8]) -> Result, StoreError>; async fn compare_exchange( &self, @@ -51,12 +85,16 @@ pub trait CatalogStore: Send + Sync { pub struct RoutedCatalogStore { client: Arc, + counters: OperationCounters, } impl RoutedCatalogStore { #[must_use] pub fn new(client: Arc) -> Self { - Self { client } + Self { + client, + counters: OperationCounters::default(), + } } pub(crate) async fn delete_mapping_if( @@ -93,6 +131,7 @@ impl RoutedCatalogStore { return Err(ValidationError::Key.into()); } validate_value(expected)?; + self.counters.conditional_delete.fetch_add(1, Ordering::Relaxed); let response = self .client .execute_with_identity( @@ -143,6 +182,7 @@ impl RoutedCatalogStore { { return Err(StoreError::Response); } + self.counters.scan.fetch_add(1, Ordering::Relaxed); let page = self.client.scan(request).await?; if let Some(failure) = page.terminal_failure { return Err(StoreError::Rejected(failure)); @@ -160,8 +200,13 @@ impl RoutedCatalogStore { #[async_trait] impl CatalogStore for RoutedCatalogStore { + fn operation_counts(&self) -> Option { + Some(self.counters.snapshot()) + } + async fn get(&self, key: &[u8]) -> Result, StoreError> { IcebergKey::decode(key)?; + self.counters.get.fetch_add(1, Ordering::Relaxed); let response = self.client.get(key.to_vec(), None).await?; match response.result.map_err(StoreError::Rejected)? { OperationResult::Value(value) => value.map(|value| stored(key, value)).transpose(), @@ -192,6 +237,7 @@ impl CatalogStore for RoutedCatalogStore { value: value.to_vec(), }, }; + self.counters.compare_exchange.fetch_add(1, Ordering::Relaxed); let response = self .client .execute_with_identity(operation, None, identity) diff --git a/lib/crowdb-access-iceberg/src/commit/provenance.rs b/lib/crowdb-access-iceberg/src/commit/provenance.rs index 36c883e0e..2f09b5a5c 100644 --- a/lib/crowdb-access-iceberg/src/commit/provenance.rs +++ b/lib/crowdb-access-iceberg/src/commit/provenance.rs @@ -127,6 +127,7 @@ impl PriorManifestSource { let payload = match &record.content { FileContent::Inline { bytes, .. } => bytes.capacity(), FileContent::Chunks { .. } => 0, + FileContent::Locations { bytes, etag } => bytes.capacity() + etag.capacity(), }; let bytes = std::mem::size_of::() + 128 + path.len() * 2 + payload; *retained = retained diff --git a/lib/crowdb-access-iceberg/src/file.rs b/lib/crowdb-access-iceberg/src/file.rs index 98dcff43f..74aa2bd3a 100644 --- a/lib/crowdb-access-iceberg/src/file.rs +++ b/lib/crowdb-access-iceberg/src/file.rs @@ -60,7 +60,7 @@ pub(crate) use location::validate_relative_key; pub use location::{FileLocation, TableLocation, MAX_OBJECT_KEY_BYTES}; pub use multipart::{ MultipartCompletion, MultipartLimits, MultipartPart, MultipartPartMutation, MultipartPhase, - MultipartSession, + MultipartSession, MultipartStreamPart, }; pub use multipart_admission::{ MultipartAdmissionLimits, MultipartAdmissionRecord, MultipartCredit, MultipartCreditAction, @@ -73,7 +73,7 @@ pub use multipart_recovery::{ MultipartRecoveryStore, }; pub use multipart_repository::{MultipartRepository, MultipartWorkError}; -pub use multipart_selection::{MultipartSelection, SelectedPart}; +pub use multipart_selection::{MultipartSelection, SelectedPart, SelectedStreamPart}; #[cfg(feature = "test-util")] pub use parquet::{ read_parquet_integer_column_for_tests, read_parquet_nullable_integer_column_for_tests, diff --git a/lib/crowdb-access-iceberg/src/file/blocks.rs b/lib/crowdb-access-iceberg/src/file/blocks.rs index 6ca7d8af1..86e3ac7d1 100644 --- a/lib/crowdb-access-iceberg/src/file/blocks.rs +++ b/lib/crowdb-access-iceberg/src/file/blocks.rs @@ -31,11 +31,30 @@ pub enum FileIoError { #[async_trait] pub trait FileBlockStore: Send + Sync { + fn stream_client(&self) -> Option<&ChunkIoClient> { + None + } + + async fn read_locations( + &self, + _locations: &[Location], + _start: u64, + _end: u64, + ) -> Result, FileIoError> { + Err(FileIoError::Bounds) + } + async fn put(&self, owner: FileIdentity, height: u8, bytes: &[u8]) -> Result; async fn read(&self, root: &ChunkRoot) -> Result, FileIoError>; async fn reclaim(&self, _root: &ChunkRoot) -> Result { Ok(crowdb_chunk_client::ReclaimOutcome::Deferred) } + async fn reclaim_location( + &self, + _location: &Location, + ) -> Result { + Ok(crowdb_chunk_client::ReclaimOutcome::Deferred) + } } #[derive(Clone)] @@ -53,6 +72,27 @@ impl NativeFileBlocks { #[async_trait] impl FileBlockStore for NativeFileBlocks { + async fn reclaim_location( + &self, + location: &Location, + ) -> Result { + let (allocator, _) = self.client.storage_parts(); + Ok(crowdb_chunk_client::reclaim_location(allocator.as_ref(), location).await?) + } + + fn stream_client(&self) -> Option<&ChunkIoClient> { + Some(&self.client) + } + + async fn read_locations( + &self, + locations: &[Location], + start: u64, + end: u64, + ) -> Result, FileIoError> { + Ok(self.client.read_range(locations, start, end).await?.concat()) + } + async fn reclaim(&self, root: &ChunkRoot) -> Result { root.validate()?; let location = Location { @@ -116,9 +156,9 @@ impl FileBlockStore for NativeFileBlocks { logical_offset: root.logical_offset, logical_length: root.logical_length, }; - let bytes = self.client.read_object(&[location]).await?; + let bytes = self.client.read_object(&[location]).await?.concat(); verify_block(root, &bytes)?; - Ok(bytes.to_vec()) + Ok(bytes) } } diff --git a/lib/crowdb-access-iceberg/src/file/content.rs b/lib/crowdb-access-iceberg/src/file/content.rs index f2a6bf11f..bdef743af 100644 --- a/lib/crowdb-access-iceberg/src/file/content.rs +++ b/lib/crowdb-access-iceberg/src/file/content.rs @@ -1,3 +1,4 @@ +use crowdb_protocol::chunkdb::rpc::Location; use crowdb_protocol::common::ChunkId; use sha2::{Digest, Sha256}; @@ -50,9 +51,54 @@ impl ChunkRoot { pub enum FileContent { Inline { codec: InlineCodec, bytes: Vec }, Chunks { root: Option }, + Locations { bytes: Vec, etag: String }, } impl FileContent { + /// Encodes complete Chunk locations once, after the whole file is durable. + /// # Errors + /// Rejects gaps, overlapping locations and metadata that cannot fit one record. + pub fn from_locations( + locations: &[Location], + length: u64, + etag: String, + ) -> Result { + validate_locations(locations, length)?; + validate_etag(&etag)?; + let bytes = bincode::serialize(locations).map_err(|_| ValidationError::Record)?; + if bytes.len() > crate::record::MAX_RECORD_BYTES - 4096 { + return Err(ValidationError::RecordTooLarge); + } + Ok(Self::Locations { bytes, etag }) + } + + /// # Errors + /// Rejects malformed location encodings and inconsistent logical ranges. + pub fn locations(&self, length: u64) -> Result>, ValidationError> { + let Self::Locations { bytes, etag } = self else { + return Ok(None); + }; + validate_etag(etag)?; + if bytes.len() < 8 || bytes.len() > crate::record::MAX_RECORD_BYTES - 4096 { + return Err(ValidationError::RecordTooLarge); + } + let count = u64::from_le_bytes(bytes[..8].try_into().map_err(|_| ValidationError::Record)?); + if count > 1250 { + return Err(ValidationError::RecordTooLarge); + } + let locations: Vec = bincode::deserialize(bytes).map_err(|_| ValidationError::Record)?; + validate_locations(&locations, length)?; + Ok(Some(locations)) + } + + #[must_use] + pub fn etag(&self) -> Option<&str> { + match self { + Self::Locations { etag, .. } => Some(etag), + _ => None, + } + } + pub(crate) fn validate(&self, length: u64, digest: &[u8; 32]) -> Result<(), ValidationError> { match self { Self::Inline { .. } => { @@ -69,6 +115,12 @@ impl FileContent { return Err(ValidationError::Record); } } + Self::Locations { .. } => { + if *digest != [0; 32] { + return Err(ValidationError::Record); + } + self.locations(length)?; + } } Ok(()) } @@ -120,3 +172,41 @@ impl FileContent { Ok(Some(decoded)) } } + +fn validate_locations(locations: &[Location], length: u64) -> Result<(), ValidationError> { + let mut cursor = 0; + for location in locations { + if location.chunk_id.is_none() + || location.length == 0 + || location.logical_length == 0 + || location.logical_offset != cursor + || location.offset.checked_add(location.length).is_none() + { + return Err(ValidationError::Record); + } + cursor = cursor + .checked_add(location.logical_length) + .ok_or(ValidationError::Record)?; + } + if cursor != length { + return Err(ValidationError::Record); + } + Ok(()) +} + +fn validate_etag(etag: &str) -> Result<(), ValidationError> { + let (digest, count) = etag + .split_once('-') + .map_or((etag, None), |(digest, count)| (digest, Some(count))); + if digest.len() != 32 + || !digest + .bytes() + .all(|byte| byte.is_ascii_hexdigit() && !byte.is_ascii_uppercase()) + || count.is_some_and(|count| { + count.is_empty() || count.starts_with('0') || count.parse::().map_or(true, |n| n == 0) + }) + { + return Err(ValidationError::Record); + } + Ok(()) +} diff --git a/lib/crowdb-access-iceberg/src/file/multipart.rs b/lib/crowdb-access-iceberg/src/file/multipart.rs index c7345469a..0c75546cd 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart.rs @@ -2,6 +2,8 @@ use crate::catalog::CatalogContext; use crate::error::ValidationError; use crate::key::{CatalogScope, FileId, IcebergKey, OperationId}; use crate::operation::PayloadReference; +use sha2::{Digest, Sha256}; +use std::fmt::Write; use super::{AssemblyProgress, FileContent, FileDigest, FileIdentity, FileLocation, FileTree}; @@ -114,19 +116,21 @@ impl MultipartSession { if let Some(pending) = &self.pending { pending.validate(self)?; } - let candidate = self - .completion - .as_ref() - .and_then(|completion| completion.candidate.as_ref()); + let completion = self.completion.as_ref(); let valid = match self.phase { MultipartPhase::Open => self.completion.is_none() && self.published.is_none(), MultipartPhase::Completing => { - self.completion.is_some() && candidate.is_none() && self.published.is_none() + completion.is_some_and(|completion| completion.publication.is_none()) + && self.published.is_none() } MultipartPhase::Publishing | MultipartPhase::Conflicted => { - candidate.is_some() && self.published.is_none() + completion.is_some_and(|completion| completion.publication.is_some()) + && self.published.is_none() + } + MultipartPhase::Published => { + completion.is_some_and(|completion| completion.publication.is_some()) + && self.published.is_some() } - MultipartPhase::Published => candidate.is_some() && self.published.is_some(), MultipartPhase::Aborted => self.published.is_none(), }; if !valid { @@ -145,14 +149,16 @@ impl MultipartCompletion { || self.selection.length == 0 || self.selection.digest != progress.selection || self.selected_parts == 0 - || self.selected_parts > session.part_count + || self.selected_parts > session.limits.max_parts || progress.next_part > self.selected_parts || progress.completed_bytes > session.limits.max_file_bytes - || progress.completed_bytes > session.staged_bytes + || (session.staged_bytes != 0 && progress.completed_bytes > session.staged_bytes) || progress.part_offset > progress.completed_bytes || progress.active.is_some() != progress.part_digest.is_some() || progress.active.is_some() != (progress.part_offset > 0) - || (progress.writer.is_none() && (progress.next_part != 0 || progress.completed_bytes != 0)) + || (progress.writer.is_none() + && self.publication.is_none() + && (progress.next_part != 0 || progress.completed_bytes != 0)) { return Err(ValidationError::Record); } @@ -183,7 +189,10 @@ impl MultipartCompletion { return Err(ValidationError::Record); } } - if self.candidate.is_some() != self.publication.is_some() { + if self.candidate.is_none() + && self.publication.is_some() + && (progress.writer.is_some() || progress.next_part != self.selected_parts) + { return Err(ValidationError::Record); } if let Some(publication) = &self.publication { @@ -207,7 +216,14 @@ pub struct MultipartPart { pub revision: u64, pub modified_ms: u64, pub owner: FileIdentity, - pub tree: FileTree, + pub tree: Option, + pub stream: Option, +} + +#[derive(Clone, Debug, Eq, PartialEq)] +pub struct MultipartStreamPart { + pub length: u64, + pub content: FileContent, } impl MultipartPart { @@ -228,7 +244,58 @@ impl MultipartPart { if self.number == 0 || self.number > 10_000 || self.revision == 0 || self.modified_ms == 0 { return Err(ValidationError::Record); } - validate_tree(&self.tree) + match (&self.tree, &self.stream) { + (Some(tree), None) => validate_tree(tree), + (None, Some(stream)) => stream.content.validate(stream.length, &[0; 32]), + _ => Err(ValidationError::Record), + } + } + + #[must_use] + pub fn length(&self) -> u64 { + self.tree.as_ref().map_or_else( + || self.stream.as_ref().map_or(0, |stream| stream.length), + |tree| tree.length, + ) + } + + #[must_use] + pub fn etag(&self) -> String { + self.stream + .as_ref() + .and_then(|stream| stream.content.etag()) + .map_or_else( + || { + let mut etag = String::with_capacity(64); + if let Some(tree) = &self.tree { + for byte in tree.digest { + write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); + } + } + etag + }, + str::to_owned, + ) + } + + #[must_use] + /// # Panics + /// Panics if the part has neither a legacy tree nor a streamed descriptor. + pub fn selection_digest(&self) -> [u8; 32] { + if let Some(tree) = &self.tree { + return tree.digest; + } + let stream = self + .stream + .as_ref() + .expect("validated multipart part has content"); + let mut digest = Sha256::new(); + digest.update(stream.length.to_le_bytes()); + if let FileContent::Locations { bytes, etag } = &stream.content { + digest.update(bytes); + digest.update(etag.as_bytes()); + } + digest.finalize().into() } /// # Errors @@ -244,7 +311,7 @@ impl MultipartPart { || self.owner.table != session.owner.table || self.owner.file == session.owner.file || self.number > session.limits.max_parts - || self.tree.length > session.limits.max_part_bytes + || self.length() > session.limits.max_part_bytes { return Err(ValidationError::Record); } @@ -263,7 +330,7 @@ impl MultipartPartMutation { self.after.validate_binding(session)?; if session.phase != MultipartPhase::Open || session.part_count == 0 - || session.staged_bytes < self.after.tree.length + || session.staged_bytes < self.after.length() { return Err(ValidationError::Record); } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_recovery.rs b/lib/crowdb-access-iceberg/src/file/multipart_recovery.rs index caac87869..b916cc691 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_recovery.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_recovery.rs @@ -154,6 +154,19 @@ impl MultipartRecovery { self.repository.abort(session).await? } else if session.phase == MultipartPhase::Completing { let completion = session.completion.as_ref().ok_or(ValidationError::Record)?; + if completion.progress.next_part == 0 { + if let Some(changed) = self + .repository + .prepare_stream_publication(session, now_ms) + .await? + { + return Ok(if changed { + RecoveryAction::Progressed + } else { + RecoveryAction::Deferred + }); + } + } if completion.progress.next_part == completion.selected_parts { return Ok(RecoveryAction::AwaitingSeal); } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/completion.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/completion.rs index 59bd6e8bf..9a65b9348 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/completion.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/completion.rs @@ -21,6 +21,25 @@ pub enum MultipartWorkError { } impl MultipartRepository { + /// Loads the durable Complete selection, including any streamed part snapshots. + /// # Errors + /// Rejects missing, corrupt or foreign selection payloads. + pub async fn load_selection( + &self, + session: &MultipartSession, + ) -> Result { + session.validate()?; + let completion = session.completion.as_ref().ok_or(ValidationError::Record)?; + let bytes = PayloadStore::new(self.store.clone()) + .get(&completion.selection) + .await?; + let selection = MultipartSelection::decode(&bytes)?; + if selection.count() != completion.selected_parts { + return Err(ValidationError::Record.into()); + } + Ok(selection) + } + /// Freezes a caller-selected part revision list; byte work verifies each selected part. /// # Errors /// Rejects expired sessions, unresolved mutations and invalid selection bounds. @@ -38,7 +57,7 @@ impl MultipartRepository { if session.pending.is_some() { return Err(CatalogError::Busy); } - if selection.parts().len() > usize::from(session.part_count) + if (selection.snapshots().is_none() && selection.parts().len() > usize::from(session.part_count)) || selection .parts() .iter() @@ -103,7 +122,7 @@ impl MultipartRepository { .part(session, selected.number) .await? .ok_or(ValidationError::Record)?; - if part.revision != selected.revision || part.tree.digest != selected.digest { + if part.revision != selected.revision || part.selection_digest() != selected.digest { return Err(ValidationError::Record.into()); } let progress = assembly @@ -112,7 +131,7 @@ impl MultipartRepository { &AssemblyPart { ordinal: completion.progress.next_part, owner: part.owner, - tree: part.tree, + tree: part.tree.ok_or(ValidationError::Record)?, }, ) .await?; diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs index 2e0f29319..178c87ed6 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/parts.rs @@ -8,6 +8,105 @@ use crate::record::StorageRecord; use super::{check_live, increment, MultipartRepository}; impl MultipartRepository { + /// Publishes a streamed part with one CAS on that part number. The session + /// is read for admission, but distinct part numbers do not write it. + /// # Errors + /// Rejects closed sessions, invalid parts, stale contexts and storage failures. + pub async fn put_stream_part( + &self, + session: &MultipartSession, + part: &MultipartPart, + now_ms: u64, + ) -> Result, CatalogError> { + session.validate()?; + let current = self + .load(session.context, session.upload) + .await? + .ok_or(CatalogError::Conflict)?; + check_live(¤t, now_ms)?; + if current.phase != MultipartPhase::Open || current.pending.is_some() { + return Err(CatalogError::Conflict); + } + if u64::from(current.limits.max_parts) + .checked_mul(current.limits.max_part_bytes) + .map_or(true, |bytes| bytes > current.limits.max_staged_bytes) + || part.stream.is_none() + || part.tree.is_some() + { + return Err(ValidationError::Record.into()); + } + part.validate_for(¤t)?; + let before = self.read_part(&part.key()).await?; + if let Some(before) = &before { + before.validate_for(¤t)?; + } + let mut after = part.clone(); + after.revision = before + .as_ref() + .map_or(Some(1), |before| before.revision.checked_add(1)) + .ok_or(ValidationError::Record)?; + after.modified_ms = now_ms; + after.validate_for(¤t)?; + let key = after.key().encode()?; + let expected = before.as_ref().map(encode_part).transpose()?; + let value = encode_part(&after)?; + let outcome = self + .store + .compare_exchange( + &key, + expected.as_deref(), + &value, + mutation_identity(&key, expected.as_deref(), &value), + ) + .await; + let written = match outcome { + Ok(CasOutcome::Applied(_)) => Some(after), + Ok(CasOutcome::Conflict(Some(existing))) if existing.bytes == value => Some(after), + Ok(CasOutcome::Conflict(_)) => None, + Err(error) => { + if self.read_part(&after.key()).await?.as_ref() == Some(&after) { + Some(after) + } else { + return Err(error.into()); + } + } + }; + if written.is_some() { + check_context(self.store.as_ref(), current.context).await?; + } + Ok(written) + } + + /// Reads the current part for an `UploadPart` replacement. The later session + /// compare-and-swap rejects a stale snapshot before the part becomes visible. + /// # Errors + /// Rejects invalid part numbers, phases and corrupt stored parts. + pub async fn part_for_upload( + &self, + session: &MultipartSession, + number: u16, + ) -> Result, CatalogError> { + session.validate()?; + if session.phase != MultipartPhase::Open || session.pending.is_some() { + return Err(CatalogError::Busy); + } + if number == 0 || number > session.limits.max_parts { + return Err(ValidationError::Record.into()); + } + let mut suffix = session.upload.as_bytes().to_vec(); + suffix.extend_from_slice(&number.to_be_bytes()); + let key = IcebergKey::Catalog { + catalog: session.context.catalog, + scope: CatalogScope::MultipartPart, + suffix, + }; + let part = self.read_part(&key).await?; + if let Some(part) = &part { + part.validate_for(session)?; + } + Ok(part) + } + /// Reads a committed part only while the supplied session snapshot stays current. /// # Errors /// Rejects unresolved mutations, stale snapshots, invalid numbers and corrupt parts. @@ -52,6 +151,18 @@ impl MultipartRepository { part: &MultipartPart, now_ms: u64, ) -> Result { + Ok(self.reserve_part_state(session, part, now_ms).await?.is_some()) + } + + /// Reserves a part and returns the exact pending session written by the CAS. + /// # Errors + /// Rejects stale revisions, expired sessions, exhausted limits and pending mutations. + pub async fn reserve_part_state( + &self, + session: &MultipartSession, + part: &MultipartPart, + now_ms: u64, + ) -> Result, CatalogError> { check_live(session, now_ms)?; let mut after = part.clone(); after.modified_ms = now_ms; @@ -75,11 +186,11 @@ impl MultipartRepository { .ok_or(ValidationError::Record)?; next.staged_bytes = session .staged_bytes - .checked_sub(before.as_ref().map_or(0, |part| part.tree.length)) - .and_then(|bytes| bytes.checked_add(part.tree.length)) + .checked_sub(before.as_ref().map_or(0, MultipartPart::length)) + .and_then(|bytes| bytes.checked_add(part.length())) .ok_or(ValidationError::Record)?; next.pending = Some(MultipartPartMutation { before, after }); - self.exchange(session, &next).await + Ok(self.exchange(session, &next).await?.then_some(next)) } /// Helps one durable part mutation and then clears its session fence. diff --git a/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs b/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs index 2d374d6ae..943407a69 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_repository/publication.rs @@ -1,14 +1,141 @@ use crate::catalog::CatalogError; use crate::error::ValidationError; use crate::file::{ - file_key, FileContent, FileRecord, FileRepository, FileTree, MultipartPhase, MultipartSession, + file_key, ContentFormat, FileContent, FileKind, FileRecord, FileRepository, FileTree, MultipartPhase, + MultipartSelection, MultipartSession, MultipartStreamPart, SelectedPart, SelectedStreamPart, }; use crate::operation::PayloadStore; use crate::record::StorageRecord; +use crowdb_protocol::chunkdb::rpc::Location; +use md5::{Digest, Md5}; +use std::fmt::Write; use super::{check_live, increment, MultipartRepository}; impl MultipartRepository { + /// Publishes selected durable part locations without reading or rewriting part bytes. + /// # Errors + /// Rejects changed parts, mixed storage formats, oversized descriptors and stale sessions. + pub async fn prepare_stream_publication( + &self, + session: &MultipartSession, + now_ms: u64, + ) -> Result, CatalogError> { + session.validate()?; + check_live(session, now_ms)?; + if session.phase != MultipartPhase::Completing { + return Err(CatalogError::Conflict); + } + let completion = session.completion.as_ref().ok_or(ValidationError::Record)?; + if completion.progress.next_part != 0 { + return Ok(None); + } + if self.load(session.context, session.upload).await?.as_ref() != Some(session) { + return Ok(Some(false)); + } + let bytes = PayloadStore::new(self.store.clone()) + .get(&completion.selection) + .await?; + let selection = MultipartSelection::decode(&bytes)?; + let mut locations = Vec::::new(); + let mut length = 0_u64; + let mut md5 = Md5::new(); + for (index, selected) in selection.parts().iter().enumerate() { + let snapshot = selection.snapshots().and_then(|snapshots| snapshots.get(index)); + let Some(stream) = self.selected_stream(session, selected, snapshot).await? else { + return Ok(None); + }; + let etag = stream.content.etag().ok_or(ValidationError::Record)?; + for pair in etag.as_bytes().chunks_exact(2) { + let pair = std::str::from_utf8(pair).map_err(|_| ValidationError::Record)?; + md5.update([u8::from_str_radix(pair, 16).map_err(|_| ValidationError::Record)?]); + } + for mut location in stream + .content + .locations(stream.length)? + .ok_or(ValidationError::Record)? + { + location.logical_offset = location + .logical_offset + .checked_add(length) + .ok_or(ValidationError::Record)?; + locations.push(location); + } + length = length + .checked_add(stream.length) + .filter(|length| *length <= session.limits.max_file_bytes) + .ok_or(ValidationError::Record)?; + } + let mut etag = String::with_capacity(40); + for byte in md5.finalize() { + write!(&mut etag, "{byte:02x}").expect("string write cannot fail"); + } + write!(&mut etag, "-{}", selection.count()).expect("string write cannot fail"); + let content = FileContent::from_locations(&locations, length, etag)?; + let path = session.location.relative_key(); + let extension = std::path::Path::new(path).extension(); + let has_extension = |wanted: &str| extension.is_some_and(|value| value.eq_ignore_ascii_case(wanted)); + let (kind, format) = if has_extension("json") { + (FileKind::Metadata, ContentFormat::Json) + } else if has_extension("avro") { + (FileKind::Unbound, ContentFormat::Avro) + } else if has_extension("parquet") { + (FileKind::Unbound, ContentFormat::Parquet) + } else if has_extension("orc") { + (FileKind::Unbound, ContentFormat::Orc) + } else if has_extension("puffin") { + (FileKind::Unbound, ContentFormat::Puffin) + } else { + (FileKind::Unbound, ContentFormat::Opaque) + }; + let record = FileRecord { + file: session.owner.file, + location: session.location.clone(), + kind, + format, + length, + digest: [0; 32], + content, + hint: None, + }; + record.validate()?; + let value = StorageRecord::File(Box::new(record)).encode()?; + let publication = PayloadStore::new(self.store.clone()) + .put(session.context.catalog, session.upload, &value) + .await?; + let mut next = increment(session)?; + next.phase = MultipartPhase::Publishing; + let completion = next.completion.as_mut().ok_or(ValidationError::Record)?; + completion.progress.next_part = selection.count(); + completion.progress.completed_bytes = length; + completion.publication = Some(publication); + Ok(Some(self.exchange(session, &next).await?)) + } + + async fn selected_stream( + &self, + session: &MultipartSession, + selected: &SelectedPart, + snapshot: Option<&SelectedStreamPart>, + ) -> Result, CatalogError> { + if let Some(snapshot) = snapshot { + return Ok(Some(MultipartStreamPart { + length: snapshot.length, + content: FileContent::Locations { + bytes: snapshot.bytes.clone(), + etag: snapshot.etag.clone(), + }, + })); + } + let part = self + .part(session, selected.number) + .await? + .ok_or(ValidationError::Record)?; + if part.revision != selected.revision || part.selection_digest() != selected.digest { + return Err(ValidationError::Record.into()); + } + Ok(part.stream) + } /// Freezes a semantically sealed record; callers must validate its canonical format first. /// # Errors /// Rejects incomplete assembly, changed byte identity, expiry and invalid file records. @@ -67,6 +194,7 @@ impl MultipartRepository { || selected.digest != candidate.digest || selected.kind != candidate.kind || selected.format != candidate.format + || selected.content != candidate.content { return Err(ValidationError::Record.into()); } @@ -97,6 +225,7 @@ impl MultipartRepository { || selected.digest != candidate.digest || selected.kind != candidate.kind || selected.format != candidate.format + || selected.content != candidate.content { let mut next = increment(session)?; next.phase = MultipartPhase::Conflicted; @@ -114,11 +243,15 @@ impl MultipartRepository { let StorageRecord::File(record) = StorageRecord::decode(&key, &bytes)? else { return Err(ValidationError::Record.into()); }; - validate_candidate( - session, - completion.candidate.as_ref().ok_or(ValidationError::Record)?, - &record, - )?; + if let Some(tree) = &completion.candidate { + validate_candidate(session, tree, &record)?; + } else if record.file != session.owner.file + || record.location != session.location + || record.length != completion.progress.completed_bytes + || !matches!(record.content, FileContent::Locations { .. }) + { + return Err(ValidationError::Record.into()); + } Ok(*record) } } diff --git a/lib/crowdb-access-iceberg/src/file/multipart_selection.rs b/lib/crowdb-access-iceberg/src/file/multipart_selection.rs index 5d8a6597d..8ae47ec83 100644 --- a/lib/crowdb-access-iceberg/src/file/multipart_selection.rs +++ b/lib/crowdb-access-iceberg/src/file/multipart_selection.rs @@ -1,6 +1,13 @@ use crate::error::ValidationError; +use crate::operation::MAX_PAYLOAD_BYTES; +use bincode::Options; +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; -const MAGIC: &[u8; 5] = b"ICMS\x01"; +use super::{FileContent, MultipartPart}; + +const MAGIC_V1: &[u8; 5] = b"ICMS\x01"; +const MAGIC_V2: &[u8; 5] = b"ICMS\x02"; const ENTRY_BYTES: usize = 42; #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -14,6 +21,14 @@ pub struct SelectedPart { pub struct MultipartSelection { parts: Vec, count: u16, + snapshots: Option>, +} + +#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)] +pub struct SelectedStreamPart { + pub length: u64, + pub bytes: Vec, + pub etag: String, } impl MultipartSelection { @@ -31,7 +46,60 @@ impl MultipartSelection { previous = part.number; } let count = u16::try_from(parts.len()).map_err(|_| ValidationError::Record)?; - Ok(Self { parts, count }) + Ok(Self { + parts, + count, + snapshots: None, + }) + } + + /// Captures complete streamed part locations at Complete so later part + /// replacements cannot change the selected file or its GC references. + /// # Errors + /// Rejects mixed legacy parts, invalid descriptors or incoherent digests. + pub fn with_stream_parts(parts: &[MultipartPart]) -> Result { + let mut estimated_bytes = 7_usize + .checked_add(parts.len().saturating_mul(ENTRY_BYTES + 64)) + .ok_or(ValidationError::RecordTooLarge)?; + for part in parts { + let stream = part.stream.as_ref().ok_or(ValidationError::Record)?; + let FileContent::Locations { bytes, etag } = &stream.content else { + return Err(ValidationError::Record); + }; + estimated_bytes = estimated_bytes + .checked_add(bytes.len()) + .and_then(|size| size.checked_add(etag.len())) + .ok_or(ValidationError::RecordTooLarge)?; + if estimated_bytes > MAX_PAYLOAD_BYTES { + return Err(ValidationError::RecordTooLarge); + } + } + let selected = parts + .iter() + .map(|part| SelectedPart { + number: part.number, + revision: part.revision, + digest: part.selection_digest(), + }) + .collect(); + let mut selection = Self::new(selected)?; + let snapshots = parts + .iter() + .map(|part| { + let stream = part.stream.as_ref().ok_or(ValidationError::Record)?; + let FileContent::Locations { bytes, etag } = &stream.content else { + return Err(ValidationError::Record); + }; + Ok(SelectedStreamPart { + length: stream.length, + bytes: bytes.clone(), + etag: etag.clone(), + }) + }) + .collect::, ValidationError>>()?; + validate_snapshots(&selection.parts, &snapshots)?; + selection.snapshots = Some(snapshots); + Ok(selection) } #[must_use] @@ -44,30 +112,53 @@ impl MultipartSelection { self.count } + #[must_use] + pub fn snapshots(&self) -> Option<&[SelectedStreamPart]> { + self.snapshots.as_deref() + } + + /// # Panics + /// Panics if bincode cannot serialize an already validated in-memory snapshot. #[must_use] pub fn encode(&self) -> Vec { let mut bytes = Vec::with_capacity(7 + ENTRY_BYTES * self.parts.len()); - bytes.extend_from_slice(MAGIC); + bytes.extend_from_slice(if self.snapshots.is_some() { + MAGIC_V2 + } else { + MAGIC_V1 + }); bytes.extend_from_slice(&self.count.to_be_bytes()); for part in &self.parts { bytes.extend_from_slice(&part.number.to_be_bytes()); bytes.extend_from_slice(&part.revision.to_be_bytes()); bytes.extend_from_slice(&part.digest); } + if let Some(snapshots) = &self.snapshots { + bytes.extend_from_slice(&bincode::serialize(snapshots).expect("validated snapshots serialize")); + } bytes } /// # Errors /// Rejects unknown versions, invalid framing and noncanonical part sequences. pub fn decode(bytes: &[u8]) -> Result { - if bytes.len() < 7 || bytes.get(..5) != Some(MAGIC) { + if bytes.len() > MAX_PAYLOAD_BYTES { + return Err(ValidationError::RecordTooLarge); + } + let version = bytes.get(..5).ok_or(ValidationError::Record)?; + if bytes.len() < 7 || (version != MAGIC_V1 && version != MAGIC_V2) { return Err(ValidationError::Record); } let count = usize::from(u16::from_be_bytes([bytes[5], bytes[6]])); - if count == 0 || count > 10_000 || bytes.len() != 7 + count * ENTRY_BYTES { + let entries_end = 7 + count * ENTRY_BYTES; + if count == 0 + || count > 10_000 + || bytes.len() < entries_end + || (version == MAGIC_V1 && bytes.len() != entries_end) + { return Err(ValidationError::Record); } - let parts = bytes[7..] + let parts = bytes[7..entries_end] .chunks_exact(ENTRY_BYTES) .map(|entry| { Ok(SelectedPart { @@ -79,6 +170,52 @@ impl MultipartSelection { }) }) .collect::>()?; - Self::new(parts) + let mut selection = Self::new(parts)?; + if version == MAGIC_V2 { + let encoded = &bytes[entries_end..]; + let snapshot_count = u64::from_le_bytes( + encoded + .get(..8) + .ok_or(ValidationError::Record)? + .try_into() + .map_err(|_| ValidationError::Record)?, + ); + if snapshot_count != count as u64 { + return Err(ValidationError::Record); + } + let snapshots: Vec = bincode::DefaultOptions::new() + .with_fixint_encoding() + .with_limit(MAX_PAYLOAD_BYTES as u64) + .reject_trailing_bytes() + .deserialize(encoded) + .map_err(|_| ValidationError::Record)?; + validate_snapshots(&selection.parts, &snapshots)?; + selection.snapshots = Some(snapshots); + } + Ok(selection) + } +} + +fn validate_snapshots( + parts: &[SelectedPart], + snapshots: &[SelectedStreamPart], +) -> Result<(), ValidationError> { + if parts.len() != snapshots.len() { + return Err(ValidationError::Record); + } + for (part, snapshot) in parts.iter().zip(snapshots) { + let content = FileContent::Locations { + bytes: snapshot.bytes.clone(), + etag: snapshot.etag.clone(), + }; + content.validate(snapshot.length, &[0; 32])?; + let mut digest = Sha256::new(); + digest.update(snapshot.length.to_le_bytes()); + digest.update(&snapshot.bytes); + digest.update(snapshot.etag.as_bytes()); + if <[u8; 32]>::from(digest.finalize()) != part.digest { + return Err(ValidationError::Record); + } } + Ok(()) } diff --git a/lib/crowdb-access-iceberg/src/file/reader.rs b/lib/crowdb-access-iceberg/src/file/reader.rs index 739f96a97..1c17e717f 100644 --- a/lib/crowdb-access-iceberg/src/file/reader.rs +++ b/lib/crowdb-access-iceberg/src/file/reader.rs @@ -21,6 +21,7 @@ pub struct FileReader { cached: Option<(u64, Vec)>, leaf_directory: Option<(ChunkRoot, Vec)>, digest: Option, + locations: Option>, failed: bool, } @@ -94,7 +95,9 @@ impl FileReader { .content .inline_bytes(record.length, &record.digest)? .map(|bytes| (0, bytes)); - let digest = (range.start == 0 && range.end == record.length).then(Sha256::new); + let digest = (range.start == 0 && range.end == record.length && record.content.etag().is_none()) + .then(Sha256::new); + let locations = record.content.locations(record.length)?; Ok(Self { store, record, @@ -104,6 +107,7 @@ impl FileReader { cached, leaf_directory: None, digest, + locations, failed: false, }) } @@ -161,6 +165,15 @@ impl FileReader { } async fn select_leaf(&mut self) -> Result<(u64, Vec), FileIoError> { + if let Some(locations) = &self.locations { + let start = self.cursor; + let end = start.saturating_add(self.frame_bytes as u64).min(self.end); + let bytes = self.store.read_locations(locations, start, end).await?; + if bytes.len() as u64 != end - start { + return Err(ValidationError::Record.into()); + } + return Ok((start, bytes)); + } let FileContent::Chunks { root: Some(root) } = &self.record.content else { return Err(ValidationError::Record.into()); }; diff --git a/lib/crowdb-access-iceberg/src/file/record.rs b/lib/crowdb-access-iceberg/src/file/record.rs index f635862c1..705b4ea69 100644 --- a/lib/crowdb-access-iceberg/src/file/record.rs +++ b/lib/crowdb-access-iceberg/src/file/record.rs @@ -32,6 +32,7 @@ pub enum ContentFormat { Parquet = 2, Orc = 3, Puffin = 4, + Opaque = 5, } #[derive(Clone, Copy, Debug, Eq, PartialEq)] @@ -69,7 +70,11 @@ impl FileRecord { FileKind::Statistics => matches!(self.format, ContentFormat::Puffin | ContentFormat::Parquet), FileKind::Unbound => matches!( self.format, - ContentFormat::Avro | ContentFormat::Parquet | ContentFormat::Orc | ContentFormat::Puffin + ContentFormat::Avro + | ContentFormat::Parquet + | ContentFormat::Orc + | ContentFormat::Puffin + | ContentFormat::Opaque ), }; if !valid_format { diff --git a/lib/crowdb-access-iceberg/src/file/repository.rs b/lib/crowdb-access-iceberg/src/file/repository.rs index f445d1ffe..6d348cb74 100644 --- a/lib/crowdb-access-iceberg/src/file/repository.rs +++ b/lib/crowdb-access-iceberg/src/file/repository.rs @@ -31,7 +31,7 @@ impl FileRepository { Ok(result) } - /// Publishes a candidate; callers must verify chunk bytes and format before calling. + /// Publishes a candidate after callers verify transfer integrity and durable storage. /// # Errors /// Rejects invalid records, changed content, retired contexts and uncertain writes. pub async fn publish( @@ -40,46 +40,7 @@ impl FileRepository { candidate: &FileRecord, ) -> Result { candidate.validate()?; - let pin = Box::pin(self.publication_pin(context, &candidate.location)).await?; - let result = self.publish_inner(context, candidate).await; - if result.is_ok() { - if let Some(pin) = pin { - crate::gc::ReaderPins::new(self.store.clone()) - .release(&pin) - .await?; - } - } - result - } - - async fn publication_pin( - &self, - context: CatalogContext, - location: &FileLocation, - ) -> Result, CatalogError> { - let key = crate::table::head_key(context.catalog, location.table().table); - if self.store.get(&key.encode()?).await?.is_none() { - return Ok(None); - } - let now_ms = u64::try_from( - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| ValidationError::Deadline)? - .as_millis(), - ) - .map_err(|_| ValidationError::Deadline)?; - let pins = crate::gc::ReaderPins::new(self.store.clone()); - let expiry = pins.request_expiry(context, now_ms).await?; - Ok(Some( - pins.protect_files( - context, - location.table().table, - "file-publication", - expiry, - now_ms, - ) - .await?, - )) + self.publish_inner(context, candidate).await } async fn publish_inner( @@ -241,6 +202,7 @@ fn compatible(existing: FileRecord, candidate: &FileRecord) -> Result Ok(None), } } } diff --git a/lib/crowdb-access-iceberg/src/gc.rs b/lib/crowdb-access-iceberg/src/gc.rs index 6c4b6ed4d..934b24a58 100644 --- a/lib/crowdb-access-iceberg/src/gc.rs +++ b/lib/crowdb-access-iceberg/src/gc.rs @@ -9,9 +9,7 @@ mod limits; mod mark; mod node; mod page; -mod pins; mod proof; -mod protection; mod reachability; mod repository; mod retirement; @@ -25,7 +23,6 @@ pub use limits::GcLimits; pub use mark::GcMarkError; pub use node::GcNode; pub use page::GcPage; -pub use pins::{GcPin, ReaderPins}; pub use proof::GcProofState; pub use reachability::{ avro_links, metadata_links, AvroMarkCursor, AvroMarkLimits, AvroMarkPage, ReachableFile, ReachableKind, diff --git a/lib/crowdb-access-iceberg/src/gc/candidate.rs b/lib/crowdb-access-iceberg/src/gc/candidate.rs index 6ae1121ad..515420b52 100644 --- a/lib/crowdb-access-iceberg/src/gc/candidate.rs +++ b/lib/crowdb-access-iceberg/src/gc/candidate.rs @@ -75,6 +75,7 @@ impl GcCandidate { owner: session.owner, frames: Vec::new(), pending: None, + next_location: 0, }) } else { TreeReclaimCursor::new(&self.file) @@ -89,16 +90,26 @@ impl GcCandidate { .owner .table .file(&format!("gc-parts/{}/{:05}.parquet", part.upload, part.number))?; + let (length, digest, content) = if let Some(tree) = &part.tree { + ( + tree.length, + tree.digest, + FileContent::Chunks { + root: tree.root.clone(), + }, + ) + } else { + let stream = part.stream.as_ref().ok_or(ValidationError::Record)?; + (stream.length, [0; 32], stream.content.clone()) + }; let file = FileRecord { file: part.owner.file, location, kind: FileKind::Unbound, format: ContentFormat::Parquet, - length: part.tree.length, - digest: part.tree.digest, - content: FileContent::Chunks { - root: part.tree.root.clone(), - }, + length, + digest, + content, hint: None, }; file.validate()?; @@ -133,6 +144,16 @@ impl GcCandidate { pub fn validate(&self) -> Result<(), ValidationError> { self.file.validate()?; self.cursor.validate()?; + if let Some(locations) = self.file.content.locations(self.file.length)? { + if usize::from(self.cursor.next_location) > locations.len() + || !self.cursor.frames.is_empty() + || self.cursor.pending.is_some() + { + return Err(ValidationError::Record); + } + } else if self.cursor.next_location != 0 { + return Err(ValidationError::Record); + } if let Some(session) = &self.assembly { if self.part.is_some() || Self::assembly_file(session)? != self.file diff --git a/lib/crowdb-access-iceberg/src/gc/discovery.rs b/lib/crowdb-access-iceberg/src/gc/discovery.rs index 2ba1f7ded..5f64fcfdc 100644 --- a/lib/crowdb-access-iceberg/src/gc/discovery.rs +++ b/lib/crowdb-access-iceberg/src/gc/discovery.rs @@ -232,6 +232,39 @@ impl GcRepository { self.session_is_abandoned(task, &session, now_ms).await } + pub(super) async fn part_is_selected( + &self, + task: &GcTask, + part: &crate::file::MultipartPart, + ) -> Result { + let key = IcebergKey::Catalog { + catalog: task.context.catalog, + scope: CatalogScope::MultipartSession, + suffix: part.upload.as_bytes().to_vec(), + }; + let value = self + .store + .get(&key.encode()?) + .await? + .ok_or(ValidationError::Record)?; + let StorageRecord::MultipartSession(session) = StorageRecord::decode(&key, &value.bytes)? else { + return Err(ValidationError::Record.into()); + }; + if session.phase != MultipartPhase::Published { + return Ok(false); + } + let completion = session.completion.as_ref().ok_or(ValidationError::Record)?; + let bytes = crate::operation::PayloadStore::new(self.store.clone()) + .get(&completion.selection) + .await?; + let selection = crate::file::MultipartSelection::decode(&bytes)?; + Ok(selection.parts().iter().any(|selected| { + selected.number == part.number + && selected.revision == part.revision + && selected.digest == part.selection_digest() + })) + } + pub(super) async fn session_is_abandoned( &self, task: &GcTask, diff --git a/lib/crowdb-access-iceberg/src/gc/pins.rs b/lib/crowdb-access-iceberg/src/gc/pins.rs deleted file mode 100644 index c5570a12d..000000000 --- a/lib/crowdb-access-iceberg/src/gc/pins.rs +++ /dev/null @@ -1,165 +0,0 @@ -use std::sync::Arc; - -use crate::{ - catalog::{check_context, CasOutcome, CatalogContext, CatalogError, CatalogStore}, - error::ValidationError, - key::{CatalogScope, IcebergKey, OperationId}, - operation::mutation_identity, - record::StorageRecord, - table::{head_key, TableHead, TableLifecycle}, -}; - -#[derive(Clone, Debug, Eq, PartialEq)] -pub struct GcPin { - pub context: CatalogContext, - pub identity: OperationId, - pub head: TableHead, - pub principal: String, - pub expires_ms: u64, - pub released: bool, - pub operator: bool, - pub protects_uploads: bool, -} - -#[derive(Clone)] -pub struct ReaderPins { - pub(super) store: Arc, -} - -impl ReaderPins { - #[must_use] - pub fn new(store: Arc) -> Self { - Self { store } - } - - /// # Errors - /// Rejects malformed or mismatched persisted pins. - pub async fn get( - &self, - catalog: crate::key::CatalogId, - table: crate::key::TableId, - identity: OperationId, - ) -> Result, CatalogError> { - let mut suffix = table.as_bytes().to_vec(); - suffix.extend_from_slice(identity.as_bytes()); - let key = IcebergKey::Catalog { - catalog, - scope: CatalogScope::GcPin, - suffix, - }; - let Some(value) = self.store.get(&key.encode()?).await? else { - return Ok(None); - }; - let StorageRecord::GcPin(pin) = StorageRecord::decode(&key, &value.bytes)? else { - return Err(ValidationError::Record.into()); - }; - if pin.context.catalog != catalog || pin.head.table != table || pin.identity != identity { - return Err(ValidationError::IdentityMismatch.into()); - } - Ok(Some(*pin)) - } - - /// Persists protection before checking the selected head against a concurrent sweep. - /// # Errors - /// Rejects retired catalogs, changed heads and reused identities. - pub async fn acquire(&self, pin: &GcPin) -> Result<(), CatalogError> { - self.acquire_selected(pin, false).await - } - - pub(super) async fn acquire_selected( - &self, - pin: &GcPin, - allow_tombstone: bool, - ) -> Result<(), CatalogError> { - pin.validate()?; - if pin.released - || pin.head.lifecycle == TableLifecycle::Reclaiming - || (!pin.operator && !allow_tombstone && pin.head.lifecycle != TableLifecycle::Ready) - { - return Err(CatalogError::Busy); - } - check_context(self.store.as_ref(), pin.context).await?; - let key = pin.key().encode()?; - let bytes = StorageRecord::GcPin(Box::new(pin.clone())).encode()?; - match self - .store - .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) - .await? - { - CasOutcome::Applied(_) => {} - CasOutcome::Conflict(Some(existing)) if existing.bytes == bytes => {} - CasOutcome::Conflict(_) => return Err(CatalogError::Conflict), - } - let key = head_key(pin.head.catalog, pin.head.table); - let current = self - .store - .get(&key.encode()?) - .await? - .ok_or(CatalogError::Conflict)?; - if StorageRecord::decode(&key, ¤t.bytes)? - != StorageRecord::TableHead(Box::new(pin.head.clone())) - { - return Err(CatalogError::Busy); - } - check_context(self.store.as_ref(), pin.context).await - } - - /// # Errors - /// Rejects a changed pin; retries preserve the same release result. - pub async fn release(&self, pin: &GcPin) -> Result<(), CatalogError> { - pin.validate()?; - let mut released = pin.clone(); - released.released = true; - let key = pin.key().encode()?; - let before = StorageRecord::GcPin(Box::new(pin.clone())).encode()?; - let after = StorageRecord::GcPin(Box::new(released)).encode()?; - match self - .store - .compare_exchange( - &key, - Some(&before), - &after, - mutation_identity(&key, Some(&before), &after), - ) - .await? - { - CasOutcome::Applied(_) => Ok(()), - CasOutcome::Conflict(Some(existing)) if existing.bytes == after => Ok(()), - CasOutcome::Conflict(_) => Err(CatalogError::Conflict), - } - } -} - -impl GcPin { - #[must_use] - pub fn key(&self) -> IcebergKey { - let mut suffix = self.head.table.as_bytes().to_vec(); - suffix.extend_from_slice(self.identity.as_bytes()); - IcebergKey::Catalog { - catalog: self.context.catalog, - scope: CatalogScope::GcPin, - suffix, - } - } - - /// # Errors - /// Rejects mismatched roots, unbounded principals and unbounded reader lifetimes. - pub fn validate(&self) -> Result<(), ValidationError> { - self.context.validate()?; - self.head.validate()?; - if self.head.catalog != self.context.catalog - || self.principal.is_empty() - || self.principal.len() > 256 - || self.principal.contains('\0') - || (!self.operator && self.expires_ms == 0) - { - return Err(ValidationError::Record); - } - Ok(()) - } - - #[must_use] - pub const fn protects(&self, now_ms: u64) -> bool { - !self.released && (self.expires_ms == 0 || now_ms < self.expires_ms) - } -} diff --git a/lib/crowdb-access-iceberg/src/gc/protection.rs b/lib/crowdb-access-iceberg/src/gc/protection.rs deleted file mode 100644 index b7866a30a..000000000 --- a/lib/crowdb-access-iceberg/src/gc/protection.rs +++ /dev/null @@ -1,150 +0,0 @@ -use crate::{ - catalog::{check_context, CatalogContext, CatalogError, CatalogLifecycle}, - commit::{TableCreateJournal, TableCreatePhase}, - error::ValidationError, - key::{CatalogScope, IcebergKey, OperationId, TableId}, - operation::mutation_identity, - record::StorageRecord, - table::head_key, -}; - -use super::{GcPin, ReaderPins}; - -impl ReaderPins { - /// # Errors - /// Rejects inactive authority and overflowing persisted protection bounds. - pub async fn request_expiry(&self, context: CatalogContext, starts_ms: u64) -> Result { - check_context(self.store.as_ref(), context).await?; - let key = IcebergKey::Catalog { - catalog: context.catalog, - scope: CatalogScope::Authority, - suffix: Vec::new(), - }; - let value = self - .store - .get(&key.encode()?) - .await? - .ok_or(ValidationError::Record)?; - let StorageRecord::Authority(authority) = StorageRecord::decode(&key, &value.bytes)? else { - return Err(ValidationError::Record.into()); - }; - if authority.lifecycle != CatalogLifecycle::Ready || authority.admission_bounds.request_ms == 0 { - return Err(CatalogError::Busy); - } - starts_ms - .checked_add(authority.admission_bounds.request_ms) - .and_then(|deadline| deadline.checked_add(authority.admission_bounds.clock_skew_ms)) - .ok_or(ValidationError::Deadline.into()) - } - - /// # Errors - /// Rejects fenced tables and missing or changed staged-create authority. - pub async fn protect_files( - &self, - context: CatalogContext, - table: TableId, - principal: &str, - expires_ms: u64, - now_ms: u64, - ) -> Result { - self.protect_selected_files(context, table, principal, expires_ms, now_ms, false) - .await - } - - /// Protects an authenticated file read, including a logically dropped table. - /// # Errors - /// Rejects reclamation fences and changed authority after persisting the pin. - pub async fn protect_file_reads( - &self, - context: CatalogContext, - table: TableId, - principal: &str, - expires_ms: u64, - now_ms: u64, - ) -> Result { - self.protect_selected_files(context, table, principal, expires_ms, now_ms, true) - .await - } - - async fn protect_selected_files( - &self, - context: CatalogContext, - table: TableId, - principal: &str, - expires_ms: u64, - now_ms: u64, - allow_tombstone: bool, - ) -> Result { - if expires_ms <= now_ms { - return Err(ValidationError::Deadline.into()); - } - let key = head_key(context.catalog, table); - let current = self.store.get(&key.encode()?).await?; - let stage = if current.is_none() { - let identity = OperationId::from_bytes(table.as_bytes())?; - let operation = TableCreateJournal::new(self.store.clone()) - .load(context, identity) - .await? - .ok_or(CatalogError::Conflict)?; - if operation.phase != TableCreatePhase::Staged - || operation.candidate.table != table - || operation.stage.as_ref().map_or(true, |stage| { - u64::try_from(stage.expires_ms).map_or(true, |expiry| now_ms >= expiry) - }) - { - return Err(CatalogError::Busy); - } - Some(operation) - } else { - None - }; - let head = if let Some(value) = current { - let StorageRecord::TableHead(head) = StorageRecord::decode(&key, &value.bytes)? else { - return Err(ValidationError::Record.into()); - }; - *head - } else { - stage.as_ref().ok_or(ValidationError::Record)?.candidate.clone() - }; - let pin = GcPin { - context, - identity: OperationId::random(), - head, - principal: principal.into(), - expires_ms, - released: false, - operator: false, - protects_uploads: true, - }; - if let Some(operation) = stage { - pin.validate()?; - let key = pin.key().encode()?; - let bytes = StorageRecord::GcPin(Box::new(pin.clone())).encode()?; - match self - .store - .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) - .await? - { - crate::catalog::CasOutcome::Applied(_) => {} - crate::catalog::CasOutcome::Conflict(_) => return Err(CatalogError::Conflict), - } - if TableCreateJournal::new(self.store.clone()) - .load(context, operation.identity.operation) - .await? - .as_ref() - != Some(&operation) - || self - .store - .get(&head_key(context.catalog, table).encode()?) - .await? - .is_some() - { - return Err(CatalogError::Busy); - } - check_context(self.store.as_ref(), context).await?; - } else { - self.acquire_selected(&pin, allow_tombstone).await?; - } - Ok(pin) - } -} diff --git a/lib/crowdb-access-iceberg/src/gc/tree.rs b/lib/crowdb-access-iceberg/src/gc/tree.rs index 2027e0205..b86373984 100644 --- a/lib/crowdb-access-iceberg/src/gc/tree.rs +++ b/lib/crowdb-access-iceberg/src/gc/tree.rs @@ -17,6 +17,7 @@ pub struct TreeReclaimCursor { pub owner: FileIdentity, pub frames: Vec, pub pending: Option, + pub next_location: u16, } #[derive(Clone, Debug, Eq, PartialEq)] @@ -46,6 +47,7 @@ impl TreeReclaimCursor { }, frames, pending: None, + next_location: 0, }) } diff --git a/lib/crowdb-access-iceberg/src/gc/worker/cleanup.rs b/lib/crowdb-access-iceberg/src/gc/worker/cleanup.rs index e618b637e..5071ae863 100644 --- a/lib/crowdb-access-iceberg/src/gc/worker/cleanup.rs +++ b/lib/crowdb-access-iceberg/src/gc/worker/cleanup.rs @@ -209,7 +209,6 @@ impl GcWorker { .await? .is_none()) } - StorageRecord::GcPin(pin) => Ok(!pin.protects(now_ms)), _ => Ok(true), } } diff --git a/lib/crowdb-access-iceberg/src/gc/worker/inactive.rs b/lib/crowdb-access-iceberg/src/gc/worker/inactive.rs index 734481530..420532401 100644 --- a/lib/crowdb-access-iceberg/src/gc/worker/inactive.rs +++ b/lib/crowdb-access-iceberg/src/gc/worker/inactive.rs @@ -129,7 +129,6 @@ pub(super) fn protects(task: &GcTask, record: &StorageRecord, now_ms: u64, grace .saturating_add(grace_ms) }; match record { - StorageRecord::GcPin(pin) => owns(pin.head.table) && pin.protects(now_ms), StorageRecord::TableCommitOperation(operation) => { owns(operation.before.table) && (retained(operation.identity.issued_ms) diff --git a/lib/crowdb-access-iceberg/src/gc/worker/live.rs b/lib/crowdb-access-iceberg/src/gc/worker/live.rs index bd36069d5..33372a6f1 100644 --- a/lib/crowdb-access-iceberg/src/gc/worker/live.rs +++ b/lib/crowdb-access-iceberg/src/gc/worker/live.rs @@ -105,18 +105,6 @@ impl GcWorker { continue; } let record = StorageRecord::decode(&key, &item.value)?; - if let StorageRecord::GcPin(pin) = &record { - if pin.protects(now_ms) - && task - .head - .as_ref() - .is_some_and(|head| head.table == pin.head.table) - && !pin.protects_uploads - { - self.repository.push_proof_root(&mut next, &pin.head).await?; - continue; - } - } if super::inactive::protects(task, &record, now_ms, grace_ms) { next.stalled = GcStalledReason::Protected; next.scan_after.clear(); diff --git a/lib/crowdb-access-iceberg/src/gc/worker/sweep.rs b/lib/crowdb-access-iceberg/src/gc/worker/sweep.rs index 01d8bcba7..37d291926 100644 --- a/lib/crowdb-access-iceberg/src/gc/worker/sweep.rs +++ b/lib/crowdb-access-iceberg/src/gc/worker/sweep.rs @@ -1,4 +1,5 @@ use crate::{catalog::CasOutcome, operation::mutation_identity}; +use crowdb_protocol::chunkdb::rpc::Location; use super::{ file_key, location_key, CandidatePhase, CatalogError, GcCandidate, GcPhase, GcStalledReason, GcTask, @@ -179,6 +180,23 @@ impl GcWorker { self.repository.candidate(Some(candidate), &next).await?; return Ok(DeleteProgress::Advanced); } + if candidate.part.as_ref().is_some_and(|part| part.stream.is_some()) + && self + .repository + .part_is_selected(task, candidate.part.as_ref().ok_or(ValidationError::Record)?) + .await? + { + self.remove_file_authority(candidate).await?; + next.phase = CandidatePhase::Complete; + next.completed_round = sweep_round; + self.repository.candidate(Some(candidate), &next).await?; + return Ok(DeleteProgress::Complete); + } + if let Some(locations) = candidate.file.content.locations(candidate.file.length)? { + return self + .delete_location_candidate(candidate, next, &locations, sweep_round) + .await; + } if let Some(root) = &candidate.cursor.pending { if self.blocks.reclaim(root).await? == ReclaimOutcome::Deferred { next.phase = CandidatePhase::Deferred; @@ -213,6 +231,37 @@ impl GcWorker { }) } + async fn delete_location_candidate( + &self, + candidate: &GcCandidate, + mut next: GcCandidate, + locations: &[Location], + sweep_round: u64, + ) -> Result { + if let Some(location) = locations.get(usize::from(candidate.cursor.next_location)) { + if self.blocks.reclaim_location(location).await? == ReclaimOutcome::Deferred { + next.phase = CandidatePhase::Deferred; + self.repository.candidate(Some(candidate), &next).await?; + return Ok(DeleteProgress::Deferred); + } + next.cursor.next_location = next + .cursor + .next_location + .checked_add(1) + .ok_or(ValidationError::GenerationExhausted)?; + } else { + self.remove_file_authority(candidate).await?; + next.phase = CandidatePhase::Complete; + next.completed_round = sweep_round; + } + self.repository.candidate(Some(candidate), &next).await?; + Ok(if next.phase == CandidatePhase::Complete { + DeleteProgress::Complete + } else { + DeleteProgress::Advanced + }) + } + async fn remove_file_authority(&self, candidate: &GcCandidate) -> Result<(), GcWorkError> { if candidate.assembly.is_some() { return Ok(()); diff --git a/lib/crowdb-access-iceberg/src/key/codec.rs b/lib/crowdb-access-iceberg/src/key/codec.rs index 4e57022fd..a717a3c82 100644 --- a/lib/crowdb-access-iceberg/src/key/codec.rs +++ b/lib/crowdb-access-iceberg/src/key/codec.rs @@ -44,7 +44,6 @@ pub enum CatalogScope { GcTask = 18, GcCandidate = 19, GcPage = 20, - GcPin = 21, GcNode = 22, GcPending = 23, GcClaim = 24, @@ -194,7 +193,6 @@ fn catalog_scope(value: u8) -> Result { 18 => Ok(CatalogScope::GcTask), 19 => Ok(CatalogScope::GcCandidate), 20 => Ok(CatalogScope::GcPage), - 21 => Ok(CatalogScope::GcPin), 22 => Ok(CatalogScope::GcNode), 23 => Ok(CatalogScope::GcPending), 24 => Ok(CatalogScope::GcClaim), @@ -281,13 +279,6 @@ fn validate_catalog(scope: CatalogScope, suffix: &[u8]) -> Result<(), Validation } super::OperationId::from_bytes(&suffix[..16]).map(|_| ()) } - CatalogScope::GcPin => { - if suffix.len() != 32 { - return Err(ValidationError::Key); - } - super::TableId::from_bytes(&suffix[..16])?; - super::OperationId::from_bytes(&suffix[16..]).map(|_| ()) - } CatalogScope::GcNode | CatalogScope::GcPending => { if suffix.len() != 32 { return Err(ValidationError::Key); diff --git a/lib/crowdb-access-iceberg/src/record/envelope.rs b/lib/crowdb-access-iceberg/src/record/envelope.rs index 5ac1e5d8e..eb0581cd2 100644 --- a/lib/crowdb-access-iceberg/src/record/envelope.rs +++ b/lib/crowdb-access-iceberg/src/record/envelope.rs @@ -21,7 +21,6 @@ pub enum StorageRecord { GcTask(Box), GcCandidate(Box), GcPage(Box), - GcPin(Box), TableLifecycleOperation(Box), TablePurgeTask(Box), TableCreateOperation(Box), @@ -78,7 +77,7 @@ impl StorageRecord { FBRecordValue::FBFileWriteIntent, super::write_intent::encode(builder, intent)?.as_union_value(), ), - Self::GcNode(_) | Self::GcTask(_) | Self::GcCandidate(_) | Self::GcPage(_) | Self::GcPin(_) => { + Self::GcNode(_) | Self::GcTask(_) | Self::GcCandidate(_) | Self::GcPage(_) => { return Err(ValidationError::Record); } Self::TableLifecycleOperation(operation) => ( @@ -186,10 +185,6 @@ impl StorageRecord { FBRecordValue::FBGcPage, super::gc::encode_page(builder, page)?.as_union_value(), ), - Self::GcPin(pin) => ( - FBRecordValue::FBGcPin, - super::gc::encode_pin(builder, pin)?.as_union_value(), - ), _ => return Ok(None), })) } @@ -243,11 +238,6 @@ impl StorageRecord { envelope.value_as_fbgc_page().ok_or(ValidationError::Record)?, )?))) } - FBRecordValue::FBGcPin => { - return Ok(Self::GcPin(Box::new(super::gc::decode_pin( - envelope.value_as_fbgc_pin().ok_or(ValidationError::Record)?, - )?))) - } _ => {} } if matches!( diff --git a/lib/crowdb-access-iceberg/src/record/file.rs b/lib/crowdb-access-iceberg/src/record/file.rs index 1018e90e2..366b80cc7 100644 --- a/lib/crowdb-access-iceberg/src/record/file.rs +++ b/lib/crowdb-access-iceberg/src/record/file.rs @@ -18,7 +18,7 @@ pub(super) fn encode<'buffer>( let file_id = builder.create_vector(record.file.as_bytes()); let location = builder.create_string(&record.location.to_string()); let digest = builder.create_vector(&record.digest); - let (storage, inline_bytes, root) = match &record.content { + let (storage, inline_bytes, root, locations, etag) = match &record.content { FileContent::Inline { codec, bytes } => ( match codec { InlineCodec::Raw => 0, @@ -26,8 +26,23 @@ pub(super) fn encode<'buffer>( }, Some(builder.create_vector(bytes)), None, + None, + None, + ), + FileContent::Chunks { root } => ( + 2, + None, + root.as_ref().map(|root| encode_root(builder, root)), + None, + None, + ), + FileContent::Locations { bytes, etag } => ( + 3, + None, + None, + Some(builder.create_vector(bytes)), + Some(builder.create_string(etag)), ), - FileContent::Chunks { root } => (2, None, root.as_ref().map(|root| encode_root(builder, root))), }; Ok(FBFileRecord::create( builder, @@ -44,13 +59,21 @@ pub(super) fn encode<'buffer>( has_hint: record.hint.is_some(), hint_offset: record.hint.map_or(0, |hint| hint.offset), hint_length: record.hint.map_or(0, |hint| hint.length), + locations, + etag, }, )) } pub(super) fn decode(value: FBFileRecord<'_>) -> Result { - let content = match (value.storage(), value.inline_bytes(), value.root()) { - (codec @ (0 | 1), Some(bytes), None) => FileContent::Inline { + let content = match ( + value.storage(), + value.inline_bytes(), + value.root(), + value.locations(), + value.etag(), + ) { + (codec @ (0 | 1), Some(bytes), None, None, None) => FileContent::Inline { codec: if codec == 0 { InlineCodec::Raw } else { @@ -58,9 +81,13 @@ pub(super) fn decode(value: FBFileRecord<'_>) -> Result FileContent::Chunks { + (2, None, root, None, None) => FileContent::Chunks { root: root.map(decode_root).transpose()?, }, + (3, None, None, Some(locations), Some(etag)) => FileContent::Locations { + bytes: locations.bytes().to_vec(), + etag: etag.to_owned(), + }, _ => return Err(ValidationError::Record), }; let record = FileRecord { @@ -84,6 +111,7 @@ pub(super) fn decode(value: FBFileRecord<'_>) -> Result ContentFormat::Parquet, 3 => ContentFormat::Orc, 4 => ContentFormat::Puffin, + 5 => ContentFormat::Opaque, _ => return Err(ValidationError::Record), }, length: value.length(), diff --git a/lib/crowdb-access-iceberg/src/record/gc.rs b/lib/crowdb-access-iceberg/src/record/gc.rs index 137c5c9f7..fbcac4242 100644 --- a/lib/crowdb-access-iceberg/src/record/gc.rs +++ b/lib/crowdb-access-iceberg/src/record/gc.rs @@ -1,6 +1,6 @@ use crowdb_protocol::iceberg_fb::{ FBGcCandidate, FBGcCandidateArgs, FBGcEntry, FBGcEntryArgs, FBGcFrame, FBGcFrameArgs, FBGcPage, - FBGcPageArgs, FBGcPin, FBGcPinArgs, FBGcTask, FBGcTaskArgs, + FBGcPageArgs, FBGcTask, FBGcTaskArgs, }; use flatbuffers::{FlatBufferBuilder, WIPOffset}; @@ -9,8 +9,8 @@ use crate::{ error::ValidationError, file::FileIdentity, gc::{ - CandidatePhase, GcCandidate, GcPage, GcPhase, GcPin, GcStalledReason, GcTask, GcTaskKind, - ReclaimFrame, TreeReclaimCursor, + CandidatePhase, GcCandidate, GcPage, GcPhase, GcStalledReason, GcTask, GcTaskKind, ReclaimFrame, + TreeReclaimCursor, }, key::{CatalogId, OperationId}, }; @@ -212,6 +212,7 @@ pub(super) fn encode_candidate<'buffer>( next_root: candidate.next_root, frames: Some(frames), pending, + next_location: candidate.cursor.next_location, }, )) } @@ -266,6 +267,7 @@ pub(super) fn decode_candidate(value: FBGcCandidate<'_>) -> Result) -> Result( - builder: &mut FlatBufferBuilder<'buffer>, - pin: &GcPin, -) -> Result>, ValidationError> { - pin.validate()?; - let catalog = builder.create_vector(pin.context.catalog.as_bytes()); - let identity = builder.create_vector(pin.identity.as_bytes()); - let head = super::table::encode_head(builder, &pin.head)?; - let principal = builder.create_string(&pin.principal); - Ok(FBGcPin::create( - builder, - &FBGcPinArgs { - catalog: Some(catalog), - activation_epoch: pin.context.activation_epoch, - identity: Some(identity), - head: Some(head), - principal: Some(principal), - expires_ms: pin.expires_ms, - released: pin.released, - operator_pin: pin.operator, - protects_uploads: pin.protects_uploads, - }, - )) -} - -pub(super) fn decode_pin(value: FBGcPin<'_>) -> Result { - let pin = GcPin { - context: CatalogContext { - catalog: CatalogId::from_bytes(value.catalog().bytes())?, - activation_epoch: value.activation_epoch(), - }, - identity: OperationId::from_bytes(value.identity().bytes())?, - head: super::table::decode_head(value.head())?, - principal: value.principal().to_owned(), - expires_ms: value.expires_ms(), - released: value.released(), - operator: value.operator_pin(), - protects_uploads: value.protects_uploads(), - }; - pin.validate()?; - Ok(pin) -} diff --git a/lib/crowdb-access-iceberg/src/record/gc_key.rs b/lib/crowdb-access-iceberg/src/record/gc_key.rs index d846f07aa..157575319 100644 --- a/lib/crowdb-access-iceberg/src/record/gc_key.rs +++ b/lib/crowdb-access-iceberg/src/record/gc_key.rs @@ -29,7 +29,6 @@ impl StorageRecord { Ok(()) } (Self::GcPage(page), key) if *key == page.key() => Ok(()), - (Self::GcPin(pin), key) if *key == pin.key() => Ok(()), _ => return None, }) } diff --git a/lib/crowdb-access-iceberg/src/record/multipart.rs b/lib/crowdb-access-iceberg/src/record/multipart.rs index 7cbba18aa..6ad337baf 100644 --- a/lib/crowdb-access-iceberg/src/record/multipart.rs +++ b/lib/crowdb-access-iceberg/src/record/multipart.rs @@ -6,7 +6,10 @@ use flatbuffers::{FlatBufferBuilder, WIPOffset}; use crate::catalog::CatalogContext; use crate::error::ValidationError; -use crate::file::{MultipartLimits, MultipartPart, MultipartPartMutation, MultipartPhase, MultipartSession}; +use crate::file::{ + FileContent, MultipartLimits, MultipartPart, MultipartPartMutation, MultipartPhase, MultipartSession, + MultipartStreamPart, +}; use crate::key::{FileId, OperationId}; mod completion; @@ -154,7 +157,19 @@ pub(super) fn encode_part<'buffer>( part.validate()?; let upload = builder.create_vector(part.upload.as_bytes()); let owner = fields::encode_owner(builder, part.owner); - let tree = fields::encode_tree(builder, &part.tree); + let tree = part.tree.as_ref().map(|tree| fields::encode_tree(builder, tree)); + let (locations, etag, stream_length) = if let Some(stream) = &part.stream { + let FileContent::Locations { bytes, etag } = &stream.content else { + return Err(ValidationError::Record); + }; + ( + Some(builder.create_vector(bytes)), + Some(builder.create_string(etag)), + stream.length, + ) + } else { + (None, None, 0) + }; Ok(FBMultipartPart::create( builder, &FBMultipartPartArgs { @@ -162,8 +177,11 @@ pub(super) fn encode_part<'buffer>( number: part.number, revision: part.revision, owner: Some(owner), - tree: Some(tree), + tree, modified_ms: part.modified_ms, + locations, + etag, + stream_length, }, )) } @@ -175,7 +193,18 @@ pub(super) fn decode_part(value: FBMultipartPart<'_>) -> Result Some(MultipartStreamPart { + length: value.stream_length(), + content: FileContent::Locations { + bytes: locations.bytes().to_vec(), + etag: etag.to_owned(), + }, + }), + (None, None) => None, + _ => return Err(ValidationError::Record), + }, }; part.validate()?; Ok(part) diff --git a/lib/crowdb-access-iceberg/src/table/load.rs b/lib/crowdb-access-iceberg/src/table/load.rs index 1a7631d57..5c153db1e 100644 --- a/lib/crowdb-access-iceberg/src/table/load.rs +++ b/lib/crowdb-access-iceberg/src/table/load.rs @@ -49,8 +49,6 @@ pub struct TableLoader { blocks: Arc, limits: TableMetadataLimits, projections: ProjectionStore, - pins: crate::gc::ReaderPins, - pin_lifetime_ms: Option, #[cfg(feature = "test-util")] projection_hits: std::sync::atomic::AtomicUsize, } @@ -63,8 +61,6 @@ impl TableLoader { limits: TableMetadataLimits, ) -> Self { Self { - pins: crate::gc::ReaderPins::new(store.clone()), - pin_lifetime_ms: None, namespaces: NamespaceRepository::new(store.clone()), projections: ProjectionStore::new(store.clone(), blocks.clone()), tables: TableRepository::new(store), @@ -75,22 +71,6 @@ impl TableLoader { } } - /// # Errors - /// Rejects zero or excessive request protection lifetimes. - pub fn with_reader_pins(mut self, lifetime_ms: u64) -> Result { - if lifetime_ms == 0 || lifetime_ms > 24 * 60 * 60 * 1000 { - return Err(crate::error::ValidationError::Deadline); - } - self.pin_lifetime_ms = Some(lifetime_ms); - Ok(self) - } - - #[must_use] - pub fn with_catalog_reader_pins(mut self) -> Self { - self.pin_lifetime_ms = Some(0); - self - } - /// Resolves live namespace identity before table selection. No table publisher is implied. /// # Errors /// Corruption, retirement and changed namespace/head identity remain errors, not absence. @@ -181,35 +161,6 @@ impl TableLoader { { return Err(TableLoadError::UnsupportedVersion); } - let pin = if let Some(lifetime_ms) = self.pin_lifetime_ms { - let now_ms = u64::try_from( - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .map_err(|_| TableMetadataError::Bounds)? - .as_millis(), - ) - .map_err(|_| TableMetadataError::Bounds)?; - let pin = crate::gc::GcPin { - context, - identity: crate::key::OperationId::random(), - head: selected.head.clone(), - principal: "catalog-metadata-reader".into(), - expires_ms: if lifetime_ms == 0 { - self.pins.request_expiry(context, now_ms).await? - } else { - now_ms - .checked_add(lifetime_ms) - .ok_or(TableMetadataError::Bounds)? - }, - released: false, - operator: false, - protects_uploads: false, - }; - self.pins.acquire(&pin).await?; - Some(pin) - } else { - None - }; let canonical = super::metadata::read_table_metadata_bytes(self.blocks.clone(), &selected, self.limits).await?; let etag = etag(&selected.head, mode); @@ -220,9 +171,6 @@ impl TableLoader { self.tables.ensure_current(context, &selected).await?; self.check_namespace(context, namespace, parent.namespace, parent.name_epoch) .await?; - if let Some(pin) = &pin { - self.pins.release(pin).await?; - } if if_none_match.is_some_and(|header| { header.split(',').any(|tag| { let tag = tag.trim(); diff --git a/lib/crowdb-access-iceberg/tests/common/gc_blocks.rs b/lib/crowdb-access-iceberg/tests/common/gc_blocks.rs index 4e7ead9e9..4984afadb 100644 --- a/lib/crowdb-access-iceberg/tests/common/gc_blocks.rs +++ b/lib/crowdb-access-iceberg/tests/common/gc_blocks.rs @@ -3,12 +3,14 @@ use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; use async_trait::async_trait; use crowdb_access_iceberg::file::{ChunkRoot, FileBlockStore, FileIdentity, FileIoError}; use crowdb_chunk_client::ReclaimOutcome; +use crowdb_protocol::chunkdb::rpc::Location; #[derive(Default)] pub struct TestReclaimBlocks { pub blocks: crate::blocks::TestBlocks, pub deferred: AtomicBool, pub deletes: AtomicUsize, + pub location_deletes: AtomicUsize, pub reply_loss: AtomicBool, pub delay_ms: AtomicUsize, } @@ -42,4 +44,19 @@ impl FileBlockStore for TestReclaimBlocks { } Ok(ReclaimOutcome::Reclaimed) } + + async fn reclaim_location(&self, location: &Location) -> Result { + if self.deferred.load(Ordering::Relaxed) { + return Ok(ReclaimOutcome::Deferred); + } + let chunk = location.chunk_id.ok_or(FileIoError::Bounds)?; + self.blocks.values.rcu(|values| { + let mut next = (**values).clone(); + next.remove(&chunk.low); + next + }); + self.deletes.fetch_add(1, Ordering::Relaxed); + self.location_deletes.fetch_add(1, Ordering::Relaxed); + Ok(ReclaimOutcome::Reclaimed) + } } diff --git a/lib/crowdb-access-iceberg/tests/common/store.rs b/lib/crowdb-access-iceberg/tests/common/store.rs index 87ec5f440..37a81c1ce 100644 --- a/lib/crowdb-access-iceberg/tests/common/store.rs +++ b/lib/crowdb-access-iceberg/tests/common/store.rs @@ -18,6 +18,7 @@ pub struct TestStore { #[allow(dead_code)] pub gc_delete_reply_loss: AtomicBool, pub writes: AtomicUsize, + pub reads: AtomicUsize, pub fencing_delay_ms: AtomicUsize, pub fencing_barrier: Option>, pub fencing_visits: AtomicUsize, @@ -111,6 +112,7 @@ impl TestStore { #[async_trait] impl CatalogStore for TestStore { async fn get(&self, key: &[u8]) -> Result, StoreError> { + self.reads.fetch_add(1, Ordering::SeqCst); Ok(self.values.load().get(key).cloned()) } diff --git a/lib/crowdb-access-iceberg/tests/file_record_test.rs b/lib/crowdb-access-iceberg/tests/file_record_test.rs index 1e7e1376e..ed251e8a8 100644 --- a/lib/crowdb-access-iceberg/tests/file_record_test.rs +++ b/lib/crowdb-access-iceberg/tests/file_record_test.rs @@ -4,6 +4,7 @@ use crowdb_access_iceberg::file::{ }; use crowdb_access_iceberg::key::{CatalogId, FileId, TableId}; use crowdb_access_iceberg::record::{StorageRecord, MAX_RECORD_BYTES}; +use crowdb_protocol::chunkdb::rpc::Location; use crowdb_protocol::common::ChunkId; use sha2::{Digest, Sha256}; @@ -25,6 +26,45 @@ fn record(input: &[u8]) -> FileRecord { } } +#[test] +fn complete_chunk_locations_round_trip_without_a_file_sha256() { + let locations = vec![ + Location { + chunk_id: Some(ChunkId { high: 1, low: 2 }), + offset: 0, + length: 65_536, + logical_offset: 0, + logical_length: 65_500, + }, + Location { + chunk_id: Some(ChunkId { high: 3, low: 4 }), + offset: 128, + length: 80, + logical_offset: 65_500, + logical_length: 44, + }, + ]; + let length = 65_544; + let content = + FileContent::from_locations(&locations, length, "d41d8cd98f00b204e9800998ecf8427e".into()).unwrap(); + let mut file = record(b"{}"); + file.kind = FileKind::Unbound; + file.format = ContentFormat::Opaque; + file.length = length; + file.digest = [0; 32]; + file.content = content; + file.validate().unwrap(); + let key = file_key(file.location.table().catalog, file.file); + let encoded = StorageRecord::File(Box::new(file.clone())).encode().unwrap(); + assert_eq!( + StorageRecord::decode(&key, &encoded).unwrap(), + StorageRecord::File(Box::new(file)) + ); + let mut broken = locations; + broken[1].logical_offset += 1; + assert!(FileContent::from_locations(&broken, length, "d41d8cd98f00b204e9800998ecf8427e".into()).is_err()); +} + #[test] fn inline_selection_enforces_kind_stored_size_and_compression_input_bounds() { for kind in [FileKind::Metadata, FileKind::ManifestList, FileKind::Manifest] { diff --git a/lib/crowdb-access-iceberg/tests/file_repository_test.rs b/lib/crowdb-access-iceberg/tests/file_repository_test.rs index 8881cf24f..5a4705297 100644 --- a/lib/crowdb-access-iceberg/tests/file_repository_test.rs +++ b/lib/crowdb-access-iceberg/tests/file_repository_test.rs @@ -6,9 +6,11 @@ mod fixture; use std::sync::{atomic::Ordering, Arc}; use crowdb_access_iceberg::catalog::{CatalogContext, CatalogError, CatalogStore, RootState}; -use crowdb_access_iceberg::file::{file_key, location_key, FileRepository}; +use crowdb_access_iceberg::file::{file_key, location_key, FileContent, FileRepository}; use crowdb_access_iceberg::key::CatalogId; use crowdb_access_iceberg::operation::mutation_identity; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; use fixture::TestFile; #[tokio::test] @@ -47,6 +49,43 @@ async fn immutable_publication_replays_equal_bytes_without_overwriting_or_extra_ ); } +#[tokio::test] +async fn streamed_publication_replays_equal_etag_with_different_chunk_locations() { + let fixture = TestFile::new(common::TestStore::default()).await; + let repository = FileRepository::new(fixture.store.clone()); + let make = |chunk_low, etag: &str| { + let mut record = fixture.record("metadata/stream.json", b"{}"); + record.digest = [0; 32]; + record.content = FileContent::from_locations( + &[Location { + chunk_id: Some(ChunkId { + high: 7, + low: chunk_low, + }), + offset: 0, + length: 36, + logical_offset: 0, + logical_length: 2, + }], + 2, + etag.to_owned(), + ) + .unwrap(); + record + }; + let first = make(1, "99914b932bd37a50b983c5e7c90ae93b"); + let replay = make(2, "99914b932bd37a50b983c5e7c90ae93b"); + let changed = make(3, "f111cdacaa915d85831037cae3622d59"); + assert_eq!(repository.publish(fixture.context, &first).await.unwrap(), first); + let writes = fixture.store.writes.load(Ordering::SeqCst); + assert_eq!(repository.publish(fixture.context, &replay).await.unwrap(), first); + assert!(matches!( + repository.publish(fixture.context, &changed).await, + Err(CatalogError::Conflict) + )); + assert_eq!(fixture.store.writes.load(Ordering::SeqCst), writes); +} + #[tokio::test] async fn every_lost_file_publication_reply_recovers_on_another_repository() { for lost in 1..=2 { diff --git a/lib/crowdb-access-iceberg/tests/gc_fence_test.rs b/lib/crowdb-access-iceberg/tests/gc_fence_test.rs index 2527ab6bc..21a673934 100644 --- a/lib/crowdb-access-iceberg/tests/gc_fence_test.rs +++ b/lib/crowdb-access-iceberg/tests/gc_fence_test.rs @@ -1,6 +1,6 @@ use crowdb_access_iceberg::{ catalog::{CasOutcome, CatalogStore}, - gc::{GcPhase, GcPin, GcRepository, GcStalledReason, GcTask, GcTaskKind, ReaderPins}, + gc::{GcPhase, GcRepository, GcStalledReason, GcTask, GcTaskKind}, key::{NamespaceId, OperationId}, operation::mutation_identity, record::StorageRecord, @@ -170,74 +170,6 @@ async fn lost_legacy_fence_release_reply_is_recovered_on_retry() { assert!(!retired.fenced); } -#[tokio::test] -async fn new_reader_pin_cannot_be_acknowledged_during_sweep() { - let (fixture, task) = fixture().await; - let pins = ReaderPins::new(fixture.store.clone()); - let pin = GcPin { - context: fixture.context, - identity: OperationId::random(), - head: task.head.clone().unwrap(), - principal: "reader".into(), - expires_ms: 2000, - released: false, - operator: false, - protects_uploads: false, - }; - pins.acquire(&pin).await.unwrap(); - let repository = GcRepository::new(fixture.store.clone()); - seed_legacy_task(&fixture.store, &task).await; - repository.fence_table(&task).await.unwrap(); - let mut newcomer = pin.clone(); - newcomer.identity = OperationId::random(); - assert!(pins.acquire(&newcomer).await.is_err()); - assert!(fixture - .store - .get(&pin.key().encode().unwrap()) - .await - .unwrap() - .is_some()); - assert!(pin.protects(1999)); - assert!(!pin.protects(2000)); - pins.release(&pin).await.unwrap(); - pins.release(&pin).await.unwrap(); - repository.release_table_fence(&task).await.unwrap(); -} - -#[tokio::test] -async fn operator_pin_can_be_inspected_and_released_after_restart() { - let (fixture, task) = fixture().await; - let pins = ReaderPins::new(fixture.store.clone()); - let pin = GcPin { - context: fixture.context, - identity: OperationId::random(), - head: task.head.unwrap(), - principal: "manager".into(), - expires_ms: 0, - released: false, - operator: true, - protects_uploads: true, - }; - pins.acquire(&pin).await.unwrap(); - let restarted = ReaderPins::new(fixture.store); - assert_eq!( - restarted - .get(pin.context.catalog, pin.head.table, pin.identity) - .await - .unwrap(), - Some(pin.clone()) - ); - restarted.release(&pin).await.unwrap(); - assert!( - restarted - .get(pin.context.catalog, pin.head.table, pin.identity) - .await - .unwrap() - .unwrap() - .released - ); -} - #[tokio::test] async fn changed_table_generation_rejects_gc_before_deletion() { let (fixture, task) = fixture().await; diff --git a/lib/crowdb-access-iceberg/tests/gc_proof_test.rs b/lib/crowdb-access-iceberg/tests/gc_proof_test.rs index 640f9f166..b8e27f752 100644 --- a/lib/crowdb-access-iceberg/tests/gc_proof_test.rs +++ b/lib/crowdb-access-iceberg/tests/gc_proof_test.rs @@ -415,149 +415,6 @@ async fn live_sweep_keeps_proven_files_and_removes_only_an_unreferenced_file() { } } -#[tokio::test] -async fn request_protection_uses_persisted_bounds_and_rejects_fenced_heads() { - let (store, mut task, _) = fixture(3, 0).await; - let pins = crowdb_access_iceberg::gc::ReaderPins::new(store.clone()); - let key = IcebergKey::Catalog { - catalog: task.context.catalog, - scope: CatalogScope::Authority, - suffix: Vec::new(), - }; - let before = store.get(&key.encode().unwrap()).await.unwrap().unwrap(); - let StorageRecord::Authority(mut authority) = StorageRecord::decode(&key, &before.bytes).unwrap() else { - panic!() - }; - authority.admission_bounds.request_ms = 600_000; - authority.admission_bounds.clock_skew_ms = 70_000; - let bytes = StorageRecord::Authority(authority).encode().unwrap(); - let key = key.encode().unwrap(); - store - .compare_exchange( - &key, - Some(&before.bytes), - &bytes, - mutation_identity(&key, Some(&before.bytes), &bytes), - ) - .await - .unwrap(); - assert_eq!(pins.request_expiry(task.context, 100).await.unwrap(), 670_100); - assert!(pins.request_expiry(task.context, u64::MAX - 100).await.is_err()); - let pin = pins - .protect_files( - task.context, - task.head.as_ref().unwrap().table, - "reader", - 670_100, - 100, - ) - .await - .unwrap(); - assert!(pin.protects_uploads); - pins.release(&pin).await.unwrap(); - task = finish(store.clone(), task).await; - GcRepository::new(store).fence_table(&task).await.unwrap(); - assert!(pins - .protect_files( - task.context, - task.head.as_ref().unwrap().table, - "reader", - 670_100, - 100 - ) - .await - .is_err()); -} - -#[tokio::test] -async fn late_credentials_cancel_sweep_and_release_the_table() { - let (store, task, _) = fixture(3, 2).await; - let mut task = finish(store.clone(), task).await; - let pins = crowdb_access_iceberg::gc::ReaderPins::new(store.clone()); - pins.protect_files( - task.context, - task.head.as_ref().unwrap().table, - "late-credentials", - 2_000_000, - 1_000_000, - ) - .await - .unwrap(); - let worker = GcWorker::new( - GcRepository::new(store.clone()), - Arc::new(blocks::TestBlocks::default()), - GcLimits::default(), - ) - .unwrap(); - for _ in 0..100 { - task = worker.step(&task, 1_000_000).await.unwrap(); - if task.phase == GcPhase::Complete { - break; - } - } - assert_eq!(task.phase, GcPhase::Complete); - assert_eq!( - task.stalled, - crowdb_access_iceberg::gc::GcStalledReason::Protected - ); - assert!(!task.fenced); - assert_eq!(task.deleted, 0); - pins.protect_files( - task.context, - task.head.as_ref().unwrap().table, - "new-request", - 2_000_000, - 1_000_000, - ) - .await - .unwrap(); -} - -#[tokio::test] -async fn dropped_table_file_reads_remain_protected_until_reclamation_fences_them() { - let (store, task, _) = fixture(3, 0).await; - let pins = crowdb_access_iceberg::gc::ReaderPins::new(store.clone()); - let mut head = task.head.unwrap(); - let key = head_key(head.catalog, head.table).encode().unwrap(); - for lifecycle in [TableLifecycle::Tombstone, TableLifecycle::Reclaiming] { - let before = store.get(&key).await.unwrap().unwrap(); - head.lifecycle = lifecycle; - head.pending_operation = Some(OperationId::random()); - let after = StorageRecord::TableHead(Box::new(head.clone())).encode().unwrap(); - store - .compare_exchange( - &key, - Some(&before.bytes), - &after, - mutation_identity(&key, Some(&before.bytes), &after), - ) - .await - .unwrap(); - let read = pins - .protect_file_reads(task.context, head.table, "file-reader", 2000, 100) - .await; - if lifecycle == TableLifecycle::Tombstone { - let pin = read.unwrap(); - assert!(pin.protects(1999)); - assert!(!pin.protects(2000)); - assert!(pin.protects_uploads); - assert_eq!( - pins.get(head.catalog, head.table, pin.identity).await.unwrap(), - Some(pin) - ); - } else { - assert!(matches!( - read, - Err(crowdb_access_iceberg::catalog::CatalogError::Busy) - )); - } - assert!(pins - .protect_files(task.context, head.table, "file-writer", 2000, 100) - .await - .is_err()); - } -} - #[tokio::test] async fn proof_and_fence_resume_after_lost_durable_write_responses() { use std::sync::atomic::Ordering; @@ -597,123 +454,6 @@ async fn proof_and_fence_resume_after_lost_durable_write_responses() { } } -#[tokio::test] -async fn historical_reader_root_is_retained_after_the_current_head_changes() { - let (store, old, old_files) = fixture(3, 1).await; - let old_head = old.head.as_ref().unwrap(); - let pin = crowdb_access_iceberg::gc::GcPin { - context: old.context, - identity: OperationId::random(), - head: old_head.clone(), - principal: "historical-reader".into(), - expires_ms: 2_000_000, - released: false, - operator: false, - protects_uploads: false, - }; - crowdb_access_iceberg::gc::ReaderPins::new(store.clone()) - .acquire(&pin) - .await - .unwrap(); - let bytes = serde_json::to_vec(&metadata::metadata(3)).unwrap(); - let mut head = old_head.clone(); - head.generation += 1; - head.operation_fence += 1; - head.metadata_file = FileId::random(); - head.metadata_location = metadata::table().file("metadata/new.json").unwrap(); - head.metadata_digest = Sha256::digest(&bytes).into(); - let file = FileRecord { - file: head.metadata_file, - location: head.metadata_location.clone(), - kind: FileKind::Metadata, - format: ContentFormat::Json, - length: bytes.len() as u64, - digest: head.metadata_digest, - content: FileContent::select_inline(FileKind::Metadata, &bytes).unwrap(), - hint: None, - }; - FileRepository::new(store.clone()) - .publish(old.context, &file) - .await - .unwrap(); - let key = head_key(head.catalog, head.table).encode().unwrap(); - let before = StorageRecord::TableHead(Box::new(old_head.clone())) - .encode() - .unwrap(); - let after = StorageRecord::TableHead(Box::new(head.clone())).encode().unwrap(); - store - .compare_exchange( - &key, - Some(&before), - &after, - mutation_identity(&key, Some(&before), &after), - ) - .await - .unwrap(); - let task = GcTask::plan( - old.context, - OperationId::random(), - Some(head), - 1, - GcLimits::default(), - ) - .unwrap(); - let repository = GcRepository::new(store.clone()); - put(&store, task.key(), StorageRecord::GcTask(Box::new(task.clone()))).await; - let task = finish(store, task).await; - for file in old_files { - assert!(repository.proof_contains(&task, file).await.unwrap()); - } - assert!(repository.proof_contains(&task, file.file).await.unwrap()); -} - -#[tokio::test] -async fn publication_pin_prevents_sweep_while_location_publication_is_in_flight() { - use std::sync::atomic::Ordering; - let (store, mut task, _) = fixture(3, 0).await; - let file = FileRecord { - file: FileId::random(), - location: metadata::table().file("metadata/in-flight.json").unwrap(), - kind: FileKind::Metadata, - format: ContentFormat::Json, - length: 2, - digest: Sha256::digest(b"{}").into(), - content: FileContent::select_inline(FileKind::Metadata, b"{}").unwrap(), - hint: None, - }; - store.file_mapping_pause.store(true, Ordering::SeqCst); - let publishing = tokio::spawn({ - let store = store.clone(); - let file = file.clone(); - let context = task.context; - async move { FileRepository::new(store).publish(context, &file).await } - }); - store.file_mapping_entered.notified().await; - let worker = GcWorker::new( - GcRepository::new(store.clone()), - Arc::new(blocks::TestBlocks::default()), - GcLimits { - minimum_retention_ms: 1, - ..GcLimits::default() - }, - ) - .unwrap(); - for _ in 0..100 { - task = worker.step(&task, 1_000_000).await.unwrap(); - if task.phase == GcPhase::Waiting { - break; - } - } - assert_eq!(task.phase, GcPhase::Waiting); - assert_eq!( - task.stalled, - crowdb_access_iceberg::gc::GcStalledReason::Protected - ); - assert_eq!(task.deleted, 0); - store.file_mapping_release.notify_one(); - assert_eq!(publishing.await.unwrap().unwrap(), file); -} - #[tokio::test] async fn live_sweep_recovers_a_lost_final_fence_release_response() { use crowdb_access_iceberg::gc::GcScan; diff --git a/lib/crowdb-access-iceberg/tests/gc_tree_test.rs b/lib/crowdb-access-iceberg/tests/gc_tree_test.rs index 8cf81ab5f..5e2def6aa 100644 --- a/lib/crowdb-access-iceberg/tests/gc_tree_test.rs +++ b/lib/crowdb-access-iceberg/tests/gc_tree_test.rs @@ -111,6 +111,7 @@ async fn corrupt_directory_never_authorizes_a_child_deletion() { next_child: 0, }], pending: None, + next_location: 0, }; assert!(cursor.next(&blocks).await.is_err()); assert!(cursor.pending.is_none()); diff --git a/lib/crowdb-access-iceberg/tests/gc_worker_test.rs b/lib/crowdb-access-iceberg/tests/gc_worker_test.rs index df3e7661b..cecbf8809 100644 --- a/lib/crowdb-access-iceberg/tests/gc_worker_test.rs +++ b/lib/crowdb-access-iceberg/tests/gc_worker_test.rs @@ -8,6 +8,7 @@ use crowdb_access_iceberg::{ operation::mutation_identity, record::StorageRecord, }; +use crowdb_protocol::{chunkdb::rpc::Location, common::ChunkId}; #[path = "common/gc_adoption.rs"] mod adoption; @@ -123,6 +124,81 @@ async fn retired_file_reclamation_survives_worker_restart_at_every_step() { .is_none()); } +#[tokio::test] +async fn retired_streamed_file_reclaims_location_after_worker_restart() { + let (fixture, blocks, mut task, limits, _) = fixture(true).await; + let chunk = ChunkId { high: 1, low: 999 }; + let location = Location { + chunk_id: Some(chunk), + offset: 0, + length: 128, + logical_offset: 0, + logical_length: 64, + }; + blocks.blocks.values.rcu(|values| { + let mut next = (**values).clone(); + next.insert(chunk.low, Arc::new(vec![7; 64])); + next + }); + let file = crowdb_access_iceberg::file::FileRecord { + file: FileId::random(), + location: fixture.table.file("data/streamed.parquet").unwrap(), + kind: FileKind::Data, + format: ContentFormat::Parquet, + length: 64, + digest: [0; 32], + content: FileContent::from_locations(&[location], 64, "00000000000000000000000000000000".into()) + .unwrap(), + hint: None, + }; + let mapping = crowdb_access_iceberg::file::FileMapping { + file: file.file, + location: file.location.clone(), + }; + for (key, record) in [ + ( + file_key(fixture.context.catalog, file.file), + StorageRecord::File(Box::new(file.clone())), + ), + ( + crowdb_access_iceberg::file::location_key(&file.location), + StorageRecord::FileMapping(mapping), + ), + ] { + let key = key.encode().unwrap(); + let bytes = record.encode().unwrap(); + fixture + .store + .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) + .await + .unwrap(); + } + for _ in 0..300 { + let repository = GcRepository::new(fixture.store.clone()); + let worker = GcWorker::new(repository.clone(), blocks.clone(), limits).unwrap(); + task = worker.step(&task, 2000_u64.max(task.retry_at_ms)).await.unwrap(); + task = repository + .task(task.context.catalog, task.identity) + .await + .unwrap() + .unwrap(); + if task.phase == GcPhase::Complete { + break; + } + } + assert_eq!(task.phase, GcPhase::Complete, "{task:?}"); + assert_eq!(task.deleted, 2); + assert_eq!(task.reclaimed_bytes, 4096 + 64); + assert_eq!(blocks.location_deletes.load(Ordering::Relaxed), 1); + assert!(blocks.blocks.values.load().is_empty()); + assert!(fixture + .store + .get(&file_key(fixture.context.catalog, file.file).encode().unwrap()) + .await + .unwrap() + .is_none()); +} + #[tokio::test] async fn inactive_discovery_preserves_file_when_gc_workspace_is_unavailable() { let (fixture, blocks, task, limits, file) = fixture(true).await; @@ -180,7 +256,8 @@ async fn expired_aborted_multipart_part_reclaims_tree_before_session_record() { revision: 1, modified_ms: 100, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, }; for (key, record) in [ ( @@ -924,137 +1001,3 @@ async fn late_candidate_discovery_starts_a_fresh_retention_window() { assert_eq!(task.phase, GcPhase::Complete, "{task:?}"); assert_eq!(task.deleted, 1); } - -#[tokio::test] -async fn purge_fences_new_readers_and_waits_for_reader_and_delegated_pins() { - let (fixture, blocks, _, limits, _) = fixture(false).await; - let (head, pin, delegated, pins) = create_purge_with_pins(&fixture).await; - let repository = GcRepository::new(fixture.store.clone()); - let mut task = GcTask::plan(fixture.context, OperationId::random(), Some(head), 1000, limits).unwrap(); - repository.create(&task).await.unwrap(); - let worker = GcWorker::new(repository.clone(), blocks.clone(), limits).unwrap(); - for _ in 0..50 { - task = worker.step(&task, 2000_u64.max(task.retry_at_ms)).await.unwrap(); - if task.phase == GcPhase::Waiting { - break; - } - } - assert!(task.fenced); - assert_eq!(task.stalled, GcStalledReason::Protected); - assert_eq!(blocks.deletes.load(Ordering::Relaxed), 0); - let mut newcomer = pin.clone(); - newcomer.identity = OperationId::random(); - assert!(pins.acquire(&newcomer).await.is_err()); - pins.release(&pin).await.unwrap(); - let restarted = GcWorker::new(repository.clone(), blocks.clone(), limits).unwrap(); - for _ in 0..30 { - task = restarted - .step(&task, 3000_u64.max(task.retry_at_ms)) - .await - .unwrap(); - assert_eq!(blocks.deletes.load(Ordering::Relaxed), 0); - if task.stalled == GcStalledReason::Protected { - break; - } - } - assert_eq!(task.stalled, GcStalledReason::Protected); - pins.release(&delegated).await.unwrap(); - for _ in 0..300 { - task = restarted - .step(&task, 10_000_u64.max(task.retry_at_ms)) - .await - .unwrap(); - if task.phase == GcPhase::Complete { - break; - } - } - assert_eq!(task.phase, GcPhase::Complete); - assert_eq!(task.deleted, 2); - assert!(!task.fenced); - assert!(blocks.blocks.values.load().is_empty()); -} - -async fn create_purge_with_pins( - fixture: &common::file::TestFile, -) -> ( - crowdb_access_iceberg::table::TableHead, - crowdb_access_iceberg::gc::GcPin, - crowdb_access_iceberg::gc::GcPin, - crowdb_access_iceberg::gc::ReaderPins, -) { - use crowdb_access_iceberg::{ - gc::{GcPin, ReaderPins}, - key::NamespaceId, - table::{head_key, TableHead, TableLifecycle, TablePurgeTask}, - }; - let metadata = fixture.record("metadata/table.json", b"{}"); - FileRepository::new(fixture.store.clone()) - .publish(fixture.context, &metadata) - .await - .unwrap(); - let mut head = TableHead { - catalog: fixture.context.catalog, - table: fixture.table.table, - namespace: NamespaceId::random(), - name: "purged".into(), - name_epoch: 1, - lifecycle: TableLifecycle::Ready, - generation: 1, - metadata_file: metadata.file, - metadata_location: metadata.location, - metadata_digest: metadata.digest, - format_version: 1, - table_uuid: None, - operation_fence: 1, - pending_operation: None, - }; - let key = head_key(head.catalog, head.table).encode().unwrap(); - let before = StorageRecord::TableHead(Box::new(head.clone())).encode().unwrap(); - fixture - .store - .compare_exchange(&key, None, &before, mutation_identity(&key, None, &before)) - .await - .unwrap(); - let pin = GcPin { - context: fixture.context, - identity: OperationId::random(), - head: head.clone(), - principal: "reader".into(), - expires_ms: 5000, - released: false, - operator: false, - protects_uploads: true, - }; - let pins = ReaderPins::new(fixture.store.clone()); - pins.acquire(&pin).await.unwrap(); - let mut delegated = pin.clone(); - delegated.identity = OperationId::random(); - delegated.principal = "delegated-credential".into(); - pins.acquire(&delegated).await.unwrap(); - head.lifecycle = TableLifecycle::Tombstone; - head.operation_fence += 1; - head.pending_operation = Some(OperationId::random()); - let after = StorageRecord::TableHead(Box::new(head.clone())).encode().unwrap(); - fixture - .store - .compare_exchange( - &key, - Some(&before), - &after, - mutation_identity(&key, Some(&before), &after), - ) - .await - .unwrap(); - let purge = TablePurgeTask { - activation_epoch: fixture.context.activation_epoch, - head: head.clone(), - }; - let key = purge.key().encode().unwrap(); - let bytes = StorageRecord::TablePurgeTask(Box::new(purge)).encode().unwrap(); - fixture - .store - .compare_exchange(&key, None, &bytes, mutation_identity(&key, None, &bytes)) - .await - .unwrap(); - (head, pin, delegated, pins) -} diff --git a/lib/crowdb-access-iceberg/tests/multipart_completion_test.rs b/lib/crowdb-access-iceberg/tests/multipart_completion_test.rs index 75f8b5f4b..457234549 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_completion_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_completion_test.rs @@ -10,8 +10,8 @@ mod fixtures; use std::sync::{atomic::Ordering, Arc}; use crowdb_access_iceberg::file::{ - FileIdentity, FileReader, FileTreeWriter, MultipartPart, MultipartPhase, MultipartRepository, - MultipartSelection, MultipartSession, SelectedPart, + FileContent, FileIdentity, FileReader, FileTreeWriter, MultipartPart, MultipartPhase, + MultipartRepository, MultipartSelection, MultipartSession, MultipartStreamPart, SelectedPart, }; use crowdb_access_iceberg::key::FileId; @@ -46,12 +46,13 @@ async fn setup() -> ( revision: 1, modified_ms: 101, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, }; selected.push(SelectedPart { number, revision: 1, - digest: part.tree.digest, + digest: part.selection_digest(), }); assert!(repository.reserve_part(&session, &part, 101).await.unwrap()); session = load(&repository, &session).await; @@ -106,6 +107,34 @@ fn frozen_selection_is_ordered_versioned_and_independently_bounded() { .is_err()); } +#[test] +fn streamed_selection_rejects_oversized_snapshot_before_copying_part_locations() { + let session = fixtures::session(); + let part = MultipartPart { + upload: session.upload, + number: 1, + revision: 1, + modified_ms: 100, + owner: session.owner, + tree: None, + stream: Some(MultipartStreamPart { + length: 1, + content: FileContent::Locations { + bytes: vec![0; crowdb_access_iceberg::operation::MAX_PAYLOAD_BYTES], + etag: "00000000000000000000000000000000".into(), + }, + }), + }; + assert!(matches!( + MultipartSelection::with_stream_parts(&[part]), + Err(crowdb_access_iceberg::error::ValidationError::RecordTooLarge) + )); + assert!(matches!( + MultipartSelection::decode(&vec![0; crowdb_access_iceberg::operation::MAX_PAYLOAD_BYTES + 1]), + Err(crowdb_access_iceberg::error::ValidationError::RecordTooLarge) + )); +} + #[tokio::test] async fn lost_selection_and_progress_replies_resume_exact_bytes_on_new_instances() { for lost in 1..=2 { diff --git a/lib/crowdb-access-iceberg/tests/multipart_list_test.rs b/lib/crowdb-access-iceberg/tests/multipart_list_test.rs index a26fdc12c..3a66be7d5 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_list_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_list_test.rs @@ -42,11 +42,12 @@ async fn setup() -> (file::TestFile, MultipartSession, Vec) { file: FileId::random(), ..session.owner }, - tree: FileTree { + tree: Some(FileTree { root: None, length: 0, digest: Sha256::digest([]).into(), - }, + }), + stream: None, }) .collect(); fixture.store.values.rcu(|values| { diff --git a/lib/crowdb-access-iceberg/tests/multipart_model_test.rs b/lib/crowdb-access-iceberg/tests/multipart_model_test.rs index c7cd7af6f..cd407ccf1 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_model_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_model_test.rs @@ -91,11 +91,12 @@ fn staged_parts_are_physical_bytes_bound_to_one_upload_table_and_revision() { file: FileId::random(), ..session.owner }, - tree: FileTree { + tree: Some(FileTree { root: None, length: 0, digest: Sha256::digest([]).into(), - }, + }), + stream: None, }; part.validate_for(&session).unwrap(); part.number = 0; diff --git a/lib/crowdb-access-iceberg/tests/multipart_observed_recovery_test.rs b/lib/crowdb-access-iceberg/tests/multipart_observed_recovery_test.rs index 8ed112822..1227df778 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_observed_recovery_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_observed_recovery_test.rs @@ -41,7 +41,8 @@ async fn completing(fixture: &file::TestFile, blocks: Arc) - revision: 1, modified_ms: 101, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, }; repository.reserve_part(&session, &part, 101).await.unwrap(); session = repository @@ -58,7 +59,7 @@ async fn completing(fixture: &file::TestFile, blocks: Arc) - let selection = MultipartSelection::new(vec![SelectedPart { number: 1, revision: 1, - digest: part.tree.digest, + digest: part.selection_digest(), }]) .unwrap(); repository diff --git a/lib/crowdb-access-iceberg/tests/multipart_publication_test.rs b/lib/crowdb-access-iceberg/tests/multipart_publication_test.rs index 3fefe515f..628271c05 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_publication_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_publication_test.rs @@ -9,15 +9,21 @@ mod fixtures; #[path = "common/multipart_recovery_store.rs"] mod scan; +use std::fmt::Write as _; use std::sync::{atomic::Ordering, Arc}; use crowdb_access_iceberg::catalog::{CatalogError, CatalogStore}; use crowdb_access_iceberg::file::{ - FileIdentity, FileRecord, FileRepository, FileTree, FileTreeWriter, MultipartPart, MultipartPhase, - MultipartRecovery, MultipartRepository, MultipartSelection, MultipartSession, SelectedPart, + FileContent, FileIdentity, FileRecord, FileRepository, FileTree, FileTreeWriter, MultipartPart, + MultipartPhase, MultipartRecovery, MultipartRepository, MultipartSelection, MultipartSession, + MultipartStreamPart, SelectedPart, }; use crowdb_access_iceberg::key::FileId; use crowdb_access_iceberg::operation::mutation_identity; +use crowdb_access_iceberg::record::StorageRecord; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; +use md5::{Digest, Md5}; async fn setup() -> ( file::TestFile, @@ -50,7 +56,8 @@ async fn setup() -> ( revision: 1, modified_ms: 101, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, }; repository.reserve_part(&session, &part, 101).await.unwrap(); session = load(&repository, &session).await; @@ -59,7 +66,7 @@ async fn setup() -> ( let selection = MultipartSelection::new(vec![SelectedPart { number: 1, revision: 1, - digest: part.tree.digest, + digest: part.selection_digest(), }]) .unwrap(); repository @@ -87,6 +94,146 @@ async fn load(repository: &MultipartRepository, session: &MultipartSession) -> M .unwrap() } +#[tokio::test] +async fn upload_part_lookup_uses_one_catalog_read_before_the_session_cas() { + let fixture = file::TestFile::new(common::TestStore::default()).await; + let repository = MultipartRepository::new(fixture.store.clone()); + let mut session = fixtures::session(); + session.context = fixture.context; + session.owner.table = fixture.table; + session.location = fixture.table.file("data/part.parquet").unwrap(); + repository.begin(&session, 100).await.unwrap(); + let before = fixture.store.reads.load(Ordering::SeqCst); + assert!(repository.part_for_upload(&session, 1).await.unwrap().is_none()); + assert_eq!(fixture.store.reads.load(Ordering::SeqCst) - before, 1); +} + +async fn replace_selected_part(fixture: &file::TestFile, first: &MultipartPart) { + let mut replacement = first.clone(); + replacement.revision = 2; + replacement.owner.file = FileId::random(); + replacement.stream.as_mut().unwrap().content = FileContent::from_locations( + &[Location { + chunk_id: Some(ChunkId { high: 7, low: 99 }), + offset: 100, + length: 41, + logical_offset: 0, + logical_length: 7, + }], + 7, + "33333333333333333333333333333333".into(), + ) + .unwrap(); + let key = first.key().encode().unwrap(); + let before = StorageRecord::MultipartPart(Box::new(first.clone())) + .encode() + .unwrap(); + let after = StorageRecord::MultipartPart(Box::new(replacement)) + .encode() + .unwrap(); + fixture + .store + .compare_exchange( + &key, + Some(&before), + &after, + mutation_identity(&key, Some(&before), &after), + ) + .await + .unwrap(); +} + +#[tokio::test] +async fn streamed_parts_complete_by_composing_locations_and_saved_md5_only() { + let fixture = file::TestFile::new(common::TestStore::default()).await; + let repository = MultipartRepository::new(fixture.store.clone()); + let mut session = fixtures::session(); + session.context = fixture.context; + session.owner.table = fixture.table; + session.location = fixture.table.file("data/composed.parquet").unwrap(); + repository.begin(&session, 100).await.unwrap(); + let mut parts = Vec::new(); + for (number, size, md5) in [ + (1_u16, 7_u64, "11111111111111111111111111111111"), + (2, 5, "22222222222222222222222222222222"), + ] { + let location = Location { + chunk_id: Some(ChunkId { + high: 7, + low: u64::from(number), + }), + offset: 100, + length: size + 34, + logical_offset: 0, + logical_length: size, + }; + let part = MultipartPart { + upload: session.upload, + number, + revision: 1, + modified_ms: 101, + owner: FileIdentity { + file: FileId::random(), + ..session.owner + }, + tree: None, + stream: Some(MultipartStreamPart { + length: size, + content: FileContent::from_locations(&[location], size, md5.into()).unwrap(), + }), + }; + repository.reserve_part(&session, &part, 101).await.unwrap(); + session = load(&repository, &session).await; + repository.settle_part(&session).await.unwrap(); + session = load(&repository, &session).await; + parts.push(part); + } + let selection = MultipartSelection::with_stream_parts(&parts).unwrap(); + let encoded = selection.encode(); + assert_eq!(MultipartSelection::decode(&encoded).unwrap(), selection); + let mut oversized_count = encoded.clone(); + let count_offset = 7 + 42 * parts.len(); + oversized_count[count_offset..count_offset + 8].copy_from_slice(&u64::MAX.to_le_bytes()); + assert!(MultipartSelection::decode(&oversized_count).is_err()); + let mut trailing = encoded; + trailing.push(0); + assert!(MultipartSelection::decode(&trailing).is_err()); + repository + .freeze_completion(&session, &selection, 102) + .await + .unwrap(); + session = load(&repository, &session).await; + assert_eq!(repository.load_selection(&session).await.unwrap(), selection); + replace_selected_part(&fixture, &parts[0]).await; + assert_eq!( + repository + .prepare_stream_publication(&session, 103) + .await + .unwrap(), + Some(true) + ); + session = load(&repository, &session).await; + let published = repository.publish(&session).await.unwrap().unwrap(); + let locations = published.content.locations(published.length).unwrap().unwrap(); + assert_eq!(published.length, 12); + assert_eq!( + locations + .iter() + .map(|location| location.logical_offset) + .collect::>(), + vec![0, 7] + ); + let mut digest = Md5::new(); + digest.update([0x11; 16]); + digest.update([0x22; 16]); + let mut expected = String::new(); + for byte in digest.finalize() { + write!(&mut expected, "{byte:02x}").unwrap(); + } + expected.push_str("-2"); + assert_eq!(published.content.etag(), Some(expected.as_str())); +} + #[tokio::test] async fn every_lost_publication_reply_recovers_the_same_seal_and_file_identity() { for lost in 1..=5 { diff --git a/lib/crowdb-access-iceberg/tests/multipart_record_test.rs b/lib/crowdb-access-iceberg/tests/multipart_record_test.rs index 47dd23fdd..22a1507ba 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_record_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_record_test.rs @@ -65,7 +65,8 @@ async fn multipart_records_round_trip_open_partial_publishing_published_and_abor revision: 2, modified_ms: 101, owner, - tree: part.tree, + tree: Some(part.tree), + stream: None, }; let record = StorageRecord::MultipartPart(Box::new(part.clone())); let bytes = record.encode().unwrap(); diff --git a/lib/crowdb-access-iceberg/tests/multipart_recovery_budget_test.rs b/lib/crowdb-access-iceberg/tests/multipart_recovery_budget_test.rs index 37c1ea3db..798e28ac8 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_recovery_budget_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_recovery_budget_test.rs @@ -68,7 +68,8 @@ async fn slow_first_session_does_not_starve_later_expiry_or_advance_unfinished_b revision: 1, modified_ms: 101, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, }; repository.reserve_part(&first, &part, 101).await.unwrap(); first = repository @@ -85,7 +86,7 @@ async fn slow_first_session_does_not_starve_later_expiry_or_advance_unfinished_b let selection = MultipartSelection::new(vec![SelectedPart { number: 1, revision: 1, - digest: part.tree.digest, + digest: part.selection_digest(), }]) .unwrap(); repository diff --git a/lib/crowdb-access-iceberg/tests/multipart_recovery_test.rs b/lib/crowdb-access-iceberg/tests/multipart_recovery_test.rs index fc469218d..4cb1454a1 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_recovery_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_recovery_test.rs @@ -51,7 +51,8 @@ async fn bounded_sweeps_settle_abandoned_part_mutations_then_expire_without_dele revision: 1, modified_ms: 101, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, }; assert!(repository.reserve_part(&session, &part, 101).await.unwrap()); uploads.push((session, part)); @@ -110,7 +111,8 @@ async fn recovery_advances_one_byte_window_per_visit_and_reports_unpublished_sea revision: 1, modified_ms: 101, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, }; repository.reserve_part(&initial, &part, 101).await.unwrap(); let pending = repository @@ -127,7 +129,7 @@ async fn recovery_advances_one_byte_window_per_visit_and_reports_unpublished_sea let selection = MultipartSelection::new(vec![SelectedPart { number: 1, revision: 1, - digest: part.tree.digest, + digest: part.selection_digest(), }]) .unwrap(); repository diff --git a/lib/crowdb-access-iceberg/tests/multipart_repository_test.rs b/lib/crowdb-access-iceberg/tests/multipart_repository_test.rs index 9c43ff34d..b06e55655 100644 --- a/lib/crowdb-access-iceberg/tests/multipart_repository_test.rs +++ b/lib/crowdb-access-iceberg/tests/multipart_repository_test.rs @@ -11,11 +11,14 @@ use std::sync::{atomic::Ordering, Arc}; use crowdb_access_iceberg::catalog::{CatalogError, CatalogStore, RootState}; use crowdb_access_iceberg::file::{ - FileIdentity, FileTreeWriter, MultipartPart, MultipartPhase, MultipartRepository, MultipartSession, + FileContent, FileIdentity, FileTreeWriter, MultipartPart, MultipartPhase, MultipartRepository, + MultipartSession, MultipartStreamPart, }; use crowdb_access_iceberg::key::FileId; use crowdb_access_iceberg::operation::mutation_identity; use crowdb_access_iceberg::record::StorageRecord; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; async fn setup() -> (file::TestFile, MultipartSession) { let fixture = file::TestFile::new(common::TestStore::default()).await; @@ -42,10 +45,99 @@ async fn part(session: &MultipartSession, number: u16, revision: u64, length: us revision, modified_ms: 101, owner, - tree: writer.finish().await.unwrap(), + tree: Some(writer.finish().await.unwrap()), + stream: None, } } +fn stream_part(session: &MultipartSession, number: u16, chunk_low: u64) -> MultipartPart { + MultipartPart { + upload: session.upload, + number, + revision: 1, + modified_ms: 101, + owner: FileIdentity { + file: FileId::random(), + ..session.owner + }, + tree: None, + stream: Some(MultipartStreamPart { + length: 10, + content: FileContent::from_locations( + &[Location { + chunk_id: Some(ChunkId { + high: 7, + low: chunk_low, + }), + offset: 0, + length: 44, + logical_offset: 0, + logical_length: 10, + }], + 10, + "0123456789abcdef0123456789abcdef".into(), + ) + .unwrap(), + }), + } +} + +#[tokio::test] +async fn streamed_parts_commit_independently_without_session_mutations() { + let (fixture, initial) = setup().await; + let repository = MultipartRepository::new(fixture.store.clone()); + repository.begin(&initial, 100).await.unwrap(); + let writes = fixture.store.writes.load(Ordering::SeqCst); + let first_input = stream_part(&initial, 1, 1); + let second_input = stream_part(&initial, 2, 2); + let (first, second) = tokio::join!( + repository.put_stream_part(&initial, &first_input, 101), + repository.put_stream_part(&initial, &second_input, 101), + ); + let first = first.unwrap().unwrap(); + let second = second.unwrap().unwrap(); + assert_eq!(first.revision, 1); + assert_eq!(second.revision, 1); + assert_eq!(fixture.store.writes.load(Ordering::SeqCst) - writes, 2); + assert_eq!(load(&repository, &initial).await, initial); + let replacement = repository + .put_stream_part(&initial, &stream_part(&initial, 1, 3), 102) + .await + .unwrap() + .unwrap(); + assert_eq!(replacement.revision, 2); + assert_eq!( + repository.part_for_upload(&initial, 1).await.unwrap(), + Some(replacement) + ); + assert!(repository.abort(&initial).await.unwrap()); + assert!(repository + .put_stream_part(&initial, &stream_part(&initial, 3, 4), 103) + .await + .is_err()); +} + +#[tokio::test] +async fn streamed_part_lost_cas_reply_resolves_from_the_part_record() { + let (fixture, initial) = setup().await; + let repository = MultipartRepository::new(fixture.store.clone()); + repository.begin(&initial, 100).await.unwrap(); + fixture + .store + .fail_after + .store(fixture.store.writes.load(Ordering::SeqCst) + 1, Ordering::SeqCst); + let published = repository + .put_stream_part(&initial, &stream_part(&initial, 1, 1), 101) + .await + .unwrap() + .unwrap(); + assert_eq!( + repository.part_for_upload(&initial, 1).await.unwrap(), + Some(published) + ); + assert_eq!(load(&repository, &initial).await, initial); +} + #[tokio::test] async fn reservation_stamps_durable_part_time_from_admission_clock() { let (fixture, initial) = setup().await; diff --git a/lib/crowdb-access-s3/src/integrity.rs b/lib/crowdb-access-s3/src/integrity.rs index b7f87e502..208390972 100644 --- a/lib/crowdb-access-s3/src/integrity.rs +++ b/lib/crowdb-access-s3/src/integrity.rs @@ -28,23 +28,31 @@ pub enum IntegrityError { /// future contract. pub struct SinglePartIntegrity { md5: md5::Context, - sha256: Sha256, + sha256: Option, } impl Default for SinglePartIntegrity { fn default() -> Self { + Self::new(false) + } +} + +impl SinglePartIntegrity { + /// Enables payload SHA-256 only when the request declares that digest. + #[must_use] + pub fn new(check_payload_sha256: bool) -> Self { Self { md5: md5::Context::new(), - sha256: Sha256::new(), + sha256: check_payload_sha256.then(Sha256::new), } } -} -impl SinglePartIntegrity { /// Adds one immutable body frame without copying it. pub fn update(&mut self, bytes: &Bytes) { self.md5.consume(bytes); - self.sha256.update(bytes); + if let Some(sha256) = &mut self.sha256 { + sha256.update(bytes); + } } /// Returns the persisted single-part `ETag` and its raw checksum bytes. @@ -95,6 +103,7 @@ impl SinglePartIntegrity { if expected.len() != 64 || !expected.bytes().all(|byte| byte.is_ascii_hexdigit()) { return Err(IntegrityError::InvalidPayloadDigest); } + let sha256 = sha256.ok_or(IntegrityError::InvalidPayloadDigest)?; let actual = format!("{:x}", sha256.finalize()); if !actual.eq_ignore_ascii_case(expected) { return Err(IntegrityError::PayloadMismatch); diff --git a/lib/crowdb-access-s3/src/metrics.rs b/lib/crowdb-access-s3/src/metrics.rs index 154cc72fe..4084ab3ce 100644 --- a/lib/crowdb-access-s3/src/metrics.rs +++ b/lib/crowdb-access-s3/src/metrics.rs @@ -7,7 +7,7 @@ use std::fmt::Write as _; use std::sync::atomic::{AtomicBool, AtomicU64, AtomicU8, Ordering}; use crowdb_chunk_client::{ - ChunkIoClient, LargeWriteBufferMetricsSnapshot, LargeWriteRepairMetricsSnapshot, + ChunkIoClient, LargeWriteBufferMetricsSnapshot, LargeWriteRepairMetricsSnapshot, ReadFlowMetricsSnapshot, SmallWriteMetricsSnapshot, }; @@ -210,6 +210,7 @@ impl S3Metrics { large_write_buffers: chunks.large_write_buffer_metrics(), large_write_repairs: chunks.large_write_repair_metrics(), small_writes: chunks.small_write_metrics(), + read_flow: chunks.read_flow_metrics(), checksum_bytes: request.checksum_bytes, metadata_retries: request.metadata_retries, cleanup_enqueued: request.cleanup_enqueued, @@ -271,6 +272,8 @@ impl S3Metrics { } if let Some(chunks) = chunks { append_chunk_metrics(&mut output, chunks.large_write_buffer_metrics()); + append_small_write_metrics(&mut output, &chunks.small_write_metrics()); + append_read_flow_metrics(&mut output, chunks.read_flow_metrics()); } output } @@ -341,6 +344,107 @@ fn append_chunk_metrics(output: &mut String, buffers: LargeWriteBufferMetricsSna } } +fn append_small_write_metrics(output: &mut String, small: &SmallWriteMetricsSnapshot) { + for (name, value) in [ + ("crowdb_s3_small_write_submitted_total", small.submitted), + ("crowdb_s3_small_write_completed_total", small.completed), + ("crowdb_s3_small_write_failed_total", small.failed), + ( + "crowdb_s3_small_write_aggregate_buffers_total", + small.aggregate_write_buffers, + ), + ( + "crowdb_s3_small_write_aggregate_payload_bytes_total", + small.aggregate_write_payload_bytes, + ), + ("crowdb_s3_small_write_queue_delay_ns_total", small.queue_delay_ns), + ( + "crowdb_s3_small_write_reservation_wait_ns_total", + small.reservation_wait_ns, + ), + ] { + append_metric(output, name, "", value); + } + append_metric( + output, + "crowdb_s3_small_write_active_pipelines", + "", + small.active_pipelines, + ); +} + +fn append_read_flow_metrics(output: &mut String, flow: ReadFlowMetricsSnapshot) { + for (name, value) in [ + ( + "crowdb_s3_chunk_read_location_normalizations_total", + flow.location_normalizations, + ), + ( + "crowdb_s3_chunk_read_locations_examined_total", + flow.locations_examined, + ), + ( + "crowdb_s3_chunk_read_range_locations_examined_total", + flow.range_locations_examined, + ), + ("crowdb_s3_chunk_read_stream_windows_total", flow.stream_windows), + ( + "crowdb_s3_chunk_read_stream_units_completed_total", + flow.stream_units_completed, + ), + ( + "crowdb_s3_chunk_read_stream_out_of_order_total", + flow.stream_out_of_order, + ), + ( + "crowdb_s3_chunk_read_stream_credit_stalls_total", + flow.stream_credit_stalls, + ), + ( + "crowdb_s3_chunk_read_stream_credit_wait_ns_total", + flow.stream_credit_wait_ns, + ), + ( + "crowdb_s3_chunk_read_stream_bytes_reserved_total", + flow.stream_bytes_reserved, + ), + ( + "crowdb_s3_chunk_read_stream_bytes_released_total", + flow.stream_bytes_released, + ), + ("crowdb_s3_chunk_read_layout_queries_total", flow.layout_queries), + ( + "crowdb_s3_chunk_read_layout_query_wait_ns_total", + flow.layout_query_wait_ns, + ), + ( + "crowdb_s3_chunk_read_strip_read_wait_ns_total", + flow.strip_read_wait_ns, + ), + ( + "crowdb_s3_chunk_range_read_wait_ns_total", + flow.chunk_read_wait_ns, + ), + ( + "crowdb_s3_chunk_read_frame_decode_wait_ns_total", + flow.frame_decode_wait_ns, + ), + ( + "crowdb_s3_chunk_read_frame_parse_wait_ns_total", + flow.frame_parse_wait_ns, + ), + ] { + append_metric(output, name, "", value); + } + append_metric( + output, + "crowdb_s3_chunk_read_stream_retained_bytes", + "", + flow.stream_bytes_reserved + .saturating_sub(flow.stream_bytes_released), + ); +} + fn append_metric(output: &mut String, name: &str, labels: &str, value: u64) { if labels.is_empty() { let _ = writeln!(output, "{name} {value}"); @@ -376,6 +480,7 @@ pub struct S3DataPathMetricsSnapshot { pub large_write_buffers: LargeWriteBufferMetricsSnapshot, pub large_write_repairs: LargeWriteRepairMetricsSnapshot, pub small_writes: SmallWriteMetricsSnapshot, + pub read_flow: ReadFlowMetricsSnapshot, pub checksum_bytes: u64, pub metadata_retries: u64, pub cleanup_enqueued: u64, diff --git a/lib/crowdb-access-s3/src/streaming.rs b/lib/crowdb-access-s3/src/streaming.rs index 3089a704f..adf54f514 100644 --- a/lib/crowdb-access-s3/src/streaming.rs +++ b/lib/crowdb-access-s3/src/streaming.rs @@ -156,8 +156,7 @@ where write_body_with_checksums_metered(body, writer, expected_content_md5, expected_payload_sha256, None).await } -/// Streams one body while accounting the exact bytes presented to both -/// checksum implementations. +/// Streams one body while accounting the exact bytes presented to integrity checks. /// /// # Errors /// @@ -174,7 +173,7 @@ where B::Error: std::fmt::Display, W: ChunkIoWriter, { - let mut integrity = SinglePartIntegrity::default(); + let mut integrity = SinglePartIntegrity::new(expected_payload_sha256.is_some()); loop { if writer.input_complete() { return finish_integrity(integrity, expected_content_md5, expected_payload_sha256); @@ -201,6 +200,65 @@ where } } +/// Coalesces receive fragments into one bounded push per receive owner. +/// # Errors +/// Returns body, writer, or declared checksum errors without publishing data. +pub async fn write_body_with_checksums_buffered( + body: &mut B, + writer: &mut W, + expected_content_md5: Option<&str>, + expected_payload_sha256: Option<&str>, + receive_bytes: usize, + metrics: Option<&crate::metrics::S3Metrics>, +) -> Result<(String, Vec), PutOutcome> +where + B: Body + Unpin, + B::Error: std::fmt::Display, + W: ChunkIoWriter, +{ + if receive_bytes == 0 || receive_bytes > 1024 * 1024 { + return Err(put_error(PutErrorCode::BodyRead, "invalid receive buffer size")); + } + let mut integrity = SinglePartIntegrity::new(expected_payload_sha256.is_some()); + let mut pending = Vec::with_capacity(receive_bytes); + loop { + if writer.input_complete() { + break; + } + while !writer.require_data() && !writer.input_complete() { + writer.wait_for_capacity().await; + } + let frame = poll_fn(|context| Pin::new(&mut *body).poll_frame(context)).await; + let Some(frame) = frame else { break }; + let frame = frame.map_err(|error| put_error(PutErrorCode::BodyRead, error))?; + let Ok(mut data) = frame.into_data() else { continue }; + if let Some(metrics) = metrics { + metrics.record_checksum_bytes(data.len()); + } + integrity.update(&data); + while !data.is_empty() { + let count = (receive_bytes - pending.len()).min(data.len()); + pending.extend_from_slice(&data.split_to(count)); + if pending.len() == receive_bytes { + writer + .on_data(Bytes::from(std::mem::replace( + &mut pending, + Vec::with_capacity(receive_bytes), + ))) + .await + .map_err(|error| put_error(PutErrorCode::ChunkWrite, error))?; + } + } + } + if !pending.is_empty() { + writer + .on_data(Bytes::from(pending)) + .await + .map_err(|error| put_error(PutErrorCode::ChunkWrite, error))?; + } + finish_integrity(integrity, expected_content_md5, expected_payload_sha256) +} + /// Streams one body through a native owner provider while calculating object /// integrity over the socket-filled payload views. Full owners and the EOF /// prefix are handed to the large writer without payload copies. @@ -252,7 +310,7 @@ where B::Error: std::fmt::Display, W: ChunkIoWriter, { - let mut integrity = SinglePartIntegrity::default(); + let mut integrity = SinglePartIntegrity::new(expected_payload_sha256.is_some()); loop { while !writer.require_data() { writer.wait_for_capacity().await; diff --git a/lib/crowdb-access-s3/tests/integrity_test.rs b/lib/crowdb-access-s3/tests/integrity_test.rs index 81ba9346e..9588e569b 100644 --- a/lib/crowdb-access-s3/tests/integrity_test.rs +++ b/lib/crowdb-access-s3/tests/integrity_test.rs @@ -35,7 +35,7 @@ fn content_md5_is_checked_before_publication() { #[test] fn signed_payload_sha256_is_checked_incrementally() { - let mut valid = SinglePartIntegrity::default(); + let mut valid = SinglePartIntegrity::new(true); valid.update(&Bytes::from_static(b"abc")); assert!(valid .finish_validated_checksums( @@ -44,7 +44,7 @@ fn signed_payload_sha256_is_checked_incrementally() { ) .is_ok()); - let mut mismatch = SinglePartIntegrity::default(); + let mut mismatch = SinglePartIntegrity::new(true); mismatch.update(&Bytes::from_static(b"abc")); assert_eq!( mismatch.finish_validated_checksums( diff --git a/lib/crowdb-access-s3/tests/streaming_test.rs b/lib/crowdb-access-s3/tests/streaming_test.rs index 79c87ab64..6828d567f 100644 --- a/lib/crowdb-access-s3/tests/streaming_test.rs +++ b/lib/crowdb-access-s3/tests/streaming_test.rs @@ -13,9 +13,9 @@ use crowdb_access_s3::metadata::{BucketId, ObjectRecord}; use crowdb_access_s3::metrics::S3Metrics; use crowdb_access_s3::native_buffer::NativeBodyAllocator; use crowdb_access_s3::streaming::{ - attach_completed_locations, cleanup_after_definite_error, write_body, - write_native_body_with_checksums_metered, FailedPublicationCleanup, FailedPublicationTarget, - PutErrorCode, PutOutcome, + attach_completed_locations, cleanup_after_definite_error, write_body, write_body_with_checksums, + write_body_with_checksums_buffered, write_native_body_with_checksums_metered, FailedPublicationCleanup, + FailedPublicationTarget, PutErrorCode, PutOutcome, }; use crowdb_chunk_client::{ChunkIoWriter, FeedStatus, FramedWriteBuffer, IoError}; use crowdb_protocol::chunkdb::rpc::Location; @@ -192,6 +192,65 @@ async fn declared_length_completion_does_not_poll_for_an_extra_body_frame() { assert_eq!(body.polls.load(Ordering::Relaxed), 1); } +#[tokio::test] +async fn payload_sha256_is_verified_only_when_declared() { + let payload = Bytes::from_static(b"abc"); + let expected = "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"; + let invalid = "0".repeat(64); + for (digest, accepted) in [ + (None, true), + (Some(expected), true), + (Some(invalid.as_str()), false), + ] { + let mut body = TestBody { + frames: VecDeque::from([payload.clone()]), + polls: AtomicUsize::new(0), + }; + let mut writer = NativeOwnerWriter::default(); + let result = write_body_with_checksums(&mut body, &mut writer, None, digest).await; + assert_eq!(result.is_ok(), accepted); + assert_eq!(writer.generic_frames, 1); + } +} + +#[derive(Default)] +struct CollectingWriter(Vec); + +#[async_trait] +impl ChunkIoWriter for CollectingWriter { + async fn on_data(&mut self, buffer: Bytes) -> crowdb_chunk_client::Result { + self.0.push(buffer); + Ok(FeedStatus::Continue) + } + + async fn on_finish(&mut self) -> crowdb_chunk_client::Result> { + Ok(Vec::new()) + } + async fn on_error(&mut self) -> crowdb_chunk_client::Result> { + Ok(Vec::new()) + } + fn require_data(&self) -> bool { + true + } +} + +#[tokio::test] +async fn receive_fragments_fill_one_small_object_buffer_before_pipeline_push() { + let mut body = TestBody { + frames: VecDeque::from([ + Bytes::from_static(b"ab"), + Bytes::from_static(b"cd"), + Bytes::from_static(b"e"), + ]), + polls: AtomicUsize::new(0), + }; + let mut writer = CollectingWriter::default(); + write_body_with_checksums_buffered(&mut body, &mut writer, None, None, 5, None) + .await + .unwrap(); + assert_eq!(writer.0, vec![Bytes::from_static(b"abcde")]); +} + #[derive(Default)] struct NativeOwnerWriter { generic_frames: usize, diff --git a/lib/crowdb-chunk-client/src/benchmark.rs b/lib/crowdb-chunk-client/src/benchmark.rs index f6fdb2230..8d9f44a7d 100644 --- a/lib/crowdb-chunk-client/src/benchmark.rs +++ b/lib/crowdb-chunk-client/src/benchmark.rs @@ -8,6 +8,7 @@ use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::sync::Arc; use std::time::{Duration, Instant}; +use bytes::Bytes; use crowdb_common::metrics::perf::DramBwCounter; use serde::Serialize; use tokio::io::{AsyncRead, ReadBuf}; @@ -174,6 +175,7 @@ pub struct ReadBenchmarkResult { pub dram_read_mib_s: Option, pub dram_write_mib_s: Option, pub dram_total_mib_s: Option, + pub read_flow: crate::ReadFlowMetricsSnapshot, pub error_messages: Vec, } @@ -707,6 +709,7 @@ pub async fn run_read_benchmark(client: ChunkIoClient, config: ReadBenchmarkConf Err(error) => return failed_read_before_load(&config, &error.to_string()), }; let preparation_secs = preparation_started.elapsed().as_secs_f64(); + let read_flow_before = client.read_flow_metrics(); let started = Instant::now(); let deadline = config.duration.map(|duration| started + duration); let mut dram_bw = DramBwCounter::new(); @@ -742,6 +745,7 @@ pub async fn run_read_benchmark(client: ChunkIoClient, config: ReadBenchmarkConf let requested_reads = next_read.load(Ordering::Relaxed).min(config.request_count); let incomplete_reads = requested_reads.saturating_sub(total.reads.saturating_add(total.errors)); let (dram_read_mib_s, dram_write_mib_s, dram_total_mib_s) = sample_dram(&mut dram_bw); + let read_flow = client.read_flow_metrics().since(read_flow_before); finalize_read_result( preparation_secs, elapsed_secs, @@ -749,6 +753,7 @@ pub async fn run_read_benchmark(client: ChunkIoClient, config: ReadBenchmarkConf incomplete_reads, total, (dram_read_mib_s, dram_write_mib_s, dram_total_mib_s), + read_flow, ) } @@ -776,7 +781,7 @@ async fn run_read_worker( let object = &objects[index]; let operation_started = Instant::now(); match client.read_object(&object.locations).await { - Ok(bytes) if bytes.len() as u64 == object.logical_bytes => { + Ok(bytes) if bytes.iter().map(Bytes::len).sum::() as u64 == object.logical_bytes => { result.reads += 1; result.logical_bytes += object.logical_bytes; if use_large { @@ -793,7 +798,7 @@ async fn run_read_worker( &mut result.error_messages, format!( "read {request}: length {} != {}", - bytes.len(), + bytes.iter().map(Bytes::len).sum::(), object.logical_bytes ), ); @@ -838,6 +843,7 @@ fn finalize_read_result( incomplete_reads: u64, total: ReadWorkerResult, dram: (Option, Option, Option), + read_flow: crate::ReadFlowMetricsSnapshot, ) -> ReadBenchmarkResult { ReadBenchmarkResult { preparation_secs, @@ -866,6 +872,7 @@ fn finalize_read_result( dram_read_mib_s: dram.0, dram_write_mib_s: dram.1, dram_total_mib_s: dram.2, + read_flow, error_messages: total.error_messages, } } @@ -935,6 +942,7 @@ fn failed_read_before_load(config: &ReadBenchmarkConfig, message: &str) -> ReadB dram_read_mib_s: None, dram_write_mib_s: None, dram_total_mib_s: None, + read_flow: crate::ReadFlowMetricsSnapshot::default(), error_messages: vec![format!("prepare reads: {message}")], } } diff --git a/lib/crowdb-chunk-client/src/chunk.rs b/lib/crowdb-chunk-client/src/chunk.rs index 948fec791..de3904948 100644 --- a/lib/crowdb-chunk-client/src/chunk.rs +++ b/lib/crowdb-chunk-client/src/chunk.rs @@ -21,6 +21,7 @@ pub mod mirror_chunk_writer; pub mod mirror_flow; pub mod mirror_strip_writer; pub mod parity_writer; +pub(crate) mod read_credit; pub(crate) mod segment_writer; pub mod strip; pub mod strip_reader; diff --git a/lib/crowdb-chunk-client/src/chunk/chunk_reader.rs b/lib/crowdb-chunk-client/src/chunk/chunk_reader.rs index 57a3076ad..39c47171f 100644 --- a/lib/crowdb-chunk-client/src/chunk/chunk_reader.rs +++ b/lib/crowdb-chunk-client/src/chunk/chunk_reader.rs @@ -3,34 +3,41 @@ //! Object and range reads over current chunk layouts. +use std::collections::BTreeMap; use std::ops::Range; +use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::Arc; use std::time::{Duration, Instant}; use bytes::{Bytes, BytesMut}; use crowdb_protocol::chunkdb::rpc::{ AdHocEcRecoveryDisposition, AdHocEcRecoveryRequest, Chunk, ChunkState, Location, QueryChunkRequest, + QueryChunkResponse, }; use crowdb_protocol::common::ChunkId; use crowdb_protocol::diskdb::rpc::Segment; use crowdb_protocol::frame::{ - parse_frame, ChunkLocation, FrameMagic, FRAME_FOOTER_BYTES, FRAME_HEADER_PREFIX_BYTES, MAX_FRAME_BYTES, - MAX_FRAME_PAYLOAD_BYTES, + parse_frame, parse_frame_views, ChunkLocation, FrameMagic, FRAME_FOOTER_BYTES, FRAME_HEADER_PREFIX_BYTES, + MAX_FRAME_BYTES, MAX_FRAME_PAYLOAD_BYTES, }; use tokio::sync::Semaphore; use tokio::task::JoinSet; use super::client_recovery::ClientRecovery; +use super::read_credit::{retain, ReadBudget, ReadLease, StreamSlots}; use super::strip_reader::StripReader; +use crate::metrics::ReadFlowMetrics; use crate::{ChunkAllocator, DiskWriter, IoError, ReadError, ReadResult}; const KIB: u64 = 1024; -const DEFAULT_STREAM_WINDOW: usize = 64 * 1024 * 1024; +const DEFAULT_STREAM_WINDOW: usize = 1024 * 1024; /// Bounded retry, streaming, and EC-recovery memory policy. #[derive(Debug, Clone)] pub struct ChunkReadPolicy { pub stream_window_bytes: usize, + pub stream_slots: usize, + pub global_stream_bytes: usize, pub recovery_memory_bytes: usize, pub layout_safety_margin: Duration, pub max_layout_retries: usize, @@ -62,10 +69,30 @@ pub struct PartialReadResult { pub failures: Vec, } +#[derive(Clone)] +struct LayoutSnapshot { + chunk_id: ChunkId, + chunk: Arc, + deadline: Instant, + valid: Arc, +} + +impl LayoutSnapshot { + fn usable(&self) -> bool { + self.valid.load(Ordering::Acquire) && Instant::now() < self.deadline + } + + fn invalidate(&self) { + self.valid.store(false, Ordering::Release); + } +} + impl Default for ChunkReadPolicy { fn default() -> Self { Self { stream_window_bytes: DEFAULT_STREAM_WINDOW, + stream_slots: 3, + global_stream_bytes: 256 * 1024 * 1024, recovery_memory_bytes: 256 * 1024 * 1024, layout_safety_margin: Duration::from_millis(5), max_layout_retries: 3, @@ -79,6 +106,8 @@ impl Default for ChunkReadPolicy { impl ChunkReadPolicy { fn validate(&self) -> ReadResult<()> { if self.stream_window_bytes == 0 + || self.stream_slots == 0 + || self.global_stream_bytes < DEFAULT_STREAM_WINDOW || self.recovery_memory_bytes == 0 || self.recovery_memory_bytes > u32::MAX as usize || self.max_layout_retries == 0 @@ -98,6 +127,8 @@ pub struct ChunkReader { chunkdb: Arc, strip_reader: StripReader, policy: ChunkReadPolicy, + flow_metrics: Arc, + stream_budget: Arc, } impl ChunkReader { @@ -106,7 +137,7 @@ impl ChunkReader { disk_io: Arc, policy: ChunkReadPolicy, ) -> ReadResult { - Self::new_with_metrics(chunkdb, disk_io, policy, Arc::default()) + Self::new_with_metrics(chunkdb, disk_io, policy, Arc::default(), Arc::default()) } pub(crate) fn new_with_metrics( @@ -114,6 +145,7 @@ impl ChunkReader { disk_io: Arc, policy: ChunkReadPolicy, metrics: Arc, + flow_metrics: Arc, ) -> ReadResult { policy.validate()?; let recovery_memory = Arc::new(Semaphore::new(policy.recovery_memory_bytes)); @@ -125,34 +157,75 @@ impl ChunkReader { policy.ad_hoc_window, metrics, )); + let stream_budget = Arc::new(ReadBudget::new( + policy.global_stream_bytes, + Arc::clone(&flow_metrics), + )); Ok(Self { chunkdb, strip_reader: StripReader::new(disk_io, recovery_memory, policy.recovery_memory_bytes) .with_ad_hoc(ad_hoc), policy, + flow_metrics, + stream_budget, + }) + } + + /// Cumulative work counters for the shared read path. + pub fn flow_metrics_snapshot(&self) -> crate::metrics::ReadFlowMetricsSnapshot { + self.flow_metrics.snapshot() + } + + async fn query_chunk_timed(&self, chunk_id: ChunkId) -> crate::Result { + let started = Instant::now(); + self.flow_metrics.layout_queries.inc(); + let result = self + .chunkdb + .query_chunk(QueryChunkRequest { + chunk_id: Some(chunk_id), + }) + .await; + self.flow_metrics + .layout_query_wait_ns + .inc_by(u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)); + result + } + + async fn query_layout(&self, chunk_id: ChunkId) -> ReadResult { + let started = Instant::now(); + let response = self + .query_chunk_timed(chunk_id) + .await + .map_err(map_metadata_error)?; + let deadline = started + + Duration::from_millis(response.layout_validity_ms) + .saturating_sub(self.policy.layout_safety_margin); + let chunk = response + .chunk + .ok_or_else(|| ReadError::ChunkDeleted(format!("{}:{}", chunk_id.high, chunk_id.low)))?; + Ok(LayoutSnapshot { + chunk_id, + chunk: Arc::new(chunk), + deadline, + valid: Arc::new(AtomicBool::new(true)), }) } - pub async fn read_object(&self, locations: &[Location]) -> ReadResult { - let (_, object_length) = normalize_locations(locations)?; - self.read_range(locations, 0, object_length).await + pub async fn read_object(&self, locations: &[Location]) -> ReadResult> { + self.flow_metrics.location_normalizations.inc(); + self.flow_metrics + .locations_examined + .inc_by(u64::try_from(locations.len()).unwrap_or(u64::MAX)); + let (locations, object_length) = normalize_locations(locations)?; + let partial = self + .read_range_partial_normalized(&locations, object_length, 0, object_length) + .await?; + complete_read(partial, 0, object_length) } - pub async fn read_range(&self, locations: &[Location], start: u64, end: u64) -> ReadResult { + pub async fn read_range(&self, locations: &[Location], start: u64, end: u64) -> ReadResult> { let partial = self.read_range_partial(locations, start, end).await?; - if let Some(failure) = partial.failures.into_iter().next() { - return Err(ReadError::FailedRange { - start: failure.start, - end: failure.end, - message: failure.error.to_string(), - }); - } - let expected = usize::try_from(end - start).map_err(|_| ReadError::InvalidRange { - start, - end, - object_length: end, - })?; - assemble_ranges(partial.ranges, start, expected) + complete_read(partial, start, end) } pub async fn read_range_partial( @@ -161,7 +234,22 @@ impl ChunkReader { start: u64, end: u64, ) -> ReadResult { + self.flow_metrics.location_normalizations.inc(); + self.flow_metrics + .locations_examined + .inc_by(u64::try_from(locations.len()).unwrap_or(u64::MAX)); let (locations, object_length) = normalize_locations(locations)?; + self.read_range_partial_normalized(&locations, object_length, start, end) + .await + } + + async fn read_range_partial_normalized( + &self, + locations: &[Location], + object_length: u64, + start: u64, + end: u64, + ) -> ReadResult { if start > end || end > object_length { return Err(ReadError::InvalidRange { start, @@ -174,8 +262,16 @@ impl ChunkReader { } let mut reads = JoinSet::new(); - for location in locations { + let first = locations.partition_point(|location| { + location.logical_offset.saturating_add(location.logical_length) <= start + }); + let mut examined = 0u64; + for location in &locations[first..] { let loc_start = location.logical_offset; + if loc_start >= end { + break; + } + examined += 1; let loc_end = loc_start + location.logical_length; let overlap_start = start.max(loc_start); let overlap_end = end.min(loc_end); @@ -183,6 +279,7 @@ impl ChunkReader { continue; } let reader = self.clone(); + let location = location.clone(); reads.spawn(async move { let local_start = overlap_start - loc_start; let length = overlap_end - overlap_start; @@ -191,6 +288,7 @@ impl ChunkReader { .await }); } + self.flow_metrics.range_locations_examined.inc_by(examined); let mut partial = PartialReadResult::default(); while let Some(result) = reads.join_next().await { let location = result.map_err(|error| ReadError::DiskIo(error.to_string()))??; @@ -199,10 +297,26 @@ impl ChunkReader { } partial.ranges.sort_unstable_by_key(|range| range.start); partial.failures.sort_unstable_by_key(|range| range.start); - Ok(coalesce_partial_ranges(partial)) + let mut failures: Vec = Vec::with_capacity(partial.failures.len()); + for failure in partial.failures { + if let Some(previous) = failures + .last_mut() + .filter(|previous| previous.end == failure.start) + { + previous.end = failure.end; + } else { + failures.push(failure); + } + } + partial.failures = failures; + Ok(partial) } pub fn read_stream(&self, locations: &[Location]) -> ReadResult { + self.flow_metrics.location_normalizations.inc(); + self.flow_metrics + .locations_examined + .inc_by(u64::try_from(locations.len()).unwrap_or(u64::MAX)); let (locations, object_length) = normalize_locations(locations)?; self.range_stream(locations, 0, object_length, object_length) } @@ -214,6 +328,10 @@ impl ChunkReader { start: u64, end: u64, ) -> ReadResult { + self.flow_metrics.location_normalizations.inc(); + self.flow_metrics + .locations_examined + .inc_by(u64::try_from(locations.len()).unwrap_or(u64::MAX)); let (locations, object_length) = normalize_locations(locations)?; self.range_stream(locations, start, end, object_length) } @@ -236,9 +354,16 @@ impl ChunkReader { reader: self.clone(), locations: Arc::from(locations), cursor: start, + fetch_cursor: start, + delivery_cursor: start, end, window_bytes: self.policy.stream_window_bytes as u64, + slots: Arc::new(StreamSlots::new(self.policy.stream_slots)), + reads: JoinSet::new(), + completed: BTreeMap::new(), + layouts: Vec::new(), pending_error: None, + pending_ranges: Vec::new().into_iter(), }) } @@ -272,13 +397,7 @@ impl ChunkReader { .ok_or_else(|| ReadError::InvalidLocations("frame end overflows".into()))?; for attempt in 1..=self.policy.max_layout_retries { let query_started = Instant::now(); - let response = match self - .chunkdb - .query_chunk(QueryChunkRequest { - chunk_id: Some(chunk_id), - }) - .await - { + let response = match self.query_chunk_timed(chunk_id).await { Ok(response) => response, Err(error) => { tracing::warn!( @@ -362,7 +481,12 @@ impl ChunkReader { continue; } }; - match parse_frame(&bytes, chunk_id) { + let decode_started = Instant::now(); + let parsed_frame = parse_frame(&bytes, chunk_id); + let parse_ns = u64::try_from(decode_started.elapsed().as_nanos()).unwrap_or(u64::MAX); + self.flow_metrics.frame_decode_wait_ns.inc_by(parse_ns); + self.flow_metrics.frame_parse_wait_ns.inc_by(parse_ns); + match parsed_frame { Ok(frame) if frame.header.magic == expected_magic => { if Instant::now() >= deadline { tracing::warn!( @@ -444,10 +568,22 @@ impl ChunkReader { local_start: u64, length: u64, logical_start: u64, + ) -> ReadResult { + self.read_location_partial_cached(location, local_start, length, logical_start, None) + .await + } + + async fn read_location_partial_cached( + &self, + location: &Location, + local_start: u64, + length: u64, + logical_start: u64, + mut first_layout: Option, ) -> ReadResult { if location.length != location.logical_length { return self - .read_framed_location(location, local_start, length, logical_start) + .read_framed_location(location, local_start, length, logical_start, first_layout) .await; } let chunk_id = location @@ -458,34 +594,32 @@ impl ChunkReader { .checked_add(local_start) .ok_or_else(|| ReadError::InvalidLocations("location physical offset overflows".into()))?; for _ in 0..self.policy.max_layout_retries { - let query_started = Instant::now(); - let response = self - .chunkdb - .query_chunk(QueryChunkRequest { - chunk_id: Some(chunk_id), - }) - .await - .map_err(map_metadata_error)?; - let validity = Duration::from_millis(response.layout_validity_ms); - let usable = validity.saturating_sub(self.policy.layout_safety_margin); - let deadline = query_started + usable; - let mut chunk = response - .chunk - .ok_or_else(|| ReadError::ChunkDeleted(format!("{}:{}", chunk_id.high, chunk_id.low)))?; + let layout = match first_layout.take().filter(LayoutSnapshot::usable) { + Some(layout) => layout, + None => self.query_layout(chunk_id).await?, + }; let (partial, observations) = self - .read_chunk_range_partial(&chunk, physical_start, length, logical_start) + .read_chunk_range_partial(&layout.chunk, physical_start, length, logical_start) .await?; - if Instant::now() >= deadline { + if Instant::now() >= layout.deadline { + layout.invalidate(); continue; } - if self - .mark_observed_failures(&mut chunk, observations) - .await - .is_err() + if observations + .iter() + .any(|observation| !observation.failed_segments.is_empty()) { - continue; + layout.invalidate(); + let mut chunk = (*layout.chunk).clone(); + if self + .mark_observed_failures(&mut chunk, observations) + .await + .is_err() + { + continue; + } } - if Instant::now() < deadline { + if Instant::now() < layout.deadline { return Ok(partial); } } @@ -498,6 +632,7 @@ impl ChunkReader { local_start: u64, length: u64, logical_start: u64, + first_layout: Option, ) -> ReadResult { let chunk_id = location .chunk_id @@ -535,6 +670,7 @@ impl ChunkReader { selected_logical_start, local_start..local_start + length, logical_start, + first_layout, ) .await } @@ -546,58 +682,66 @@ impl ChunkReader { selected_logical_start: u64, requested: Range, output_logical_start: u64, + mut first_layout: Option, ) -> ReadResult { let chunk_id = framed.chunk_id; for _ in 0..self.policy.max_layout_retries { - let query_started = Instant::now(); - let response = self - .chunkdb - .query_chunk(QueryChunkRequest { - chunk_id: Some(chunk_id), - }) - .await - .map_err(map_metadata_error)?; - let validity = Duration::from_millis(response.layout_validity_ms); - let deadline = query_started + validity.saturating_sub(self.policy.layout_safety_margin); - let mut chunk = response - .chunk - .ok_or_else(|| ReadError::ChunkDeleted(format!("{}:{}", chunk_id.high, chunk_id.low)))?; + let layout = match first_layout.take().filter(LayoutSnapshot::usable) { + Some(layout) => layout, + None => self.query_layout(chunk_id).await?, + }; let (physical, observations) = self .read_chunk_range_partial( - &chunk, + &layout.chunk, physical_range.start, physical_range.end - physical_range.start, physical_range.start, ) .await?; - let parsed = match decode_framed_partial( + let decode_started = Instant::now(); + let decoded = decode_framed_partial( &physical, framed, selected_logical_start, requested.clone(), output_logical_start, - ) { + &self.flow_metrics, + ); + self.flow_metrics + .frame_decode_wait_ns + .inc_by(u64::try_from(decode_started.elapsed().as_nanos()).unwrap_or(u64::MAX)); + let parsed = match decoded { Ok(parsed) => parsed, Err(error) => { + layout.invalidate(); let corrupt = mark_served_segments_corrupt(observations); - if corrupt.is_empty() - || Instant::now() >= deadline - || self.mark_observed_failures(&mut chunk, corrupt).await.is_err() - { + if corrupt.is_empty() || Instant::now() >= layout.deadline { + return Err(error); + } + let mut chunk = (*layout.chunk).clone(); + if self.mark_observed_failures(&mut chunk, corrupt).await.is_err() { return Err(error); } continue; } }; - if Instant::now() >= deadline { + if Instant::now() >= layout.deadline { + layout.invalidate(); continue; } - if self - .mark_observed_failures(&mut chunk, observations) - .await - .is_err() + if observations + .iter() + .any(|observation| !observation.failed_segments.is_empty()) { - continue; + layout.invalidate(); + let mut chunk = (*layout.chunk).clone(); + if self + .mark_observed_failures(&mut chunk, observations) + .await + .is_err() + { + continue; + } } return Ok(parsed); } @@ -610,6 +754,23 @@ impl ChunkReader { start: u64, length: u64, logical_start: u64, + ) -> ReadResult<(PartialReadResult, Vec)> { + let started = Instant::now(); + let result = self + .read_chunk_range_partial_inner(chunk, start, length, logical_start) + .await; + self.flow_metrics + .chunk_read_wait_ns + .inc_by(u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)); + result + } + + async fn read_chunk_range_partial_inner( + &self, + chunk: &Chunk, + start: u64, + length: u64, + logical_start: u64, ) -> ReadResult<(PartialReadResult, Vec)> { if chunk.state == ChunkState::Deleted as i32 { let id = chunk.id.unwrap_or_default(); @@ -644,6 +805,7 @@ impl ChunkReader { for (part_start, part_end) in strip_read_parts(strip, cursor, overlap_end)? { let range_start = logical_start + (part_start - start); let range_end = range_start + (part_end - part_start); + let read_started = Instant::now(); let observed = self .strip_reader .read_observed( @@ -654,6 +816,9 @@ impl ChunkReader { part_end - part_start, ) .await; + self.flow_metrics + .strip_read_wait_ns + .inc_by(u64::try_from(read_started.elapsed().as_nanos()).unwrap_or(u64::MAX)); if !observed.failed_segments.is_empty() { observations.push(StripFailureObservation { strip_sequence: strip.strip_sequence, @@ -735,10 +900,7 @@ impl ChunkReader { ))); } *chunk = self - .chunkdb - .query_chunk(QueryChunkRequest { - chunk_id: Some(chunk_id), - }) + .query_chunk_timed(chunk_id) .await .map_err(|error| ReadError::Metadata(error.to_string()))? .chunk @@ -755,6 +917,7 @@ fn decode_framed_partial( selected_logical_start: u64, requested: Range, output_logical_start: u64, + flow_metrics: &ReadFlowMetrics, ) -> ReadResult { let first_frame = selected_logical_start / MAX_FRAME_PAYLOAD_BYTES as u64; let last_frame = (requested.end - 1) / MAX_FRAME_PAYLOAD_BYTES as u64; @@ -786,45 +949,74 @@ fn decode_framed_partial( }); continue; } - let bytes = extract_physical_range(&physical.ranges, frame_start..frame_end)?; - let frame = parse_frame(&bytes, framed.chunk_id) - .map_err(|error| ReadError::DataLoss(format!("invalid chunk frame: {error}")))?; - if !matches!( - frame.header.magic, - FrameMagic::RepoSmallV1 | FrameMagic::RepoLargeV1 - ) { + let views = extract_physical_views(&physical.ranges, frame_start..frame_end)?; + let parse_started = Instant::now(); + let parsed = if views.len() == 1 { + parse_frame(&views[0], framed.chunk_id).map(|frame| (frame.header, frame.physical_length)) + } else { + let slices: Vec<_> = views.iter().map(Bytes::as_ref).collect(); + parse_frame_views(&slices, framed.chunk_id).map(|frame| (frame.header, frame.physical_length)) + }; + flow_metrics + .frame_parse_wait_ns + .inc_by(u64::try_from(parse_started.elapsed().as_nanos()).unwrap_or(u64::MAX)); + let (header, physical_length) = + parsed.map_err(|error| ReadError::DataLoss(format!("invalid chunk frame: {error}")))?; + if !matches!(header.magic, FrameMagic::RepoSmallV1 | FrameMagic::RepoLargeV1) { return Err(ReadError::DataLoss( "framed location has unsupported frame kind".into(), )); } match magic { - Some(previous) if previous != frame.header.magic => { + Some(previous) if previous != header.magic => { return Err(ReadError::DataLoss("framed location mixes frame kinds".into())); } - None => magic = Some(frame.header.magic), + None => magic = Some(header.magic), Some(_) => {} } - if frame.physical_length as u64 != frame_end - frame_start { + if physical_length as u64 != frame_end - frame_start { return Err(ReadError::DataLoss("frame length disagrees with location".into())); } let payload_start = usize::try_from(wanted_start - frame_logical_start) .map_err(|_| ReadError::InvalidLocations("frame payload range overflows".into()))?; let payload_end = usize::try_from(wanted_end - frame_logical_start) .map_err(|_| ReadError::InvalidLocations("frame payload range overflows".into()))?; - result.ranges.push(ReadRangeData { - start: output_start, - end: output_end, - data: Bytes::copy_from_slice(&frame.payload[payload_start..payload_end]), - }); + let physical_payload_start = usize::from(header.payload_offset) + payload_start; + let physical_payload_end = usize::from(header.payload_offset) + payload_end; + let mut logical_cursor = output_start; + for view in slice_views(&views, physical_payload_start..physical_payload_end)? { + let view_end = logical_cursor + view.len() as u64; + result.ranges.push(ReadRangeData { + start: logical_cursor, + end: view_end, + data: view, + }); + logical_cursor = view_end; + } + if logical_cursor != output_end { + return Err(ReadError::InvalidLocations( + "frame payload views are incomplete".into(), + )); + } } Ok(result) } fn extract_physical_range(ranges: &[ReadRangeData], wanted: Range) -> ReadResult { + let first = ranges.partition_point(|range| range.end <= wanted.start); + if let Some(range) = ranges.get(first) { + if range.start <= wanted.start && range.end >= wanted.end { + let offset = usize::try_from(wanted.start - range.start) + .map_err(|_| ReadError::InvalidLocations("physical frame offset overflows".into()))?; + let length = usize::try_from(wanted.end - wanted.start) + .map_err(|_| ReadError::InvalidLocations("physical frame length overflows".into()))?; + return Ok(range.data.slice(offset..offset + length)); + } + } let mut output = BytesMut::with_capacity(usize::try_from(wanted.end - wanted.start).unwrap_or(usize::MAX)); let mut cursor = wanted.start; - for range in ranges { + for range in &ranges[first..] { if range.end <= cursor || range.start >= wanted.end { continue; } @@ -853,31 +1045,64 @@ fn extract_physical_range(ranges: &[ReadRangeData], wanted: Range) -> ReadR Ok(output.freeze()) } -fn coalesce_partial_ranges(partial: PartialReadResult) -> PartialReadResult { - let mut ranges: Vec = Vec::new(); - for range in partial.ranges { - if let Some(previous) = ranges.last_mut().filter(|previous| previous.end == range.start) { - previous.end = range.end; - let mut data = BytesMut::with_capacity(previous.data.len() + range.data.len()); - data.extend_from_slice(&previous.data); - data.extend_from_slice(&range.data); - previous.data = data.freeze(); - } else { - ranges.push(range); +fn extract_physical_views(ranges: &[ReadRangeData], wanted: Range) -> ReadResult> { + let first = ranges.partition_point(|range| range.end <= wanted.start); + let mut output = Vec::new(); + let mut cursor = wanted.start; + for range in &ranges[first..] { + if range.start >= wanted.end { + break; + } + if range.end <= cursor { + continue; + } + if range.start > cursor { + return Err(ReadError::InvalidLocations( + "physical frame data has a gap".into(), + )); + } + let end = wanted.end.min(range.end); + let offset = usize::try_from(cursor - range.start) + .map_err(|_| ReadError::InvalidLocations("physical frame offset overflows".into()))?; + let length = usize::try_from(end - cursor) + .map_err(|_| ReadError::InvalidLocations("physical frame length overflows".into()))?; + output.push(range.data.slice(offset..offset + length)); + cursor = end; + if cursor == wanted.end { + break; } } - let mut failures: Vec = Vec::new(); - for failure in partial.failures { - if let Some(previous) = failures - .last_mut() - .filter(|previous| previous.end == failure.start) - { - previous.end = failure.end; - } else { - failures.push(failure); + if cursor != wanted.end { + return Err(ReadError::InvalidLocations( + "physical frame data is incomplete".into(), + )); + } + Ok(output) +} + +fn slice_views(views: &[Bytes], wanted: Range) -> ReadResult> { + let mut output = Vec::new(); + let mut cursor = 0_usize; + let mut covered = wanted.start; + for view in views { + let view_end = cursor.saturating_add(view.len()); + if view_end > covered && cursor < wanted.end { + let start = covered.saturating_sub(cursor); + let end = (wanted.end - cursor).min(view.len()); + output.push(view.slice(start..end)); + covered += end - start; } + cursor = view_end; + if covered == wanted.end { + break; + } + } + if covered != wanted.end { + return Err(ReadError::InvalidLocations( + "frame payload views are incomplete".into(), + )); } - PartialReadResult { ranges, failures } + Ok(output) } struct StripFailureObservation { @@ -907,89 +1132,257 @@ fn mark_served_segments_corrupt( observations } -/// Pull-based stream whose emitted item never exceeds the configured window. +type StreamRead = (u64, u64, ReadResult, Arc); + +/// Ordered, bounded read pipeline whose frames retain their read credits. pub struct ChunkReadStream { reader: ChunkReader, locations: Arc<[Location]>, cursor: u64, + fetch_cursor: u64, + delivery_cursor: u64, end: u64, window_bytes: u64, + slots: Arc, + reads: JoinSet, + completed: BTreeMap, + layouts: Vec, pending_error: Option, + pending_ranges: std::vec::IntoIter, } impl ChunkReadStream { pub async fn next_chunk(&mut self) -> Option> { - if let Some(error) = self.pending_error.take() { - self.cursor = self.end; - return Some(Err(error)); + loop { + if let Some(pending) = self.next_pending() { + return Some(pending); + } + if self.delivery_cursor >= self.end { + return None; + } + let budget = Arc::clone(&self.reader.stream_budget); + let _registration = budget.register(); + let notified = budget.wake.notified(); + tokio::pin!(notified); + notified.as_mut().enable(); + if let Err(error) = self.fill_slots().await { + self.abort(error); + continue; + } + if let Some((start, end, result, lease)) = self.completed.remove(&self.delivery_cursor) { + self.load_unit(start, end, result, &lease); + continue; + } + if self.reads.is_empty() { + let started = Instant::now(); + notified.await; + self.reader + .flow_metrics + .stream_credit_wait_ns + .inc_by(u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX)); + continue; + } + match self.reads.join_next().await { + Some(Ok(completed)) => { + self.reader.flow_metrics.stream_units_completed.inc(); + if completed.0 != self.delivery_cursor { + self.reader.flow_metrics.stream_out_of_order.inc(); + } + self.completed.insert(completed.0, completed); + } + Some(Err(error)) => self.abort(ReadError::DiskIo(error.to_string())), + None => {} + } } - if self.cursor >= self.end { - return None; + } + + async fn fill_slots(&mut self) -> ReadResult<()> { + while self.fetch_cursor < self.end { + let (next, physical_bytes) = self.next_unit(self.fetch_cursor)?; + let Some(lease) = self.slots.try_reserve(&self.reader.stream_budget, physical_bytes) else { + self.reader.flow_metrics.stream_credit_stalls.inc(); + break; + }; + let start = self.fetch_cursor; + let index = self.locations.partition_point(|location| { + location.logical_offset.saturating_add(location.logical_length) <= start + }); + let location = self.locations[index].clone(); + let layout = self + .layout_for( + location + .chunk_id + .ok_or_else(|| ReadError::InvalidLocations("location has no chunk ID".into()))?, + ) + .await?; + let reader = self.reader.clone(); + self.reader.flow_metrics.stream_windows.inc(); + self.reader.flow_metrics.range_locations_examined.inc(); + self.reads.spawn(async move { + let local_start = start - location.logical_offset; + let result = reader + .read_location_partial_cached(&location, local_start, next - start, start, Some(layout)) + .await; + (start, next, result, lease) + }); + self.fetch_cursor = next; } - let next = self.cursor.saturating_add(self.window_bytes).min(self.end); - let partial = match self - .reader - .read_range_partial(&self.locations, self.cursor, next) - .await + Ok(()) + } + + async fn layout_for(&mut self, chunk_id: ChunkId) -> ReadResult { + if let Some(layout) = self + .layouts + .iter() + .find(|layout| layout.chunk_id == chunk_id && layout.usable()) { + return Ok(layout.clone()); + } + let reader = self.reader.clone(); + let layout = tokio::spawn(async move { reader.query_layout(chunk_id).await }) + .await + .map_err(|error| ReadError::Metadata(error.to_string()))??; + self.layouts.retain(|cached| cached.chunk_id != chunk_id); + if self.layouts.len() == 8 { + self.layouts.remove(0); + } + self.layouts.push(layout.clone()); + Ok(layout) + } + + fn next_unit(&self, start: u64) -> ReadResult<(u64, usize)> { + let first = self.locations.partition_point(|location| { + location.logical_offset.saturating_add(location.logical_length) <= start + }); + let location = self + .locations + .get(first) + .ok_or_else(|| ReadError::InvalidLocations("stream has no location for requested byte".into()))?; + let local = start - location.logical_offset; + let window = self.window_bytes.min(DEFAULT_STREAM_WINDOW as u64); + if location.length == location.logical_length { + let length = window.min(location.logical_length - local).min(self.end - start); + return Ok((start + length, length as usize)); + } + let frames = (window / MAX_FRAME_BYTES as u64).max(1); + let frame_index = local / MAX_FRAME_PAYLOAD_BYTES as u64; + let local_end = ((frame_index + frames) * MAX_FRAME_PAYLOAD_BYTES as u64) + .min(location.logical_length) + .min(self.end - location.logical_offset); + let frame_location = ChunkLocation { + chunk_id: location + .chunk_id + .ok_or_else(|| ReadError::InvalidLocations("location has no chunk ID".into()))?, + frame_offset: location.offset, + logical_length: location.logical_length, + }; + let physical = frame_location + .physical_range_for_subrange(local..local_end) + .map_err(|error| ReadError::InvalidLocations(error.to_string()))?; + let bytes = usize::try_from(physical.end - physical.start) + .map_err(|_| ReadError::InvalidLocations("stream unit exceeds address space".into()))?; + Ok((location.logical_offset + local_end, bytes)) + } + + fn load_unit( + &mut self, + start: u64, + end: u64, + result: ReadResult, + lease: &Arc, + ) { + self.delivery_cursor = end; + let partial = match result { Ok(partial) => partial, Err(error) => { - self.cursor = self.end; - return Some(Err(error)); + self.abort(error); + return; } }; - if let Some(failure) = partial.failures.into_iter().next() { - let error = ReadError::FailedRange { + let failure = partial.failures.into_iter().next(); + let verified_end = failure.as_ref().map_or(end, |failure| failure.start); + let mut ranges = partial.ranges; + let prefix_len = ranges.partition_point(|range| range.end <= verified_end); + let mut cursor = start; + for range in &ranges[..prefix_len] { + if range.start != cursor || range.end - range.start != range.data.len() as u64 { + self.abort(ReadError::InvalidLocations(format!( + "read ranges are not contiguous at byte {cursor}" + ))); + return; + } + cursor = range.end; + } + if cursor != verified_end { + self.abort(ReadError::InvalidLocations(format!( + "read ranges are not contiguous at byte {cursor}" + ))); + return; + } + ranges.truncate(prefix_len); + for range in &mut ranges { + range.data = retain(std::mem::take(&mut range.data), Arc::clone(lease)); + } + self.pending_ranges = ranges.into_iter(); + if let Some(failure) = failure { + self.reads.abort_all(); + self.completed.clear(); + self.fetch_cursor = self.end; + self.delivery_cursor = self.end; + self.pending_error = Some(ReadError::FailedRange { start: failure.start, end: failure.end, message: failure.error.to_string(), - }; - if failure.start == self.cursor { - self.cursor = self.end; - return Some(Err(error)); - } - let prefix_len = usize::try_from(failure.start - self.cursor).unwrap_or(usize::MAX); - let prefix = partial - .ranges - .into_iter() - .filter(|range| range.end <= failure.start) - .collect(); - let data = assemble_ranges(prefix, self.cursor, prefix_len); - self.cursor = failure.start; - self.pending_error = Some(error); - return Some(data); + }); } - let expected = usize::try_from(next - self.cursor).unwrap_or(usize::MAX); - let result = assemble_ranges(partial.ranges, self.cursor, expected); - if result.is_ok() { - self.cursor = next; - } else { + } + + fn abort(&mut self, error: ReadError) { + self.reads.abort_all(); + self.completed.clear(); + self.fetch_cursor = self.end; + self.delivery_cursor = self.end; + self.pending_error = Some(error); + } + + fn next_pending(&mut self) -> Option> { + if let Some(range) = self.pending_ranges.next() { + self.cursor = range.end; + return Some(Ok(range.data)); + } + if let Some(error) = self.pending_error.take() { self.cursor = self.end; + return Some(Err(error)); } - Some(result) + None } } -fn assemble_ranges(mut ranges: Vec, start: u64, expected: usize) -> ReadResult { - ranges.sort_unstable_by_key(|range| range.start); - let actual = ranges.iter().map(|range| range.data.len()).sum::(); - if actual != expected { - return Err(ReadError::InvalidLocations(format!( - "read assembled {actual} bytes, expected {expected}" - ))); +fn complete_read(partial: PartialReadResult, start: u64, end: u64) -> ReadResult> { + if let Some(failure) = partial.failures.into_iter().next() { + return Err(ReadError::FailedRange { + start: failure.start, + end: failure.end, + message: failure.error.to_string(), + }); } - let mut output = BytesMut::with_capacity(actual); let mut cursor = start; - for range in ranges { + let mut buffers = Vec::with_capacity(partial.ranges.len()); + for range in partial.ranges { if range.start != cursor || range.end - range.start != range.data.len() as u64 { return Err(ReadError::InvalidLocations(format!( "read ranges are not contiguous at byte {cursor}" ))); } cursor = range.end; - output.extend_from_slice(&range.data); + buffers.push(range.data); } - Ok(output.freeze()) + if cursor != end { + return Err(ReadError::InvalidLocations(format!( + "read ranges end at byte {cursor}, expected {end}" + ))); + } + Ok(buffers) } fn segment_identity(segment: &Segment) -> (u64, u64, u32, u64, u64) { diff --git a/lib/crowdb-chunk-client/src/chunk/read_credit.rs b/lib/crowdb-chunk-client/src/chunk/read_credit.rs new file mode 100644 index 000000000..26b2b41c6 --- /dev/null +++ b/lib/crowdb-chunk-client/src/chunk/read_credit.rs @@ -0,0 +1,129 @@ +// Copyright 2026-present Gian +// Licensed under the Apache License, Version 2.0. + +//! Atomic credits retained through the last HTTP-owned buffer view. + +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; + +use bytes::Bytes; +use tokio::sync::Notify; + +use crate::metrics::ReadFlowMetrics; + +pub(super) struct ReadBudget { + limit: usize, + used: AtomicUsize, + waiters: AtomicUsize, + pub wake: Notify, + metrics: Arc, +} + +impl ReadBudget { + pub fn new(limit: usize, metrics: Arc) -> Self { + Self { + limit, + used: AtomicUsize::new(0), + waiters: AtomicUsize::new(0), + wake: Notify::new(), + metrics, + } + } + + fn try_charge(&self, amount: usize) -> bool { + let accepted = self + .used + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { + used.checked_add(amount).filter(|next| *next <= self.limit) + }) + .is_ok(); + if accepted { + self.metrics.stream_bytes_reserved.inc_by(amount as u64); + } + accepted + } + + fn release(&self, amount: usize) { + self.used.fetch_sub(amount, Ordering::AcqRel); + self.metrics.stream_bytes_released.inc_by(amount as u64); + if self.waiters.load(Ordering::Acquire) != 0 { + self.wake.notify_waiters(); + } + } + + pub fn register(self: &Arc) -> WaitRegistration { + self.waiters.fetch_add(1, Ordering::AcqRel); + WaitRegistration(Arc::clone(self)) + } +} + +pub(super) struct WaitRegistration(Arc); + +impl Drop for WaitRegistration { + fn drop(&mut self) { + self.0.waiters.fetch_sub(1, Ordering::AcqRel); + } +} + +pub(super) struct StreamSlots { + limit: usize, + used: AtomicUsize, +} + +impl StreamSlots { + pub fn new(limit: usize) -> Self { + Self { + limit, + used: AtomicUsize::new(0), + } + } + + pub fn try_reserve(self: &Arc, global: &Arc, bytes: usize) -> Option> { + if self + .used + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { + used.checked_add(1).filter(|next| *next <= self.limit) + }) + .is_err() + { + return None; + } + if !global.try_charge(bytes) { + self.used.fetch_sub(1, Ordering::AcqRel); + return None; + } + Some(Arc::new(ReadLease { + slots: Arc::clone(self), + global: Arc::clone(global), + bytes, + })) + } +} + +pub(super) struct ReadLease { + slots: Arc, + global: Arc, + bytes: usize, +} + +impl Drop for ReadLease { + fn drop(&mut self) { + self.slots.used.fetch_sub(1, Ordering::AcqRel); + self.global.release(self.bytes); + } +} + +struct LeasedView { + data: Bytes, + _lease: Arc, +} + +impl AsRef<[u8]> for LeasedView { + fn as_ref(&self) -> &[u8] { + &self.data + } +} + +pub(super) fn retain(data: Bytes, lease: Arc) -> Bytes { + Bytes::from_owner(LeasedView { data, _lease: lease }) +} diff --git a/lib/crowdb-chunk-client/src/chunk/strip_reader.rs b/lib/crowdb-chunk-client/src/chunk/strip_reader.rs index 5c3b6aeef..0fa979a2f 100644 --- a/lib/crowdb-chunk-client/src/chunk/strip_reader.rs +++ b/lib/crowdb-chunk-client/src/chunk/strip_reader.rs @@ -209,7 +209,9 @@ impl StripReader { }; pieces.push((order, shard_index, local_start, read_len, result)); } - pieces.sort_unstable_by_key(|(order, _, _, _, _)| *order); + if let [(_, shard_index, _, _, Ok(data))] = pieces.as_slice() { + return Ok((data.clone(), failed_segments, vec![ec.segments[*shard_index]])); + } let mut output = BytesMut::with_capacity(usize::try_from(length).unwrap_or(usize::MAX)); let mut served_segments = Vec::new(); for (_, shard_index, local_start, read_len, result) in pieces { diff --git a/lib/crowdb-chunk-client/src/client.rs b/lib/crowdb-chunk-client/src/client.rs index 92e5ac2ac..daaaa26a9 100644 --- a/lib/crowdb-chunk-client/src/client.rs +++ b/lib/crowdb-chunk-client/src/client.rs @@ -108,6 +108,15 @@ impl ChunkIoClient { /// Discover services using an existing KV topology client shared with the /// embedding process. pub async fn connect_with_kv(config: ChunkIoClientConfig, kv: Arc) -> Result { + Self::connect_with_kv_read_policy(config, kv, ChunkReadPolicy::default()).await + } + + /// Connect with explicit shared read resource limits. + pub async fn connect_with_kv_read_policy( + config: ChunkIoClientConfig, + kv: Arc, + read_policy: ChunkReadPolicy, + ) -> Result { let service = ServiceRegistryClient::from_shared(kv.clone()); let hardware = HardwareClient::from_shared(kv.clone()); let range_binding = discover_current_range_bindings(&service, kv.clone()).await?; @@ -136,7 +145,7 @@ impl ChunkIoClient { Arc::new(SmallWriteMetrics::default()), Arc::clone(&failed_disks), )?; - let reader = ChunkReader::new(chunkdb.clone(), disk_writer.clone(), ChunkReadPolicy::default()) + let reader = ChunkReader::new(chunkdb.clone(), disk_writer.clone(), read_policy) .map_err(|error| crate::IoError::Internal(error.to_string()))?; Ok(Self { allocator: chunkdb.clone(), @@ -254,6 +263,7 @@ impl ChunkIoClient { Arc::clone(&self.disk_writer), ChunkReadPolicy::default(), Arc::clone(&metrics.read_recovery), + Arc::clone(&metrics.read_flow), ) .unwrap_or_else(|_| unreachable!("default read policy is valid")); self @@ -268,17 +278,20 @@ impl ChunkIoClient { self.metrics .as_ref() .map_or_else(Arc::default, |metrics| Arc::clone(&metrics.read_recovery)), + self.metrics + .as_ref() + .map_or_else(Arc::default, |metrics| Arc::clone(&metrics.read_flow)), )?; Ok(self) } - /// Reconstruct a complete object from writer-produced locations. - pub async fn read_object(&self, locations: &[Location]) -> ReadResult { + /// Read a complete object as verified buffers from writer-produced locations. + pub async fn read_object(&self, locations: &[Location]) -> ReadResult> { self.reader.read_object(locations).await } - /// Reconstruct the logical half-open range `[start, end)`. - pub async fn read_range(&self, locations: &[Location], start: u64, end: u64) -> ReadResult { + /// Read the logical half-open range `[start, end)` as verified buffers. + pub async fn read_range(&self, locations: &[Location], start: u64, end: u64) -> ReadResult> { self.reader.read_range(locations, start, end).await } @@ -351,6 +364,30 @@ impl ChunkIoClient { Ok(SharedObjectWriter::new(runtime, object_size, route, route_hash)) } + /// Prepares one shared object whose complete 64 KiB frame sequence has one location. + pub async fn prepare_shared_object_write_for_key( + &self, + object_size: usize, + key: &[u8], + ) -> Result { + if object_size == 0 { + return Ok(SharedObjectWriter::empty()); + } + let runtime = self.small_pool.prepare(object_size).await?; + let route_hash = stable_route_hash(key); + let route = runtime + .route_for_hash(route_hash) + .ok_or_else(|| IoError::Internal("small write has no pipeline route".into()))?; + if object_size as u64 > route.capacity_bytes { + return Err(IoError::MemoryBudgetExhausted); + } + let mut writer = SharedObjectWriter::new(runtime, object_size, route, route_hash); + while !writer.try_reserve_declared() { + writer.wait_for_route_capacity().await; + } + Ok(writer) + } + /// Stop admission, drain accepted objects, and finalize shared chunks. pub async fn shutdown_small_writes(&self) -> Result<()> { self.small_pool.shutdown().await @@ -373,6 +410,11 @@ impl ChunkIoClient { snapshot } + /// Snapshot location and layout work on the shared read path. + pub fn read_flow_metrics(&self) -> crate::ReadFlowMetricsSnapshot { + self.reader.flow_metrics_snapshot() + } + /// Snapshot in-line large-write segment replacement counters. pub fn large_write_repair_metrics(&self) -> crate::LargeWriteRepairMetricsSnapshot { self.large_write_repair.snapshot() diff --git a/lib/crowdb-chunk-client/src/config.rs b/lib/crowdb-chunk-client/src/config.rs index 91490864b..46ae37120 100644 --- a/lib/crowdb-chunk-client/src/config.rs +++ b/lib/crowdb-chunk-client/src/config.rs @@ -50,7 +50,7 @@ impl Default for SmallWritePolicy { fn default() -> Self { const MIB: usize = 1024 * 1024; Self { - object_limit: MIB, + object_limit: 8 * MIB, // 1,000 concurrent 1 MiB objects are a normal S3 small-object // workload. 3,000 and 5,000 require roughly 3.25 GiB and 5.25 // GiB respectively after pipeline and conversion headroom; set @@ -82,17 +82,17 @@ impl Default for SmallWritePolicy { impl SmallWritePolicy { pub fn validate(&self) -> Result<(), IoError> { - const HARD_LIMIT: usize = 1024 * 1024; + const HARD_LIMIT: usize = 8 * 1024 * 1024; if self.object_limit == 0 || self.object_limit > HARD_LIMIT { return Err(IoError::Internal( - "small object limit must be in 1..=1 MiB".into(), + "small object limit must be in 1..=8 MiB".into(), )); } - let shadow_budget = self.max_pipelines.saturating_mul(HARD_LIMIT); + let shadow_budget = self.max_pipelines.saturating_mul(1024 * 1024); let conversion_budget = if self.conversion_enabled { self.conversion_data_num .saturating_add(self.conversion_code_num) - .saturating_mul(HARD_LIMIT) + .saturating_mul(1024 * 1024) } else { 0 }; diff --git a/lib/crowdb-chunk-client/src/lib.rs b/lib/crowdb-chunk-client/src/lib.rs index fb37f22d5..72371e3d7 100644 --- a/lib/crowdb-chunk-client/src/lib.rs +++ b/lib/crowdb-chunk-client/src/lib.rs @@ -53,7 +53,7 @@ pub use error::{IoError, ReadError, ReadResult, Result}; pub use io::{BackpressurePolicy, ChunkIoWriter, FeedStatus, FramedWriteBuffer}; pub use metrics::{ ChunkClientMetrics, LargeWriteBufferMetricsSnapshot, LargeWriteRepairMetricsSnapshot, - SmallWriteMetricsSnapshot, + ReadFlowMetricsSnapshot, SmallWriteMetricsSnapshot, }; pub use negative_list::FailedDiskList; pub use reclamation::{reclaim_location, ReclaimOutcome}; diff --git a/lib/crowdb-chunk-client/src/metrics.rs b/lib/crowdb-chunk-client/src/metrics.rs index f5ecebfb8..f95ac5555 100644 --- a/lib/crowdb-chunk-client/src/metrics.rs +++ b/lib/crowdb-chunk-client/src/metrics.rs @@ -84,6 +84,7 @@ pub struct ChunkClientMetrics { pub large_write_buffer: Arc, pub small_write: Arc, pub read_recovery: Arc, + pub read_flow: Arc, } impl ChunkClientMetrics { @@ -107,6 +108,149 @@ impl ChunkClientMetrics { large_write_buffer: Arc::new(LargeWriteBufferMetrics::register(registry)), small_write: Arc::new(SmallWriteMetrics::register(registry)), read_recovery: Arc::new(ReadRecoveryMetrics::register(registry)), + read_flow: Arc::new(ReadFlowMetrics::register(registry)), + } + } +} + +/// Aggregate read-path work counters without object or chunk labels. +#[derive(Debug)] +pub struct ReadFlowMetrics { + pub(crate) location_normalizations: Arc, + pub(crate) locations_examined: Arc, + pub(crate) range_locations_examined: Arc, + pub(crate) stream_windows: Arc, + pub(crate) stream_units_completed: Arc, + pub(crate) stream_out_of_order: Arc, + pub(crate) stream_credit_stalls: Arc, + pub(crate) stream_credit_wait_ns: Arc, + pub(crate) stream_bytes_reserved: Arc, + pub(crate) stream_bytes_released: Arc, + pub(crate) layout_queries: Arc, + pub(crate) layout_query_wait_ns: Arc, + pub(crate) strip_read_wait_ns: Arc, + pub(crate) chunk_read_wait_ns: Arc, + pub(crate) frame_decode_wait_ns: Arc, + pub(crate) frame_parse_wait_ns: Arc, +} + +/// Cumulative read-path work visible to access-server metrics. +#[derive(Clone, Copy, Debug, Default, Eq, PartialEq, serde::Serialize)] +pub struct ReadFlowMetricsSnapshot { + pub location_normalizations: u64, + pub locations_examined: u64, + pub range_locations_examined: u64, + pub stream_windows: u64, + pub stream_units_completed: u64, + pub stream_out_of_order: u64, + pub stream_credit_stalls: u64, + pub stream_credit_wait_ns: u64, + pub stream_bytes_reserved: u64, + pub stream_bytes_released: u64, + pub layout_queries: u64, + pub layout_query_wait_ns: u64, + pub strip_read_wait_ns: u64, + pub chunk_read_wait_ns: u64, + pub frame_decode_wait_ns: u64, + pub frame_parse_wait_ns: u64, +} + +impl ReadFlowMetricsSnapshot { + pub(crate) fn since(self, earlier: Self) -> Self { + Self { + location_normalizations: self + .location_normalizations + .saturating_sub(earlier.location_normalizations), + locations_examined: self.locations_examined.saturating_sub(earlier.locations_examined), + range_locations_examined: self + .range_locations_examined + .saturating_sub(earlier.range_locations_examined), + stream_windows: self.stream_windows.saturating_sub(earlier.stream_windows), + stream_units_completed: self + .stream_units_completed + .saturating_sub(earlier.stream_units_completed), + stream_out_of_order: self + .stream_out_of_order + .saturating_sub(earlier.stream_out_of_order), + stream_credit_stalls: self + .stream_credit_stalls + .saturating_sub(earlier.stream_credit_stalls), + stream_credit_wait_ns: self + .stream_credit_wait_ns + .saturating_sub(earlier.stream_credit_wait_ns), + stream_bytes_reserved: self + .stream_bytes_reserved + .saturating_sub(earlier.stream_bytes_reserved), + stream_bytes_released: self + .stream_bytes_released + .saturating_sub(earlier.stream_bytes_released), + layout_queries: self.layout_queries.saturating_sub(earlier.layout_queries), + layout_query_wait_ns: self + .layout_query_wait_ns + .saturating_sub(earlier.layout_query_wait_ns), + strip_read_wait_ns: self.strip_read_wait_ns.saturating_sub(earlier.strip_read_wait_ns), + chunk_read_wait_ns: self.chunk_read_wait_ns.saturating_sub(earlier.chunk_read_wait_ns), + frame_decode_wait_ns: self + .frame_decode_wait_ns + .saturating_sub(earlier.frame_decode_wait_ns), + frame_parse_wait_ns: self + .frame_parse_wait_ns + .saturating_sub(earlier.frame_parse_wait_ns), + } + } +} + +impl Default for ReadFlowMetrics { + fn default() -> Self { + Self::new(|name| Arc::new(Counter::new(name.into()))) + } +} + +impl ReadFlowMetrics { + fn register(registry: &mut MetricsRegistry) -> Self { + Self::new(|name| registry.register_counter(name)) + } + + fn new(mut counter: impl FnMut(&'static str) -> Arc) -> Self { + Self { + location_normalizations: counter("chunkio.read.location_normalizations.c"), + locations_examined: counter("chunkio.read.locations_examined.c"), + range_locations_examined: counter("chunkio.read.range_locations_examined.c"), + stream_windows: counter("chunkio.read.stream_windows.c"), + stream_units_completed: counter("chunkio.read.stream_units_completed.c"), + stream_out_of_order: counter("chunkio.read.stream_out_of_order.c"), + stream_credit_stalls: counter("chunkio.read.stream_credit_stalls.c"), + stream_credit_wait_ns: counter("chunkio.read.stream_credit_wait_ns.c"), + stream_bytes_reserved: counter("chunkio.read.stream_bytes_reserved.c"), + stream_bytes_released: counter("chunkio.read.stream_bytes_released.c"), + layout_queries: counter("chunkio.read.layout_queries.c"), + layout_query_wait_ns: counter("chunkio.read.layout_query_wait_ns.c"), + strip_read_wait_ns: counter("chunkio.read.strip_read_wait_ns.c"), + chunk_read_wait_ns: counter("chunkio.read.chunk_read_wait_ns.c"), + frame_decode_wait_ns: counter("chunkio.read.frame_decode_wait_ns.c"), + frame_parse_wait_ns: counter("chunkio.read.frame_parse_wait_ns.c"), + } + } + + #[must_use] + pub fn snapshot(&self) -> ReadFlowMetricsSnapshot { + ReadFlowMetricsSnapshot { + location_normalizations: self.location_normalizations.snapshot().total, + locations_examined: self.locations_examined.snapshot().total, + range_locations_examined: self.range_locations_examined.snapshot().total, + stream_windows: self.stream_windows.snapshot().total, + stream_units_completed: self.stream_units_completed.snapshot().total, + stream_out_of_order: self.stream_out_of_order.snapshot().total, + stream_credit_stalls: self.stream_credit_stalls.snapshot().total, + stream_credit_wait_ns: self.stream_credit_wait_ns.snapshot().total, + stream_bytes_reserved: self.stream_bytes_reserved.snapshot().total, + stream_bytes_released: self.stream_bytes_released.snapshot().total, + layout_queries: self.layout_queries.snapshot().total, + layout_query_wait_ns: self.layout_query_wait_ns.snapshot().total, + strip_read_wait_ns: self.strip_read_wait_ns.snapshot().total, + chunk_read_wait_ns: self.chunk_read_wait_ns.snapshot().total, + frame_decode_wait_ns: self.frame_decode_wait_ns.snapshot().total, + frame_parse_wait_ns: self.frame_parse_wait_ns.snapshot().total, } } } @@ -351,7 +495,7 @@ impl Default for SmallWriteMetrics { } } -#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, serde::Serialize)] pub struct SmallWriteMetricsSnapshot { pub submitted: u64, pub completed: u64, diff --git a/lib/crowdb-chunk-client/src/writer/shared_object.rs b/lib/crowdb-chunk-client/src/writer/shared_object.rs index e1c4768a3..8ab6add8f 100644 --- a/lib/crowdb-chunk-client/src/writer/shared_object.rs +++ b/lib/crowdb-chunk-client/src/writer/shared_object.rs @@ -8,7 +8,8 @@ use std::time::Instant; use bytes::Bytes; use crowdb_protocol::chunkdb::rpc::Location as ProtoLocation; -use tokio::sync::oneshot; +use crowdb_protocol::frame::MAX_FRAME_PAYLOAD_BYTES; +use tokio::sync::{mpsc, oneshot}; use crate::io::{ChunkIoWriter, FeedStatus}; use crate::{IoError, Result}; @@ -28,7 +29,10 @@ pub struct SharedObjectWriter { route_hash: u64, route: Option>, retained_size: usize, + full_reservation: bool, fragments: Vec, + stream_tx: Option>, + stream_result: Option>>>, finished: bool, durable_completion: bool, intent: Option>, @@ -49,7 +53,10 @@ impl SharedObjectWriter { route_hash, route: Some(route), retained_size: 0, + full_reservation: false, fragments: Vec::new(), + stream_tx: None, + stream_result: None, finished: false, durable_completion: false, intent: None, @@ -64,7 +71,10 @@ impl SharedObjectWriter { route_hash: 0, route: None, retained_size: 0, + full_reservation: false, fragments: Vec::new(), + stream_tx: None, + stream_result: None, finished: false, durable_completion: false, intent: None, @@ -79,6 +89,28 @@ impl SharedObjectWriter { } } + pub(crate) fn try_reserve_declared(&mut self) -> bool { + if self + .charge + .as_mut() + .is_some_and(|charge| charge.try_reserve(self.declared_size)) + { + self.full_reservation = true; + true + } else { + false + } + } + + pub(crate) async fn wait_for_route_capacity(&self) { + if let Some(route) = &self.route { + tokio::select! { + () = route.capacity_changed.notified() => {}, + () = tokio::time::sleep(std::time::Duration::from_millis(5)) => {}, + } + } + } + /// Completes only after the readable chunk cursor covers this object's bytes. /// # Errors /// Returns admission, physical write, metadata confirmation or size failures. @@ -87,6 +119,11 @@ impl SharedObjectWriter { self.on_finish().await } + /// Require readable metadata before a caller publishes this object's location. + pub fn require_durable_completion(&mut self) { + self.durable_completion = true; + } + /// Persists exact object ownership before any physical write for the batch. /// # Errors /// A failed intent aborts the batch without issuing its disk writes. @@ -95,6 +132,11 @@ impl SharedObjectWriter { intent: Arc, ) -> Result> { self.ensure_open()?; + if self.stream_tx.is_some() { + return Err(IoError::Internal( + "a streaming shared object cannot register an intent after writing starts".into(), + )); + } self.intent = Some(intent); self.finish_durable().await } @@ -102,6 +144,8 @@ impl SharedObjectWriter { fn fail_size(&mut self, actual: usize) -> IoError { self.finished = true; self.fragments.clear(); + self.stream_tx.take(); + self.stream_result.take(); self.charge.take(); IoError::ObjectSizeMismatch { declared: self.declared_size, @@ -119,11 +163,55 @@ impl ChunkIoWriter for SharedObjectWriter { return Err(self.fail_size(actual)); } self.retained_size = actual; - self.charge - .as_mut() - .ok_or_else(|| IoError::Internal("shared writer missing route charge".into()))? - .add(buffer.len()); - self.fragments.push(buffer); + if buffer.is_empty() { + return Ok(FeedStatus::Continue); + } + if !self.full_reservation { + self.charge + .as_mut() + .ok_or_else(|| IoError::Internal("shared writer missing route charge".into()))? + .add(buffer.len()); + } + if self.declared_size > MAX_FRAME_PAYLOAD_BYTES && self.full_reservation { + if self.stream_tx.is_none() { + let runtime = self + .runtime + .take() + .ok_or_else(|| IoError::Internal("small writer missing shared pool".into()))?; + let charge = self + .charge + .take() + .ok_or_else(|| IoError::Internal("shared writer missing route charge".into()))?; + let (sender, receiver) = mpsc::channel(2); + let (completion, result) = oneshot::channel(); + let object = PendingObject { + intent: self.intent.take(), + durable_completion: true, + route_hash: self.route_hash, + route: self + .route + .take() + .ok_or_else(|| IoError::Internal("shared writer missing route".into()))?, + fragments: Vec::new(), + stream: Some(receiver), + len: self.declared_size, + enqueued_at: Instant::now(), + completion, + charge, + }; + runtime.submit(object).await?; + self.stream_tx = Some(sender); + self.stream_result = Some(result); + } + self.stream_tx + .as_ref() + .ok_or_else(|| IoError::Internal("shared object stream missing sender".into()))? + .send(buffer) + .await + .map_err(|_| IoError::WriteFailed("shared object pipeline stopped".into()))?; + } else { + self.fragments.push(buffer); + } Ok(if self.retained_size == self.declared_size { FeedStatus::Pause } else { @@ -140,6 +228,15 @@ impl ChunkIoWriter for SharedObjectWriter { if self.declared_size == 0 { return Ok(Vec::new()); } + if let Some(sender) = self.stream_tx.take() { + drop(sender); + return self + .stream_result + .take() + .ok_or_else(|| IoError::Internal("shared object stream missing completion".into()))? + .await + .map_err(|_| IoError::WriteFailed("small-write completion was lost".into()))?; + } let runtime = self .runtime .take() @@ -158,6 +255,7 @@ impl ChunkIoWriter for SharedObjectWriter { .take() .ok_or_else(|| IoError::Internal("shared writer missing route".into()))?, fragments: std::mem::take(&mut self.fragments), + stream: None, len: self.declared_size, enqueued_at: Instant::now(), completion, @@ -173,6 +271,8 @@ impl ChunkIoWriter for SharedObjectWriter { self.ensure_open()?; self.finished = true; self.fragments.clear(); + self.stream_tx.take(); + self.stream_result.take(); self.charge.take(); Ok(Vec::new()) } @@ -180,7 +280,7 @@ impl ChunkIoWriter for SharedObjectWriter { fn require_data(&self) -> bool { !self.finished && self.retained_size < self.declared_size - && self.route.as_ref().map_or(true, |route| route.has_capacity()) + && (self.full_reservation || self.route.as_ref().map_or(true, |route| route.has_capacity())) } fn input_complete(&self) -> bool { diff --git a/lib/crowdb-chunk-client/src/writer/small_pipeline.rs b/lib/crowdb-chunk-client/src/writer/small_pipeline.rs index 9f2c9729e..55c573de8 100644 --- a/lib/crowdb-chunk-client/src/writer/small_pipeline.rs +++ b/lib/crowdb-chunk-client/src/writer/small_pipeline.rs @@ -78,6 +78,7 @@ impl ManagedPipeline { } mod batch; +mod stream_object; struct PipelineWorker { runtime: Arc, @@ -127,7 +128,13 @@ impl PipelineWorker { if dequeued { self.note_dequeue(&first); } - if let Err(error) = self.ensure_object_fits(frame_bytes(first.len)?).await { + let fit = if first.len > MAX_FRAME_PAYLOAD_BYTES { + self.ensure_stream_object_fits(stream_object::physical_bytes(first.len)?) + .await + } else { + self.ensure_object_fits(frame_bytes(first.len)?).await + }; + if let Err(error) = fit { fail_one(first, &error.to_string(), &self.runtime.metrics); self.fail_remaining(&error.to_string()).await; let _ = self.finish_chunks().await; @@ -141,6 +148,18 @@ impl PipelineWorker { .last_active_ms .store(self.runtime.now_ms(), Ordering::Relaxed); if let Err(error) = result { + if matches!(error, IoError::SourceRead(_)) { + let replacement = match self.replacement.take() { + Some(chunk) => chunk, + None => { + OwnedChunk::allocate(&self.runtime, Arc::clone(&self.route.conversion_active)) + .await? + } + }; + self.chunk.finish().await?; + self.chunk = replacement; + continue; + } self.receiver.close(); self.fail_remaining(&error.to_string()).await; let _ = self.finish_chunks().await; @@ -240,6 +259,9 @@ impl PipelineWorker { } fn collect_batch(&mut self, first: PendingObject) -> Vec { + if first.len > MAX_FRAME_PAYLOAD_BYTES { + return vec![first]; + } let mut bytes = frame_bytes(first.len).unwrap_or(usize::MAX); let mut batch = vec![first]; while batch.len() < self.runtime.policy.max_batch_objects @@ -1091,8 +1113,19 @@ impl OwnedChunk { Ok(()) } - async fn write_batch(&mut self, batch: Vec, metrics: &SmallWriteMetrics) -> Result<()> { - match self.try_write_batch(&batch, metrics).await { + async fn write_batch( + &mut self, + mut batch: Vec, + metrics: &SmallWriteMetrics, + ) -> Result<()> { + let result = if batch.len() == 1 && batch[0].len > MAX_FRAME_PAYLOAD_BYTES { + self.try_write_stream_object(&mut batch[0], metrics) + .await + .map(|location| vec![location]) + } else { + self.try_write_batch(&batch, metrics).await + }; + match result { Ok(locations) => { for (object, location) in batch.into_iter().zip(locations) { metrics.completed.fetch_add(1, Ordering::Relaxed); diff --git a/lib/crowdb-chunk-client/src/writer/small_pipeline/stream_object.rs b/lib/crowdb-chunk-client/src/writer/small_pipeline/stream_object.rs new file mode 100644 index 000000000..b8c7b3ece --- /dev/null +++ b/lib/crowdb-chunk-client/src/writer/small_pipeline/stream_object.rs @@ -0,0 +1,222 @@ +use std::sync::Arc; +use std::time::{SystemTime, UNIX_EPOCH}; + +use bytes::Bytes; +use crowdb_protocol::chunkdb::rpc::Location; +use crowdb_protocol::common::ChunkId; +use crowdb_protocol::frame::{ + encode_frame, FrameMagic, FRAME_FOOTER_BYTES, FRAME_HEADER_PREFIX_BYTES, MAX_FRAME_BYTES, + MAX_FRAME_PAYLOAD_BYTES, +}; + +use super::{ + IoError, MirrorBatchStats, OwnedChunk, PendingObject, PipelineWorker, Result, SmallWriteMetrics, +}; + +pub(super) fn physical_bytes(logical: usize) -> Result { + logical + .checked_add( + logical + .div_ceil(MAX_FRAME_PAYLOAD_BYTES) + .checked_mul(FRAME_HEADER_PREFIX_BYTES + FRAME_FOOTER_BYTES) + .ok_or_else(|| IoError::WriteFailed("shared object frame length overflow".into()))?, + ) + .ok_or_else(|| IoError::WriteFailed("shared object frame length overflow".into())) +} + +impl PipelineWorker { + pub(super) async fn ensure_stream_object_fits(&mut self, physical: usize) -> Result<()> { + if physical as u64 > self.runtime.policy.chunk_capacity { + return Err(IoError::ObjectTooLarge { + size: physical, + limit: usize::try_from(self.runtime.policy.chunk_capacity).unwrap_or(usize::MAX), + }); + } + if let Ok(strip) = self.chunk.current_strip() { + if self.chunk.cursor > u64::from(strip.chunk_offset) * 1024 { + self.chunk.close_strip(&self.runtime.metrics).await?; + } + } + self.chunk.ensure_strip().await?; + if self.chunk.remaining_in_chunk() < physical as u64 { + let replacement = match self.replacement.take() { + Some(chunk) => chunk, + None => { + OwnedChunk::allocate(&self.runtime, Arc::clone(&self.route.conversion_active)).await? + } + }; + self.chunk.finish().await?; + self.chunk = replacement; + self.chunk.ensure_strip().await?; + } + Ok(()) + } +} + +struct FragmentCursor<'a> { + receiver: &'a mut tokio::sync::mpsc::Receiver, + current: Option, + offset: usize, + received: usize, +} + +impl FragmentCursor<'_> { + async fn take(&mut self, length: usize) -> Result> { + let mut output = Vec::with_capacity(length); + while output.len() < length { + if self.current.is_none() { + self.current = self.receiver.recv().await; + self.received += 1; + } + let fragment = self.current.as_ref().ok_or_else(|| { + IoError::SourceRead("shared object body ended before its declared length".into()) + })?; + let count = (length - output.len()).min(fragment.len() - self.offset); + output.extend_from_slice(&fragment[self.offset..self.offset + count]); + self.offset += count; + if self.offset == fragment.len() { + self.current = None; + self.offset = 0; + } + } + Ok(output) + } +} + +impl OwnedChunk { + pub(super) async fn try_write_stream_object( + &mut self, + object: &mut PendingObject, + metrics: &SmallWriteMetrics, + ) -> Result { + let start = self.cursor; + let chunk_id = self + .chunk + .id + .ok_or_else(|| IoError::AllocationFailed("shared chunk missing ID".into()))?; + let physical = physical_bytes(object.len)?; + let location = Location { + chunk_id: Some(chunk_id), + offset: start, + length: physical as u64, + logical_offset: 0, + logical_length: object.len as u64, + }; + if let Some(intent) = &object.intent { + intent.before_write(&location).await?; + } + let mut source = FragmentCursor { + receiver: object + .stream + .as_mut() + .ok_or_else(|| IoError::Internal("shared object stream missing receiver".into()))?, + current: None, + offset: 0, + received: 0, + }; + let mut remaining = object.len; + let write_time_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |duration| { + u64::try_from(duration.as_millis()).unwrap_or(u64::MAX) + }); + while remaining > 0 { + self.write_stream_strip(&mut source, &mut remaining, chunk_id, write_time_ms) + .await?; + } + if self.cursor - start != physical as u64 { + return Err(IoError::Internal("shared object physical length mismatch".into())); + } + self.flush_pending_advance().await?; + if self.chunk.acknowledged_cursor < self.cursor { + self.start_pending_advance(self.cursor)?; + self.flush_pending_advance().await?; + } + if self.chunk.acknowledged_cursor < self.cursor { + return Err(IoError::MetadataConflict( + "shared object cursor was not published".into(), + )); + } + self.confirm_batch_publication(std::slice::from_ref(object), self.cursor) + .await?; + metrics.record_batch(1, object.len); + Ok(location) + } + + async fn write_stream_strip( + &mut self, + source: &mut FragmentCursor<'_>, + remaining: &mut usize, + chunk_id: ChunkId, + write_time_ms: u64, + ) -> Result<()> { + self.ensure_strip().await?; + let strip = self.current_strip()?.clone(); + let strip_start = u64::from(strip.chunk_offset) * 1024; + let strip_end = u64::from(strip.chunk_offset.saturating_add(strip.capacity)) * 1024; + let block_offset = self.cursor - strip_start; + let frame_capacity = (strip_end - self.cursor) as usize / MAX_FRAME_BYTES; + if frame_capacity == 0 { + return Err(IoError::Internal("shared object lost frame alignment".into())); + } + let mut shadow = self.take_shadow(strip.capacity as usize * 1024, block_offset as usize); + let before = shadow.len(); + let mut logical = 0usize; + for _ in 0..frame_capacity { + if *remaining == 0 { + break; + } + let length = (*remaining).min(MAX_FRAME_PAYLOAD_BYTES); + let payload = source.take(length).await?; + let frame = encode_frame(FrameMagic::RepoSmallV1, chunk_id, &payload, write_time_ms) + .map_err(|error| IoError::WriteFailed(error.to_string()))?; + shadow.extend_from_slice(&frame); + *remaining -= length; + logical += length; + } + let written = shadow.len() - before; + let end = self.cursor + written as u64; + self.consume_staged_reservation(end).await?; + let frozen = shadow.freeze(); + let view = frozen.slice(before..before + written); + let image = frozen.slice(0..before + written); + let (_, result) = self + .write_mirrors_with_repair( + &strip, + view, + image, + u64::from(strip.unit_kb) * 1024, + block_offset, + MirrorBatchStats { + object_count: 1, + buffer_count: source.received, + logical_bytes: logical, + }, + ) + .await; + self.shadow = Some( + frozen + .try_into_mut() + .unwrap_or_else(|shared| bytes::BytesMut::from(shared.as_ref())), + ); + result?; + self.cursor = end; + if end == strip_end { + let closed = self + .chunk + .strips + .iter() + .find(|current| current.strip_sequence == strip.strip_sequence) + .cloned() + .ok_or_else(|| IoError::MetadataConflict("closed mirror strip disappeared".into()))?; + self.schedule_closed_advance(end, strip.strip_sequence); + if let Err(error) = self.retain_closed_strip(closed).await { + tracing::warn!(%error, "mirror-to-EC fast path deferred to chunkdb"); + } + } else { + self.refresh_pending_advance().await?; + self.start_pending_advance(end)?; + } + Ok(()) + } +} diff --git a/lib/crowdb-chunk-client/src/writer/small_pool.rs b/lib/crowdb-chunk-client/src/writer/small_pool.rs index 46adea6f7..140cb06a8 100644 --- a/lib/crowdb-chunk-client/src/writer/small_pool.rs +++ b/lib/crowdb-chunk-client/src/writer/small_pool.rs @@ -25,6 +25,7 @@ pub(crate) struct PendingObject { pub route_hash: u64, pub route: Arc, pub fragments: Vec, + pub stream: Option>, pub len: usize, pub enqueued_at: Instant, pub completion: oneshot::Sender>>, @@ -55,6 +56,24 @@ impl RouteCharge { self.metrics.reserved_bytes.fetch_add(bytes, Ordering::Relaxed); } + pub fn try_reserve(&mut self, bytes: usize) -> bool { + let bytes = bytes as u64; + if self + .route + .used_bytes + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |used| { + used.checked_add(bytes) + .filter(|next| *next <= self.route.capacity_bytes) + }) + .is_err() + { + return false; + } + self.bytes += bytes; + self.metrics.reserved_bytes.fetch_add(bytes, Ordering::Relaxed); + true + } + fn rebind(&mut self, route: Arc) { if Arc::ptr_eq(&self.route, &route) { return; diff --git a/lib/crowdb-chunk-client/tests/benchmark_e2e.rs b/lib/crowdb-chunk-client/tests/benchmark_e2e.rs index e6105d7cc..8d260ed2e 100644 --- a/lib/crowdb-chunk-client/tests/benchmark_e2e.rs +++ b/lib/crowdb-chunk-client/tests/benchmark_e2e.rs @@ -108,11 +108,25 @@ async fn mixed_read_benchmark_prepares_and_verifies_real_frames() { ) .await; + eprintln!( + "read benchmark result: {}", + serde_json::to_string(&result).unwrap() + ); assert_eq!(result.errors, 0, "{:?}", result.error_messages); assert_eq!(result.reads, 4); assert_eq!(result.small_reads, 2); assert_eq!(result.large_reads, 2); assert_eq!(result.logical_bytes, 2 * KIB as u64 + 4 * MIB as u64); + assert_eq!(result.read_flow.location_normalizations, result.reads); + assert_eq!(result.read_flow.locations_examined, result.reads); + assert_eq!(result.read_flow.range_locations_examined, result.reads); + assert!(result.read_flow.layout_queries >= result.reads); + assert!(result.read_flow.layout_query_wait_ns > 0); + assert!(result.read_flow.strip_read_wait_ns > 0); + assert!(result.read_flow.chunk_read_wait_ns >= result.read_flow.strip_read_wait_ns); + assert!(result.read_flow.frame_decode_wait_ns > 0); + assert!(result.read_flow.frame_parse_wait_ns > 0); + assert!(result.read_flow.frame_parse_wait_ns <= result.read_flow.frame_decode_wait_ns); assert_eq!(result.incomplete_reads, 0); assert_eq!(result.stop_reason, "complete"); assert!(result.preparation_secs > 0.0); diff --git a/lib/crowdb-chunk-client/tests/chunk_reader_e2e.rs b/lib/crowdb-chunk-client/tests/chunk_reader_e2e.rs index 14beaec42..058ef5d1c 100644 --- a/lib/crowdb-chunk-client/tests/chunk_reader_e2e.rs +++ b/lib/crowdb-chunk-client/tests/chunk_reader_e2e.rs @@ -230,7 +230,11 @@ async fn ec_range_read_recovers_only_the_requested_bytes() { let (reader, fault) = reader_with_failures(&stack, vec![failed]).await; let start = MIB as u64 + 123; let end = start + 16 * KIB as u64; - let actual = reader.read_range(&result.locations, start, end).await.unwrap(); + let actual = reader + .read_range(&result.locations, start, end) + .await + .unwrap() + .concat(); assert_eq!( actual, data[usize::try_from(start).unwrap()..usize::try_from(end).unwrap()] @@ -241,7 +245,10 @@ async fn ec_range_read_recovers_only_the_requested_bytes() { let frame = parse_frame(&direct, result.locations[0].chunk_id.unwrap()).unwrap(); assert_eq!(frame.payload, &data[..frame.payload.len()]); assert!(fault.max_read.load(Ordering::Relaxed) <= 64 * KIB); - assert_eq!(reader.read_object(&result.locations).await.unwrap(), data); + assert_eq!( + reader.read_object(&result.locations).await.unwrap().concat(), + data + ); let after = stack.query_chunk(&result.locations[0]).await; assert!(after.strips[0].unavailable_segments.is_empty()); @@ -328,33 +335,42 @@ async fn partial_read_preserves_healthy_ec_shard_ranges_around_data_loss() { (partial.failures[0].start, partial.failures[0].end), (payload_per_shard, 3 * payload_per_shard) ); - assert_eq!(partial.ranges.len(), 2); + assert_eq!(partial.ranges.len(), 33); assert_eq!( - (partial.ranges[0].start, partial.ranges[0].end), + (partial.ranges[0].start, partial.ranges[15].end), (0, payload_per_shard) ); assert_eq!( - partial.ranges[0].data, + partial.ranges[..16] + .iter() + .map(|range| range.data.as_ref()) + .collect::>() + .concat(), data[..usize::try_from(payload_per_shard).unwrap()] ); assert_eq!( - (partial.ranges[1].start, partial.ranges[1].end), + (partial.ranges[16].start, partial.ranges[32].end), (3 * payload_per_shard, 4 * MIB as u64) ); assert_eq!( - partial.ranges[1].data, + partial.ranges[16..] + .iter() + .map(|range| range.data.as_ref()) + .collect::>() + .concat(), data[usize::try_from(3 * payload_per_shard).unwrap()..] ); let mut stream = reader.read_stream(&result.locations).unwrap(); - assert_eq!( - stream.next_chunk().await.unwrap().unwrap(), - data[..usize::try_from(payload_per_shard).unwrap()] - ); + let mut prefix = Vec::new(); + for _ in 0..16 { + prefix.extend_from_slice(&stream.next_chunk().await.unwrap().unwrap()); + } + assert_eq!(prefix, data[..usize::try_from(payload_per_shard).unwrap()]); assert!(matches!( stream.next_chunk().await.unwrap(), Err(ReadError::FailedRange { start, end, .. }) - if start == payload_per_shard && end == 3 * payload_per_shard + if start == payload_per_shard && end == payload_per_shard + MAX_FRAME_PAYLOAD_BYTES as u64 )); assert!(stream.next_chunk().await.is_none()); } @@ -385,7 +401,8 @@ async fn mirror_read_succeeds_and_reports_replica_loss() { .client .read_object(std::slice::from_ref(&location)) .await - .unwrap(), + .unwrap() + .concat(), data ); let (unreadable, _) = reader_with_failures(&stack, vec![primary]).await; @@ -431,7 +448,8 @@ async fn transient_mirror_read_failure_does_not_mark_segment_unavailable() { .client .read_object(std::slice::from_ref(&location)) .await - .unwrap(), + .unwrap() + .concat(), data ); stack.client.shutdown_small_writes().await.unwrap(); @@ -465,7 +483,11 @@ async fn chunkdb_restart_admits_verified_corruption_and_repairs_full_shard() { let (reader, fault) = reader_with_corruption(&stack, vec![failed_disk]).await; let length = 16 * KIB as u64; assert_eq!( - reader.read_range(&result.locations, 0, length).await.unwrap(), + reader + .read_range(&result.locations, 0, length) + .await + .unwrap() + .concat(), data[..16 * KIB] ); assert!(fault.max_read.load(Ordering::Relaxed) <= 64 * KIB); @@ -504,7 +526,8 @@ async fn chunkdb_restart_admits_verified_corruption_and_repairs_full_shard() { after_restart .read_range(&result.locations, 0, length) .await - .unwrap(), + .unwrap() + .concat(), data[..16 * KIB] ); } diff --git a/lib/crowdb-chunk-client/tests/chunk_reader_test.rs b/lib/crowdb-chunk-client/tests/chunk_reader_test.rs index c61afdb9e..df59616ed 100644 --- a/lib/crowdb-chunk-client/tests/chunk_reader_test.rs +++ b/lib/crowdb-chunk-client/tests/chunk_reader_test.rs @@ -22,8 +22,8 @@ use crowdb_protocol::chunkdb::rpc::{ }; use crowdb_protocol::common::{ChunkId, DiskId}; use crowdb_protocol::diskdb::rpc::Segment; -use crowdb_protocol::frame::{encode_frame, FrameMagic}; -use tokio::sync::Semaphore; +use crowdb_protocol::frame::{encode_frame, FrameMagic, MAX_FRAME_BYTES, MAX_FRAME_PAYLOAD_BYTES}; +use tokio::sync::{Notify, Semaphore}; const KIB: usize = 1024; const SHARD: usize = 64 * KIB; @@ -34,6 +34,37 @@ struct MemoryDiskIo { reads: Arc, } +struct BlockingDiskIo { + inner: MemoryDiskIo, + release_first: Arc, + started: Arc, +} + +#[async_trait] +impl DiskWriter for BlockingDiskIo { + async fn write(&self, _seg: &Segment, _unit_bytes: u64, _data: Bytes) -> Result<()> { + unreachable!("reader test does not write") + } + + async fn write_at_byte_offset( + &self, + _seg: &Segment, + _unit_bytes: u64, + _byte_offset: u64, + _data: Bytes, + ) -> Result<()> { + unreachable!("reader test does not write") + } + + async fn read(&self, segment: &Segment, unit_bytes: u64, offset: u64, length: u32) -> Result { + self.started.fetch_add(1, Ordering::AcqRel); + if offset == 0 { + self.release_first.notified().await; + } + self.inner.read(segment, unit_bytes, offset, length).await + } +} + struct SequenceAllocator { queries: AtomicUsize, first: Chunk, @@ -402,8 +433,396 @@ async fn object_reader_discards_bytes_from_an_expired_layout() { logical_length: 4, ..Location::default() }; - assert_eq!(reader.read_object(&[location]).await.unwrap(), b"new!".as_slice()); + assert_eq!( + reader.read_object(&[location]).await.unwrap().concat(), + b"new!".as_slice() + ); assert_eq!(allocator.queries.load(Ordering::Acquire), 2); + assert_eq!(reader.flow_metrics_snapshot().location_normalizations, 1); +} + +#[tokio::test] +async fn framed_read_stream_keeps_whole_frame_under_small_window_setting() { + let id = ChunkId { high: 13, low: 37 }; + let segment = segment(1); + let payload = b"abcdefghij"; + let frame = Bytes::from(encode_frame(FrameMagic::RepoLargeV1, id, payload, 1).unwrap()); + let chunk = Chunk { + id: Some(id), + state: ChunkState::Sealed as i32, + capacity: 1, + sealed_length: 1, + strips: vec![ChunkStrip { + unit_kb: 1, + capacity: 1, + sealed_length: 1, + sealed_ts_ms: 1, + strip_type: StripType::Mirror as i32, + strip: Some(Strip::MirrorStrip(MirrorStrip { + segments: vec![segment], + })), + ..ChunkStrip::default() + }], + ..Chunk::default() + }; + let allocator = Arc::new(SequenceAllocator { + queries: AtomicUsize::new(0), + first: chunk.clone(), + current: Mutex::new(chunk), + first_layout_validity_ms: 1_000, + full_reply: None, + full_calls: AtomicUsize::new(0), + }); + let disk_io = Arc::new(MemoryDiskIo { + shards: vec![(segment.disk_id.unwrap(), frame.clone())], + failed: Vec::new(), + reads: Arc::new(AtomicUsize::new(0)), + }); + let reader = ChunkReader::new( + allocator.clone(), + disk_io, + ChunkReadPolicy { + stream_window_bytes: 4, + ..ChunkReadPolicy::default() + }, + ) + .unwrap(); + let location = Location { + chunk_id: Some(id), + length: frame.len() as u64, + logical_length: payload.len() as u64, + ..Location::default() + }; + let mut stream = reader.read_stream(&[location]).unwrap(); + assert_eq!(stream.next_chunk().await.unwrap().unwrap(), payload.as_slice()); + assert!(stream.next_chunk().await.is_none()); + assert_eq!(allocator.queries.load(Ordering::Relaxed), 1); + let metrics = reader.flow_metrics_snapshot(); + assert_eq!(metrics.location_normalizations, 1); + assert_eq!(metrics.locations_examined, 1); + assert_eq!(metrics.range_locations_examined, 1); + assert_eq!(metrics.stream_windows, 1); + assert_eq!(metrics.layout_queries, 1); +} + +#[tokio::test] +async fn read_stream_selects_only_locations_overlapping_each_window() { + let id = ChunkId { high: 19, low: 41 }; + let segment = segment(1); + let mut physical = Vec::new(); + let mut locations = Vec::new(); + for (index, payload) in [b"abcd", b"efgh", b"ijkl"].into_iter().enumerate() { + let frame = encode_frame(FrameMagic::RepoLargeV1, id, payload, 1).unwrap(); + locations.push(Location { + chunk_id: Some(id), + offset: physical.len() as u64, + length: frame.len() as u64, + logical_offset: (index * 4) as u64, + logical_length: 4, + }); + physical.extend_from_slice(&frame); + } + let chunk = Chunk { + id: Some(id), + state: ChunkState::Sealed as i32, + capacity: 1, + sealed_length: 1, + strips: vec![ChunkStrip { + unit_kb: 1, + capacity: 1, + sealed_length: 1, + sealed_ts_ms: 1, + strip_type: StripType::Mirror as i32, + strip: Some(Strip::MirrorStrip(MirrorStrip { + segments: vec![segment], + })), + ..ChunkStrip::default() + }], + ..Chunk::default() + }; + let allocator = Arc::new(SequenceAllocator { + queries: AtomicUsize::new(0), + first: chunk.clone(), + current: Mutex::new(chunk), + first_layout_validity_ms: 1_000, + full_reply: None, + full_calls: AtomicUsize::new(0), + }); + let disk_io = Arc::new(MemoryDiskIo { + shards: vec![(segment.disk_id.unwrap(), Bytes::from(physical))], + failed: Vec::new(), + reads: Arc::new(AtomicUsize::new(0)), + }); + let reader = ChunkReader::new( + allocator.clone(), + disk_io, + ChunkReadPolicy { + stream_window_bytes: 4, + ..ChunkReadPolicy::default() + }, + ) + .unwrap(); + let mut stream = reader.read_stream(&locations).unwrap(); + for expected in [b"abcd".as_slice(), b"efgh", b"ijkl"] { + assert_eq!(stream.next_chunk().await.unwrap().unwrap(), expected); + } + assert!(stream.next_chunk().await.is_none()); + let metrics = reader.flow_metrics_snapshot(); + assert_eq!(metrics.location_normalizations, 1); + assert_eq!(metrics.locations_examined, 3); + assert_eq!(metrics.range_locations_examined, 3); + assert_eq!(metrics.stream_windows, 3); + assert_eq!(metrics.layout_queries, 1); +} + +#[tokio::test] +async fn stream_waits_for_first_result_and_retains_three_slots_until_consumer_releases_buffers() { + let id = ChunkId { high: 23, low: 51 }; + let segment = segment(1); + let mut physical = Vec::new(); + let mut locations = Vec::new(); + for (index, payload) in [b"aaaa", b"bbbb", b"cccc", b"dddd"].into_iter().enumerate() { + let frame = encode_frame(FrameMagic::RepoLargeV1, id, payload, 1).unwrap(); + locations.push(Location { + chunk_id: Some(id), + offset: physical.len() as u64, + length: frame.len() as u64, + logical_offset: (index * 4) as u64, + logical_length: 4, + }); + physical.extend_from_slice(&frame); + } + let chunk = Chunk { + id: Some(id), + state: ChunkState::Sealed as i32, + capacity: 1, + sealed_length: 1, + strips: vec![ChunkStrip { + unit_kb: 1, + capacity: 1, + sealed_length: 1, + sealed_ts_ms: 1, + strip_type: StripType::Mirror as i32, + strip: Some(Strip::MirrorStrip(MirrorStrip { + segments: vec![segment], + })), + ..ChunkStrip::default() + }], + ..Chunk::default() + }; + let allocator = Arc::new(SequenceAllocator { + queries: AtomicUsize::new(0), + first: chunk.clone(), + current: Mutex::new(chunk), + first_layout_validity_ms: 1_000, + full_reply: None, + full_calls: AtomicUsize::new(0), + }); + let started = Arc::new(AtomicUsize::new(0)); + let release_first = Arc::new(Notify::new()); + let disk_io = Arc::new(BlockingDiskIo { + inner: MemoryDiskIo { + shards: vec![(segment.disk_id.unwrap(), Bytes::from(physical))], + failed: Vec::new(), + reads: Arc::new(AtomicUsize::new(0)), + }, + release_first: Arc::clone(&release_first), + started: Arc::clone(&started), + }); + let reader = ChunkReader::new( + allocator, + disk_io, + ChunkReadPolicy { + stream_slots: 3, + stream_window_bytes: 4, + layout_safety_margin: Duration::ZERO, + ..ChunkReadPolicy::default() + }, + ) + .unwrap(); + let mut stream = reader.read_stream(&locations).unwrap(); + let waiting = tokio::spawn(async move { + let first = stream.next_chunk().await.unwrap().unwrap(); + (stream, first) + }); + tokio::time::timeout(Duration::from_secs(1), async { + while started.load(Ordering::Acquire) < 3 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(started.load(Ordering::Acquire), 3); + release_first.notify_one(); + let (mut stream, first) = waiting.await.unwrap(); + assert_eq!(&first[..], b"aaaa"); + let second = stream.next_chunk().await.unwrap().unwrap(); + let third = stream.next_chunk().await.unwrap().unwrap(); + assert_eq!(&second[..], b"bbbb"); + assert_eq!(&third[..], b"cccc"); + assert!( + tokio::time::timeout(Duration::from_millis(20), stream.next_chunk()) + .await + .is_err() + ); + assert_eq!(started.load(Ordering::Acquire), 3); + drop(first); + assert_eq!(stream.next_chunk().await.unwrap().unwrap(), b"dddd".as_slice()); + assert!(stream.next_chunk().await.is_none()); +} + +#[tokio::test] +async fn global_read_budget_remains_charged_while_http_keeps_frame_views() { + let id = ChunkId { high: 29, low: 57 }; + let mut segment = segment(1); + segment.unit_count = 2; + let payload = vec![0x5a; MAX_FRAME_PAYLOAD_BYTES]; + let frame = encode_frame(FrameMagic::RepoLargeV1, id, &payload, 1).unwrap(); + assert_eq!(frame.len(), MAX_FRAME_BYTES); + let physical = Bytes::from(frame.repeat(32)); + let chunk = Chunk { + id: Some(id), + state: ChunkState::Sealed as i32, + capacity: 2048, + sealed_length: 2048, + strips: vec![ChunkStrip { + unit_kb: 1024, + capacity: 2048, + sealed_length: 2048, + sealed_ts_ms: 1, + strip_type: StripType::Mirror as i32, + strip: Some(Strip::MirrorStrip(MirrorStrip { + segments: vec![segment], + })), + ..ChunkStrip::default() + }], + ..Chunk::default() + }; + let allocator = Arc::new(SequenceAllocator { + queries: AtomicUsize::new(0), + first: chunk.clone(), + current: Mutex::new(chunk), + first_layout_validity_ms: 1_000, + full_reply: None, + full_calls: AtomicUsize::new(0), + }); + let reads = Arc::new(AtomicUsize::new(0)); + let disk_io = Arc::new(MemoryDiskIo { + shards: vec![(segment.disk_id.unwrap(), physical)], + failed: Vec::new(), + reads: Arc::clone(&reads), + }); + let reader = ChunkReader::new( + allocator, + disk_io, + ChunkReadPolicy { + stream_slots: 3, + global_stream_bytes: 1024 * 1024, + layout_safety_margin: Duration::ZERO, + ..ChunkReadPolicy::default() + }, + ) + .unwrap(); + let location = Location { + chunk_id: Some(id), + length: 32 * MAX_FRAME_BYTES as u64, + logical_length: 32 * MAX_FRAME_PAYLOAD_BYTES as u64, + ..Location::default() + }; + let mut stream = reader.read_stream(&[location]).unwrap(); + let mut retained = Vec::new(); + for _ in 0..16 { + retained.push(stream.next_chunk().await.unwrap().unwrap()); + } + assert_eq!(reads.load(Ordering::Acquire), 1); + assert!( + tokio::time::timeout(Duration::from_millis(20), stream.next_chunk()) + .await + .is_err() + ); + drop(retained); + assert_eq!(stream.next_chunk().await.unwrap().unwrap(), payload.as_slice()); + assert_eq!(reads.load(Ordering::Acquire), 2); +} + +#[tokio::test] +async fn frame_crossing_physical_parts_returns_two_original_payload_views() { + let id = ChunkId { high: 31, low: 59 }; + let first_segment = segment(1); + let second_segment = segment(2); + let payload = b"abcdefghijklmnopqrst"; + let frame = encode_frame(FrameMagic::RepoSmallV1, id, payload, 1).unwrap(); + let frame_offset = 1000_usize; + let first_length = 1024 - frame_offset; + let mut first_data = vec![0; 1024]; + first_data[frame_offset..].copy_from_slice(&frame[..first_length]); + let first_data = Bytes::from(first_data); + let second_data = Bytes::from(frame[first_length..].to_vec()); + let strips = [first_segment, second_segment] + .into_iter() + .enumerate() + .map(|(index, segment)| ChunkStrip { + strip_sequence: u32::try_from(index).unwrap(), + chunk_offset: u32::try_from(index).unwrap(), + unit_kb: 1, + capacity: 1, + sealed_length: 1, + sealed_ts_ms: 1, + strip_type: StripType::Mirror as i32, + strip: Some(Strip::MirrorStrip(MirrorStrip { + segments: vec![segment], + })), + ..ChunkStrip::default() + }) + .collect(); + let chunk = Chunk { + id: Some(id), + state: ChunkState::Sealed as i32, + capacity: 2, + sealed_length: 2, + strips, + ..Chunk::default() + }; + let allocator = Arc::new(SequenceAllocator { + queries: AtomicUsize::new(0), + first: chunk.clone(), + current: Mutex::new(chunk), + first_layout_validity_ms: 1_000, + full_reply: None, + full_calls: AtomicUsize::new(0), + }); + let disk_io = Arc::new(MemoryDiskIo { + shards: vec![ + (first_segment.disk_id.unwrap(), first_data.clone()), + (second_segment.disk_id.unwrap(), second_data.clone()), + ], + failed: Vec::new(), + reads: Arc::new(AtomicUsize::new(0)), + }); + let reader = ChunkReader::new( + allocator, + disk_io, + ChunkReadPolicy { + layout_safety_margin: Duration::ZERO, + ..ChunkReadPolicy::default() + }, + ) + .unwrap(); + let location = Location { + chunk_id: Some(id), + offset: frame_offset as u64, + length: frame.len() as u64, + logical_length: payload.len() as u64, + ..Location::default() + }; + let views = reader + .read_range(&[location], 0, payload.len() as u64) + .await + .unwrap(); + assert_eq!(views.concat(), payload); + assert_eq!(views.len(), 2); + assert_eq!(views[0].as_ptr(), first_data.slice(frame_offset + 14..).as_ptr()); + assert_eq!(views[1].as_ptr(), second_data.as_ptr()); } #[tokio::test] @@ -456,7 +875,10 @@ async fn framed_mirror_crc_failure_uses_a_verified_fallback() { logical_length: payload.len() as u64, ..Location::default() }; - assert_eq!(reader.read_object(&[location]).await.unwrap(), payload.as_slice()); + assert_eq!( + reader.read_object(&[location]).await.unwrap().concat(), + payload.as_slice() + ); assert!(allocator.queries.load(Ordering::Acquire) >= 2); assert_eq!( allocator.current.lock().unwrap().strips[0].unavailable_segments, @@ -531,8 +953,8 @@ async fn marked_ec_fragment_shares_one_full_recovery_future() { reader.read_range(std::slice::from_ref(&location), 0, SHARD as u64), reader.read_range(std::slice::from_ref(&location), 0, SHARD as u64) ); - assert_eq!(first.unwrap(), data[..SHARD]); - assert_eq!(second.unwrap(), data[..SHARD]); + assert_eq!(first.unwrap().concat(), data[..SHARD]); + assert_eq!(second.unwrap().concat(), data[..SHARD]); assert_eq!(allocator.full_calls.load(Ordering::Acquire), 1); let bounded = ChunkReader::new( @@ -545,7 +967,11 @@ async fn marked_ec_fragment_shares_one_full_recovery_future() { ) .unwrap(); assert_eq!( - bounded.read_range(&[location], 0, SHARD as u64).await.unwrap(), + bounded + .read_range(&[location], 0, SHARD as u64) + .await + .unwrap() + .concat(), data[..SHARD] ); assert_eq!(allocator.full_calls.load(Ordering::Acquire), 1); diff --git a/lib/crowdb-chunk-client/tests/common/e2e_stack.rs b/lib/crowdb-chunk-client/tests/common/e2e_stack.rs index 776bf583b..0ca4ffdfd 100644 --- a/lib/crowdb-chunk-client/tests/common/e2e_stack.rs +++ b/lib/crowdb-chunk-client/tests/common/e2e_stack.rs @@ -211,9 +211,13 @@ impl E2eStack { drop(chunkdb); let replacement = ChunkdbProcess::start_with_options(&self.cluster.mgmt_endpoints, options); replacement.wait_for_ready().await; + let kv = Arc::new(CrowdbKvClient::new(ClientConfig::new( + self.cluster.mgmt_endpoints.clone(), + ))); + let registry = ServiceRegistryClient::from_shared(kv); + replacement.wait_for_registry_ready(®istry).await; self.chunkdb = Some(replacement); self.chunkdb_options = options; - tokio::time::sleep(Duration::from_secs(3)).await; } pub async fn query_chunk(&self, location: &Location) -> Chunk { @@ -255,6 +259,6 @@ impl E2eStack { .expect("send disk read"); let (code, data) = DiskioClient::await_read_response(response).await.unwrap(); assert_eq!(code, DiskIoRetCode::Success); - data.expect("successful disk read data") + data.expect("successful disk read data").to_vec() } } diff --git a/lib/crowdb-chunk-client/tests/large_object_writer_e2e.rs b/lib/crowdb-chunk-client/tests/large_object_writer_e2e.rs index e2f7b1996..009ddff64 100644 --- a/lib/crowdb-chunk-client/tests/large_object_writer_e2e.rs +++ b/lib/crowdb-chunk-client/tests/large_object_writer_e2e.rs @@ -20,7 +20,7 @@ use crowdb_common::ec::{encode_parity_from_shards, EcScheme}; use crowdb_kv_client::{ClientConfig, CrowdbKvClient, HardwareClient, ServiceRegistryClient}; use crowdb_protocol::chunkdb::rpc::{Chunk, ChunkState, EcState, Location, Strip}; use crowdb_protocol::diskdb::rpc::Segment; -use crowdb_protocol::frame::ChunkLocation; +use crowdb_protocol::frame::{ChunkLocation, MAX_FRAME_PAYLOAD_BYTES}; use e2e_stack::{all_binaries_available, E2eStack}; @@ -213,13 +213,15 @@ async fn large_write_multi_strip_persists_data_metadata_and_parity() { assert_eq!(chunk.strips.len(), 4); assert_ec_parity(&stack, &chunk, location).await; let read = stack.client.read_object(&result.locations).await.unwrap(); - assert_bytes_match(&read, &data); + assert!(read.len() > 1, "complete read must retain verified frame buffers"); + assert_bytes_match(&read.concat(), &data); assert_eq!( stack .client .read_range(&result.locations, 3 * MIB as u64 + 17, 9 * MIB as u64 + 31) .await - .unwrap(), + .unwrap() + .concat(), data[3 * MIB + 17..9 * MIB + 31] ); } @@ -276,14 +278,22 @@ async fn large_write_rotates_chunks_without_losing_data() { assert_ec_parity(&stack, &chunk, location).await; } - read_back.extend(stack.client.read_object(&result.locations).await.unwrap()); + read_back.extend( + stack + .client + .read_object(&result.locations) + .await + .unwrap() + .concat(), + ); assert_eq!(read_back, data); assert_eq!( stack .client .read_range(&result.locations, 7 * MIB as u64, 10 * MIB as u64) .await - .unwrap(), + .unwrap() + .concat(), data[7 * MIB..10 * MIB] ); let read_client = stack @@ -298,7 +308,7 @@ async fn large_write_rotates_chunks_without_losing_data() { let mut streamed = Vec::new(); while let Some(part) = stream.next_chunk().await { let part = part.unwrap(); - assert!(part.len() <= 3 * MIB); + assert!(part.len() <= MAX_FRAME_PAYLOAD_BYTES); streamed.extend_from_slice(&part); } assert_eq!(streamed, data); @@ -337,7 +347,15 @@ async fn large_write_unknown_size_partial_tail_is_durable() { u32::try_from(location.length.div_ceil(1024)).unwrap() ); assert!(chunk.strips.len() >= 2); - assert_eq!(stack.client.read_object(&result.locations).await.unwrap(), data); + assert_eq!( + stack + .client + .read_object(&result.locations) + .await + .unwrap() + .concat(), + data + ); } #[tokio::test] @@ -380,7 +398,10 @@ async fn large_write_replaces_failed_data_and_parity_segments_end_to_end() { 4 ); assert_eq!(client.large_write_repair_metrics().repaired_segments, 1); - assert_eq!(client.read_object(&result.locations).await.unwrap(), data); + assert_eq!( + client.read_object(&result.locations).await.unwrap().concat(), + data + ); assert_ec_parity(&stack, &chunk, &result.locations[0]).await; } } diff --git a/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs b/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs index ae96cea52..458452bfc 100644 --- a/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs +++ b/lib/crowdb-chunk-client/tests/small_object_writer_e2e.rs @@ -110,6 +110,7 @@ fn policy() -> SmallWritePolicy { control_interval: Duration::from_millis(1), cooldown: Duration::from_millis(1), chunk_capacity: 2 * MIB as u64, + object_limit: MAX_FRAME_PAYLOAD_BYTES, mirror_copies: 1, ..SmallWritePolicy::default() } @@ -121,6 +122,81 @@ async fn write_object(client: &ChunkIoClient, data: Bytes) -> Location { writer.on_finish().await.unwrap().remove(0) } +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn shared_object_spans_mirror_strips_with_one_location() { + if !all_binaries_available() { + return; + } + let mut configured = policy(); + configured.chunk_capacity = 16 * MIB as u64; + configured.object_limit = 2 * MIB; + let stack = E2eStack::start(configured).await; + let _prefix = write_object(&stack.client, Bytes::from_static(b"PAR1datafoot\x04\0\0\0PAR1")).await; + let data = Bytes::from( + (0_u32..1_200_000) + .map(|index| u8::try_from(index % 251).unwrap()) + .collect::>(), + ); + let mut writer = stack + .client + .prepare_shared_object_write_for_key(data.len(), b"shared-span") + .await + .unwrap(); + let submitted_before = stack.client.small_write_metrics().submitted; + for (index, fragment) in data.chunks(1024 * 1024).enumerate() { + writer.on_data(Bytes::copy_from_slice(fragment)).await.unwrap(); + if index == 0 { + assert_eq!(stack.client.small_write_metrics().submitted, submitted_before + 1); + } + } + let locations = writer.on_finish().await.unwrap(); + assert_eq!(locations.len(), 1); + let read = stack + .client + .read_range(&locations, 1_048_550, 1_048_601) + .await + .unwrap(); + assert_eq!(read.concat(), &data[1_048_550..1_048_601]); + let mut stream = stack + .client + .read_range_stream(&locations, 1_048_550, 1_048_601) + .unwrap(); + let mut streamed = Vec::new(); + while let Some(chunk) = stream.next_chunk().await { + streamed.extend_from_slice(&chunk.unwrap()); + } + assert_eq!(streamed, &data[1_048_550..1_048_601]); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn cancelled_streaming_shared_object_keeps_pipeline_available() { + if !all_binaries_available() { + return; + } + let mut configured = policy(); + configured.chunk_capacity = 16 * MIB as u64; + configured.object_limit = 2 * MIB; + let stack = E2eStack::start(configured).await; + let mut cancelled = stack + .client + .prepare_shared_object_write_for_key(1_200_000, b"cancelled") + .await + .unwrap(); + cancelled.on_data(Bytes::from(vec![7; MIB])).await.unwrap(); + cancelled.on_error().await.unwrap(); + + let data = Bytes::from_static(b"surviving object"); + let mut survivor = stack.client.prepare_small_write(data.len()).await.unwrap(); + survivor.on_data(data.clone()).await.unwrap(); + let location = survivor.finish_durable().await.unwrap().remove(0); + let read = stack + .client + .read_range(&[location], 0, data.len() as u64) + .await + .unwrap(); + assert_eq!(read.concat(), data); +} + async fn write_full_small_strip(client: &ChunkIoClient, value: u8) -> Vec<(Bytes, Location)> { let mut frames = Vec::with_capacity(FRAMES_PER_MIB_STRIP); for _ in 0..FRAMES_PER_MIB_STRIP { @@ -269,7 +345,8 @@ async fn small_write_batches_concurrent_objects_and_reads_them_back() { .client .read_object(std::slice::from_ref(&location)) .await - .unwrap(), + .unwrap() + .concat(), data ); let chunk = stack.query_chunk(&location).await; @@ -348,7 +425,8 @@ async fn eight_closed_mirror_strips_become_one_durable_ec_strip_without_reread() .client .read_object(std::slice::from_ref(location)) .await - .unwrap(), + .unwrap() + .concat(), *expected ); } @@ -570,7 +648,8 @@ async fn manual_chunkdb_trigger_converts_closed_active_range_end_to_end() { .client .read_object(std::slice::from_ref(location)) .await - .unwrap(), + .unwrap() + .concat(), *expected ); } @@ -580,7 +659,8 @@ async fn manual_chunkdb_trigger_converts_closed_active_range_end_to_end() { .client .read_object(std::slice::from_ref(&appended)) .await - .unwrap(), + .unwrap() + .concat(), appended_data ); stack.client.shutdown_small_writes().await.unwrap(); @@ -693,7 +773,8 @@ async fn automatic_chunkdb_scan_converts_three_groups_and_preserves_tail() { .client .read_object(std::slice::from_ref(location)) .await - .unwrap(), + .unwrap() + .concat(), *expected ); } diff --git a/lib/crowdb-chunk-stream/src/production_chunk.rs b/lib/crowdb-chunk-stream/src/production_chunk.rs index 8ea70374c..3ed2c5278 100644 --- a/lib/crowdb-chunk-stream/src/production_chunk.rs +++ b/lib/crowdb-chunk-stream/src/production_chunk.rs @@ -535,6 +535,7 @@ impl StreamChunkStore for ProductionStreamChunkStore { length, ) .await + .map(|buffers| Bytes::from(buffers.concat())) .map_err(read_error) } diff --git a/lib/crowdb-common/rust/tests/ec_test.rs b/lib/crowdb-common/rust/tests/ec_test.rs index 182f2d374..7e5c0b1ac 100644 --- a/lib/crowdb-common/rust/tests/ec_test.rs +++ b/lib/crowdb-common/rust/tests/ec_test.rs @@ -202,6 +202,49 @@ fn encode_parity_from_shards_partial_2_of_4() { assert_eq!(recovered2[1], shard1); } +#[test] +fn partial_8_2_strip_recovers_two_mib_with_virtual_zero_shards() { + let scheme = EcScheme::new(8, 2); + let shard_bytes = 1024 * 1024; + let first: Vec = (0..shard_bytes).map(|index| (index % 251) as u8).collect(); + let second: Vec = (0..shard_bytes).map(|index| ((index * 7) % 253) as u8).collect(); + + let mut encoder = IncrementalParity::new(scheme).unwrap(); + encoder.push_partial(&first).unwrap(); + encoder.push_partial(&second).unwrap(); + let parity = encoder.finish_partial().unwrap(); + assert_eq!(parity.len(), 2); + assert!(parity.iter().all(|shard| shard.len() == shard_bytes)); + + let empty = vec![0; shard_bytes]; + let full_data = [ + &first[..], + &second[..], + &empty, + &empty, + &empty, + &empty, + &empty, + &empty, + ]; + assert_eq!(parity, encode_parity_from_shards(scheme, &full_data).unwrap()); + + for missing in [[0, 1], [0, 8], [1, 9]] { + let mut blocks: Vec>> = vec![Some(first.clone()), Some(second.clone())]; + blocks.extend((0..6).map(|_| Some(vec![0; shard_bytes]))); + blocks.extend(parity.iter().cloned().map(Some)); + for index in missing { + blocks[index] = None; + } + let recovered = decode(scheme, blocks).unwrap(); + assert_eq!(recovered[0], first); + assert_eq!(recovered[1], second); + assert!(recovered[2..8] + .iter() + .all(|shard| shard.iter().all(|byte| *byte == 0))); + } +} + /// Single short shard (1 of 4, < 1 MB). Encode parity → parity length /// matches shard length. #[test] diff --git a/lib/crowdb-diskio-client/src/client.rs b/lib/crowdb-diskio-client/src/client.rs index d13fed9cb..5b43033e9 100644 --- a/lib/crowdb-diskio-client/src/client.rs +++ b/lib/crowdb-diskio-client/src/client.rs @@ -3,6 +3,7 @@ //! Internal `DiskIO` wire transport. +use bytes::Bytes; use crowdb_common::RequestIdGen; use crowdb_protocol::diskio_fb::{ FBDiskFsyncRequest, FBDiskFsyncRequestArgs, FBDiskFsyncResponse, FBDiskReadRequest, @@ -397,10 +398,10 @@ impl WireClient { /// # Errors /// /// Returns `DiskioError::Rpc` if the response is missing or invalid. - pub async fn await_read_response(fut: CallFuture) -> WireResult<(DiskIoRetCode, Option>)> { + pub async fn await_read_response(fut: CallFuture) -> WireResult<(DiskIoRetCode, Option)> { let resp = fut.await.map_err(WireError::from)?; let code = parse_ret_code(&resp)?; - let data = resp.data.map(|b| b.bytes().to_vec()); + let data = resp.data.map(Bytes::from_owner); Ok((code, data)) } diff --git a/lib/crowdb-diskio-client/src/semantic.rs b/lib/crowdb-diskio-client/src/semantic.rs index 764865ddc..30def2064 100644 --- a/lib/crowdb-diskio-client/src/semantic.rs +++ b/lib/crowdb-diskio-client/src/semantic.rs @@ -1014,7 +1014,7 @@ impl DiskioClient { data.len() ))); } - Ok(Bytes::from(data)) + Ok(data) } fn classify_wire(error: WireError, operation: OperationKind) -> DiskioError { diff --git a/lib/crowdb-protocol/Cargo.toml b/lib/crowdb-protocol/Cargo.toml index c0280a60a..54cb0ee9d 100644 --- a/lib/crowdb-protocol/Cargo.toml +++ b/lib/crowdb-protocol/Cargo.toml @@ -24,6 +24,7 @@ serde = { version = "1", features = ["derive"] } serde_json = "1" bincode = "1" crc32fast = "1" +crc32c = "0.6.8" getrandom = "0.2" xxhash-rust = { version = "0.8", features = ["xxh64"] } sha2 = "0.10" diff --git a/lib/crowdb-protocol/src/fbs/iceberg.fbs b/lib/crowdb-protocol/src/fbs/iceberg.fbs index 3b64be5c3..c8a031774 100644 --- a/lib/crowdb-protocol/src/fbs/iceberg.fbs +++ b/lib/crowdb-protocol/src/fbs/iceberg.fbs @@ -171,6 +171,8 @@ table FBFileRecord { has_hint:bool; hint_offset:ulong; hint_length:ulong; + locations:[ubyte]; + etag:string; } table FBFileMapping { @@ -252,8 +254,11 @@ table FBMultipartPart { number:ushort; revision:ulong; owner:FBFileIdentity (required); - tree:FBFileTree (required); + tree:FBFileTree; modified_ms:ulong; + locations:[ubyte]; + etag:string; + stream_length:ulong; } table FBMultipartCreditMutation { @@ -426,6 +431,7 @@ table FBGcCandidate { next_root:ushort; frames:[FBGcFrame] (required); pending:FBFileChunkRoot; + next_location:ushort; } table FBFileWriteIntent { @@ -451,18 +457,6 @@ table FBGcPage { entries:[FBGcEntry] (required); } -table FBGcPin { - catalog:[ubyte] (required); - activation_epoch:ulong; - identity:[ubyte] (required); - head:FBTableHead (required); - principal:string (required); - expires_ms:ulong; - released:bool; - operator_pin:bool; - protects_uploads:bool; -} - table FBGcNode { continuation:FBPayloadReference; head:FBTableHead; @@ -476,7 +470,7 @@ table FBGcNode { complete:bool; } -union FBRecordValue { FBActiveCatalog, FBCatalogAuthority, FBManagementOperation, FBRetryRecord, FBNamespaceAuthority, FBNamespaceMapping, FBPayloadPage, FBRetryResult, FBNamespaceOperation, FBFileRecord, FBFileMapping, FBMultipartSession, FBMultipartPart, FBMultipartAdmission, FBTableHead, FBTableMapping, FBTableCommitOperation, FBTableCreateOperation, FBTableLifecycleOperation, FBTablePurgeTask, FBGcTask, FBGcCandidate, FBGcPage, FBGcPin, FBGcNode, FBFileWriteIntent } +union FBRecordValue { FBActiveCatalog, FBCatalogAuthority, FBManagementOperation, FBRetryRecord, FBNamespaceAuthority, FBNamespaceMapping, FBPayloadPage, FBRetryResult, FBNamespaceOperation, FBFileRecord, FBFileMapping, FBMultipartSession, FBMultipartPart, FBMultipartAdmission, FBTableHead, FBTableMapping, FBTableCommitOperation, FBTableCreateOperation, FBTableLifecycleOperation, FBTablePurgeTask, FBGcTask, FBGcCandidate, FBGcPage, FBGcNode, FBFileWriteIntent } table FBIcebergRecord { schema_version:ushort; diff --git a/lib/crowdb-protocol/src/frame.rs b/lib/crowdb-protocol/src/frame.rs index 8b9ec7729..60271b299 100644 --- a/lib/crowdb-protocol/src/frame.rs +++ b/lib/crowdb-protocol/src/frame.rs @@ -81,6 +81,99 @@ pub struct ParsedFrame<'a> { pub physical_length: usize, } +/// Verified frame metadata when the physical bytes span several buffers. +pub struct ParsedFrameViews { + pub header: FrameHeaderPrefix, + pub chunk_id: ChunkId, + pub physical_length: usize, +} + +/// Verify a frame over ordered views without assembling its payload. +/// +/// # Errors +/// Returns an error for an incomplete, malformed, or corrupt frame. +pub fn parse_frame_views( + views: &[&[u8]], + expected_chunk_id: ChunkId, +) -> Result { + let mut header_bytes = [0_u8; FRAME_HEADER_PREFIX_BYTES]; + copy_view_range(views, 0, &mut header_bytes)?; + let header = parse_header(&header_bytes)?; + let length = frame_length(header)?; + let available = views + .iter() + .fold(0_usize, |total, view| total.saturating_add(view.len())); + if available < length { + return Err(FrameError::Incomplete { + required_bytes: length, + }); + } + let footer_start = length - FRAME_FOOTER_BYTES; + let mut footer = [0_u8; FRAME_FOOTER_BYTES]; + copy_view_range(views, footer_start, &mut footer)?; + let expected_crc = u32::from_le_bytes( + footer[..4] + .try_into() + .map_err(|_| FrameError::InvalidRegionLength)?, + ); + let mut crc = 0_u32; + let mut cursor = 0_usize; + for view in views { + let included = footer_start.saturating_sub(cursor).min(view.len()); + if included > 0 { + crc = !crc32c::crc32c_append(!crc, &view[..included]); + } + cursor = cursor.saturating_add(view.len()); + if cursor >= footer_start { + break; + } + } + crc = !crc32c::crc32c_append(!crc, &footer[4..]); + if crc != expected_crc { + return Err(FrameError::ChecksumMismatch); + } + let chunk_id = ChunkId { + high: u64::from_be_bytes( + footer[4..12] + .try_into() + .map_err(|_| FrameError::InvalidRegionLength)?, + ), + low: u64::from_be_bytes( + footer[12..20] + .try_into() + .map_err(|_| FrameError::InvalidRegionLength)?, + ), + }; + if chunk_id != expected_chunk_id { + return Err(FrameError::ChunkIdMismatch); + } + Ok(ParsedFrameViews { + header, + chunk_id, + physical_length: length, + }) +} + +fn copy_view_range(views: &[&[u8]], start: usize, target: &mut [u8]) -> Result<(), FrameError> { + let required_bytes = start.saturating_add(target.len()); + let mut cursor = 0_usize; + let mut copied = 0_usize; + for view in views { + let view_end = cursor.saturating_add(view.len()); + if view_end > start && copied < target.len() { + let from = start.saturating_sub(cursor); + let count = (view.len() - from).min(target.len() - copied); + target[copied..copied + count].copy_from_slice(&view[from..from + count]); + copied += count; + } + cursor = view_end; + if copied == target.len() { + return Ok(()); + } + } + Err(FrameError::Incomplete { required_bytes }) +} + /// Encode a canonical v1 frame without header extensions. /// /// # Errors @@ -405,26 +498,16 @@ fn write_header_region(region: &mut [u8], header: FrameHeaderPrefix) { region[6..14].copy_from_slice(&header.write_time_ms.to_le_bytes()); } -fn crc32c(bytes: &[u8]) -> u32 { - crc32c_parts([bytes]) -} - fn crc32c_parts<'a>(parts: impl IntoIterator) -> u32 { let mut crc = 0_u32; for bytes in parts { - for byte in bytes { - crc ^= u32::from(*byte); - for _ in 0..8 { - crc = (crc >> 1) ^ (0x82F6_3B78 & (0_u32.wrapping_sub(crc & 1))); - } - } + // The frame format stores the raw seed-zero CRC, while this API applies + // initial and final XOR. Invert around each append to preserve the wire value. + crc = !crc32c::crc32c_append(!crc, bytes); } crc } fn crc32c_frame_parts(prefix: &[u8], chunk_id: &[u8]) -> u32 { - let mut bytes = Vec::with_capacity(prefix.len() + chunk_id.len()); - bytes.extend_from_slice(prefix); - bytes.extend_from_slice(chunk_id); - crc32c(&bytes) + crc32c_parts([prefix, chunk_id]) } diff --git a/lib/crowdb-protocol/tests/frame_test.rs b/lib/crowdb-protocol/tests/frame_test.rs index 8db92ea71..e130cbe6a 100644 --- a/lib/crowdb-protocol/tests/frame_test.rs +++ b/lib/crowdb-protocol/tests/frame_test.rs @@ -3,9 +3,9 @@ use crowdb_protocol::common::ChunkId; use crowdb_protocol::frame::{ - encode_frame, encode_frame_regions, encode_frames, merge_adjacent_locations, parse_frame, ChunkLocation, - FrameError, FrameMagic, FRAME_FOOTER_BYTES, FRAME_HEADER_PREFIX_BYTES, MAX_FRAME_BYTES, - MAX_FRAME_PAYLOAD_BYTES, + encode_frame, encode_frame_regions, encode_frames, merge_adjacent_locations, parse_frame, + parse_frame_views, ChunkLocation, FrameError, FrameMagic, FRAME_FOOTER_BYTES, FRAME_HEADER_PREFIX_BYTES, + MAX_FRAME_BYTES, MAX_FRAME_PAYLOAD_BYTES, }; const CHUNK: ChunkId = ChunkId { high: 7, low: 11 }; @@ -25,6 +25,55 @@ fn frame_matches_cross_language_vector() { ); } +#[test] +fn split_frame_views_verify_without_assembling_payload() { + let payload = vec![0x6b; 1024]; + let frame = encode_frame(FrameMagic::RepoSmallV1, CHUNK, &payload, 42).unwrap(); + let boundaries = [3, 11, 127, frame.len() - 9, frame.len() - 2]; + let mut views = Vec::new(); + let mut start = 0; + for end in boundaries.into_iter().chain(std::iter::once(frame.len())) { + views.push(&frame[start..end]); + start = end; + } + let parsed = parse_frame_views(&views, CHUNK).unwrap(); + assert_eq!(parsed.header.magic, FrameMagic::RepoSmallV1); + assert_eq!(parsed.physical_length, frame.len()); + + let mut corrupted = frame; + corrupted[127] ^= 1; + let mut start = 0; + let mut damaged = Vec::new(); + for end in boundaries.into_iter().chain(std::iter::once(corrupted.len())) { + damaged.push(&corrupted[start..end]); + start = end; + } + assert_eq!( + parse_frame_views(&damaged, CHUNK).err(), + Some(FrameError::ChecksumMismatch) + ); +} + +#[test] +fn frame_crc_matches_bitwise_reference_across_header_payload_and_chunk_id() { + for length in [0, 1, 17, 1024, MAX_FRAME_PAYLOAD_BYTES] { + let payload: Vec = (0..length) + .map(|index| u8::try_from(index % 251).unwrap()) + .collect(); + let frame = encode_frame(FrameMagic::RepoLargeV1, CHUNK, &payload, 42).unwrap(); + let footer = frame.len() - FRAME_FOOTER_BYTES; + let mut crc = 0_u32; + for byte in frame[..footer].iter().chain(frame[footer + 4..].iter()) { + crc ^= u32::from(*byte); + for _ in 0..8 { + crc = (crc >> 1) ^ (0x82F6_3B78 & (0_u32.wrapping_sub(crc & 1))); + } + } + assert_eq!(frame[footer..footer + 4], crc.to_le_bytes()); + assert_eq!(parse_frame(&frame, CHUNK).unwrap().payload, payload); + } +} + #[test] fn frame_round_trips_and_has_canonical_maximum_size() { let payload = vec![0xA5; MAX_FRAME_PAYLOAD_BYTES]; diff --git a/lib/crowdb-rpc/ffi/src/buffer.rs b/lib/crowdb-rpc/ffi/src/buffer.rs index 09c90f337..946c16354 100644 --- a/lib/crowdb-rpc/ffi/src/buffer.rs +++ b/lib/crowdb-rpc/ffi/src/buffer.rs @@ -123,6 +123,12 @@ impl std::fmt::Debug for Buffer { } impl Buffer { + /// Transfer a receive buffer out of its pool, preserving its allocation. + /// Returns false for pool types that cannot detach receive memory. + pub fn detach(&mut self) -> bool { + !self.handle.is_null() && unsafe { sys::crowdb_rpc_buffer_detach(self.handle) } + } + /// Allocate a new buffer from the pool with the given capacity. /// Returns `None` if the pool is exhausted. pub fn alloc(pool: &BufferPool, capacity: u32) -> Option { @@ -242,6 +248,12 @@ impl Drop for Buffer { } } +impl AsRef<[u8]> for Buffer { + fn as_ref(&self) -> &[u8] { + self.bytes() + } +} + // Buffer is Send (C++ buffers are thread-safe via atomic refcount). // Not Sync (the write path is single-threaded per buffer). unsafe impl Send for Buffer {} diff --git a/lib/crowdb-rpc/ffi/src/client.rs b/lib/crowdb-rpc/ffi/src/client.rs index 6c02a8114..6b47ea341 100644 --- a/lib/crowdb-rpc/ffi/src/client.rs +++ b/lib/crowdb-rpc/ffi/src/client.rs @@ -520,7 +520,7 @@ unsafe extern "C" fn on_complete_cb( Ok(Response { request_id, control: standalone_buffer(control), - data: standalone_buffer(data), + data: receive_data_buffer(data), }) } else { Err(RpcError::from_status(status)) @@ -538,6 +538,19 @@ fn standalone_buffer(handle: sys::crowdb_rpc_buffer_t) -> Option { Some(Buffer::from_bytes(&bytes)) } +fn receive_data_buffer(handle: sys::crowdb_rpc_buffer_t) -> Option { + if handle.is_null() { + return None; + } + let mut buffer = Buffer::from_raw(handle); + if buffer.detach() { + Some(buffer) + } else { + let copy = Buffer::from_bytes(buffer.bytes()); + Some(copy) + } +} + // Rust-side histograms (registered with the Rust MetricsRegistry::global(), // flushed in the rust section). fn e2e_histogram() -> std::sync::Arc { diff --git a/lib/crowdb-rpc/ffi/src/sys.rs b/lib/crowdb-rpc/ffi/src/sys.rs index 653eb1fe2..6b62f1505 100644 --- a/lib/crowdb-rpc/ffi/src/sys.rs +++ b/lib/crowdb-rpc/ffi/src/sys.rs @@ -122,6 +122,7 @@ extern "C" { pub fn crowdb_rpc_buffer_write(buf: crowdb_rpc_buffer_t, data: *const u8, len: u32); pub fn crowdb_rpc_buffer_data(buf: crowdb_rpc_buffer_t) -> *const u8; pub fn crowdb_rpc_buffer_len(buf: crowdb_rpc_buffer_t) -> u32; + pub fn crowdb_rpc_buffer_detach(buf: crowdb_rpc_buffer_t) -> bool; pub fn crowdb_rpc_buffer_ref(buf: crowdb_rpc_buffer_t) -> crowdb_rpc_buffer_t; pub fn crowdb_rpc_buffer_release(buf: crowdb_rpc_buffer_t); pub fn crowdb_rpc_buffer_create(data: *const u8, len: u32) -> crowdb_rpc_buffer_t; diff --git a/lib/crowdb-rpc/ffi/tests/buffer_test.rs b/lib/crowdb-rpc/ffi/tests/buffer_test.rs index 404bc7b69..859b43869 100644 --- a/lib/crowdb-rpc/ffi/tests/buffer_test.rs +++ b/lib/crowdb-rpc/ffi/tests/buffer_test.rs @@ -2,7 +2,21 @@ // Licensed under the Apache License, Version 2.0. use bytes::Bytes; -use crowdb_rpc_ffi::{Buffer, BufferChain, BufferChainError}; +use crowdb_rpc_ffi::{Buffer, BufferChain, BufferChainError, BufferPool}; + +#[test] +fn detached_read_buffer_survives_pool_and_keeps_pointer() { + let pool = BufferPool::new(1); + let mut buffer = pool.alloc_buffer(4).expect("pool allocation"); + buffer.write(b"read"); + let original = buffer.bytes().as_ptr(); + assert!(buffer.detach()); + drop(pool); + + let bytes = Bytes::from_owner(buffer); + assert_eq!(bytes.as_ptr(), original); + assert_eq!(&bytes[..], b"read"); +} #[test] fn owned_bytes_buffer_keeps_the_original_allocation() { diff --git a/lib/crowdb-rpc/include/crowdb-rpc/buffer.h b/lib/crowdb-rpc/include/crowdb-rpc/buffer.h index ca214a00a..a83207c30 100644 --- a/lib/crowdb-rpc/include/crowdb-rpc/buffer.h +++ b/lib/crowdb-rpc/include/crowdb-rpc/buffer.h @@ -69,6 +69,14 @@ class BufferPool // Called by Buffer::release when ref == 0. Returns the buffer to the // free list for reuse. virtual void recycle(Buffer *buf) = 0; + + // Transfer an immutable receive buffer out of the pool without copying. + // The caller must hold the only live reference before detaching. + virtual bool detach(Buffer *buf) + { + (void)buf; + return false; + } }; // SystemBufferPool: direct heap allocation (posix_memalign + new/delete), @@ -85,6 +93,7 @@ class SystemBufferPool : public BufferPool Buffer *alloc(uint32_t capacity) override; void recycle(Buffer *buf) override; + bool detach(Buffer *buf) override; private: uint32_t max_buffers_; diff --git a/lib/crowdb-rpc/include/crowdb-rpc/c_api.h b/lib/crowdb-rpc/include/crowdb-rpc/c_api.h index 9735e6ffa..450f4db1b 100644 --- a/lib/crowdb-rpc/include/crowdb-rpc/c_api.h +++ b/lib/crowdb-rpc/include/crowdb-rpc/c_api.h @@ -37,6 +37,7 @@ crowdb_rpc_buffer_t crowdb_rpc_buffer_alloc(crowdb_rpc_pool_t pool, uint32_t cap void crowdb_rpc_buffer_write(crowdb_rpc_buffer_t buf, const uint8_t *data, uint32_t len); const uint8_t *crowdb_rpc_buffer_data(crowdb_rpc_buffer_t buf); uint32_t crowdb_rpc_buffer_len(crowdb_rpc_buffer_t buf); +bool crowdb_rpc_buffer_detach(crowdb_rpc_buffer_t buf); crowdb_rpc_buffer_t crowdb_rpc_buffer_ref(crowdb_rpc_buffer_t buf); void crowdb_rpc_buffer_release(crowdb_rpc_buffer_t buf); // Create a standalone buffer (not pool-allocated) from raw bytes. The diff --git a/lib/crowdb-rpc/src/buffer.cpp b/lib/crowdb-rpc/src/buffer.cpp index 4d2f67858..545317569 100644 --- a/lib/crowdb-rpc/src/buffer.cpp +++ b/lib/crowdb-rpc/src/buffer.cpp @@ -101,4 +101,14 @@ void SystemBufferPool::recycle(Buffer *buf) outstanding_.fetch_sub(1, std::memory_order_relaxed); } +bool SystemBufferPool::detach(Buffer *buf) +{ + if (buf == nullptr || buf->pool != this || buf->ref->load(std::memory_order_acquire) != 1) { + return false; + } + buf->pool = nullptr; + outstanding_.fetch_sub(1, std::memory_order_relaxed); + return true; +} + } // namespace crowdb::rpc diff --git a/lib/crowdb-rpc/src/c_api.cpp b/lib/crowdb-rpc/src/c_api.cpp index 28b2b811c..cb8cd19e0 100644 --- a/lib/crowdb-rpc/src/c_api.cpp +++ b/lib/crowdb-rpc/src/c_api.cpp @@ -110,6 +110,20 @@ uint32_t crowdb_rpc_buffer_len(crowdb_rpc_buffer_t buf) } } +bool crowdb_rpc_buffer_detach(crowdb_rpc_buffer_t buf) +{ + try { + if (buf == nullptr || buf->buf == nullptr) { + return false; + } + auto *buffer = buf->buf; + return buffer->pool == nullptr || buffer->pool->detach(buffer); + } + catch (...) { + return false; + } +} + crowdb_rpc_buffer_t crowdb_rpc_buffer_ref(crowdb_rpc_buffer_t buf) { try { diff --git a/lib/crowdb-rpc/tests/buffer_test.cpp b/lib/crowdb-rpc/tests/buffer_test.cpp index 30860defe..dab729c92 100644 --- a/lib/crowdb-rpc/tests/buffer_test.cpp +++ b/lib/crowdb-rpc/tests/buffer_test.cpp @@ -6,6 +6,7 @@ #include #include +#include using crowdb::rpc::Buffer; using crowdb::rpc::BufferType; @@ -82,3 +83,23 @@ TEST(BufferTest, ExactCapacityNoBucketing) EXPECT_EQ(buf->capacity, 200U); // exact, not bucketed to 256 buf->release(); } + +TEST(BufferTest, DetachedReceiveBufferOutlivesPoolWithoutCopy) +{ + auto pool = std::make_unique(1); + Buffer *buf = pool->alloc(4); + ASSERT_NE(buf, nullptr); + const uint8_t data[] = {1, 2, 3, 4}; + buf->write(data, 4); + uint8_t *original = buf->data; + + EXPECT_TRUE(pool->detach(buf)); + EXPECT_EQ(buf->data, original); + Buffer *next = pool->alloc(4); + ASSERT_NE(next, nullptr); + next->release(); + pool.reset(); + + EXPECT_EQ(std::memcmp(buf->data, data, 4), 0); + buf->release(); +} diff --git a/pixi.toml b/pixi.toml index 4eb1b0cb2..c6b6194d5 100644 --- a/pixi.toml +++ b/pixi.toml @@ -216,7 +216,7 @@ test-suite = { cmd = "bash tools/pixi-tasks/test-suite.sh", depends-on = ["build # ── Coverage & LOC ─────────────────────────────────────────────── coverage = { cmd = "cargo tarpaulin --workspace --out Html --output-dir target/coverage --exclude-files '*/tests/*'", depends-on = ["install-deps"] } -loc = "tokei . .github --exclude swagger-ui --exclude third-party --exclude 'lib/crowdb-tree/build-*' --exclude target --exclude dist --exclude node_modules --exclude 'doc/user-manual/*.html' --exclude 'doc/user-manual/*.py' --exclude package-lock.json" +loc = "tokei . .github --exclude swagger-ui --exclude third-party --exclude 'lib/crowdb-tree/build-*' --exclude target --exclude dist --exclude node_modules --exclude package-lock.json" # ── Clean ──────────────────────────────────────────────────────── # Removes build products and disposable namespaced runtime artifacts. Durable diff --git a/tools/pixi-tasks/test-java-iceberg-e2e.sh b/tools/pixi-tasks/test-java-iceberg-e2e.sh index 2ade415da..d60a2ff1f 100644 --- a/tools/pixi-tasks/test-java-iceberg-e2e.sh +++ b/tools/pixi-tasks/test-java-iceberg-e2e.sh @@ -6,7 +6,8 @@ cd "${PIXI_PROJECT_ROOT:?}" source tools/pixi-tasks/prepare-iceberg.sh pixi run -e iceberg-e2e -- mvn --batch-mode --no-transfer-progress \ - -f app/crowdb-access-server/tests/common/iceberg_java/pom.xml compile exec:help + -f app/crowdb-access-server/tests/common/iceberg_java/pom.xml \ + dependency:go-offline compile exec:help pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e \ --test iceberg_namespace_sdk_test official_catalog_continues_through_empty_namespace_pages -- --ignored --exact pixi run -e default -- cargo test --release -p crowdb-access-server --features iceberg-e2e \