From 829cb5a8dc1a6240ae9a336730dd85e811f6b0fa Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Tue, 6 Oct 2026 21:07:18 +0000 Subject: [PATCH 01/16] Add experimental bounded streaming ingestion primitives --- Cargo.lock | 16 + Cargo.toml | 1 + crates/lance-context-ingestion/Cargo.toml | 20 + crates/lance-context-ingestion/README.md | 84 +++ .../lance-context-ingestion/src/checkpoint.rs | 170 +++++ .../lance-context-ingestion/src/consumer.rs | 129 ++++ crates/lance-context-ingestion/src/journal.rs | 440 ++++++++++++ crates/lance-context-ingestion/src/lib.rs | 31 + .../lance-context-ingestion/src/pipeline.rs | 461 ++++++++++++ .../lance-context-ingestion/tests/pipeline.rs | 673 ++++++++++++++++++ .../tests/storage_faults.rs | 153 ++++ 11 files changed, 2178 insertions(+) create mode 100644 crates/lance-context-ingestion/Cargo.toml create mode 100644 crates/lance-context-ingestion/README.md create mode 100644 crates/lance-context-ingestion/src/checkpoint.rs create mode 100644 crates/lance-context-ingestion/src/consumer.rs create mode 100644 crates/lance-context-ingestion/src/journal.rs create mode 100644 crates/lance-context-ingestion/src/lib.rs create mode 100644 crates/lance-context-ingestion/src/pipeline.rs create mode 100644 crates/lance-context-ingestion/tests/pipeline.rs create mode 100644 crates/lance-context-ingestion/tests/storage_faults.rs diff --git a/Cargo.lock b/Cargo.lock index e44a4c5b..1d4445b9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3907,6 +3907,22 @@ dependencies = [ "uuid", ] +[[package]] +name = "lance-context-ingestion" +version = "0.1.0" +dependencies = [ + "async-trait", + "futures", + "object_store", + "serde", + "serde_json", + "sha2 0.10.9", + "tempfile", + "thiserror 2.0.18", + "tokio", + "uuid", +] + [[package]] name = "lance-context-master" version = "0.6.3" diff --git a/Cargo.toml b/Cargo.toml index 2cb7893b..a2476d21 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,6 @@ [workspace] members = [ + "crates/lance-context-ingestion", "crates/lance-context-merge", "crates/lance-context-core", "crates/lance-context-metrics", diff --git a/crates/lance-context-ingestion/Cargo.toml b/crates/lance-context-ingestion/Cargo.toml new file mode 100644 index 00000000..989d3dbe --- /dev/null +++ b/crates/lance-context-ingestion/Cargo.toml @@ -0,0 +1,20 @@ +[package] +name = "lance-context-ingestion" +version = "0.1.0" +edition.workspace = true +license.workspace = true +description = "Bounded session-ordered ingestion with recoverable WAL and independent consumers" + +[dependencies] +async-trait = "0.1" +futures = "0.3" +object_store = "0.13.2" +serde = { version = "1", features = ["derive"] } +serde_json = "1" +sha2 = "0.10" +thiserror = "2" +tokio = { version = "1", features = ["macros", "rt-multi-thread", "sync", "time"] } +uuid = { version = "1", features = ["v4"] } + +[dev-dependencies] +tempfile = "3" diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md new file mode 100644 index 00000000..d477b0e3 --- /dev/null +++ b/crates/lance-context-ingestion/README.md @@ -0,0 +1,84 @@ +# lance-context-ingestion + +Experimental streaming ingestion primitives. The implementation currently provides +the pipeline and journal protocol; integration with the existing trace alignment +adapter and a Lance table sink is **not complete**. This crate is not deployed. + +```text +replayable source / stable partition-local receipts + -> bounded concurrent history prefetch + -> ordered alignment within each virtual partition + -> bounded WAL queue / byte, count, timer flush + -> immutable payload + skip links + conditional head publication -> ACK + | | + v v + checkpoint consumer table consumer + coalesced session deltas independently grouped WAL ranges + | | + conditional session states staged fragments / fenced manifest commit + | + asynchronous indexer +``` + +## Implemented contracts + +- `Partition::start_with_loader` overlaps history loading, alignment and WAL + publication. Different virtual partitions run independently. Within a partition, + load completion may reorder, while alignment and durable publication stay ordered. + Do not change session-to-virtual-partition routing when changing worker count. +- `enqueue` reserves bytes before admission. A reservation remains held through the + durable ACK, covering input, history, bounded output and serialization. Adapter + caches, transient adapter allocation, executor and storage-client memory need their + own budgets. Sources must also bound concurrent requests waiting for admission. +- `Aligner` owns the session cache and speculative state. Prefetched checkpoints may + be stale: reconcile their revisions with pending deltas. On failure, discard the + adapter and recover; speculative changes must never become checkpoints directly. +- `Journal` stores immutable segments binding records, state deltas, exact input + digests, receipt identities, run/schema and predecessor sequence. Conditional head + updates fence stale writers. Failed or cancelled commits poison a writer. +- ACK follows head publication. An uploaded orphan is not committed. Retry the + same partition sequence, receipt, session and input bytes after an uncertain result. + A gap is rejected; older receipts are compared with the committed journal. +- Immutable skip links support bounded chronological recovery pages and historical + receipt lookup without reading unrelated record payloads. Recovery replays all + pages after the adapter's checkpoint, not just one next WAL segment. +- Named `Consumer`s have independent durable cursors and coalesce producer segments + into their own batches. Sink output and input coverage must be committed together; + cursor writes can fail after output succeeds, so repeated/regrouped input must be + idempotent. The scheduler owns exclusive consumer assignment and sink-side fencing. +- `SessionCheckpoints` provides an actual object-store checkpoint sink: group by + session, reduce ordered deltas, then write each session once with a conditional put. + A partially successful checkpoint batch can leave some session states ahead of the + global cursor. Restore each state with its own sequence and skip already applied + deltas when replaying the remaining global prefix. + +Use a backend supporting atomic conditional updates, such as a suitably configured +cloud object store. `object_store::local::LocalFileSystem` does not implement the +required update operation and is rejected; there is no unsafe local-lock fallback. +Tests use the real `object_store::memory::InMemory` implementation for CAS behavior, +with fault injection around writes. These tests do not establish cloud durability, +process-crash behavior on a real durable service, or production throughput. + +## Remaining integration + +1. Adapt the existing revisioned session history/cache and cross-call alignment + implementation. Preserve its compaction/branch identity rules and source ordering; + the generic pipeline deliberately does not invent a new turn-ID algorithm. +2. Connect the table consumer to public Lance staging/commit APIs. Persist exact WAL + input coverage with table publication; staged files alone do not justify a cursor + advance. Preserve Lance 2.2, Zstd and session ZoneMap configuration in that adapter. +3. Add source fan-out receipts and contiguous source progress. A source call spanning + partitions is complete only after every required partition ACK. +4. Add worker ownership orchestration, stage timing/queue telemetry, consumer run loops, + bounded durable backlog and safe WAL reclamation. No WAL files are deleted here. +5. Verify real compacted sessions, process crash/restart with durable storage, Lance + uncertain commits and source retry integration before a guarded production handoff. + Existing source-reader local audit history must survive that handoff. + +Run focused checks from the workspace root: + +```sh +CARGO_TARGET_DIR=/tmp/trace-streaming-target cargo test -p lance-context-ingestion --offline +CARGO_TARGET_DIR=/tmp/trace-streaming-target cargo clippy -p lance-context-ingestion --all-targets --offline -- -D warnings +cargo fmt -p lance-context-ingestion --check +``` diff --git a/crates/lance-context-ingestion/src/checkpoint.rs b/crates/lance-context-ingestion/src/checkpoint.rs new file mode 100644 index 00000000..cbfe79eb --- /dev/null +++ b/crates/lance-context-ingestion/src/checkpoint.rs @@ -0,0 +1,170 @@ +use std::collections::BTreeMap; +use std::sync::Arc; + +use async_trait::async_trait; +use futures::{stream, FutureExt, StreamExt, TryStreamExt}; +use object_store::{PutMode, UpdateVersion}; +use serde::{Deserialize, Serialize}; + +use crate::journal::digest; +use crate::{Binding, Entry, Error, Journal, Result, Sink}; + +/// A per-session checkpoint may be ahead of the consumer's coherent prefix if a +/// previous multi-session batch failed halfway. Recovery must skip replay deltas +/// through this session's sequence, not apply them again. It must still replay +/// all other sessions through the committed WAL head before accepting requests. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct SessionState { + pub binding: Binding, + pub session: String, + pub through_sequence: u64, + pub value: Vec, +} + +/// Apply the already chosen alignment delta. This operation must be deterministic +/// and must not allocate new IDs or rerun history matching. Empty state denotes +/// a session with no checkpoint. Bound transient reducer memory in the adapter. +#[async_trait] +pub trait Reducer: Send + Sync { + async fn apply(&self, session: &str, state: &[u8], delta: &[u8]) -> Result>; +} + +#[derive(Clone)] +pub struct SessionCheckpoints { + journal: Journal, + max_state_bytes: usize, +} + +impl SessionCheckpoints { + pub fn new(journal: Journal, max_state_bytes: usize) -> Result { + if max_state_bytes == 0 { + return Err(Error::Invalid("zero session checkpoint budget".into())); + } + Ok(Self { + journal, + max_state_bytes, + }) + } + + async fn read(&self, session: &str) -> Result> { + let path = self + .journal + .path(&format!("sessions/{}.json", digest(session.as_bytes()))); + match self.journal.read::(&path).await { + Ok((state, version)) => { + if state.binding != *self.journal.binding() + || state.session != session + || state.value.len() > self.max_state_bytes + || state.through_sequence == 0 + { + return Err(Error::Invalid("invalid session checkpoint".into())); + } + Ok(Some((state, version))) + } + Err(Error::Storage(object_store::Error::NotFound { .. })) => Ok(None), + Err(error) => Err(error), + } + } + + pub async fn load(&self, session: &str) -> Result> { + Ok(self.read(session).await?.map(|(state, _)| state)) + } + + /// Each session is written once per consumer batch, even if it occurred in + /// many producer WAL segments. Conditional writes never replace newer state. + pub fn sink(&self, reducer: R, concurrency: usize) -> Result> { + if concurrency == 0 { + return Err(Error::Invalid("zero checkpoint concurrency".into())); + } + Ok(CheckpointSink { + store: self.clone(), + reducer: Arc::new(reducer), + concurrency, + }) + } + + async fn apply( + &self, + reducer: &R, + session: &str, + entries: Vec<&Entry>, + ) -> Result<()> { + let (mut state, mode) = match self.read(session).await? { + Some((state, version)) => (state, PutMode::Update(version)), + None => ( + SessionState { + binding: self.journal.binding().clone(), + session: session.into(), + through_sequence: 0, + value: Vec::new(), + }, + PutMode::Create, + ), + }; + let before = state.through_sequence; + for entry in entries { + if entry.sequence <= state.through_sequence { + continue; + } + state.value = reducer + .apply(session, &state.value, &entry.transition.delta) + .await?; + if state.value.len() > self.max_state_bytes { + return Err(Error::Invalid( + "session checkpoint exceeds state budget".into(), + )); + } + state.through_sequence = entry.sequence; + } + if state.through_sequence != before { + self.journal + .put( + &self + .journal + .path(&format!("sessions/{}.json", digest(session.as_bytes()))), + &state, + mode, + ) + .await?; + } + Ok(()) + } +} + +pub struct CheckpointSink { + store: SessionCheckpoints, + reducer: Arc, + concurrency: usize, +} + +#[async_trait] +impl Sink for CheckpointSink { + async fn apply(&mut self, binding: &Binding, entries: &[Entry]) -> Result<()> { + if binding != self.store.journal.binding() + || entries + .windows(2) + .any(|pair| pair[0].sequence >= pair[1].sequence) + { + return Err(Error::Invalid( + "checkpoint binding or sequence order mismatch".into(), + )); + } + let mut sessions = BTreeMap::<&str, Vec<&Entry>>::new(); + for entry in entries { + sessions.entry(&entry.session).or_default().push(entry); + } + let mut work = Vec::with_capacity(sessions.len()); + for (session, entries) in sessions { + work.push( + self.store + .apply(self.reducer.as_ref(), session, entries) + .boxed(), + ); + } + stream::iter(work) + .buffer_unordered(self.concurrency) + .try_collect::>() + .await?; + Ok(()) + } +} diff --git a/crates/lance-context-ingestion/src/consumer.rs b/crates/lance-context-ingestion/src/consumer.rs new file mode 100644 index 00000000..a47c918b --- /dev/null +++ b/crates/lance-context-ingestion/src/consumer.rs @@ -0,0 +1,129 @@ +use async_trait::async_trait; +use object_store::{PutMode, UpdateVersion}; +use serde::{Deserialize, Serialize}; + +use crate::{Binding, Entry, Error, Journal, Position, Result}; + +/// Apply ordered WAL entries idempotently using (binding, entry.sequence). +/// On restart or uncertain progress writes, a batch can repeat or be regrouped. +/// Return success only after outputs AND their coverage are durable together. +/// For Lance, staged fragments alone are insufficient: manifest publication must +/// atomically record the covered input range. This callback never owns WAL ACKs. +#[async_trait] +pub trait Sink: Send { + async fn apply(&mut self, binding: &Binding, entries: &[Entry]) -> Result<()>; +} + +#[derive(Serialize, Deserialize)] +struct Cursor { + binding: Binding, + position: Position, +} + +/// Separate names (for example `checkpoint` and `table`) consume the same WAL +/// independently. Scheduler must assign a single active worker per consumer and +/// partition; cursor CAS detects ownership races but does not fence sink writes. +pub struct Consumer { + journal: Journal, + name: String, + position: Position, + version: UpdateVersion, + poisoned: bool, +} + +impl Consumer { + pub async fn open(journal: Journal, name: &str) -> Result { + if name.is_empty() + || !name + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || byte == b'-') + { + return Err(Error::Invalid("invalid consumer name".into())); + } + let path = journal.path(&format!("consumers/{name}.json")); + let (cursor, version): (Cursor, _) = match journal.read(&path).await { + Ok(pair) => pair, + Err(Error::Storage(object_store::Error::NotFound { .. })) => { + let cursor = Cursor { + binding: journal.binding().clone(), + position: Position::default(), + }; + let version = journal.put(&path, &cursor, PutMode::Create).await?; + (cursor, version) + } + Err(error) => return Err(error), + }; + if &cursor.binding != journal.binding() { + return Err(Error::Invalid( + "consumer run/schema/partition mismatch".into(), + )); + } + Ok(Self { + journal, + name: name.into(), + position: cursor.position, + version, + poisoned: false, + }) + } + + pub fn position(&self) -> &Position { + &self.position + } + + /// Coalesce up to `max_segments` / `max_bytes` independently of producer batch + /// sizes. No empty progress writes. A segment larger than the budget fails; + /// it is never silently admitted. Cancellation requires reopening the cursor. + pub async fn consume( + &mut self, + sink: &mut S, + max_segments: usize, + max_bytes: usize, + ) -> Result { + if self.poisoned { + return Err(Error::Fenced); + } + if max_segments == 0 || max_bytes == 0 { + return Err(Error::Invalid("zero consumer limit".into())); + } + let through = self.journal.position().await?; + let positions = self.journal.pending(&self.position, &through).await?; + let mut entries = Vec::new(); + let mut bytes = 0; + let mut last = self.position.clone(); + for position in positions.into_iter().take(max_segments) { + let batch = self.journal.entries(&position).await?; + let size = serde_json::to_vec(&batch)?.len(); + if size > max_bytes { + return Err(Error::Invalid("WAL segment exceeds consumer budget".into())); + } + if size > max_bytes - bytes { + break; + } + bytes += size; + entries.extend(batch); + last = position; + } + if entries.is_empty() { + return Ok(0); + } + self.poisoned = true; + sink.apply(self.journal.binding(), &entries).await?; + let next = Cursor { + binding: self.journal.binding().clone(), + position: last.clone(), + }; + let version = self + .journal + .put( + &self.journal.path(&format!("consumers/{}.json", self.name)), + &next, + PutMode::Update(self.version.clone()), + ) + .await?; + self.position = last; + self.version = version; + self.poisoned = false; + Ok(entries.len()) + } +} diff --git a/crates/lance-context-ingestion/src/journal.rs b/crates/lance-context-ingestion/src/journal.rs new file mode 100644 index 00000000..a2589781 --- /dev/null +++ b/crates/lance-context-ingestion/src/journal.rs @@ -0,0 +1,440 @@ +use std::sync::Arc; + +use object_store::{path::Path, ObjectStore, ObjectStoreExt, PutMode, PutOptions, UpdateVersion}; +use serde::{de::DeserializeOwned, Deserialize, Serialize}; +use sha2::{Digest, Sha256}; +use uuid::Uuid; + +use crate::{Error, Result}; + +/// A namespace is permanently bound to one run, schema and virtual partition. +/// Worker count may change; these virtual partition identities must not. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct Binding { + pub run: String, + pub schema: String, + pub partition: u32, +} + +/// An opaque immutable segment reference. Sequence zero denotes the empty WAL. +#[derive(Clone, Debug, Default, PartialEq, Eq, Serialize, Deserialize)] +pub struct Position { + pub sequence: u64, + pub generation: u64, + pub segment: Option, +} + +/// Adapter-defined state delta and output records are committed together. +/// Deltas must replay without rerunning alignment or choosing new turn IDs. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct Transition { + pub delta: Vec, + pub records: Vec, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct Entry { + pub sequence: u64, + pub session: String, + pub receipt: String, + pub input_digest: String, + pub transition: Transition, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +struct Head { + binding: Binding, + epoch: u64, + position: Position, +} + +#[derive(Clone, Debug, Serialize, Deserialize)] +struct Segment { + binding: Binding, + predecessor: Position, + entries: Vec, +} + +/// Small immutable skip links let lagging consumers seek a bounded page without +/// loading intervening record payloads or accumulating the whole WAL inventory. +#[derive(Clone, Debug, Serialize, Deserialize)] +struct Link { + binding: Binding, + position: Position, + ancestors: Vec, + payload_digest: String, +} + +#[derive(Clone)] +pub struct Journal { + store: Arc, + prefix: Path, + binding: Binding, + max_segment_bytes: usize, + max_replay_segments: usize, +} + +pub struct Writer { + journal: Journal, + head: Head, + version: UpdateVersion, + poisoned: bool, +} + +pub(crate) fn digest(bytes: &[u8]) -> String { + format!("{:x}", Sha256::digest(bytes)) +} + +impl Journal { + /// Storage must support atomic conditional puts. No process-local fallback is + /// used. The replay limit bounds each page, not total recoverable backlog. + /// This module deliberately performs no WAL garbage collection. + pub fn new( + store: Arc, + prefix: Path, + binding: Binding, + max_segment_bytes: usize, + max_replay_segments: usize, + ) -> Result { + if binding.run.is_empty() + || binding.schema.is_empty() + || max_segment_bytes == 0 + || max_replay_segments == 0 + { + return Err(Error::Invalid("empty binding or zero journal limit".into())); + } + Ok(Self { + store, + prefix, + binding, + max_segment_bytes, + max_replay_segments, + }) + } + + pub fn binding(&self) -> &Binding { + &self.binding + } + + pub(crate) fn path(&self, child: &str) -> Path { + child + .split('/') + .fold(self.prefix.clone(), |path, part| path.join(part)) + } + + pub(crate) async fn read( + &self, + path: &Path, + ) -> Result<(T, UpdateVersion)> { + let result = self.store.get(path).await?; + let version = UpdateVersion { + e_tag: result.meta.e_tag.clone(), + version: result.meta.version.clone(), + }; + if result.meta.size > self.max_segment_bytes as u64 { + return Err(Error::Invalid("journal object exceeds read budget".into())); + } + let bytes = result.bytes().await?; + Ok((serde_json::from_slice(&bytes)?, version)) + } + + pub(crate) async fn put( + &self, + path: &Path, + value: &T, + mode: PutMode, + ) -> Result { + let bytes = serde_json::to_vec(value)?; + if bytes.len() > self.max_segment_bytes { + return Err(Error::Invalid("journal object exceeds write budget".into())); + } + Ok(self + .store + .put_opts(path, bytes.into(), PutOptions::from(mode)) + .await? + .into()) + } + + async fn head(&self) -> Result<(Head, UpdateVersion)> { + let (head, version): (Head, _) = self.read(&self.path("head.json")).await?; + if head.binding != self.binding { + return Err(Error::Invalid("WAL run/schema/partition mismatch".into())); + } + validate_position(&head.position)?; + Ok((head, version)) + } + + pub async fn position(&self) -> Result { + Ok(self.head().await?.0.position) + } + + /// Explicitly fence any previous publisher using a conditional head update. + /// The caller must own scheduling for this partition. Old tasks can still + /// upload orphan files, but cannot publish or ACK them after this CAS. + /// Reopen after uncertain I/O; never resume speculative alignment state. + pub async fn acquire(&self) -> Result { + let path = self.path("head.json"); + let (mut head, mode) = match self.head().await { + Ok((head, version)) => (head, PutMode::Update(version)), + Err(Error::Storage(object_store::Error::NotFound { .. })) => ( + Head { + binding: self.binding.clone(), + epoch: 0, + position: Position::default(), + }, + PutMode::Create, + ), + Err(error) => return Err(error), + }; + head.epoch = head + .epoch + .checked_add(1) + .ok_or_else(|| Error::Invalid("epoch overflow".into()))?; + let version = self.put(&path, &head, mode).await?; + Ok(Writer { + journal: self.clone(), + head, + version, + poisoned: false, + }) + } + + async fn segment(&self, position: &Position) -> Result { + validate_position(position)?; + let name = position + .segment + .as_ref() + .ok_or_else(|| Error::Invalid("empty segment".into()))?; + let (segment, _): (Segment, _) = self + .read(&self.path(&format!("segments/{name}.json"))) + .await?; + let link = self.link(position).await?; + if digest(&serde_json::to_vec(&segment)?) != link.payload_digest + || segment.predecessor != link.ancestors[0] + { + return Err(Error::Invalid("WAL payload integrity mismatch".into())); + } + if segment.binding != self.binding || segment.entries.is_empty() { + return Err(Error::Invalid( + "invalid WAL segment binding or empty entries".into(), + )); + } + validate_position(&segment.predecessor)?; + let mut expected = segment.predecessor.sequence; + for entry in &segment.entries { + expected = expected + .checked_add(1) + .ok_or_else(|| Error::Invalid("sequence overflow".into()))?; + if entry.sequence != expected { + return Err(Error::Invalid("noncontiguous WAL sequence".into())); + } + } + if expected != position.sequence { + return Err(Error::Invalid("WAL segment/head mismatch".into())); + } + Ok(segment) + } + + async fn link(&self, position: &Position) -> Result { + validate_position(position)?; + let name = position + .segment + .as_ref() + .ok_or_else(|| Error::Invalid("empty link".into()))?; + let (link, _): (Link, _) = self.read(&self.path(&format!("links/{name}.json"))).await?; + let expected = u64::BITS - position.generation.leading_zeros(); + if link.binding != self.binding + || link.position != *position + || link.ancestors.len() != expected as usize + { + return Err(Error::Invalid("invalid WAL link".into())); + } + for (level, ancestor) in link.ancestors.iter().enumerate() { + validate_position(ancestor)?; + if ancestor.generation != position.generation - (1_u64 << level) + || ancestor.sequence >= position.sequence + { + return Err(Error::Invalid("invalid WAL ancestor".into())); + } + } + Ok(link) + } + + /// Returns the next chronological page after `after`, bounded by + /// max_replay_segments. Only the supplied committed chain is traversed; + /// object listings and orphan uploads never establish durability. + pub async fn pending(&self, after: &Position, through: &Position) -> Result> { + validate_position(after)?; + validate_position(through)?; + if after.generation > through.generation { + return Err(Error::Invalid("cursor ahead of durable WAL".into())); + } + let mut cursor = through.clone(); + let page_end = after + .generation + .saturating_add(self.max_replay_segments as u64) + .min(through.generation); + while cursor.generation > page_end { + let link = self.link(&cursor).await?; + cursor = link + .ancestors + .into_iter() + .rev() + .find(|ancestor| ancestor.generation >= page_end) + .ok_or_else(|| Error::Invalid("missing WAL skip link".into()))?; + } + let mut segments = Vec::new(); + while cursor != *after { + if cursor.sequence <= after.sequence { + return Err(Error::Invalid( + "cursor is not on the committed WAL chain".into(), + )); + } + let link = self.link(&cursor).await?; + segments.push(cursor); + cursor = link.ancestors[0].clone(); + } + segments.reverse(); + Ok(segments) + } + + /// Read a segment previously obtained from `pending`; not an authorization + /// to consume arbitrary uncommitted object paths. + pub async fn entries(&self, position: &Position) -> Result> { + Ok(self.segment(position).await?.entries) + } + + pub(crate) async fn receipt(&self, sequence: u64, through: &Position) -> Result { + let mut cursor = through.clone(); + while sequence > 0 && cursor.sequence >= sequence { + let link = self.link(&cursor).await?; + if sequence > link.ancestors[0].sequence { + return self + .segment(&cursor) + .await? + .entries + .into_iter() + .find(|entry| entry.sequence == sequence) + .ok_or_else(|| Error::Invalid("receipt missing from WAL".into())); + } + cursor = link + .ancestors + .into_iter() + .rev() + .find(|ancestor| ancestor.sequence >= sequence) + .ok_or_else(|| Error::Invalid("missing receipt skip link".into()))?; + } + Err(Error::Invalid("receipt is outside committed WAL".into())) + } +} + +fn validate_position(position: &Position) -> Result<()> { + if (position.sequence == 0) != position.segment.is_none() + || (position.generation == 0) != position.segment.is_none() + || position.sequence < position.generation + || position + .segment + .as_ref() + .is_some_and(|id| Uuid::parse_str(id).is_err()) + { + return Err(Error::Invalid("invalid WAL position".into())); + } + Ok(()) +} + +impl Writer { + pub fn position(&self) -> &Position { + &self.head.position + } + + pub fn journal(&self) -> &Journal { + &self.journal + } + + /// Any storage failure poisons this writer, including ambiguous success. + /// Recovery follows the head and verifies the original receipt before ACK. + pub async fn append(&mut self, entries: Vec) -> Result { + if self.poisoned { + return Err(Error::Fenced); + } + if entries.is_empty() { + return Err(Error::Invalid("cannot append empty WAL batch".into())); + } + let mut sequence = self.head.position.sequence; + for entry in &entries { + sequence = sequence + .checked_add(1) + .ok_or_else(|| Error::Invalid("sequence overflow".into()))?; + if entry.sequence != sequence { + return Err(Error::Invalid("append sequence gap or duplicate".into())); + } + } + let position = Position { + sequence, + generation: self + .head + .position + .generation + .checked_add(1) + .ok_or_else(|| Error::Invalid("generation overflow".into()))?, + segment: Some(Uuid::new_v4().to_string()), + }; + let segment = Segment { + binding: self.journal.binding.clone(), + predecessor: self.head.position.clone(), + entries, + }; + // Set before the first await: cancelling this future also fences reuse. + self.poisoned = true; + let mut ancestors = vec![self.head.position.clone()]; + let mut level = 1; + while let Some(ancestor) = ancestors.last().filter(|ancestor| ancestor.generation > 0) { + let link = self.journal.link(ancestor).await?; + match link.ancestors.get(level - 1) { + Some(next) => ancestors.push(next.clone()), + None => break, + } + level += 1; + } + let link = Link { + binding: self.journal.binding.clone(), + position: position.clone(), + ancestors, + payload_digest: digest(&serde_json::to_vec(&segment)?), + }; + self.journal + .put( + &self.journal.path(&format!( + "segments/{}.json", + position.segment.as_ref().unwrap() + )), + &segment, + PutMode::Create, + ) + .await?; + self.journal + .put( + &self.journal.path(&format!( + "links/{}.json", + position.segment.as_ref().unwrap() + )), + &link, + PutMode::Create, + ) + .await?; + let mut next = self.head.clone(); + next.position = position.clone(); + let version = self + .journal + .put( + &self.journal.path("head.json"), + &next, + PutMode::Update(self.version.clone()), + ) + .await?; + self.head = next; + self.version = version; + self.poisoned = false; + Ok(position) + } +} diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs new file mode 100644 index 00000000..6c7e3927 --- /dev/null +++ b/crates/lance-context-ingestion/src/lib.rs @@ -0,0 +1,31 @@ +//! Session-ordered alignment and durable WAL publication run in separate tasks. +//! Checkpoint and table consumers have independent, durable progress cursors. +//! A successful submission means the WAL is recoverable, not that it is indexed. + +mod checkpoint; +mod consumer; +mod journal; +mod pipeline; + +pub use checkpoint::{CheckpointSink, Reducer, SessionCheckpoints, SessionState}; +pub use consumer::{Consumer, Sink}; +pub use journal::{Binding, Entry, Journal, Position, Transition, Writer}; +pub use pipeline::{Ack, Aligner, BatchPolicy, HistoryLoader, Partition, PipelineConfig, Request}; + +pub type Result = std::result::Result; + +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("object storage: {0}")] + Storage(#[from] object_store::Error), + #[error("WAL encoding: {0}")] + Encoding(#[from] serde_json::Error), + #[error("invalid ingestion state: {0}")] + Invalid(String), + #[error("writer fenced or commit outcome uncertain; reopen and recover before retrying")] + Fenced, + #[error("pipeline stopped; retry the same receipt after recovery")] + Stopped, + #[error("stage failed: {0}")] + Stage(String), +} diff --git a/crates/lance-context-ingestion/src/pipeline.rs b/crates/lance-context-ingestion/src/pipeline.rs new file mode 100644 index 00000000..198eaa42 --- /dev/null +++ b/crates/lance-context-ingestion/src/pipeline.rs @@ -0,0 +1,461 @@ +use std::sync::Arc; +use std::time::Duration; + +use async_trait::async_trait; +use futures::{stream, StreamExt}; +use tokio::sync::{mpsc, oneshot, watch, OwnedSemaphorePermit, Semaphore}; +use tokio::task::JoinHandle; +use tokio::time::{timeout_at, Instant}; + +use crate::journal::digest; +use crate::{Binding, Entry, Error, Journal, Position, Result, Transition, Writer}; + +#[derive(Clone, Debug)] +pub struct Request { + /// Contiguous, stable sequence within this virtual partition, starting at 1. + /// A source fan-out must wait for every partition ACK before advancing its + /// source checkpoint. Retries preserve sequence, receipt, session and bytes. + pub sequence: u64, + pub session: String, + pub receipt: String, + pub payload: Vec, +} + +/// Adapter owns session state and its cache budget. Alignment can run ahead of +/// WAL durability, so an error discards this adapter and its speculative suffix. +/// Checkpoints restore exact committed deltas, never freshly recomputed IDs. +#[async_trait] +pub trait Aligner: Send + 'static { + async fn restore(&mut self, binding: &Binding) -> Result; + async fn replay(&mut self, entry: &Entry) -> Result<()>; + /// Prefetched history is an immutable checkpoint, potentially behind this + /// adapter's speculative state. Reconcile its revision before using it; + /// never replace newer same-session state with an older prefetch result. + async fn align(&mut self, request: &Request, history: &[u8]) -> Result; +} + +/// Concurrent, read-only history loading. The adapter must respect `max_bytes` +/// while fetching/decoding, not only after allocation. Loaded state carries its +/// revision in the adapter's encoding so alignment can reconcile stale loads. +#[async_trait] +pub trait HistoryLoader: Send + Sync + 'static { + async fn load(&self, request: &Request, max_bytes: usize) -> Result>; +} + +struct NoHistory; + +#[async_trait] +impl HistoryLoader for NoHistory { + async fn load(&self, _request: &Request, _max_bytes: usize) -> Result> { + Ok(Vec::new()) + } +} + +#[derive(Clone, Debug)] +pub struct BatchPolicy { + pub max_entries: usize, + /// Serialized Entry bytes, excluding the bounded segment envelope. + pub max_bytes: usize, + pub max_delay: Duration, +} + +impl BatchPolicy { + pub(crate) fn validate(&self) -> Result<()> { + if self.max_entries == 0 || self.max_bytes == 0 || self.max_delay.is_zero() { + return Err(Error::Invalid( + "batch count, bytes and delay must be positive".into(), + )); + } + Ok(()) + } +} + +#[derive(Clone, Debug)] +pub struct PipelineConfig { + pub queue_entries: usize, + pub load_concurrency: usize, + /// Bounds reserved input/output/serialization bytes across both queues and + /// active batches. Adapter state, runtime and object-store buffers are extra. + pub memory_bytes: u32, + pub max_input_bytes: usize, + pub max_transition_bytes: usize, + pub max_history_bytes: usize, + pub wal: BatchPolicy, +} + +struct Input { + request: Request, + ack: oneshot::Sender>, + permit: OwnedSemaphorePermit, +} + +struct Prepared { + input: Input, + history: Vec, +} + +struct Aligned { + entry: Entry, + encoded_bytes: usize, + ack: oneshot::Sender>, + _permit: OwnedSemaphorePermit, +} + +/// Dropping an ACK waiter does not cancel already admitted durable work. +pub struct Ack(oneshot::Receiver>); + +impl Ack { + pub async fn wait(self) -> Result { + self.0.await.map_err(|_| Error::Stopped)? + } +} + +/// One stable virtual partition, with independent alignment and WAL tasks. +/// Run different partitions on independent workers. A session must always route +/// to the same partition; changing worker count must not change that mapping. +pub struct Partition { + input: Option>, + loading: Option>>, + alignment: Option>>, + wal: Option>>, + budget: Arc, + config: PipelineConfig, + durable: watch::Receiver, +} + +impl Partition { + pub async fn start( + writer: Writer, + aligner: A, + config: PipelineConfig, + ) -> Result { + Self::start_with_loader(writer, aligner, NoHistory, config).await + } + + pub async fn start_with_loader( + writer: Writer, + mut aligner: A, + loader: L, + config: PipelineConfig, + ) -> Result { + config.wal.validate()?; + let reserve = reservation( + config.max_input_bytes, + config.max_transition_bytes, + config.max_history_bytes, + )?; + if config.queue_entries == 0 + || config.load_concurrency == 0 + || reserve > config.memory_bytes + { + return Err(Error::Invalid( + "queue empty or maximum request exceeds memory budget".into(), + )); + } + let journal = writer.journal().clone(); + let mut checkpoint = aligner.restore(journal.binding()).await?; + while &checkpoint != writer.position() { + for position in journal.pending(&checkpoint, writer.position()).await? { + for entry in journal.entries(&position).await? { + aligner.replay(&entry).await?; + } + checkpoint = position; + } + } + let (input_tx, input_rx) = mpsc::channel(config.queue_entries); + let (loaded_tx, loaded_rx) = mpsc::channel(config.queue_entries); + let (wal_tx, wal_rx) = mpsc::channel(config.queue_entries); + let (durable_tx, durable_rx) = watch::channel(writer.position().clone()); + let loading = tokio::spawn(load_loop(loader, input_rx, loaded_tx, config.clone())); + let alignment = tokio::spawn(align_loop( + aligner, + journal, + loaded_rx, + wal_tx, + durable_rx.clone(), + config.clone(), + )); + let wal = tokio::spawn(wal_loop(writer, wal_rx, durable_tx, config.wal.clone())); + Ok(Self { + input: Some(input_tx), + loading: Some(loading), + alignment: Some(alignment), + wal: Some(wal), + budget: Arc::new(Semaphore::new(config.memory_bytes as usize)), + config, + durable: durable_rx, + }) + } + + /// Admission is bounded; the returned ACK resolves only after a conditional + /// durable head publication. Caller must serialize admission order within a + /// partition. Cancellation before admission leaves the receipt uncommitted. + pub async fn enqueue(&self, request: Request) -> Result { + let input_bytes = request + .payload + .capacity() + .checked_add(request.session.capacity()) + .and_then(|n| n.checked_add(request.receipt.capacity())) + .ok_or_else(|| Error::Invalid("input size overflow".into()))?; + if input_bytes > self.config.max_input_bytes + || request.session.is_empty() + || request.receipt.is_empty() + { + return Err(Error::Invalid( + "input exceeds limit or lacks session/receipt".into(), + )); + } + let permit = self + .budget + .clone() + .acquire_many_owned(reservation( + input_bytes, + self.config.max_transition_bytes, + self.config.max_history_bytes, + )?) + .await + .map_err(|_| Error::Stopped)?; + let (ack, receiver) = oneshot::channel(); + self.input + .as_ref() + .ok_or(Error::Stopped)? + .send(Input { + request, + ack, + permit, + }) + .await + .map_err(|_| Error::Stopped)?; + Ok(Ack(receiver)) + } + + pub fn durable_position(&self) -> Position { + self.durable.borrow().clone() + } + + /// Stop admission, drain alignment and flush even a partially filled WAL. + /// Dropping Partition instead aborts tasks; recovery reconciles any uncertain + /// head write. Neither path advances checkpoint or merge consumer cursors. + pub async fn shutdown(mut self) -> Result<()> { + self.input.take(); + let loading = self + .loading + .as_mut() + .unwrap() + .await + .map_err(|error| Error::Stage(error.to_string()))?; + self.loading.take(); + let alignment = self + .alignment + .as_mut() + .unwrap() + .await + .map_err(|error| Error::Stage(error.to_string()))?; + self.alignment.take(); + let wal = self + .wal + .as_mut() + .unwrap() + .await + .map_err(|error| Error::Stage(error.to_string()))?; + self.wal.take(); + loading?; + alignment?; + wal + } +} + +impl Drop for Partition { + fn drop(&mut self) { + if let Some(task) = &self.loading { + task.abort(); + } + if let Some(task) = &self.alignment { + task.abort(); + } + if let Some(task) = &self.wal { + task.abort(); + } + } +} + +fn reservation(input: usize, output: usize, history: usize) -> Result { + // Include retained buffer capacities, escaped strings (up to six bytes per + // byte), geometric serializer capacity growth and fixed envelope space. + input + .checked_add(output) + .and_then(|n| n.checked_add(history)) + .and_then(|n| n.checked_mul(16)) + .and_then(|n| n.checked_add(4096)) + .and_then(|n| u32::try_from(n).ok()) + .ok_or_else(|| Error::Invalid("request reservation overflow".into())) +} + +async fn load_loop( + loader: L, + input: mpsc::Receiver, + output: mpsc::Sender, + config: PipelineConfig, +) -> Result<()> { + let loader = Arc::new(loader); + let mut prepared = stream::unfold(input, |mut receiver| async move { + receiver.recv().await.map(|item| (item, receiver)) + }) + .map(|input| { + let loader = loader.clone(); + async move { + let history = loader + .load(&input.request, config.max_history_bytes) + .await?; + if history.capacity() > config.max_history_bytes { + return Err(Error::Invalid( + "history loader exceeded reserved bytes".into(), + )); + } + Ok(Prepared { input, history }) + } + }) + .buffered(config.load_concurrency) + .boxed(); + while let Some(item) = prepared.next().await { + output.send(item?).await.map_err(|_| Error::Stopped)?; + } + Ok(()) +} + +async fn align_loop( + mut aligner: A, + journal: Journal, + mut input: mpsc::Receiver, + wal: mpsc::Sender, + mut durable: watch::Receiver, + config: PipelineConfig, +) -> Result<()> { + let mut next = durable + .borrow() + .sequence + .checked_add(1) + .ok_or_else(|| Error::Invalid("sequence overflow".into()))?; + while let Some(input) = input.recv().await { + let Prepared { input, history } = input; + let Input { + request, + ack, + permit, + } = input; + let input_digest = digest(&request.payload); + if request.sequence < next && request.sequence > 0 { + // A repeated in-flight receipt waits for its original commit without + // applying alignment twice. Closed writer watch means uncertain I/O. + while durable.borrow().sequence < request.sequence { + durable.changed().await.map_err(|_| Error::Stopped)?; + } + let through = durable.borrow().clone(); + let original = journal.receipt(request.sequence, &through).await?; + let response = if original.session == request.session + && original.receipt == request.receipt + && original.input_digest == input_digest + { + Ok(through) + } else { + Err(Error::Invalid("retry receipt identity changed".into())) + }; + let _ = ack.send(response); + continue; + } + if request.sequence != next { + let _ = ack.send(Err(Error::Invalid(format!( + "expected partition sequence {next}" + )))); + continue; + } + let transition = aligner.align(&request, &history).await?; + if transition + .delta + .capacity() + .saturating_add(transition.records.capacity()) + > config.max_transition_bytes + { + return Err(Error::Invalid( + "aligner exceeded reserved output bytes".into(), + )); + } + let entry = Entry { + sequence: next, + session: request.session, + receipt: request.receipt, + input_digest, + transition, + }; + let encoded_bytes = serde_json::to_vec(&entry)?.len(); + if encoded_bytes > config.wal.max_bytes { + return Err(Error::Invalid( + "single transition exceeds WAL batch limit".into(), + )); + } + wal.send(Aligned { + entry, + encoded_bytes, + ack, + _permit: permit, + }) + .await + .map_err(|_| Error::Stopped)?; + next = next + .checked_add(1) + .ok_or_else(|| Error::Invalid("sequence overflow".into()))?; + } + Ok(()) +} + +async fn wal_loop( + mut writer: Writer, + mut input: mpsc::Receiver, + durable: watch::Sender, + policy: BatchPolicy, +) -> Result<()> { + let mut carry: Option = None; + loop { + let first = match carry.take() { + Some(first) => first, + None => match input.recv().await { + Some(first) => first, + None => return Ok(()), + }, + }; + let deadline = Instant::now() + policy.max_delay; + let mut bytes = first.encoded_bytes; + let mut batch = vec![first]; + while batch.len() < policy.max_entries && bytes < policy.max_bytes { + match timeout_at(deadline, input.recv()).await { + Ok(Some(next)) if next.encoded_bytes <= policy.max_bytes - bytes => { + bytes += next.encoded_bytes; + batch.push(next); + } + Ok(Some(next)) => { + carry = Some(next); + break; + } + Ok(None) | Err(_) => break, + } + } + let entries = batch + .iter_mut() + .map(|item| Entry { + sequence: item.entry.sequence, + session: std::mem::take(&mut item.entry.session), + receipt: std::mem::take(&mut item.entry.receipt), + input_digest: std::mem::take(&mut item.entry.input_digest), + transition: Transition { + delta: std::mem::take(&mut item.entry.transition.delta), + records: std::mem::take(&mut item.entry.transition.records), + }, + }) + .collect(); + let position = writer.append(entries).await?; + durable.send_replace(position.clone()); + for item in batch { + let _ = item.ack.send(Ok(position.clone())); + } + } +} diff --git a/crates/lance-context-ingestion/tests/pipeline.rs b/crates/lance-context-ingestion/tests/pipeline.rs new file mode 100644 index 00000000..26918d5b --- /dev/null +++ b/crates/lance-context-ingestion/tests/pipeline.rs @@ -0,0 +1,673 @@ +use std::collections::BTreeMap; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::{Arc, Mutex}; +use std::time::Duration; + +use async_trait::async_trait; +use lance_context_ingestion::{ + Aligner, BatchPolicy, Binding, Consumer, Entry, Error, HistoryLoader, Journal, Partition, + PipelineConfig, Position, Reducer, Request, Result, SessionCheckpoints, Sink, Transition, +}; +use object_store::{memory::InMemory, path::Path, ObjectStore, ObjectStoreExt}; +use tokio::sync::Semaphore; +use tokio::time::timeout; + +#[tokio::test] +async fn lagging_recovery_and_consumers_page_without_loading_entire_wal() { + let store = Arc::new(InMemory::new()); + let journal = Journal::new(store, Path::from("paged"), binding(0), 1 << 20, 3).unwrap(); + let mut writer = journal.acquire().await.unwrap(); + for sequence in 1..=35 { + writer + .append(vec![Entry { + sequence, + session: "s".into(), + receipt: format!("r-{sequence}"), + input_digest: "digest".into(), + transition: Transition { + delta: serde_json::to_vec(&sequence).unwrap(), + records: vec![], + }, + }]) + .await + .unwrap(); + } + let head = writer.position().clone(); + let mut position = Position::default(); + let mut seen = Vec::new(); + while position != head { + let page = journal.pending(&position, &head).await.unwrap(); + assert!(!page.is_empty() && page.len() <= 3); + seen.extend(page.iter().map(|p| p.sequence)); + position = page.last().unwrap().clone(); + } + assert_eq!(seen, (1..=35).collect::>()); + let mut consumer = Consumer::open(journal.clone(), "table").await.unwrap(); + let mut sink = Collect::new(); + while consumer.consume(&mut sink, 2, 16384).await.unwrap() != 0 {} + assert_eq!(sink.entries.lock().unwrap().len(), 35); + let observed: Arc>> = Arc::default(); + let pipeline = Partition::start( + journal.acquire().await.unwrap(), + Counter::new(observed.clone()), + config(), + ) + .await + .unwrap(); + let mut next = request(36); + next.session = "s".into(); + pipeline.enqueue(next).await.unwrap().wait().await.unwrap(); + assert_eq!(*observed.lock().unwrap(), vec![(36, 36)]); + pipeline.shutdown().await.unwrap(); +} + +struct AddDelta { + fail_b: Arc, + a_calls: Arc, +} + +#[async_trait] +impl Reducer for AddDelta { + async fn apply(&self, session: &str, state: &[u8], delta: &[u8]) -> Result> { + if session == "b" && self.fail_b.load(Ordering::SeqCst) { + return Err(Error::Stage("injected checkpoint failure".into())); + } + if session == "a" { + self.a_calls.fetch_add(1, Ordering::SeqCst); + } + let value = if state.is_empty() { + 0 + } else { + serde_json::from_slice::(state)? + }; + Ok(serde_json::to_vec( + &(value + serde_json::from_slice::(delta)?), + )?) + } +} + +#[tokio::test] +async fn partial_checkpoint_batch_recovery_skips_already_applied_session_deltas() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut writer = journal.acquire().await.unwrap(); + for (sequence, session) in [(1, "a"), (2, "b"), (3, "a")] { + writer + .append(vec![Entry { + sequence, + session: session.into(), + receipt: format!("r-{sequence}"), + input_digest: "digest".into(), + transition: Transition { + delta: b"1".to_vec(), + records: vec![], + }, + }]) + .await + .unwrap(); + } + let fail_b = Arc::new(std::sync::atomic::AtomicBool::new(true)); + let a_calls = Arc::new(AtomicUsize::new(0)); + let checkpoints = SessionCheckpoints::new(journal.clone(), 1024).unwrap(); + let mut sink = checkpoints + .sink( + AddDelta { + fail_b: fail_b.clone(), + a_calls: a_calls.clone(), + }, + 1, + ) + .unwrap(); + let mut consumer = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + assert!(consumer.consume(&mut sink, 10, 16384).await.is_err()); + assert_eq!(consumer.position().sequence, 0); + let a = checkpoints.load("a").await.unwrap().unwrap(); + assert_eq!(a.through_sequence, 3); + assert_eq!(a.value, b"2"); + assert!(checkpoints.load("b").await.unwrap().is_none()); + fail_b.store(false, Ordering::SeqCst); + let mut consumer = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + assert_eq!(consumer.consume(&mut sink, 10, 16384).await.unwrap(), 3); + assert_eq!(consumer.position().sequence, 3); + assert_eq!(checkpoints.load("a").await.unwrap().unwrap(), a); + assert_eq!( + a_calls.load(Ordering::SeqCst), + 2, + "retry must not apply delta to a twice" + ); + assert_eq!(checkpoints.load("b").await.unwrap().unwrap().value, b"1"); +} + +fn binding(partition: u32) -> Binding { + Binding { + run: "run-1".into(), + schema: "test-delta-v1".into(), + partition, + } +} + +fn journal(store: Arc, partition: u32) -> Journal { + Journal::new( + store, + Path::from(format!("run/partition-{partition}")), + binding(partition), + 1 << 20, + 100, + ) + .unwrap() +} + +fn config() -> PipelineConfig { + PipelineConfig { + queue_entries: 4, + load_concurrency: 4, + memory_bytes: 1 << 20, + max_input_bytes: 1024, + max_transition_bytes: 1024, + max_history_bytes: 0, + wal: BatchPolicy { + max_entries: 3, + max_bytes: 16 * 1024, + max_delay: Duration::from_millis(10), + }, + } +} + +fn request(sequence: u64) -> Request { + Request { + sequence, + session: "session-a".into(), + receipt: format!("receipt-{sequence}"), + payload: vec![sequence as u8], + } +} + +struct Counter { + states: BTreeMap, + observed: Arc>>, + align_gate: Option>, +} + +struct ReorderedLoads { + first: Arc, + second_loaded: Arc, +} + +#[async_trait] +impl HistoryLoader for ReorderedLoads { + async fn load(&self, request: &Request, _max_bytes: usize) -> Result> { + if request.sequence == 1 { + self.first.acquire().await.unwrap().forget(); + } else { + self.second_loaded.add_permits(1); + } + Ok(Vec::new()) + } +} + +#[tokio::test] +async fn history_prefetch_is_concurrent_but_same_session_alignment_keeps_input_order() { + let first = Arc::new(Semaphore::new(0)); + let second_loaded = Arc::new(Semaphore::new(0)); + let observed: Arc>> = Arc::default(); + let journal = journal(Arc::new(InMemory::new()), 0); + let pipeline = Partition::start_with_loader( + journal.acquire().await.unwrap(), + Counter::new(observed.clone()), + ReorderedLoads { + first: first.clone(), + second_loaded: second_loaded.clone(), + }, + config(), + ) + .await + .unwrap(); + let a = pipeline.enqueue(request(1)).await.unwrap(); + let b = pipeline.enqueue(request(2)).await.unwrap(); + timeout(Duration::from_secs(2), second_loaded.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + assert!( + observed.lock().unwrap().is_empty(), + "prefetch finishing out of order cannot reorder alignment" + ); + first.add_permits(1); + a.wait().await.unwrap(); + b.wait().await.unwrap(); + assert_eq!(*observed.lock().unwrap(), vec![(1, 1), (2, 2)]); + pipeline.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn dropping_partition_discards_uncommitted_suffix_before_recovery() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut aligner = Counter::new(Arc::default()); + aligner.align_gate = Some(Arc::new(Semaphore::new(0))); + let pipeline = Partition::start(journal.acquire().await.unwrap(), aligner, config()) + .await + .unwrap(); + let ack = pipeline.enqueue(request(1)).await.unwrap(); + drop(pipeline); + assert!(timeout(Duration::from_secs(2), ack.wait()) + .await + .unwrap() + .is_err()); + assert_eq!(journal.position().await.unwrap(), Position::default()); + let observed: Arc>> = Arc::default(); + let pipeline = Partition::start( + journal.acquire().await.unwrap(), + Counter::new(observed.clone()), + config(), + ) + .await + .unwrap(); + pipeline + .enqueue(request(1)) + .await + .unwrap() + .wait() + .await + .unwrap(); + assert_eq!(*observed.lock().unwrap(), vec![(1, 1)]); + pipeline.shutdown().await.unwrap(); +} + +impl Counter { + fn new(observed: Arc>>) -> Self { + Self { + states: BTreeMap::new(), + observed, + align_gate: None, + } + } +} + +#[async_trait] +impl Aligner for Counter { + async fn restore(&mut self, _binding: &Binding) -> Result { + Ok(Position::default()) + } + + async fn replay(&mut self, entry: &Entry) -> Result<()> { + self.states.insert( + entry.session.clone(), + serde_json::from_slice(&entry.transition.delta)?, + ); + Ok(()) + } + + async fn align(&mut self, request: &Request, _history: &[u8]) -> Result { + if let Some(gate) = &self.align_gate { + gate.acquire().await.unwrap().forget(); + } + let state = self.states.entry(request.session.clone()).or_default(); + *state += 1; + self.observed + .lock() + .unwrap() + .push((request.sequence, *state)); + Ok(Transition { + delta: serde_json::to_vec(state)?, + records: request.payload.clone(), + }) + } +} + +struct Collect { + entries: Arc>>, + gate: Option>, + calls: Arc, + fail_after_apply: bool, +} + +impl Collect { + fn new() -> Self { + Self { + entries: Arc::default(), + gate: None, + calls: Arc::default(), + fail_after_apply: false, + } + } +} + +#[async_trait] +impl Sink for Collect { + async fn apply(&mut self, _binding: &Binding, entries: &[Entry]) -> Result<()> { + self.calls.fetch_add(1, Ordering::SeqCst); + if let Some(gate) = &self.gate { + gate.acquire().await.unwrap().forget(); + } + for entry in entries { + if let Some(previous) = self + .entries + .lock() + .unwrap() + .insert(entry.sequence, entry.clone()) + { + assert_eq!(previous, *entry); + } + } + if self.fail_after_apply { + return Err(Error::Stage("injected lost sink response".into())); + } + Ok(()) + } +} + +#[tokio::test] +async fn durable_ack_and_alignment_continue_while_checkpoint_is_blocked() { + let journal = journal(Arc::new(InMemory::new()), 0); + let observed = Arc::default(); + let pipeline = Partition::start( + journal.acquire().await.unwrap(), + Counter::new(observed), + config(), + ) + .await + .unwrap(); + let first = pipeline.enqueue(request(1)).await.unwrap(); + assert_eq!(first.wait().await.unwrap().sequence, 1); + + let gate = Arc::new(Semaphore::new(0)); + let mut checkpoint = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + let mut sink = Collect::new(); + sink.gate = Some(gate.clone()); + let calls = sink.calls.clone(); + let stalled = tokio::spawn(async move { + checkpoint.consume(&mut sink, 10, 64 * 1024).await.unwrap(); + checkpoint.position().clone() + }); + timeout(Duration::from_secs(2), async { + while calls.load(Ordering::SeqCst) == 0 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + + let second = pipeline.enqueue(request(2)).await.unwrap(); + assert_eq!( + timeout(Duration::from_secs(2), second.wait()) + .await + .unwrap() + .unwrap() + .sequence, + 2 + ); + assert!(!stalled.is_finished()); + let mut merge = Consumer::open(journal.clone(), "table").await.unwrap(); + let mut table = Collect::new(); + assert_eq!(merge.consume(&mut table, 10, 64 * 1024).await.unwrap(), 2); + assert_eq!( + table.calls.load(Ordering::SeqCst), + 1, + "merge independently coalesces two WAL batches" + ); + assert_eq!(merge.position().sequence, 2); + gate.add_permits(1); + assert_eq!(stalled.await.unwrap().sequence, 1); + pipeline.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn restart_replays_committed_suffix_and_retry_does_not_align_twice() { + let journal = journal(Arc::new(InMemory::new()), 0); + let observed = Arc::default(); + let pipeline = Partition::start( + journal.acquire().await.unwrap(), + Counter::new(observed), + config(), + ) + .await + .unwrap(); + let ack = pipeline.enqueue(request(1)).await.unwrap(); + drop(ack); // client disappeared; admitted work still commits + pipeline.shutdown().await.unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 1); + + let observed: Arc>> = Arc::default(); + let pipeline = Partition::start( + journal.acquire().await.unwrap(), + Counter::new(observed.clone()), + config(), + ) + .await + .unwrap(); + assert_eq!( + pipeline + .enqueue(request(1)) + .await + .unwrap() + .wait() + .await + .unwrap() + .sequence, + 1 + ); + let mut changed = request(1); + changed.payload = vec![99]; + assert!(pipeline + .enqueue(changed) + .await + .unwrap() + .wait() + .await + .is_err()); + assert!(observed.lock().unwrap().is_empty()); + assert!( + pipeline + .enqueue(request(3)) + .await + .unwrap() + .wait() + .await + .is_err(), + "cannot skip missing source receipt" + ); + assert_eq!( + pipeline + .enqueue(request(2)) + .await + .unwrap() + .wait() + .await + .unwrap() + .sequence, + 2 + ); + assert_eq!( + *observed.lock().unwrap(), + vec![(2, 2)], + "restored state determines next aligned ID" + ); + pipeline.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn independent_partition_progresses_while_another_alignment_is_stalled() { + let store = Arc::new(InMemory::new()); + let gate = Arc::new(Semaphore::new(0)); + let mut slow = Counter::new(Arc::default()); + slow.align_gate = Some(gate.clone()); + let a = Partition::start( + journal(store.clone(), 0).acquire().await.unwrap(), + slow, + config(), + ) + .await + .unwrap(); + let b = Partition::start( + journal(store, 1).acquire().await.unwrap(), + Counter::new(Arc::default()), + config(), + ) + .await + .unwrap(); + let blocked = a.enqueue(request(1)).await.unwrap(); + let fast = b.enqueue(request(1)).await.unwrap(); + assert_eq!( + timeout(Duration::from_secs(2), fast.wait()) + .await + .unwrap() + .unwrap() + .sequence, + 1 + ); + assert_eq!(a.durable_position().sequence, 0); + gate.add_permits(1); + assert_eq!(blocked.wait().await.unwrap().sequence, 1); + a.shutdown().await.unwrap(); + b.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn stale_writer_upload_is_not_recoverable_or_acknowledged() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut old = journal.acquire().await.unwrap(); + let mut new = journal.acquire().await.unwrap(); + let entry = Entry { + sequence: 1, + session: "s".into(), + receipt: "r".into(), + input_digest: "hash".into(), + transition: Transition { + delta: vec![1], + records: vec![2], + }, + }; + assert!(old.append(vec![entry.clone()]).await.is_err()); + assert!(matches!( + old.append(vec![entry.clone()]).await, + Err(Error::Fenced) + )); + assert_eq!(journal.position().await.unwrap().sequence, 0); + let committed = new.append(vec![entry.clone()]).await.unwrap(); + let positions = journal + .pending(&Position::default(), &committed) + .await + .unwrap(); + assert_eq!(positions, vec![committed.clone()]); + assert_eq!(journal.entries(&committed).await.unwrap(), vec![entry]); +} + +#[tokio::test] +async fn consumer_uncertain_apply_replays_without_skipping_or_duplicating_output() { + let journal = journal(Arc::new(InMemory::new()), 0); + let pipeline = Partition::start( + journal.acquire().await.unwrap(), + Counter::new(Arc::default()), + config(), + ) + .await + .unwrap(); + pipeline + .enqueue(request(1)) + .await + .unwrap() + .wait() + .await + .unwrap(); + let mut consumer = Consumer::open(journal.clone(), "table").await.unwrap(); + let mut sink = Collect::new(); + sink.fail_after_apply = true; + assert!(consumer.consume(&mut sink, 5, 32 * 1024).await.is_err()); + assert_eq!(consumer.position().sequence, 0); + assert!(matches!( + consumer.consume(&mut sink, 5, 32 * 1024).await, + Err(Error::Fenced) + )); + let mut recovered = Consumer::open(journal.clone(), "table").await.unwrap(); + assert_eq!(recovered.position().sequence, 0); + sink.fail_after_apply = false; + recovered.consume(&mut sink, 1, 32 * 1024).await.unwrap(); + assert_eq!(sink.entries.lock().unwrap().len(), 1); + assert_eq!(sink.calls.load(Ordering::SeqCst), 2); + pipeline.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn byte_admission_is_bounded_and_sparse_wal_flushes_on_timer() { + let mut cfg = config(); + // Exactly one maximum-sized input/output reservation fits. + cfg.memory_bytes = ((cfg.max_input_bytes + cfg.max_transition_bytes) * 16 + 4096) as u32; + cfg.wal.max_delay = Duration::from_millis(50); + let gate = Arc::new(Semaphore::new(0)); + let mut aligner = Counter::new(Arc::default()); + aligner.align_gate = Some(gate.clone()); + let pipeline = Partition::start( + journal(Arc::new(InMemory::new()), 0) + .acquire() + .await + .unwrap(), + aligner, + cfg, + ) + .await + .unwrap(); + let first = pipeline.enqueue(request(1)).await.unwrap(); + assert!( + timeout(Duration::from_millis(30), pipeline.enqueue(request(2))) + .await + .is_err() + ); + gate.add_permits(1); + assert_eq!( + timeout(Duration::from_secs(2), first.wait()) + .await + .unwrap() + .unwrap() + .sequence, + 1 + ); + let second = pipeline.enqueue(request(2)).await.unwrap(); + gate.add_permits(1); + assert_eq!(second.wait().await.unwrap().sequence, 2); + pipeline.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn corrupt_committed_segment_fails_recovery_and_binding_cannot_change() { + let store = Arc::new(InMemory::new()); + let journal = journal(store.clone(), 0); + let pipeline = Partition::start( + journal.acquire().await.unwrap(), + Counter::new(Arc::default()), + config(), + ) + .await + .unwrap(); + let position = pipeline + .enqueue(request(1)) + .await + .unwrap() + .wait() + .await + .unwrap(); + pipeline.shutdown().await.unwrap(); + let wrong = Journal::new( + store.clone(), + Path::from("run/partition-0"), + binding(1), + 1 << 20, + 100, + ) + .unwrap(); + assert!(wrong.acquire().await.is_err()); + let segment = Path::from(format!( + "run/partition-0/segments/{}.json", + position.segment.unwrap() + )); + store + .put(&segment, b"corrupt".to_vec().into()) + .await + .unwrap(); + assert!(Partition::start( + journal.acquire().await.unwrap(), + Counter::new(Arc::default()), + config() + ) + .await + .is_err()); +} diff --git a/crates/lance-context-ingestion/tests/storage_faults.rs b/crates/lance-context-ingestion/tests/storage_faults.rs new file mode 100644 index 00000000..38e64c73 --- /dev/null +++ b/crates/lance-context-ingestion/tests/storage_faults.rs @@ -0,0 +1,153 @@ +use std::fmt; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::sync::Arc; + +use async_trait::async_trait; +use futures::stream::BoxStream; +use lance_context_ingestion::{Binding, Entry, Error, Journal, Position, Transition}; +use object_store::{ + memory::InMemory, path::Path, CopyOptions, GetOptions, GetResult, ListResult, MultipartUpload, + ObjectMeta, ObjectStore, PutMode, PutMultipartOptions, PutOptions, PutPayload, PutResult, +}; + +/// Real conditional-put storage with an injected response loss, not a replacement +/// journal. Tests distinguish an orphan upload from a successful head mutation. +#[derive(Debug)] +struct FaultStore { + inner: InMemory, + head_failure: AtomicUsize, +} + +impl fmt::Display for FaultStore { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "fault-store") + } +} + +fn injected() -> object_store::Error { + object_store::Error::Generic { + store: "fault-store", + source: std::io::Error::other("injected response loss").into(), + } +} + +#[async_trait] +impl ObjectStore for FaultStore { + async fn put_opts( + &self, + path: &Path, + payload: PutPayload, + opts: PutOptions, + ) -> object_store::Result { + let failure = + if path.filename() == Some("head.json") && matches!(opts.mode, PutMode::Update(_)) { + self.head_failure.swap(0, Ordering::SeqCst) + } else { + 0 + }; + if failure == 1 { + return Err(injected()); + } + let result = self.inner.put_opts(path, payload, opts).await?; + if failure == 2 { + return Err(injected()); + } + Ok(result) + } + + async fn put_multipart_opts( + &self, + path: &Path, + opts: PutMultipartOptions, + ) -> object_store::Result> { + self.inner.put_multipart_opts(path, opts).await + } + async fn get_opts(&self, path: &Path, opts: GetOptions) -> object_store::Result { + self.inner.get_opts(path, opts).await + } + fn delete_stream( + &self, + paths: BoxStream<'static, object_store::Result>, + ) -> BoxStream<'static, object_store::Result> { + self.inner.delete_stream(paths) + } + fn list(&self, prefix: Option<&Path>) -> BoxStream<'static, object_store::Result> { + self.inner.list(prefix) + } + async fn list_with_delimiter(&self, prefix: Option<&Path>) -> object_store::Result { + self.inner.list_with_delimiter(prefix).await + } + async fn copy_opts( + &self, + from: &Path, + to: &Path, + opts: CopyOptions, + ) -> object_store::Result<()> { + self.inner.copy_opts(from, to, opts).await + } +} + +fn binding() -> Binding { + Binding { + run: "r".into(), + schema: "s".into(), + partition: 0, + } +} + +fn entry() -> Entry { + Entry { + sequence: 1, + session: "session".into(), + receipt: "source-receipt".into(), + input_digest: "exact-input-digest".into(), + transition: Transition { + delta: vec![7], + records: vec![9], + }, + } +} + +#[tokio::test] +async fn uncertain_head_write_is_reconciled_from_storage_not_from_upload_success() { + for failure in [1, 2] { + let store = Arc::new(FaultStore { + inner: InMemory::new(), + head_failure: AtomicUsize::new(0), + }); + let journal = Journal::new(store.clone(), Path::from("wal"), binding(), 65536, 4).unwrap(); + let mut writer = journal.acquire().await.unwrap(); + store.head_failure.store(failure, Ordering::SeqCst); + assert!(writer.append(vec![entry()]).await.is_err()); + assert!(matches!( + writer.append(vec![entry()]).await, + Err(Error::Fenced) + )); + let mut recovered = journal.acquire().await.unwrap(); + assert_eq!( + recovered.position().sequence, + if failure == 2 { 1 } else { 0 } + ); + if failure == 1 { + recovered.append(vec![entry()]).await.unwrap(); + } + let committed = journal + .pending(&Position::default(), recovered.position()) + .await + .unwrap(); + assert_eq!(committed.len(), 1); + assert_eq!(journal.entries(&committed[0]).await.unwrap(), vec![entry()]); + } +} + +#[tokio::test] +async fn local_filesystem_without_conditional_updates_is_rejected_not_silently_emulated() { + let dir = tempfile::tempdir().unwrap(); + let store = + Arc::new(object_store::local::LocalFileSystem::new_with_prefix(dir.path()).unwrap()); + let journal = Journal::new(store, Path::from("wal"), binding(), 65536, 4).unwrap(); + let mut writer = journal.acquire().await.unwrap(); + assert!(writer.append(vec![entry()]).await.is_err()); + assert_eq!(journal.position().await.unwrap(), Position::default()); + assert!(journal.acquire().await.is_err()); +} From 976361eb31b793aec0d9ed8a1eecd21bc341b162 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Tue, 6 Oct 2026 21:35:32 +0000 Subject: [PATCH 02/16] Add Lance sink with atomic ingestion coverage and version fencing --- Cargo.lock | 8 + crates/lance-context-ingestion/Cargo.toml | 14 +- crates/lance-context-ingestion/README.md | 25 +- .../lance-context-ingestion/src/lance_sink.rs | 516 ++++++++++++++++++ crates/lance-context-ingestion/src/lib.rs | 2 + .../tests/lance_sink.rs | 345 ++++++++++++ 6 files changed, 902 insertions(+), 8 deletions(-) create mode 100644 crates/lance-context-ingestion/src/lance_sink.rs create mode 100644 crates/lance-context-ingestion/tests/lance_sink.rs diff --git a/Cargo.lock b/Cargo.lock index 1d4445b9..968ded52 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3911,8 +3911,16 @@ dependencies = [ name = "lance-context-ingestion" version = "0.1.0" dependencies = [ + "arrow-array", + "arrow-ipc", + "arrow-schema", "async-trait", "futures", + "lance", + "lance-file", + "lance-index", + "lance-io", + "lance-table", "object_store", "serde", "serde_json", diff --git a/crates/lance-context-ingestion/Cargo.toml b/crates/lance-context-ingestion/Cargo.toml index 989d3dbe..46c30962 100644 --- a/crates/lance-context-ingestion/Cargo.toml +++ b/crates/lance-context-ingestion/Cargo.toml @@ -5,8 +5,20 @@ edition.workspace = true license.workspace = true description = "Bounded session-ordered ingestion with recoverable WAL and independent consumers" +[features] +default = [] +lance = ["dep:arrow-array", "dep:arrow-ipc", "dep:arrow-schema", "dep:lance", "dep:lance-file", "dep:lance-index", "dep:lance-io", "dep:lance-table"] + [dependencies] async-trait = "0.1" +arrow-array = { version = "58", optional = true } +arrow-ipc = { version = "58", optional = true } +arrow-schema = { version = "58", optional = true } +lance = { version = "9.0.0", optional = true } +lance-file = { version = "9.0.0", optional = true } +lance-index = { version = "9.0.0", optional = true } +lance-io = { version = "9.0.0", optional = true } +lance-table = { version = "9.0.0", optional = true } futures = "0.3" object_store = "0.13.2" serde = { version = "1", features = ["derive"] } @@ -14,7 +26,7 @@ serde_json = "1" sha2 = "0.10" thiserror = "2" tokio = { version = "1", features = ["macros", "rt-multi-thread", "sync", "time"] } -uuid = { version = "1", features = ["v4"] } +uuid = { version = "1", features = ["v4", "v5"] } [dev-dependencies] tempfile = "3" diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index d477b0e3..88b108b8 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -1,8 +1,9 @@ # lance-context-ingestion Experimental streaming ingestion primitives. The implementation currently provides -the pipeline and journal protocol; integration with the existing trace alignment -adapter and a Lance table sink is **not complete**. This crate is not deployed. +the pipeline, journal protocol and an optional Lance table sink. Application +alignment/source adapters and deployment orchestration remain separate. This crate +is not deployed. ```text replayable source / stable partition-local receipts @@ -52,6 +53,16 @@ replayable source / stable partition-local receipts global cursor. Restore each state with its own sequence and skip already applied deltas when replaying the remaining global prefix. +- With the `lance` feature, `lance_sink::stage` writes immutable Lance 2.2 files + using a Zstd-annotated schema. Lance's constant-valued pages use scalar + encoding before codec selection; even a single large string can take that path. `LanceTableSink::commit_staged` coalesces staged + partitions and atomically publishes rows plus each partition's covered sequence. + Fully covered retries are skipped; gaps and partial overlaps are rejected. + A caller-supplied ownership-guarded commit handler is wrapped by an exact version + pin to reject implicit rebasing. Uncertain publication poisons the sink; reopen + and reconcile the table watermarks before retrying. The sink does not acquire + table ownership or schedule index maintenance. + Use a backend supporting atomic conditional updates, such as a suitably configured cloud object store. `object_store::local::LocalFileSystem` does not implement the required update operation and is rejected; there is no unsafe local-lock fallback. @@ -64,9 +75,9 @@ process-crash behavior on a real durable service, or production throughput. 1. Adapt the existing revisioned session history/cache and cross-call alignment implementation. Preserve its compaction/branch identity rules and source ordering; the generic pipeline deliberately does not invent a new turn-ID algorithm. -2. Connect the table consumer to public Lance staging/commit APIs. Persist exact WAL - input coverage with table publication; staged files alone do not justify a cursor - advance. Preserve Lance 2.2, Zstd and session ZoneMap configuration in that adapter. +2. Wire table ownership and independently scheduled session ZoneMap maintenance. + Staging is independent of publication; a distributed worker transport must + carry validated staged results and retain ownership through manifest commit. 3. Add source fan-out receipts and contiguous source progress. A source call spanning partitions is complete only after every required partition ACK. 4. Add worker ownership orchestration, stage timing/queue telemetry, consumer run loops, @@ -78,7 +89,7 @@ process-crash behavior on a real durable service, or production throughput. Run focused checks from the workspace root: ```sh -CARGO_TARGET_DIR=/tmp/trace-streaming-target cargo test -p lance-context-ingestion --offline -CARGO_TARGET_DIR=/tmp/trace-streaming-target cargo clippy -p lance-context-ingestion --all-targets --offline -- -D warnings +CARGO_TARGET_DIR=/tmp/trace-streaming-target cargo test -p lance-context-ingestion --features lance --offline +CARGO_TARGET_DIR=/tmp/trace-streaming-target cargo clippy -p lance-context-ingestion --features lance --all-targets --offline -- -D warnings cargo fmt -p lance-context-ingestion --check ``` diff --git a/crates/lance-context-ingestion/src/lance_sink.rs b/crates/lance-context-ingestion/src/lance_sink.rs new file mode 100644 index 00000000..dc71677b --- /dev/null +++ b/crates/lance-context-ingestion/src/lance_sink.rs @@ -0,0 +1,516 @@ +//! Immutable file staging plus a single version-pinned table publisher. +//! The caller holds the table's durable maintenance ownership across publication +//! and supplies its guarded commit handler. This module does not acquire leases. +use std::collections::{HashMap, HashSet}; +use std::io::Cursor; +use std::sync::Arc; + +use arrow_array::RecordBatch; +use arrow_ipc::{reader::StreamReader, writer::StreamWriter}; +use arrow_schema::Schema; +use async_trait::async_trait; +use futures::stream::BoxStream; +use lance::dataset::{ + transaction::{Operation, Transaction}, + CommitBuilder, Dataset, InsertBuilder, WriteMode, WriteParams, +}; +use lance::index::DatasetIndexExt; +use lance_file::version::LanceFileVersion; +use lance_index::mem_wal::{MemWalIndexDetails, MergedGeneration, MEM_WAL_INDEX_NAME}; +use lance_io::object_store::ObjectStore as LanceObjectStore; +use lance_table::format::{pb, Fragment, IndexMetadata, Manifest}; +use lance_table::io::commit::{ + CommitError, CommitHandler, ManifestLocation, ManifestNamingScheme, ManifestWriter, +}; +use object_store::{path::Path, ObjectStore}; +use uuid::Uuid; + +use crate::{Binding, Entry, Error, Result, Sink}; + +pub const RUN_METADATA: &str = "lance-context.ingestion.run"; +pub const SCHEMA_METADATA: &str = "lance-context.ingestion.schema"; +const SHARD_NAMESPACE: Uuid = Uuid::from_u128(0x8370a74b_3aaf_4083_8ed0_610b9bd6e8da); + +fn failure(error: impl std::fmt::Display) -> Error { + Error::Stage(error.to_string()) +} + +/// Declare the run identity and the encoding policy before table creation. +/// All batches must use this same Arrow schema; this never alters turn IDs. +/// Zstd applies where Lance selects a compression codec. Lance 2.2 constant +/// pages use a scalar layout instead, even for a single large string. +pub fn table_schema(schema: &Schema, run: &str, identity_schema: &str) -> Schema { + let fields = schema + .fields + .iter() + .map(|field| { + let mut metadata = field.metadata().clone(); + for (key, value) in [ + ("lance-encoding:compression", "zstd"), + ("lance-encoding:compression-level", "3"), + ("lance-encoding:dict-values-compression", "zstd"), + ("lance-encoding:dict-values-compression-level", "3"), + ] { + metadata.insert(key.into(), value.into()); + } + Arc::new(field.as_ref().clone().with_metadata(metadata)) + }) + .collect::>(); + let mut metadata = schema.metadata.clone(); + metadata.insert(RUN_METADATA.into(), run.into()); + metadata.insert(SCHEMA_METADATA.into(), identity_schema.into()); + Schema::new_with_metadata(fields, metadata) +} + +/// Encode already aligned output records. The WAL also persists the matching +/// state delta, so recovery does not run alignment or regenerate IDs. +pub fn encode_records(batches: &[RecordBatch]) -> Result> { + let Some(first) = batches.first() else { + return Ok(Vec::new()); + }; + let mut output = Vec::new(); + { + let mut writer = StreamWriter::try_new(&mut output, &first.schema()).map_err(failure)?; + for batch in batches { + writer.write(batch).map_err(failure)?; + } + writer.finish().map_err(failure)?; + } + Ok(output) +} + +pub struct Staged { + dataset_uri: String, + schema: lance::datatypes::Schema, + binding: Binding, + first: u64, + last: u64, + fragments: Vec, + rows: usize, +} + +impl Staged { + pub fn rows(&self) -> usize { + self.rows + } + pub fn first_sequence(&self) -> u64 { + self.first + } + pub fn last_sequence(&self) -> u64 { + self.last + } +} + +fn validate_dataset(dataset: &Dataset, binding: &Binding) -> Result<()> { + let schema = Schema::from(dataset.schema()); + if schema.metadata.get(RUN_METADATA) != Some(&binding.run) + || schema.metadata.get(SCHEMA_METADATA) != Some(&binding.schema) + || dataset.manifest().data_storage_format.version != "2.2" + { + return Err(Error::Invalid( + "Lance run/schema/2.2 format mismatch".into(), + )); + } + if schema.fields.iter().any(|field| { + field + .metadata() + .get("lance-encoding:compression") + .map(String::as_str) + != Some("zstd") + }) { + return Err(Error::Invalid( + "Lance schema must request Zstd encoding".into(), + )); + } + Ok(()) +} + +fn shard(binding: &Binding) -> Result { + Ok(Uuid::new_v5( + &SHARD_NAMESPACE, + &serde_json::to_vec(binding)?, + )) +} + +async fn watermarks(dataset: &Dataset) -> Result> { + let indices = dataset.load_indices().await.map_err(failure)?; + let Some(index) = indices + .iter() + .find(|index| index.name == MEM_WAL_INDEX_NAME) + else { + return Ok(HashMap::new()); + }; + let details = index + .index_details + .as_ref() + .ok_or_else(|| Error::Invalid("missing Lance WAL watermark details".into()))?; + let details = MemWalIndexDetails::try_from( + details + .to_msg::() + .map_err(failure)?, + ) + .map_err(failure)?; + Ok(details + .merged_generations + .into_iter() + .map(|entry| (entry.shard_id, entry.generation)) + .collect()) +} + +/// Decode a bounded WAL range and prepare immutable Lance 2.2 files only. This +/// is safe to distribute across workers: no table manifest or WAL cursor changes. +/// `max_bytes` bounds decoded Arrow data retained by this stage; input WAL and +/// encoder working memory require separate reservations in the calling worker. +pub async fn stage( + dataset: &Dataset, + binding: &Binding, + entries: &[Entry], + max_bytes: usize, +) -> Result { + validate_dataset(dataset, binding)?; + if entries.is_empty() + || max_bytes == 0 + || entries[0].sequence == 0 + || entries + .windows(2) + .any(|pair| pair[0].sequence.checked_add(1) != Some(pair[1].sequence)) + { + return Err(Error::Invalid("invalid Lance staging input range".into())); + } + let schema = Schema::from(dataset.schema()); + let mut batches = Vec::new(); + let mut bytes = 0_usize; + for entry in entries { + if entry.transition.records.is_empty() { + continue; + } + let reader = + StreamReader::try_new(Cursor::new(&entry.transition.records), None).map_err(failure)?; + if reader.schema().as_ref() != &schema { + return Err(Error::Invalid( + "WAL Arrow schema differs from target table".into(), + )); + } + for batch in reader { + let batch = batch.map_err(failure)?; + bytes = bytes + .checked_add(batch.get_array_memory_size()) + .ok_or_else(|| Error::Invalid("Arrow byte count overflow".into()))?; + if bytes > max_bytes { + return Err(Error::Invalid( + "staging decoded Arrow budget exceeded".into(), + )); + } + batches.push(batch); + } + } + let rows = batches.iter().map(RecordBatch::num_rows).sum(); + let fragments = if rows == 0 { + Vec::new() + } else { + let params = WriteParams { + mode: WriteMode::Append, + data_storage_version: Some(LanceFileVersion::V2_2), + max_bytes_per_file: max_bytes, + ..Default::default() + }; + let transaction = InsertBuilder::new(Arc::new(dataset.clone())) + .with_params(¶ms) + .execute_uncommitted(batches) + .await + .map_err(failure)?; + let Operation::Append { fragments } = transaction.operation else { + return Err(Error::Invalid("unexpected staging operation".into())); + }; + fragments + }; + Ok(Staged { + dataset_uri: dataset.uri().to_owned(), + schema: dataset.schema().clone(), + binding: binding.clone(), + first: entries[0].sequence, + last: entries.last().unwrap().sequence, + fragments, + rows, + }) +} + +/// A single manifest publisher. Supply the same guarded handler used by the +/// owner to open this dataset; the version pin wraps, never replaces, that guard. +/// After any uncertain commit, discard this sink, reopen the table and reconcile +/// its atomic watermarks. Do not retry a transaction against a newer base. +pub struct LanceTableSink { + dataset: Dataset, + handler: Arc, + max_stage_bytes: usize, + poisoned: bool, +} + +impl LanceTableSink { + pub fn new( + dataset: Dataset, + handler: Arc, + max_stage_bytes: usize, + ) -> Result { + if max_stage_bytes == 0 { + return Err(Error::Invalid("zero Lance staging budget".into())); + } + Ok(Self { + dataset, + handler, + max_stage_bytes, + poisoned: false, + }) + } + + pub fn dataset(&self) -> &Dataset { + &self.dataset + } + + pub async fn covered_sequence(&mut self, binding: &Binding) -> Result { + self.dataset.checkout_latest().await.map_err(failure)?; + validate_dataset(&self.dataset, binding)?; + Ok(watermarks(&self.dataset) + .await? + .get(&shard(binding)?) + .copied() + .unwrap_or(0)) + } + + pub async fn commit_staged(&mut self, staged: Vec) -> Result { + if self.poisoned { + return Err(Error::Fenced); + } + self.dataset.checkout_latest().await.map_err(failure)?; + let marks = watermarks(&self.dataset).await?; + let mut seen = HashSet::new(); + let mut fragments = Vec::new(); + let mut merged = Vec::new(); + let mut rows = 0; + for part in staged { + validate_dataset(&self.dataset, &part.binding)?; + let id = shard(&part.binding)?; + if !seen.insert(id) || part.dataset_uri != self.dataset.uri() { + return Err(Error::Invalid( + "duplicate partition or different staged dataset".into(), + )); + } + let high = marks.get(&id).copied().unwrap_or(0); + if part.last <= high { + continue; + } + if high.checked_add(1) != Some(part.first) { + return Err(Error::Invalid( + "staged range overlaps or skips merged WAL; restage from durable coverage" + .into(), + )); + } + let options = lance::datatypes::SchemaCompareOptions { + compare_metadata: true, + compare_field_ids: true, + ..Default::default() + }; + if Schema::from(&part.schema) != Schema::from(self.dataset.schema()) + || part + .schema + .check_compatible(self.dataset.schema(), &options) + .is_err() + || part + .fragments + .iter() + .map(|fragment| fragment.physical_rows.unwrap_or(0)) + .sum::() + != part.rows + { + return Err(Error::Invalid("staged schema or row count changed".into())); + } + rows += part.rows; + fragments.extend(part.fragments); + merged.push(MergedGeneration::new(id, part.last)); + } + if merged.is_empty() { + return Ok(0); + } + let base = self.dataset.version().version; + let operation = Operation::Update { + removed_fragment_ids: Vec::new(), + updated_fragments: Vec::new(), + new_fragments: fragments, + fields_modified: Vec::new(), + merged_generations: merged, + fields_for_preserving_frag_bitmap: Vec::new(), + update_mode: None, + inserted_rows_filter: None, + updated_fragment_offsets: None, + }; + let handler = Arc::new(PinnedCommit { + delegate: self.handler.clone(), + next_version: base + .checked_add(1) + .ok_or_else(|| Error::Invalid("table version overflow".into()))?, + }); + self.poisoned = true; + self.dataset = CommitBuilder::new(Arc::new(self.dataset.clone())) + .with_commit_handler(handler) + .with_max_retries(0) + .execute(Transaction::new(base, operation, None)) + .await + .map_err(failure)?; + self.poisoned = false; + Ok(rows) + } +} + +#[async_trait] +impl Sink for LanceTableSink { + async fn apply(&mut self, binding: &Binding, entries: &[Entry]) -> Result<()> { + if self.poisoned { + return Err(Error::Fenced); + } + if entries.is_empty() { + return Ok(()); + } + let high = self.covered_sequence(binding).await?; + let remaining = entries.partition_point(|entry| entry.sequence <= high); + if remaining == entries.len() { + return Ok(()); + } + let staged = stage( + &self.dataset, + binding, + &entries[remaining..], + self.max_stage_bytes, + ) + .await?; + self.commit_staged(vec![staged]).await?; + Ok(()) + } +} + +#[derive(Debug)] +struct PinnedCommit { + delegate: Arc, + next_version: u64, +} + +#[async_trait] +impl CommitHandler for PinnedCommit { + async fn resolve_latest_location( + &self, + base: &Path, + store: &LanceObjectStore, + ) -> lance::Result { + self.delegate.resolve_latest_location(base, store).await + } + async fn resolve_version_location( + &self, + base: &Path, + version: u64, + store: &dyn ObjectStore, + ) -> lance::Result { + self.delegate + .resolve_version_location(base, version, store) + .await + } + async fn version_exists( + &self, + base: &Path, + version: u64, + store: &dyn ObjectStore, + scheme: ManifestNamingScheme, + ) -> lance::Result { + self.delegate + .version_exists(base, version, store, scheme) + .await + } + fn list_detached_manifest_locations<'a>( + &self, + base: &Path, + store: &'a LanceObjectStore, + ) -> BoxStream<'a, lance::Result> { + self.delegate.list_detached_manifest_locations(base, store) + } + fn list_manifest_locations<'a>( + &self, + base: &Path, + store: &'a LanceObjectStore, + descending: bool, + ) -> BoxStream<'a, lance::Result> { + self.delegate + .list_manifest_locations(base, store, descending) + } + async fn commit( + &self, + manifest: &mut Manifest, + indices: Option>, + base: &Path, + store: &LanceObjectStore, + writer: ManifestWriter, + scheme: ManifestNamingScheme, + transaction: Option, + ) -> std::result::Result { + if manifest.version != self.next_version { + return Err(CommitError::OtherError(lance::Error::invalid_input( + "ingestion commit base changed; reopen and reconcile WAL coverage", + ))); + } + self.delegate + .commit(manifest, indices, base, store, writer, scheme, transaction) + .await + } + async fn delete(&self, base: &Path) -> lance::Result<()> { + self.delegate.delete(base).await + } +} + +#[cfg(test)] +mod tests { + use super::*; + use arrow_array::RecordBatchIterator; + use arrow_schema::{DataType, Field}; + + #[tokio::test] + async fn pinned_handler_rejects_initial_rebase_even_with_zero_retries() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().to_str().unwrap(); + let schema = Arc::new(table_schema( + &Schema::new(vec![Field::new("id", DataType::Utf8, false)]), + "run", + "schema", + )); + let old = Dataset::write( + RecordBatchIterator::new(vec![Ok(RecordBatch::new_empty(schema.clone()))], schema), + uri, + Some(WriteParams { + data_storage_version: Some(LanceFileVersion::V2_2), + ..Default::default() + }), + ) + .await + .unwrap(); + let mut current = old.clone(); + current + .update_metadata([("competing", "publication")]) + .await + .unwrap(); + let version = current.version().version; + let handler = Arc::new(PinnedCommit { + delegate: lance_table::io::commit::commit_handler_from_url(uri, &None) + .await + .unwrap(), + next_version: old.version().version + 1, + }); + let transaction = Transaction::new( + old.version().version, + Operation::Append { fragments: vec![] }, + None, + ); + assert!(CommitBuilder::new(Arc::new(old)) + .with_commit_handler(handler) + .with_max_retries(0) + .execute(transaction) + .await + .is_err()); + assert_eq!(Dataset::open(uri).await.unwrap().version().version, version); + } +} diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs index 6c7e3927..2ce5bd47 100644 --- a/crates/lance-context-ingestion/src/lib.rs +++ b/crates/lance-context-ingestion/src/lib.rs @@ -5,6 +5,8 @@ mod checkpoint; mod consumer; mod journal; +#[cfg(feature = "lance")] +pub mod lance_sink; mod pipeline; pub use checkpoint::{CheckpointSink, Reducer, SessionCheckpoints, SessionState}; diff --git a/crates/lance-context-ingestion/tests/lance_sink.rs b/crates/lance-context-ingestion/tests/lance_sink.rs new file mode 100644 index 00000000..3b949649 --- /dev/null +++ b/crates/lance-context-ingestion/tests/lance_sink.rs @@ -0,0 +1,345 @@ +#![cfg(feature = "lance")] + +use std::sync::Arc; + +use arrow_array::{Array, RecordBatch, RecordBatchIterator, StringArray}; +use arrow_schema::{DataType, Field, Schema}; +use futures::TryStreamExt; +use lance::dataset::{Dataset, WriteParams}; +use lance_context_ingestion::lance_sink::{encode_records, stage, table_schema, LanceTableSink}; +use lance_context_ingestion::{Binding, Consumer, Entry, Journal, Sink, Transition}; +use lance_file::version::LanceFileVersion; +use lance_table::io::commit::commit_handler_from_url; +use object_store::{memory::InMemory, path::Path}; + +fn binding(partition: u32) -> Binding { + Binding { + run: "isolated-run".into(), + schema: "aligned-v2".into(), + partition, + } +} + +async fn table(uri: &str) -> Dataset { + let schema = Arc::new(table_schema( + &Schema::new(vec![Field::new("id", DataType::Utf8, false)]), + &binding(0).run, + &binding(0).schema, + )); + let empty = RecordBatch::new_empty(schema.clone()); + Dataset::write( + RecordBatchIterator::new(vec![Ok(empty)], schema), + uri, + Some(WriteParams { + data_storage_version: Some(LanceFileVersion::V2_2), + ..Default::default() + }), + ) + .await + .unwrap() +} + +async fn sink(dataset: Dataset) -> LanceTableSink { + let handler = commit_handler_from_url(dataset.uri(), &None).await.unwrap(); + LanceTableSink::new(dataset, handler, 8 << 20).unwrap() +} + +fn entry(dataset: &Dataset, sequence: u64, id: Option<&str>) -> Entry { + let records = id + .map(|id| { + let batch = RecordBatch::try_new( + Arc::new(Schema::from(dataset.schema())), + vec![Arc::new(StringArray::from(vec![id]))], + ) + .unwrap(); + encode_records(&[batch]).unwrap() + }) + .unwrap_or_default(); + Entry { + sequence, + session: "session".into(), + receipt: format!("receipt-{sequence}"), + input_digest: format!("input-{sequence}"), + transition: Transition { + delta: vec![], + records, + }, + } +} + +async fn ids(dataset: &Dataset) -> Vec { + let batches = dataset + .scan() + .try_into_stream() + .await + .unwrap() + .try_collect::>() + .await + .unwrap(); + let mut ids = batches + .iter() + .flat_map(|batch| { + let strings = batch + .column(0) + .as_any() + .downcast_ref::() + .unwrap(); + (0..strings.len()) + .map(|i| strings.value(i).to_owned()) + .collect::>() + }) + .collect::>(); + ids.sort(); + ids +} + +#[tokio::test] +async fn immutable_staging_and_atomic_multi_partition_publication() { + let dir = tempfile::tempdir().unwrap(); + let dataset = table(dir.path().to_str().unwrap()).await; + let original = dataset.version().version; + let a = stage( + &dataset, + &binding(0), + &[entry(&dataset, 1, Some("a")), entry(&dataset, 2, None)], + 8 << 20, + ) + .await + .unwrap(); + let b = stage( + &dataset, + &binding(1), + &[entry(&dataset, 1, Some("b"))], + 8 << 20, + ) + .await + .unwrap(); + assert_eq!((a.rows(), a.first_sequence(), a.last_sequence()), (1, 1, 2)); + let unchanged = Dataset::open(dataset.uri()).await.unwrap(); + assert_eq!(unchanged.version().version, original); + assert!(ids(&unchanged).await.is_empty()); + let mut sink = sink(dataset).await; + assert_eq!(sink.commit_staged(vec![a, b]).await.unwrap(), 2); + assert_eq!(sink.dataset().version().version, original + 1); + assert_eq!(sink.covered_sequence(&binding(0)).await.unwrap(), 2); + assert_eq!(sink.covered_sequence(&binding(1)).await.unwrap(), 1); + assert_eq!(ids(sink.dataset()).await, vec!["a", "b"]); + assert_eq!(sink.dataset().manifest().data_storage_format.version, "2.2"); +} + +#[tokio::test] +async fn reopen_and_regroup_retries_preserve_exact_output_and_empty_coverage() { + let dir = tempfile::tempdir().unwrap(); + let dataset = table(dir.path().to_str().unwrap()).await; + let entries = vec![ + entry(&dataset, 1, Some("a")), + entry(&dataset, 2, None), + entry(&dataset, 3, Some("c")), + entry(&dataset, 4, None), + ]; + let mut first = sink(dataset).await; + first.apply(&binding(0), &entries[..2]).await.unwrap(); + let mut restarted = sink(Dataset::open(dir.path().to_str().unwrap()).await.unwrap()).await; + restarted.apply(&binding(0), &entries).await.unwrap(); + let version = restarted.dataset().version().version; + restarted.apply(&binding(0), &entries).await.unwrap(); + assert_eq!(restarted.dataset().version().version, version); + assert_eq!(restarted.covered_sequence(&binding(0)).await.unwrap(), 4); + assert_eq!(ids(restarted.dataset()).await, vec!["a", "c"]); +} + +#[tokio::test] +async fn wal_batches_merge_independently_and_lost_consumer_cursor_does_not_duplicate() { + let dir = tempfile::tempdir().unwrap(); + let dataset = table(dir.path().to_str().unwrap()).await; + let journal = Journal::new( + Arc::new(InMemory::new()), + Path::from("test"), + binding(0), + 1 << 20, + 100, + ) + .unwrap(); + let mut writer = journal.acquire().await.unwrap(); + for sequence in 1..=5 { + writer + .append(vec![entry( + &dataset, + sequence, + Some(&format!("id-{sequence}")), + )]) + .await + .unwrap(); + } + let mut first = sink(dataset).await; + let mut consumer = Consumer::open(journal.clone(), "table").await.unwrap(); + assert_eq!(consumer.consume(&mut first, 3, 8 << 20).await.unwrap(), 3); + let version = first.dataset().version().version; + // A different cursor starts at zero, modeling a table commit whose cursor ACK was lost. + let mut retry = Consumer::open(journal, "lost-cursor").await.unwrap(); + let mut restarted = sink(Dataset::open(dir.path().to_str().unwrap()).await.unwrap()).await; + assert_eq!(retry.consume(&mut restarted, 5, 8 << 20).await.unwrap(), 5); + assert_eq!(restarted.dataset().version().version, version + 1); + assert_eq!(restarted.covered_sequence(&binding(0)).await.unwrap(), 5); + assert_eq!(ids(restarted.dataset()).await.len(), 5); +} + +#[tokio::test] +async fn gaps_partial_overlaps_and_wrong_identity_cannot_advance_coverage() { + let dir = tempfile::tempdir().unwrap(); + let dataset = table(dir.path().to_str().unwrap()).await; + let entries = vec![entry(&dataset, 1, Some("a")), entry(&dataset, 2, Some("b"))]; + let overlap = stage(&dataset, &binding(0), &entries, 8 << 20) + .await + .unwrap(); + let gap = stage( + &dataset, + &binding(0), + &[entry(&dataset, 3, Some("c"))], + 8 << 20, + ) + .await + .unwrap(); + let mut wrong = binding(0); + wrong.run = "different-run".into(); + assert!(stage(&dataset, &wrong, &entries, 8 << 20).await.is_err()); + assert!(stage( + &dataset, + &binding(0), + &[entries[1].clone(), entries[0].clone()], + 8 << 20 + ) + .await + .is_err()); + let mut sink = sink(dataset).await; + sink.apply(&binding(0), &entries[..1]).await.unwrap(); + let version = sink.dataset().version().version; + assert!(sink.commit_staged(vec![overlap]).await.is_err()); + assert!(sink.commit_staged(vec![gap]).await.is_err()); + assert_eq!(sink.dataset().version().version, version); + assert_eq!(sink.covered_sequence(&binding(0)).await.unwrap(), 1); + assert_eq!(ids(sink.dataset()).await, vec!["a"]); +} + +#[tokio::test] +async fn duplicate_staging_workers_cannot_insert_same_partition_twice() { + let dir = tempfile::tempdir().unwrap(); + let dataset = table(dir.path().to_str().unwrap()).await; + let entries = vec![entry(&dataset, 1, Some("a"))]; + let a = stage(&dataset, &binding(0), &entries, 8 << 20) + .await + .unwrap(); + let b = stage(&dataset, &binding(0), &entries, 8 << 20) + .await + .unwrap(); + let mut sink = sink(dataset).await; + assert_eq!(sink.commit_staged(vec![a]).await.unwrap(), 1); + let version = sink.dataset().version().version; + assert_eq!(sink.commit_staged(vec![b]).await.unwrap(), 0); + assert_eq!(sink.dataset().version().version, version); + assert_eq!(ids(sink.dataset()).await, vec!["a"]); +} + +#[tokio::test] +async fn schema_change_and_decoding_budget_fail_before_publication() { + let dir = tempfile::tempdir().unwrap(); + let mut dataset = table(dir.path().to_str().unwrap()).await; + let entries = vec![entry(&dataset, 1, Some("a"))]; + assert!(stage(&dataset, &binding(0), &entries, 1).await.is_err()); + let staged = stage(&dataset, &binding(0), &entries, 8 << 20) + .await + .unwrap(); + dataset + .update_schema_metadata([("changed", "schema")]) + .await + .unwrap(); + let mut sink = sink(dataset).await; + assert!(sink.commit_staged(vec![staged]).await.is_err()); + assert_eq!(sink.covered_sequence(&binding(0)).await.unwrap(), 0); + assert!(ids(sink.dataset()).await.is_empty()); +} + +#[derive(Debug)] +struct CommitThenFail { + delegate: Arc, +} + +#[async_trait::async_trait] +impl lance_table::io::commit::CommitHandler for CommitThenFail { + async fn commit( + &self, + manifest: &mut lance_table::format::Manifest, + indices: Option>, + base: &Path, + store: &lance_io::object_store::ObjectStore, + writer: lance_table::io::commit::ManifestWriter, + scheme: lance_table::io::commit::ManifestNamingScheme, + transaction: Option, + ) -> std::result::Result< + lance_table::io::commit::ManifestLocation, + lance_table::io::commit::CommitError, + > { + self.delegate + .commit(manifest, indices, base, store, writer, scheme, transaction) + .await?; + Err(lance_table::io::commit::CommitError::OtherError( + lance::Error::invalid_input("injected lost successful commit response"), + )) + } +} + +#[tokio::test] +async fn lost_manifest_ack_poisons_sink_then_recovers_atomic_coverage() { + let dir = tempfile::tempdir().unwrap(); + let dataset = table(dir.path().to_str().unwrap()).await; + let entries = vec![entry(&dataset, 1, Some("a"))]; + let handler = Arc::new(CommitThenFail { + delegate: commit_handler_from_url(dataset.uri(), &None).await.unwrap(), + }); + let mut failed = LanceTableSink::new(dataset, handler, 8 << 20).unwrap(); + assert!(failed.apply(&binding(0), &entries).await.is_err()); + assert!(matches!( + failed.apply(&binding(0), &entries).await, + Err(lance_context_ingestion::Error::Fenced) + )); + let mut recovered = sink(Dataset::open(dir.path().to_str().unwrap()).await.unwrap()).await; + assert_eq!(recovered.covered_sequence(&binding(0)).await.unwrap(), 1); + recovered.apply(&binding(0), &entries).await.unwrap(); + assert_eq!(ids(recovered.dataset()).await, vec!["a"]); +} + +#[tokio::test] +async fn physical_files_are_22_and_contain_zstd_frames_with_lossless_roundtrip() { + let dir = tempfile::tempdir().unwrap(); + let dataset = table(dir.path().to_str().unwrap()).await; + let payload = + "a repeated compressible string with unicode 你好 and escaped quotes \"\\\n".repeat(8000); + // A constant-valued page uses Lance's scalar layout before compression selection. + // Distinct values exercise the actual configurable Zstd path. + let second = format!("{payload}second"); + let entries = vec![ + entry(&dataset, 1, Some(&payload)), + entry(&dataset, 2, Some(&second)), + ]; + let mut sink = sink(dataset).await; + sink.apply(&binding(0), &entries).await.unwrap(); + assert_eq!(ids(sink.dataset()).await, vec![payload.clone(), second]); + let mut zstd_frame_found = false; + for fragment in sink.dataset().get_fragments() { + for file in &fragment.metadata().files { + assert_eq!((file.file_major_version, file.file_minor_version), (2, 2)); + let bytes = std::fs::read(dir.path().join("data").join(&file.path)).unwrap(); + zstd_frame_found |= bytes + .windows(4) + .any(|window| window == [0x28, 0xb5, 0x2f, 0xfd]); + assert!( + bytes.len() < payload.len() / 4, + "large repeated value should be physically compressed" + ); + } + } + assert!( + zstd_frame_found, + "actual file must contain a Zstandard frame, not merely schema annotations" + ); +} From 5f340600b70d6740aab2d4198ba55af0b50eaab2 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Tue, 6 Oct 2026 22:36:35 +0000 Subject: [PATCH 03/16] Bound durable ingestion backlog by required consumer progress --- crates/lance-context-ingestion/README.md | 15 +- .../lance-context-ingestion/src/consumer.rs | 44 +++- crates/lance-context-ingestion/src/journal.rs | 90 +++++++ crates/lance-context-ingestion/src/lib.rs | 2 +- .../lance-context-ingestion/tests/pipeline.rs | 236 +++++++++++++++++- 5 files changed, 375 insertions(+), 12 deletions(-) diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index 88b108b8..82d2c389 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -3,7 +3,7 @@ Experimental streaming ingestion primitives. The implementation currently provides the pipeline, journal protocol and an optional Lance table sink. Application alignment/source adapters and deployment orchestration remain separate. This crate -is not deployed. +does not provide a complete distributed ingestion service. ```text replayable source / stable partition-local receipts @@ -47,6 +47,17 @@ replayable source / stable partition-local receipts into their own batches. Sink output and input coverage must be committed together; cursor writes can fail after output succeeds, so repeated/regrouped input must be idempotent. The scheduler owns exclusive consumer assignment and sink-side fencing. +- `Writer::with_backlog` optionally limits committed segments outstanding for + every required consumer. A consumer that has not started is at zero; table and + checkpoint progress are both required when both are configured. The publisher + waits before writing another segment, retaining bounded pipeline reservations; + other partitions and already durable retries remain independent. Consumers must + continue running while producers drain. Cancellation or ownership transfer fences + a paused writer. Reapply the same policy on every acquire/restart. The gate checks + durable cursor metadata and its committed ancestry; it adds storage reads and is + not itself a throughput optimization. It bounds unconsumed payload bytes by + `max_segments * max_segment_bytes`, not retained history, orphan uploads or total + storage. No WAL garbage collection or scheduling is implied. - `SessionCheckpoints` provides an actual object-store checkpoint sink: group by session, reduce ordered deltas, then write each session once with a conditional put. A partially successful checkpoint batch can leave some session states ahead of the @@ -81,7 +92,7 @@ process-crash behavior on a real durable service, or production throughput. 3. Add source fan-out receipts and contiguous source progress. A source call spanning partitions is complete only after every required partition ACK. 4. Add worker ownership orchestration, stage timing/queue telemetry, consumer run loops, - bounded durable backlog and safe WAL reclamation. No WAL files are deleted here. + deployment of the backlog policy and safe WAL reclamation. No WAL files are deleted here. 5. Verify real compacted sessions, process crash/restart with durable storage, Lance uncertain commits and source retry integration before a guarded production handoff. Existing source-reader local audit history must survive that handoff. diff --git a/crates/lance-context-ingestion/src/consumer.rs b/crates/lance-context-ingestion/src/consumer.rs index a47c918b..60739c5a 100644 --- a/crates/lance-context-ingestion/src/consumer.rs +++ b/crates/lance-context-ingestion/src/consumer.rs @@ -20,6 +20,42 @@ struct Cursor { position: Position, } +pub(crate) fn validate_consumer_name(name: &str) -> Result<()> { + if name.is_empty() + || !name + .bytes() + .all(|byte| byte.is_ascii_alphanumeric() || byte == b'-') + { + return Err(Error::Invalid("invalid consumer name".into())); + } + Ok(()) +} + +impl Journal { + /// Read durable consumer progress without creating or advancing its cursor. + /// A consumer that has not started is at zero. The returned position must + /// still be checked against the publisher's committed chain before use. + pub async fn consumer_position(&self, name: &str) -> Result { + validate_consumer_name(name)?; + let (cursor, _): (Cursor, _) = match self + .read(&self.path(&format!("consumers/{name}.json"))) + .await + { + Ok(value) => value, + Err(Error::Storage(object_store::Error::NotFound { .. })) => { + return Ok(Position::default()); + } + Err(error) => return Err(error), + }; + if &cursor.binding != self.binding() { + return Err(Error::Invalid( + "consumer run/schema/partition mismatch".into(), + )); + } + Ok(cursor.position) + } +} + /// Separate names (for example `checkpoint` and `table`) consume the same WAL /// independently. Scheduler must assign a single active worker per consumer and /// partition; cursor CAS detects ownership races but does not fence sink writes. @@ -33,13 +69,7 @@ pub struct Consumer { impl Consumer { pub async fn open(journal: Journal, name: &str) -> Result { - if name.is_empty() - || !name - .bytes() - .all(|byte| byte.is_ascii_alphanumeric() || byte == b'-') - { - return Err(Error::Invalid("invalid consumer name".into())); - } + validate_consumer_name(name)?; let path = journal.path(&format!("consumers/{name}.json")); let (cursor, version): (Cursor, _) = match journal.read(&path).await { Ok(pair) => pair, diff --git a/crates/lance-context-ingestion/src/journal.rs b/crates/lance-context-ingestion/src/journal.rs index a2589781..b2ed064e 100644 --- a/crates/lance-context-ingestion/src/journal.rs +++ b/crates/lance-context-ingestion/src/journal.rs @@ -1,12 +1,43 @@ use std::sync::Arc; +use std::time::Duration; use object_store::{path::Path, ObjectStore, ObjectStoreExt, PutMode, PutOptions, UpdateVersion}; use serde::{de::DeserializeOwned, Deserialize, Serialize}; use sha2::{Digest, Sha256}; use uuid::Uuid; +use crate::consumer::validate_consumer_name; use crate::{Error, Result}; +/// Limit committed WAL segments not yet acknowledged by every named consumer. +/// This bounds outstanding payload bytes by `max_segments * max_segment_bytes` +/// for this journal; retained consumed history, orphan uploads and metadata are +/// not reclaimed or included. All scheduled publishers must use the same policy. +#[derive(Clone, Debug)] +pub struct BacklogPolicy { + pub consumers: Vec, + pub max_segments: u64, + pub poll_interval: Duration, +} + +impl BacklogPolicy { + fn validate(&self) -> Result<()> { + if self.consumers.is_empty() || self.max_segments == 0 || self.poll_interval.is_zero() { + return Err(Error::Invalid( + "empty consumers or zero backlog limit".into(), + )); + } + let mut names = std::collections::HashSet::new(); + for name in &self.consumers { + validate_consumer_name(name)?; + if !names.insert(name) { + return Err(Error::Invalid("duplicate backlog consumer".into())); + } + } + Ok(()) + } +} + /// A namespace is permanently bound to one run, schema and virtual partition. /// Worker count may change; these virtual partition identities must not. #[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] @@ -79,6 +110,7 @@ pub struct Writer { head: Head, version: UpdateVersion, poisoned: bool, + backlog: Option, } pub(crate) fn digest(bytes: &[u8]) -> String { @@ -196,6 +228,7 @@ impl Journal { head, version, poisoned: false, + backlog: None, }) } @@ -298,6 +331,28 @@ impl Journal { Ok(segments) } + async fn validate_ancestor(&self, ancestor: &Position, through: &Position) -> Result<()> { + validate_position(ancestor)?; + validate_position(through)?; + let mut cursor = through.clone(); + while cursor.generation > ancestor.generation { + cursor = self + .link(&cursor) + .await? + .ancestors + .into_iter() + .rev() + .find(|position| position.generation >= ancestor.generation) + .ok_or_else(|| Error::Invalid("missing cursor skip link".into()))?; + } + if cursor != *ancestor { + return Err(Error::Invalid( + "consumer cursor is outside committed WAL".into(), + )); + } + Ok(()) + } + /// Read a segment previously obtained from `pending`; not an authorization /// to consume arbitrary uncommitted object paths. pub async fn entries(&self, position: &Position) -> Result> { @@ -343,6 +398,40 @@ fn validate_position(position: &Position) -> Result<()> { } impl Writer { + /// Apply backpressure before publishing another segment. Existing backlog + /// above the limit is drained, never discarded. Required consumers must keep + /// running while a pipeline shuts down; dropping a blocked append fences it. + /// This policy is process configuration and must be reapplied after acquire. + pub fn with_backlog(mut self, policy: BacklogPolicy) -> Result { + policy.validate()?; + self.backlog = Some(policy); + Ok(self) + } + + async fn wait_for_backlog(&self) -> Result<()> { + let Some(policy) = &self.backlog else { + return Ok(()); + }; + loop { + let (head, _) = self.journal.head().await?; + if head.epoch != self.head.epoch || head.position != self.head.position { + return Err(Error::Fenced); + } + let mut full = false; + for name in &policy.consumers { + let cursor = self.journal.consumer_position(name).await?; + self.journal + .validate_ancestor(&cursor, &head.position) + .await?; + full |= head.position.generation - cursor.generation >= policy.max_segments; + } + if !full { + return Ok(()); + } + tokio::time::sleep(policy.poll_interval).await; + } + } + pub fn position(&self) -> &Position { &self.head.position } @@ -386,6 +475,7 @@ impl Writer { }; // Set before the first await: cancelling this future also fences reuse. self.poisoned = true; + self.wait_for_backlog().await?; let mut ancestors = vec![self.head.position.clone()]; let mut level = 1; while let Some(ancestor) = ancestors.last().filter(|ancestor| ancestor.generation > 0) { diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs index 2ce5bd47..78510f53 100644 --- a/crates/lance-context-ingestion/src/lib.rs +++ b/crates/lance-context-ingestion/src/lib.rs @@ -11,7 +11,7 @@ mod pipeline; pub use checkpoint::{CheckpointSink, Reducer, SessionCheckpoints, SessionState}; pub use consumer::{Consumer, Sink}; -pub use journal::{Binding, Entry, Journal, Position, Transition, Writer}; +pub use journal::{BacklogPolicy, Binding, Entry, Journal, Position, Transition, Writer}; pub use pipeline::{Ack, Aligner, BatchPolicy, HistoryLoader, Partition, PipelineConfig, Request}; pub type Result = std::result::Result; diff --git a/crates/lance-context-ingestion/tests/pipeline.rs b/crates/lance-context-ingestion/tests/pipeline.rs index 26918d5b..3f10da39 100644 --- a/crates/lance-context-ingestion/tests/pipeline.rs +++ b/crates/lance-context-ingestion/tests/pipeline.rs @@ -5,13 +5,245 @@ use std::time::Duration; use async_trait::async_trait; use lance_context_ingestion::{ - Aligner, BatchPolicy, Binding, Consumer, Entry, Error, HistoryLoader, Journal, Partition, - PipelineConfig, Position, Reducer, Request, Result, SessionCheckpoints, Sink, Transition, + Aligner, BacklogPolicy, BatchPolicy, Binding, Consumer, Entry, Error, HistoryLoader, Journal, + Partition, PipelineConfig, Position, Reducer, Request, Result, SessionCheckpoints, Sink, + Transition, }; + use object_store::{memory::InMemory, path::Path, ObjectStore, ObjectStoreExt}; use tokio::sync::Semaphore; use tokio::time::timeout; +fn backlog(consumers: &[&str], max_segments: u64) -> BacklogPolicy { + BacklogPolicy { + consumers: consumers.iter().map(|name| (*name).into()).collect(), + max_segments, + poll_interval: Duration::from_millis(1), + } +} + +fn wal_entry(sequence: u64) -> Entry { + Entry { + sequence, + session: "session-a".into(), + receipt: format!("receipt-{sequence}"), + input_digest: "digest".into(), + transition: Transition { + delta: serde_json::to_vec(&sequence).unwrap(), + records: vec![], + }, + } +} + +#[tokio::test] +async fn backlog_waits_for_every_required_consumer_and_reopens_without_reset() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut writer = journal.acquire().await.unwrap(); + for n in 1..=3 { + writer.append(vec![wal_entry(n)]).await.unwrap(); + } + // Enabling the limit on an existing backlog must preserve all its data. + let mut writer = journal + .acquire() + .await + .unwrap() + .with_backlog(backlog(&["table", "checkpoint"], 2)) + .unwrap(); + assert_eq!( + journal.consumer_position("table").await.unwrap(), + Position::default() + ); + let mut append = Box::pin(writer.append(vec![wal_entry(4)])); + assert!(timeout(Duration::from_millis(20), &mut append) + .await + .is_err()); + let mut table = Consumer::open(journal.clone(), "table").await.unwrap(); + let mut checkpoint = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + let mut table_sink = Collect::new(); + let mut checkpoint_sink = Collect::new(); + assert_eq!(table.consume(&mut table_sink, 3, 16384).await.unwrap(), 3); + assert!(timeout(Duration::from_millis(20), &mut append) + .await + .is_err()); + assert_eq!( + checkpoint + .consume(&mut checkpoint_sink, 1, 16384) + .await + .unwrap(), + 1 + ); + assert!(timeout(Duration::from_millis(20), &mut append) + .await + .is_err()); + assert_eq!( + checkpoint + .consume(&mut checkpoint_sink, 1, 16384) + .await + .unwrap(), + 1 + ); + assert_eq!( + timeout(Duration::from_secs(2), append) + .await + .unwrap() + .unwrap() + .sequence, + 4 + ); + assert_eq!(journal.position().await.unwrap().generation, 4); + assert_eq!( + journal + .consumer_position("checkpoint") + .await + .unwrap() + .generation, + 2 + ); +} + +#[tokio::test] +async fn cancelling_or_reassigning_a_backlog_paused_writer_fences_it() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut writer = journal + .acquire() + .await + .unwrap() + .with_backlog(backlog(&["table"], 1)) + .unwrap(); + writer.append(vec![wal_entry(1)]).await.unwrap(); + // This timeout drops the actual append future, unlike the retained futures above. + assert!( + timeout(Duration::from_millis(20), writer.append(vec![wal_entry(2)])) + .await + .is_err() + ); + assert!(matches!( + writer.append(vec![wal_entry(2)]).await, + Err(Error::Fenced) + )); + let mut writer = journal + .acquire() + .await + .unwrap() + .with_backlog(backlog(&["table"], 1)) + .unwrap(); + let mut append = Box::pin(writer.append(vec![wal_entry(2)])); + assert!(timeout(Duration::from_millis(20), &mut append) + .await + .is_err()); + let mut replacement = journal + .acquire() + .await + .unwrap() + .with_backlog(backlog(&["table"], 1)) + .unwrap(); + assert!(matches!( + timeout(Duration::from_secs(2), append).await.unwrap(), + Err(Error::Fenced) + )); + let mut table = Consumer::open(journal.clone(), "table").await.unwrap(); + assert_eq!( + table.consume(&mut Collect::new(), 1, 16384).await.unwrap(), + 1 + ); + replacement.append(vec![wal_entry(2)]).await.unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 2); +} + +#[tokio::test] +async fn pipeline_backpressure_keeps_retries_live_and_other_partitions_independent() { + let store = Arc::new(InMemory::new()); + let j = journal(store.clone(), 0); + let mut cfg = config(); + cfg.wal.max_entries = 1; + let pipeline = Partition::start( + j.acquire() + .await + .unwrap() + .with_backlog(backlog(&["table"], 1)) + .unwrap(), + Counter::new(Arc::default()), + cfg, + ) + .await + .unwrap(); + pipeline + .enqueue(request(1)) + .await + .unwrap() + .wait() + .await + .unwrap(); + let mut next = Box::pin(pipeline.enqueue(request(2)).await.unwrap().wait()); + assert!(timeout(Duration::from_millis(30), &mut next).await.is_err()); + // Already durable retries must not wait for a table consumer to catch up. + timeout( + Duration::from_secs(2), + pipeline.enqueue(request(1)).await.unwrap().wait(), + ) + .await + .unwrap() + .unwrap(); + let other = journal(store, 1); + other + .acquire() + .await + .unwrap() + .with_backlog(backlog(&["table"], 1)) + .unwrap() + .append(vec![wal_entry(1)]) + .await + .unwrap(); + assert_eq!(j.position().await.unwrap().sequence, 1); + let mut consumer = Consumer::open(j.clone(), "table").await.unwrap(); + consumer + .consume(&mut Collect::new(), 1, 16384) + .await + .unwrap(); + timeout(Duration::from_secs(2), &mut next) + .await + .unwrap() + .unwrap(); + pipeline.shutdown().await.unwrap(); + assert_eq!(j.position().await.unwrap().sequence, 2); +} + +#[tokio::test] +async fn backlog_rejects_forged_cursor_even_when_its_generation_would_release_capacity() { + let store = Arc::new(InMemory::new()); + let j = journal(store.clone(), 0); + let mut writer = j + .acquire() + .await + .unwrap() + .with_backlog(backlog(&["table"], 1)) + .unwrap(); + let committed = writer.append(vec![wal_entry(1)]).await.unwrap(); + let forged = Position { + segment: Some(uuid::Uuid::new_v4().to_string()), + ..committed.clone() + }; + let path = Path::from("run/partition-0/consumers/table.json"); + store + .put( + &path, + serde_json::to_vec(&serde_json::json!({"binding":binding(0),"position":forged})) + .unwrap() + .into(), + ) + .await + .unwrap(); + assert!(matches!( + writer.append(vec![wal_entry(2)]).await, + Err(Error::Invalid(_)) + )); + assert_eq!(j.position().await.unwrap(), committed); + assert!(matches!( + writer.append(vec![wal_entry(2)]).await, + Err(Error::Fenced) + )); +} + #[tokio::test] async fn lagging_recovery_and_consumers_page_without_loading_entire_wal() { let store = Arc::new(InMemory::new()); From 034819eb971ce1b2760abf2e5c7ee0d1984cc335 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Tue, 6 Oct 2026 23:01:56 +0000 Subject: [PATCH 04/16] Allow coalesced session checkpoint reduction --- crates/lance-context-ingestion/README.md | 6 + .../lance-context-ingestion/src/checkpoint.rs | 46 ++++- .../lance-context-ingestion/tests/pipeline.rs | 175 ++++++++++++++++++ 3 files changed, 218 insertions(+), 9 deletions(-) diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index 82d2c389..92fd65b0 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -63,6 +63,12 @@ replayable source / stable partition-local receipts A partially successful checkpoint batch can leave some session states ahead of the global cursor. Restore each state with its own sequence and skip already applied deltas when replaying the remaining global prefix. + `Reducer::apply_batch` receives only a session's ordered, unapplied deltas and + lets an adapter decode its state once and encode it once per consumer batch. + Its default preserves per-delta `apply` behavior and intermediate size checks; + overrides must preserve those semantics and bound intermediate state themselves. + The sink also checks final output size before writing. This interface alone does + not accelerate existing reducers or change the deployed ingestion adapter. - With the `lance` feature, `lance_sink::stage` writes immutable Lance 2.2 files using a Zstd-annotated schema. Lance's constant-valued pages use scalar diff --git a/crates/lance-context-ingestion/src/checkpoint.rs b/crates/lance-context-ingestion/src/checkpoint.rs index cbfe79eb..a43ed368 100644 --- a/crates/lance-context-ingestion/src/checkpoint.rs +++ b/crates/lance-context-ingestion/src/checkpoint.rs @@ -27,6 +27,32 @@ pub struct SessionState { #[async_trait] pub trait Reducer: Send + Sync { async fn apply(&self, session: &str, state: &[u8], delta: &[u8]) -> Result>; + + /// Reduce one session's ordered, unapplied deltas. Adapters may decode state + /// once and encode it once; the result must equal repeated `apply` calls. + /// Preserve delta order and enforce `max_state_bytes` at every intermediate + /// state, including in overrides. Errors publish no state for this session. + async fn apply_batch( + &self, + session: &str, + state: &[u8], + deltas: &[&[u8]], + max_state_bytes: usize, + ) -> Result> { + let mut result = None; + for delta in deltas { + let next = self + .apply(session, result.as_deref().unwrap_or(state), delta) + .await?; + if next.len() > max_state_bytes { + return Err(Error::Invalid( + "session checkpoint exceeds state budget".into(), + )); + } + result = Some(next); + } + Ok(result.unwrap_or_else(|| state.to_vec())) + } } #[derive(Clone)] @@ -101,22 +127,24 @@ impl SessionCheckpoints { PutMode::Create, ), }; - let before = state.through_sequence; - for entry in entries { - if entry.sequence <= state.through_sequence { - continue; - } + let pending = entries + .into_iter() + .filter(|entry| entry.sequence > state.through_sequence) + .collect::>(); + if let Some(last) = pending.last() { + let deltas = pending + .iter() + .map(|entry| entry.transition.delta.as_slice()) + .collect::>(); state.value = reducer - .apply(session, &state.value, &entry.transition.delta) + .apply_batch(session, &state.value, &deltas, self.max_state_bytes) .await?; if state.value.len() > self.max_state_bytes { return Err(Error::Invalid( "session checkpoint exceeds state budget".into(), )); } - state.through_sequence = entry.sequence; - } - if state.through_sequence != before { + state.through_sequence = last.sequence; self.journal .put( &self diff --git a/crates/lance-context-ingestion/tests/pipeline.rs b/crates/lance-context-ingestion/tests/pipeline.rs index 3f10da39..ff044241 100644 --- a/crates/lance-context-ingestion/tests/pipeline.rs +++ b/crates/lance-context-ingestion/tests/pipeline.rs @@ -369,6 +369,181 @@ async fn partial_checkpoint_batch_recovery_skips_already_applied_session_deltas( assert_eq!(checkpoints.load("b").await.unwrap().unwrap().value, b"1"); } +type SessionDeltas = (String, Vec); + +struct OrderedBatch { + batches: Arc>>, + fail_b: Arc, +} + +#[async_trait] +impl Reducer for OrderedBatch { + async fn apply(&self, _: &str, _: &[u8], _: &[u8]) -> Result> { + Err(Error::Stage("unexpected per-delta reduction".into())) + } + + async fn apply_batch( + &self, + session: &str, + state: &[u8], + deltas: &[&[u8]], + max_state_bytes: usize, + ) -> Result> { + let deltas = deltas + .iter() + .map(|delta| serde_json::from_slice::(delta)) + .collect::, _>>()?; + self.batches + .lock() + .unwrap() + .push((session.into(), deltas.clone())); + let mut values: Vec = if state.is_empty() { + Vec::new() + } else { + serde_json::from_slice(state)? + }; + values.extend(deltas); + let result = serde_json::to_vec(&values)?; + // This representation only grows, so the final bound covers each prefix. + if result.len() > max_state_bytes { + return Err(Error::Invalid("test state budget exceeded".into())); + } + if session == "b" && self.fail_b.load(Ordering::SeqCst) { + return Err(Error::Stage("injected batch reduction failure".into())); + } + Ok(result) + } +} + +#[tokio::test] +async fn checkpoint_batch_override_coalesces_ordered_deltas_and_skips_committed_prefix() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut writer = journal.acquire().await.unwrap(); + for sequence in 1..=5 { + let mut entry = wal_entry(sequence); + entry.session = if sequence % 2 == 1 { "a" } else { "b" }.into(); + writer.append(vec![entry]).await.unwrap(); + } + let batches = Arc::new(Mutex::new(Vec::new())); + let fail_b = Arc::new(std::sync::atomic::AtomicBool::new(true)); + let checkpoints = SessionCheckpoints::new(journal.clone(), 1024).unwrap(); + let mut sink = checkpoints + .sink( + OrderedBatch { + batches: batches.clone(), + fail_b: fail_b.clone(), + }, + 1, + ) + .unwrap(); + let mut consumer = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + assert!(consumer.consume(&mut sink, 10, 16384).await.is_err()); + assert_eq!(consumer.position(), &Position::default()); + let a = checkpoints.load("a").await.unwrap().unwrap(); + assert_eq!(a.through_sequence, 5); + assert_eq!(a.value, b"[1,3,5]"); + assert!(checkpoints.load("b").await.unwrap().is_none()); + assert_eq!( + *batches.lock().unwrap(), + vec![("a".into(), vec![1, 3, 5]), ("b".into(), vec![2, 4])] + ); + + fail_b.store(false, Ordering::SeqCst); + let mut consumer = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + // Regroup the retry across different consumer batch boundaries. Session a + // is ahead of the global prefix and must not be reduced again, even partly. + assert_eq!(consumer.consume(&mut sink, 2, 16384).await.unwrap(), 2); + assert_eq!(consumer.consume(&mut sink, 3, 16384).await.unwrap(), 3); + assert_eq!(checkpoints.load("a").await.unwrap().unwrap(), a); + assert_eq!( + checkpoints.load("b").await.unwrap().unwrap().value, + b"[2,4]" + ); + assert_eq!( + *batches.lock().unwrap(), + vec![ + ("a".into(), vec![1, 3, 5]), + ("b".into(), vec![2, 4]), + ("b".into(), vec![2]), + ("b".into(), vec![4]), + ] + ); + + let mut next = wal_entry(6); + next.session = "a".into(); + writer.append(vec![next]).await.unwrap(); + assert_eq!(consumer.consume(&mut sink, 10, 16384).await.unwrap(), 1); + let a = checkpoints.load("a").await.unwrap().unwrap(); + assert_eq!(a.through_sequence, 6); + assert_eq!(a.value, b"[1,3,5,6]"); + assert_eq!(batches.lock().unwrap().last(), Some(&("a".into(), vec![6]))); + assert_eq!(consumer.consume(&mut sink, 10, 16384).await.unwrap(), 0); +} + +struct ReplaceDelta; + +#[async_trait] +impl Reducer for ReplaceDelta { + async fn apply(&self, _: &str, _: &[u8], delta: &[u8]) -> Result> { + Ok(delta.to_vec()) + } +} + +struct OversizedBatch; + +#[async_trait] +impl Reducer for OversizedBatch { + async fn apply(&self, _: &str, _: &[u8], _: &[u8]) -> Result> { + unreachable!("batch override is required") + } + + async fn apply_batch(&self, _: &str, _: &[u8], _: &[&[u8]], _: usize) -> Result> { + // Deliberately violate the callback's budget contract. The sink still + // rejects oversized output without publishing it or advancing progress. + Ok(vec![0; 9]) + } +} + +async fn checkpoint_budget_rejects_without_advancing(reducer: R) { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut writer = journal.acquire().await.unwrap(); + let mut entries = vec![wal_entry(1), wal_entry(2)]; + entries[0].transition.delta = vec![1; 9]; + entries[1].transition.delta = vec![2; 1]; + writer.append(entries).await.unwrap(); + let checkpoints = SessionCheckpoints::new(journal.clone(), 8).unwrap(); + let mut sink = checkpoints.sink(reducer, 1).unwrap(); + let mut consumer = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + assert!(matches!( + consumer.consume(&mut sink, 10, 16384).await, + Err(Error::Invalid(_)) + )); + assert_eq!(consumer.position(), &Position::default()); + assert!(checkpoints.load("session-a").await.unwrap().is_none()); + assert_eq!( + journal.consumer_position("checkpoint").await.unwrap(), + Position::default() + ); + // Recovery with sufficient budget consumes the unchanged WAL normally. + let checkpoints = SessionCheckpoints::new(journal.clone(), 16).unwrap(); + let mut sink = checkpoints.sink(ReplaceDelta, 1).unwrap(); + let mut consumer = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + assert_eq!(consumer.consume(&mut sink, 10, 16384).await.unwrap(), 2); + let state = checkpoints.load("session-a").await.unwrap().unwrap(); + assert_eq!(state.through_sequence, 2); + assert_eq!(state.value, vec![2]); +} + +#[tokio::test] +async fn checkpoint_default_batch_rejects_oversized_intermediate_state() { + checkpoint_budget_rejects_without_advancing(ReplaceDelta).await; +} + +#[tokio::test] +async fn checkpoint_batch_override_cannot_publish_oversized_output() { + checkpoint_budget_rejects_without_advancing(OversizedBatch).await; +} + fn binding(partition: u32) -> Binding { Binding { run: "run-1".into(), From 09590f70d330993a0a86b9b454f4969ebb977248 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 01:46:24 +0000 Subject: [PATCH 05/16] Recover cold ingestion sessions from checkpoints and committed WAL --- crates/lance-context-ingestion/README.md | 7 + .../lance-context-ingestion/src/checkpoint.rs | 77 ++++++++++- crates/lance-context-ingestion/src/lib.rs | 2 +- .../lance-context-ingestion/tests/pipeline.rs | 130 ++++++++++++++++++ 4 files changed, 214 insertions(+), 2 deletions(-) diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index 92fd65b0..cc2a200d 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -70,6 +70,13 @@ replayable source / stable partition-local receipts The sink also checks final output size before writing. This interface alone does not accelerate existing reducers or change the deployed ingestion adapter. +- `SessionCheckpoints::recover` reconstructs one evicted or cold session from the + checkpoint consumer cursor plus committed WAL suffix, one segment at a time. + It returns the captured WAL position separately from the session mutation + sequence and skips partially published checkpoint deltas. This is read-only; + callers must still reconcile their newer speculative state and bound the cache. + Use the checkpoint consumer name, never an unrelated table consumer cursor. + - With the `lance` feature, `lance_sink::stage` writes immutable Lance 2.2 files using a Zstd-annotated schema. Lance's constant-valued pages use scalar encoding before codec selection; even a single large string can take that path. `LanceTableSink::commit_staged` coalesces staged diff --git a/crates/lance-context-ingestion/src/checkpoint.rs b/crates/lance-context-ingestion/src/checkpoint.rs index a43ed368..fb0435d0 100644 --- a/crates/lance-context-ingestion/src/checkpoint.rs +++ b/crates/lance-context-ingestion/src/checkpoint.rs @@ -7,7 +7,7 @@ use object_store::{PutMode, UpdateVersion}; use serde::{Deserialize, Serialize}; use crate::journal::digest; -use crate::{Binding, Entry, Error, Journal, Result, Sink}; +use crate::{Binding, Entry, Error, Journal, Position, Result, Sink}; /// A per-session checkpoint may be ahead of the consumer's coherent prefix if a /// previous multi-session batch failed halfway. Recovery must skip replay deltas @@ -21,6 +21,15 @@ pub struct SessionState { pub value: Vec, } +/// One session reconstructed through an exact committed WAL position. The state +/// sequence is its last mutation, which can be older than `position.sequence`. +/// This does not include a publisher's uncommitted alignment suffix. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct RecoveredSession { + pub state: SessionState, + pub position: Position, +} + /// Apply the already chosen alignment delta. This operation must be deterministic /// and must not allocate new IDs or rerun history matching. Empty state denotes /// a session with no checkpoint. Bound transient reducer memory in the adapter. @@ -96,6 +105,72 @@ impl SessionCheckpoints { Ok(self.read(session).await?.map(|(state, _)| state)) } + /// Lazily recover one session after restart or cache eviction. `consumer` + /// must name the consumer using this checkpoint store, never a table cursor. + /// Read its coherent prefix before the session and the WAL head after it: + /// a partially committed checkpoint batch can leave this session ahead of + /// the prefix. Such deltas are skipped, not applied twice. No storage writes + /// occur. Replay holds one WAL segment and one session state at a time; + /// reducer transient allocations remain the adapter's responsibility. + pub async fn recover( + &self, + consumer: &str, + session: &str, + reducer: &R, + ) -> Result { + if session.is_empty() { + return Err(Error::Invalid("empty recovery session".into())); + } + let mut after = self.journal.consumer_position(consumer).await?; + let mut state = self.load(session).await?.unwrap_or_else(|| SessionState { + binding: self.journal.binding().clone(), + session: session.into(), + through_sequence: 0, + value: Vec::new(), + }); + let position = self.journal.position().await?; + if state.through_sequence > position.sequence { + return Err(Error::Invalid( + "session checkpoint is ahead of committed WAL".into(), + )); + } + // pending also validates that the consumer's opaque position belongs to + // this exact committed chain, including when there is nothing to replay. + loop { + let pending = self.journal.pending(&after, &position).await?; + if pending.is_empty() { + break; + } + for segment in pending { + let entries = self.journal.entries(&segment).await?; + let unapplied = entries + .iter() + .filter(|entry| { + entry.session == session && entry.sequence > state.through_sequence + }) + .collect::>(); + if let Some(last) = unapplied.last() { + let deltas = unapplied + .iter() + .map(|entry| entry.transition.delta.as_slice()) + .collect::>(); + let value = reducer + .apply_batch(session, &state.value, &deltas, self.max_state_bytes) + .await?; + if value.len() > self.max_state_bytes { + return Err(Error::Invalid( + "recovered session exceeds state budget".into(), + )); + } + state.value = value; + state.through_sequence = last.sequence; + } + after = segment; + } + } + Ok(RecoveredSession { state, position }) + } + /// Each session is written once per consumer batch, even if it occurred in /// many producer WAL segments. Conditional writes never replace newer state. pub fn sink(&self, reducer: R, concurrency: usize) -> Result> { diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs index 78510f53..5988386c 100644 --- a/crates/lance-context-ingestion/src/lib.rs +++ b/crates/lance-context-ingestion/src/lib.rs @@ -9,7 +9,7 @@ mod journal; pub mod lance_sink; mod pipeline; -pub use checkpoint::{CheckpointSink, Reducer, SessionCheckpoints, SessionState}; +pub use checkpoint::{CheckpointSink, RecoveredSession, Reducer, SessionCheckpoints, SessionState}; pub use consumer::{Consumer, Sink}; pub use journal::{BacklogPolicy, Binding, Entry, Journal, Position, Transition, Writer}; pub use pipeline::{Ack, Aligner, BatchPolicy, HistoryLoader, Partition, PipelineConfig, Request}; diff --git a/crates/lance-context-ingestion/tests/pipeline.rs b/crates/lance-context-ingestion/tests/pipeline.rs index ff044241..aae5dbb6 100644 --- a/crates/lance-context-ingestion/tests/pipeline.rs +++ b/crates/lance-context-ingestion/tests/pipeline.rs @@ -369,6 +369,136 @@ async fn partial_checkpoint_batch_recovery_skips_already_applied_session_deltas( assert_eq!(checkpoints.load("b").await.unwrap().unwrap().value, b"1"); } +#[tokio::test] +async fn lazy_session_recovery_reads_only_suffix_and_handles_uncached_sessions() { + let store = Arc::new(InMemory::new()); + // One-segment pages force recovery to consume several pages. + let journal = Journal::new(store.clone(), Path::from("lazy"), binding(0), 1 << 20, 1).unwrap(); + let mut writer = journal.acquire().await.unwrap(); + let first = writer.append(vec![wal_entry(1)]).await.unwrap(); + let checkpoints = SessionCheckpoints::new(journal.clone(), 1024).unwrap(); + let reducer = || AddDelta { + fail_b: Arc::new(std::sync::atomic::AtomicBool::new(false)), + a_calls: Arc::new(AtomicUsize::new(0)), + }; + let mut sink = checkpoints.sink(reducer(), 1).unwrap(); + let mut consumer = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + consumer.consume(&mut sink, 10, 16384).await.unwrap(); + // The checkpoint replaces these payload reads, but not immutable link metadata. + // Deliberately deleting data here is a regression test, not a GC protocol. + store + .delete(&Path::from(format!( + "lazy/segments/{}.json", + first.segment.unwrap() + ))) + .await + .unwrap(); + for sequence in 2..=5 { + writer.append(vec![wal_entry(sequence)]).await.unwrap(); + } + let recovered = checkpoints + .recover("checkpoint", "session-a", &reducer()) + .await + .unwrap(); + assert_eq!(recovered.state.value, b"15"); + assert_eq!(recovered.state.through_sequence, 5); + assert_eq!(recovered.position, *writer.position()); + assert_eq!( + checkpoints.load("session-a").await.unwrap().unwrap().value, + b"1", + "readonly recovery must not publish a new checkpoint" + ); + assert_eq!( + journal + .consumer_position("checkpoint") + .await + .unwrap() + .sequence, + 1 + ); + let missing = checkpoints + .recover("checkpoint", "new-session", &reducer()) + .await + .unwrap(); + assert!(missing.state.value.is_empty()); + assert_eq!(missing.state.through_sequence, 0); + assert_eq!(missing.position, recovered.position); +} + +#[tokio::test] +async fn lazy_recovery_skips_partially_published_state_and_propagates_reducer_errors() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut writer = journal.acquire().await.unwrap(); + for (sequence, session) in [(1, "a"), (2, "b"), (3, "a")] { + let mut entry = wal_entry(sequence); + entry.session = session.into(); + entry.transition.delta = b"1".to_vec(); + writer.append(vec![entry]).await.unwrap(); + } + let checkpoints = SessionCheckpoints::new(journal.clone(), 1024).unwrap(); + let fail_b = Arc::new(std::sync::atomic::AtomicBool::new(true)); + let a_calls = Arc::new(AtomicUsize::new(0)); + let reducer = || AddDelta { + fail_b: fail_b.clone(), + a_calls: a_calls.clone(), + }; + let mut sink = checkpoints.sink(reducer(), 1).unwrap(); + let mut consumer = Consumer::open(journal.clone(), "checkpoint").await.unwrap(); + assert!(consumer.consume(&mut sink, 10, 16384).await.is_err()); + assert_eq!(consumer.position().sequence, 0); + let a = checkpoints + .recover("checkpoint", "a", &reducer()) + .await + .unwrap(); + assert_eq!(a.state.value, b"2"); + assert_eq!( + a_calls.load(Ordering::SeqCst), + 2, + "already checkpointed deltas must not replay" + ); + assert!(checkpoints + .recover("checkpoint", "b", &reducer()) + .await + .is_err()); + fail_b.store(false, Ordering::SeqCst); + let b = checkpoints + .recover("checkpoint", "b", &reducer()) + .await + .unwrap(); + assert_eq!(b.state.value, b"1"); + assert_eq!(b.state.through_sequence, 2); + assert_eq!(b.position.sequence, 3); + assert!(checkpoints.load("b").await.unwrap().is_none()); +} + +#[tokio::test] +async fn lazy_recovery_rejects_wrong_chain_cursor_and_oversized_reduction() { + let store = Arc::new(InMemory::new()); + let journal = journal(store.clone(), 0); + let mut writer = journal.acquire().await.unwrap(); + let head = writer.append(vec![wal_entry(1)]).await.unwrap(); + let checkpoints = SessionCheckpoints::new(journal.clone(), 1).unwrap(); + assert!(checkpoints + .recover("checkpoint", "session-a", &OversizedBatch) + .await + .is_err()); + let mut wrong = head; + wrong.segment = Some(uuid::Uuid::new_v4().to_string()); + store + .put( + &Path::from("run/partition-0/consumers/checkpoint.json"), + serde_json::to_vec(&serde_json::json!({"binding": binding(0), "position": wrong})) + .unwrap() + .into(), + ) + .await + .unwrap(); + assert!(checkpoints + .recover("checkpoint", "new-session", &ReplaceDelta) + .await + .is_err()); +} + type SessionDeltas = (String, Vec); struct OrderedBatch { From d14c089f562d45e397e6c6d6684b07232061b952 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 01:56:22 +0000 Subject: [PATCH 06/16] Scope macro-generated must-use lint allowance to async ingestion traits --- crates/lance-context-ingestion/src/checkpoint.rs | 4 ++++ crates/lance-context-ingestion/src/consumer.rs | 4 ++++ crates/lance-context-ingestion/src/pipeline.rs | 8 ++++++++ 3 files changed, 16 insertions(+) diff --git a/crates/lance-context-ingestion/src/checkpoint.rs b/crates/lance-context-ingestion/src/checkpoint.rs index fb0435d0..c6eec322 100644 --- a/crates/lance-context-ingestion/src/checkpoint.rs +++ b/crates/lance-context-ingestion/src/checkpoint.rs @@ -33,6 +33,10 @@ pub struct RecoveredSession { /// Apply the already chosen alignment delta. This operation must be deterministic /// and must not allocate new IDs or rerun history matching. Empty state denotes /// a session with no checkpoint. Bound transient reducer memory in the adapter. +#[allow( + clippy::double_must_use, + reason = "async_trait adds must_use to boxed futures" +)] #[async_trait] pub trait Reducer: Send + Sync { async fn apply(&self, session: &str, state: &[u8], delta: &[u8]) -> Result>; diff --git a/crates/lance-context-ingestion/src/consumer.rs b/crates/lance-context-ingestion/src/consumer.rs index 60739c5a..145711b7 100644 --- a/crates/lance-context-ingestion/src/consumer.rs +++ b/crates/lance-context-ingestion/src/consumer.rs @@ -9,6 +9,10 @@ use crate::{Binding, Entry, Error, Journal, Position, Result}; /// Return success only after outputs AND their coverage are durable together. /// For Lance, staged fragments alone are insufficient: manifest publication must /// atomically record the covered input range. This callback never owns WAL ACKs. +#[allow( + clippy::double_must_use, + reason = "async_trait adds must_use to boxed futures" +)] #[async_trait] pub trait Sink: Send { async fn apply(&mut self, binding: &Binding, entries: &[Entry]) -> Result<()>; diff --git a/crates/lance-context-ingestion/src/pipeline.rs b/crates/lance-context-ingestion/src/pipeline.rs index 198eaa42..a5a18215 100644 --- a/crates/lance-context-ingestion/src/pipeline.rs +++ b/crates/lance-context-ingestion/src/pipeline.rs @@ -24,6 +24,10 @@ pub struct Request { /// Adapter owns session state and its cache budget. Alignment can run ahead of /// WAL durability, so an error discards this adapter and its speculative suffix. /// Checkpoints restore exact committed deltas, never freshly recomputed IDs. +#[allow( + clippy::double_must_use, + reason = "async_trait adds must_use to boxed futures" +)] #[async_trait] pub trait Aligner: Send + 'static { async fn restore(&mut self, binding: &Binding) -> Result; @@ -37,6 +41,10 @@ pub trait Aligner: Send + 'static { /// Concurrent, read-only history loading. The adapter must respect `max_bytes` /// while fetching/decoding, not only after allocation. Loaded state carries its /// revision in the adapter's encoding so alignment can reconcile stale loads. +#[allow( + clippy::double_must_use, + reason = "async_trait adds must_use to boxed futures" +)] #[async_trait] pub trait HistoryLoader: Send + Sync + 'static { async fn load(&self, request: &Request, max_bytes: usize) -> Result>; From c9344a1e8fb7919a7cbe3cbdb390cea76713a2c7 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 02:17:06 +0000 Subject: [PATCH 07/16] Run session alignment lanes with supervised pipeline stages --- crates/lance-context-ingestion/README.md | 8 + .../lance-context-ingestion/src/pipeline.rs | 266 ++++++---- .../lance-context-ingestion/tests/pipeline.rs | 462 ++++++++++++++++++ 3 files changed, 654 insertions(+), 82 deletions(-) diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index cc2a200d..52b413d2 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -43,6 +43,14 @@ replayable source / stable partition-local receipts - Immutable skip links support bounded chronological recovery pages and historical receipt lookup without reading unrelated record payloads. Recovery replays all pages after the adapter's checkpoint, not just one next WAL segment. +- `Partition::start_with_aligners` can run several session alignment lanes inside + one stable durable partition, independently of history-loader concurrency and WAL + batch size. Same-session calls stay ordered; completed lanes rejoin the original + sequence before publication. Changing lane count on restart reroutes replayed + session deltas without changing partition identity. An error or cancellation + stops all speculative lanes. Lane count must fit the queue-entry budget; adapter + caches are additional to the shared input/output byte budget. This does not + schedule workers on other machines or remove the WAL ordering barrier. - Named `Consumer`s have independent durable cursors and coalesce producer segments into their own batches. Sink output and input coverage must be committed together; cursor writes can fail after output succeeds, so repeated/regrouped input must be diff --git a/crates/lance-context-ingestion/src/pipeline.rs b/crates/lance-context-ingestion/src/pipeline.rs index a5a18215..fc3707fa 100644 --- a/crates/lance-context-ingestion/src/pipeline.rs +++ b/crates/lance-context-ingestion/src/pipeline.rs @@ -3,8 +3,9 @@ use std::time::Duration; use async_trait::async_trait; use futures::{stream, StreamExt}; +use sha2::{Digest, Sha256}; use tokio::sync::{mpsc, oneshot, watch, OwnedSemaphorePermit, Semaphore}; -use tokio::task::JoinHandle; +use tokio::task::{JoinHandle, JoinSet}; use tokio::time::{timeout_at, Instant}; use crate::journal::digest; @@ -123,9 +124,7 @@ impl Ack { /// to the same partition; changing worker count must not change that mapping. pub struct Partition { input: Option>, - loading: Option>>, - alignment: Option>>, - wal: Option>>, + supervisor: Option>>, budget: Arc, config: PipelineConfig, durable: watch::Receiver, @@ -142,7 +141,21 @@ impl Partition { pub async fn start_with_loader( writer: Writer, - mut aligner: A, + aligner: A, + loader: L, + config: PipelineConfig, + ) -> Result { + Self::start_with_aligners(writer, vec![aligner], loader, config).await + } + + /// Independent session alignment lanes within one durable partition. Every + /// adapter must implement the same per-session transition rules. Routing is + /// session-stable for this process; lanes may change only after shutdown and + /// recovery. Their state/cache allocations are additional to `memory_bytes`. + /// WAL output remains globally ordered even when later sessions finish first. + pub async fn start_with_aligners( + writer: Writer, + mut aligners: Vec, loader: L, config: PipelineConfig, ) -> Result { @@ -152,6 +165,11 @@ impl Partition { config.max_transition_bytes, config.max_history_bytes, )?; + if aligners.is_empty() || aligners.len() > config.queue_entries { + return Err(Error::Invalid( + "alignment lanes must fit the nonempty queue budget".into(), + )); + } if config.queue_entries == 0 || config.load_concurrency == 0 || reserve > config.memory_bytes @@ -161,34 +179,46 @@ impl Partition { )); } let journal = writer.journal().clone(); - let mut checkpoint = aligner.restore(journal.binding()).await?; - while &checkpoint != writer.position() { - for position in journal.pending(&checkpoint, writer.position()).await? { - for entry in journal.entries(&position).await? { - aligner.replay(&entry).await?; + let lanes = aligners.len(); + for (lane, aligner) in aligners.iter_mut().enumerate() { + let mut checkpoint = aligner.restore(journal.binding()).await?; + while &checkpoint != writer.position() { + for position in journal.pending(&checkpoint, writer.position()).await? { + for entry in journal.entries(&position).await? { + if alignment_lane(&entry.session, lanes) == lane { + aligner.replay(&entry).await?; + } + } + checkpoint = position; } - checkpoint = position; } } let (input_tx, input_rx) = mpsc::channel(config.queue_entries); let (loaded_tx, loaded_rx) = mpsc::channel(config.queue_entries); let (wal_tx, wal_rx) = mpsc::channel(config.queue_entries); let (durable_tx, durable_rx) = watch::channel(writer.position().clone()); - let loading = tokio::spawn(load_loop(loader, input_rx, loaded_tx, config.clone())); - let alignment = tokio::spawn(align_loop( - aligner, + let mut stages = JoinSet::new(); + stages.spawn(load_loop(loader, input_rx, loaded_tx, config.clone())); + stages.spawn(align_loop( + aligners, journal, loaded_rx, wal_tx, durable_rx.clone(), config.clone(), )); - let wal = tokio::spawn(wal_loop(writer, wal_rx, durable_tx, config.wal.clone())); + stages.spawn(wal_loop(writer, wal_rx, durable_tx, config.wal.clone())); + let supervisor = tokio::spawn(async move { + // Observe every stage independently. On error, dropping this JoinSet + // cancels even a loader or alignment lane blocked on unrelated work. + while let Some(result) = stages.join_next().await { + result.map_err(|error| Error::Stage(error.to_string()))??; + } + Ok(()) + }); Ok(Self { input: Some(input_tx), - loading: Some(loading), - alignment: Some(alignment), - wal: Some(wal), + supervisor: Some(supervisor), budget: Arc::new(Semaphore::new(config.memory_bytes as usize)), config, durable: durable_rx, @@ -246,42 +276,20 @@ impl Partition { /// head write. Neither path advances checkpoint or merge consumer cursors. pub async fn shutdown(mut self) -> Result<()> { self.input.take(); - let loading = self - .loading + let result = self + .supervisor .as_mut() .unwrap() .await .map_err(|error| Error::Stage(error.to_string()))?; - self.loading.take(); - let alignment = self - .alignment - .as_mut() - .unwrap() - .await - .map_err(|error| Error::Stage(error.to_string()))?; - self.alignment.take(); - let wal = self - .wal - .as_mut() - .unwrap() - .await - .map_err(|error| Error::Stage(error.to_string()))?; - self.wal.take(); - loading?; - alignment?; - wal + self.supervisor.take(); + result } } impl Drop for Partition { fn drop(&mut self) { - if let Some(task) = &self.loading { - task.abort(); - } - if let Some(task) = &self.alignment { - task.abort(); - } - if let Some(task) = &self.wal { + if let Some(task) = &self.supervisor { task.abort(); } } @@ -331,21 +339,74 @@ async fn load_loop( Ok(()) } +struct AlignmentWork { + request: Request, + history: Vec, + input_digest: String, + ack: oneshot::Sender>, + permit: OwnedSemaphorePermit, + result: oneshot::Sender>, +} + +fn alignment_lane(session: &str, lanes: usize) -> usize { + let hash = Sha256::digest(session.as_bytes()); + (u64::from_le_bytes(hash[..8].try_into().expect("SHA-256 prefix")) % lanes as u64) as usize +} + async fn align_loop( - mut aligner: A, + aligners: Vec, journal: Journal, - mut input: mpsc::Receiver, + input: mpsc::Receiver, wal: mpsc::Sender, - mut durable: watch::Receiver, + durable: watch::Receiver, config: PipelineConfig, +) -> Result<()> { + let mut workers = JoinSet::new(); + let mut lanes = Vec::with_capacity(aligners.len()); + for aligner in aligners { + let (sender, receiver) = mpsc::channel(config.queue_entries); + lanes.push(sender); + workers.spawn(alignment_worker_loop(aligner, receiver, config.clone())); + } + let (ordered, mut results) = + mpsc::channel::>>(config.queue_entries); + // JoinSet owns every lane: error, cancellation or dropping this supervisor + // aborts the entire speculative suffix, including other sessions' workers. + tokio::try_join!( + async { + // Supervision must progress independently of ordered output. A later + // lane can fail while an earlier request is still blocked. + while let Some(result) = workers.join_next().await { + result.map_err(|error| Error::Stage(error.to_string()))??; + } + Ok(()) + }, + dispatch_alignment(lanes, journal, input, ordered, durable), + async move { + while let Some(result) = results.recv().await { + let aligned = result.await.map_err(|_| Error::Stopped)??; + wal.send(aligned).await.map_err(|_| Error::Stopped)?; + } + Ok(()) + } + )?; + Ok(()) +} + +async fn dispatch_alignment( + lanes: Vec>, + journal: Journal, + mut input: mpsc::Receiver, + ordered: mpsc::Sender>>, + mut durable: watch::Receiver, ) -> Result<()> { let mut next = durable .borrow() .sequence .checked_add(1) .ok_or_else(|| Error::Invalid("sequence overflow".into()))?; - while let Some(input) = input.recv().await { - let Prepared { input, history } = input; + while let Some(prepared) = input.recv().await { + let Prepared { input, history } = prepared; let Input { request, ack, @@ -377,38 +438,20 @@ async fn align_loop( )))); continue; } - let transition = aligner.align(&request, &history).await?; - if transition - .delta - .capacity() - .saturating_add(transition.records.capacity()) - > config.max_transition_bytes - { - return Err(Error::Invalid( - "aligner exceeded reserved output bytes".into(), - )); - } - let entry = Entry { - sequence: next, - session: request.session, - receipt: request.receipt, - input_digest, - transition, - }; - let encoded_bytes = serde_json::to_vec(&entry)?.len(); - if encoded_bytes > config.wal.max_bytes { - return Err(Error::Invalid( - "single transition exceeds WAL batch limit".into(), - )); - } - wal.send(Aligned { - entry, - encoded_bytes, - ack, - _permit: permit, - }) - .await - .map_err(|_| Error::Stopped)?; + let lane = alignment_lane(&request.session, lanes.len()); + let (result, receiver) = oneshot::channel(); + lanes[lane] + .send(AlignmentWork { + request, + history, + input_digest, + ack, + permit, + result, + }) + .await + .map_err(|_| Error::Stopped)?; + ordered.send(receiver).await.map_err(|_| Error::Stopped)?; next = next .checked_add(1) .ok_or_else(|| Error::Invalid("sequence overflow".into()))?; @@ -416,6 +459,65 @@ async fn align_loop( Ok(()) } +async fn alignment_worker_loop( + mut aligner: A, + mut input: mpsc::Receiver, + config: PipelineConfig, +) -> Result<()> { + while let Some(work) = input.recv().await { + let AlignmentWork { + request, + history, + input_digest, + ack, + permit, + result, + } = work; + let transition = aligner.align(&request, &history).await; + let aligned = transition.and_then(|transition| { + if transition + .delta + .capacity() + .saturating_add(transition.records.capacity()) + > config.max_transition_bytes + { + return Err(Error::Invalid( + "aligner exceeded reserved output bytes".into(), + )); + } + let entry = Entry { + sequence: request.sequence, + session: request.session, + receipt: request.receipt, + input_digest, + transition, + }; + let encoded_bytes = serde_json::to_vec(&entry)?.len(); + if encoded_bytes > config.wal.max_bytes { + return Err(Error::Invalid( + "single transition exceeds WAL batch limit".into(), + )); + } + Ok(Aligned { + entry, + encoded_bytes, + ack, + _permit: permit, + }) + }); + match aligned { + Ok(aligned) => result.send(Ok(aligned)).map_err(|_| Error::Stopped)?, + Err(error) => { + // Report out of order to the supervisor as well as to the + // ordered waiter; the latter may be stuck on an earlier lane. + let _ = result.send(Err(Error::Stage(error.to_string()))); + return Err(error); + } + } + } + Ok(()) +} + async fn wal_loop( mut writer: Writer, mut input: mpsc::Receiver, diff --git a/crates/lance-context-ingestion/tests/pipeline.rs b/crates/lance-context-ingestion/tests/pipeline.rs index aae5dbb6..75de03c4 100644 --- a/crates/lance-context-ingestion/tests/pipeline.rs +++ b/crates/lance-context-ingestion/tests/pipeline.rs @@ -1208,3 +1208,465 @@ async fn corrupt_committed_segment_fails_recovery_and_binding_cannot_change() { .await .is_err()); } + +struct EmptyHistory; + +#[async_trait] +impl HistoryLoader for EmptyHistory { + async fn load(&self, _: &Request, _: usize) -> Result> { + Ok(Vec::new()) + } +} + +struct LaneCounter { + counter: Counter, + slow: Arc, + fast_done: Arc, + failure: Option<(u64, bool)>, + slow_started: Arc, +} + +#[async_trait] +impl Aligner for LaneCounter { + async fn restore(&mut self, binding: &Binding) -> Result { + self.counter.restore(binding).await + } + async fn replay(&mut self, entry: &Entry) -> Result<()> { + self.counter.replay(entry).await + } + async fn align(&mut self, request: &Request, history: &[u8]) -> Result { + if request.session == "slow" { + self.slow_started.add_permits(1); + self.slow.acquire().await.unwrap().forget(); + } + if let Some((sequence, panic)) = self.failure { + if request.sequence == sequence { + assert!(!panic, "injected later lane panic"); + return Err(Error::Stage("injected lane failure".into())); + } + } + let transition = self.counter.align(request, history).await?; + if request.session == "fast" { + self.fast_done.add_permits(1); + } + Ok(transition) + } +} + +fn lane_request(sequence: u64, session: &str) -> Request { + Request { + session: session.into(), + ..request(sequence) + } +} + +#[tokio::test] +async fn independent_alignment_lanes_overlap_but_wal_and_same_session_stay_ordered() { + let journal = journal(Arc::new(InMemory::new()), 0); + let observed = Arc::new(Mutex::new(Vec::new())); + let slow = Arc::new(Semaphore::new(0)); + let fast_done = Arc::new(Semaphore::new(0)); + let aligners = (0..2) + .map(|_| LaneCounter { + counter: Counter::new(observed.clone()), + slow: slow.clone(), + fast_done: fast_done.clone(), + failure: None, + slow_started: Arc::new(Semaphore::new(0)), + }) + .collect(); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + aligners, + EmptyHistory, + config(), + ) + .await + .unwrap(); + let mut acks = Vec::new(); + for (sequence, session) in [(1, "slow"), (2, "fast"), (3, "slow")] { + acks.push( + pipeline + .enqueue(lane_request(sequence, session)) + .await + .unwrap(), + ); + } + timeout(Duration::from_secs(2), fast_done.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + assert_eq!( + *observed.lock().unwrap(), + vec![(2, 1)], + "fast session must execute while earlier slow session is blocked" + ); + assert_eq!( + journal.position().await.unwrap().sequence, + 0, + "later completed lane cannot publish ahead of the first entry" + ); + slow.add_permits(2); + for ack in acks { + timeout(Duration::from_secs(2), ack.wait()) + .await + .unwrap() + .unwrap(); + } + pipeline.shutdown().await.unwrap(); + let mut consumer = Consumer::open(journal.clone(), "check").await.unwrap(); + let mut sink = Collect::new(); + consumer.consume(&mut sink, 100, 16384).await.unwrap(); + { + let rows = sink.entries.lock().unwrap(); + assert_eq!(rows.keys().copied().collect::>(), vec![1, 2, 3]); + assert_eq!( + rows.values() + .map(|v| serde_json::from_slice::(&v.transition.delta).unwrap()) + .collect::>(), + vec![1, 1, 2] + ); + } + // Lane count is process-local: replay routes each durable session to its new + // lane without changing the journal binding or allocating its IDs again. + let observed = Arc::new(Mutex::new(Vec::new())); + let aligners = (0..3).map(|_| Counter::new(observed.clone())).collect(); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + aligners, + EmptyHistory, + config(), + ) + .await + .unwrap(); + pipeline + .enqueue(lane_request(3, "slow")) + .await + .unwrap() + .wait() + .await + .unwrap(); + assert!( + observed.lock().unwrap().is_empty(), + "durable retry must not realign" + ); + pipeline + .enqueue(lane_request(4, "slow")) + .await + .unwrap() + .wait() + .await + .unwrap(); + assert_eq!(*observed.lock().unwrap(), vec![(4, 3)]); + pipeline.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn failed_alignment_lane_discards_later_speculation_and_reopens_cleanly() { + let journal = journal(Arc::new(InMemory::new()), 0); + let slow = Arc::new(Semaphore::new(0)); + let fast_done = Arc::new(Semaphore::new(0)); + let aligners = (0..2) + .map(|_| LaneCounter { + counter: Counter::new(Arc::default()), + slow: slow.clone(), + fast_done: fast_done.clone(), + failure: Some((1, false)), + slow_started: Arc::new(Semaphore::new(0)), + }) + .collect(); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + aligners, + EmptyHistory, + config(), + ) + .await + .unwrap(); + let first = pipeline.enqueue(lane_request(1, "slow")).await.unwrap(); + let second = pipeline.enqueue(lane_request(2, "fast")).await.unwrap(); + timeout(Duration::from_secs(2), fast_done.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + slow.add_permits(1); + assert!(timeout(Duration::from_secs(2), first.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), second.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), pipeline.shutdown()) + .await + .unwrap() + .is_err()); + assert_eq!(journal.position().await.unwrap().sequence, 0); + let observed = Arc::new(Mutex::new(Vec::new())); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + (0..2).map(|_| Counter::new(observed.clone())).collect(), + EmptyHistory, + config(), + ) + .await + .unwrap(); + for (sequence, session) in [(1, "slow"), (2, "fast")] { + pipeline + .enqueue(lane_request(sequence, session)) + .await + .unwrap() + .wait() + .await + .unwrap(); + } + assert_eq!(*observed.lock().unwrap(), vec![(1, 1), (2, 1)]); + pipeline.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn dropping_alignment_supervisor_cancels_blocked_lanes_without_publishing_a_gap() { + let journal = journal(Arc::new(InMemory::new()), 0); + let slow = Arc::new(Semaphore::new(0)); + let fast_done = Arc::new(Semaphore::new(0)); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + (0..2) + .map(|_| LaneCounter { + counter: Counter::new(Arc::default()), + slow: slow.clone(), + fast_done: fast_done.clone(), + failure: None, + slow_started: Arc::new(Semaphore::new(0)), + }) + .collect(), + EmptyHistory, + config(), + ) + .await + .unwrap(); + let first = pipeline.enqueue(lane_request(1, "slow")).await.unwrap(); + let second = pipeline.enqueue(lane_request(2, "fast")).await.unwrap(); + timeout(Duration::from_secs(2), fast_done.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + drop(pipeline); + assert!(timeout(Duration::from_secs(2), first.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), second.wait()) + .await + .unwrap() + .is_err()); + assert_eq!(journal.position().await.unwrap().sequence, 0); +} + +#[tokio::test] +async fn later_lane_error_or_panic_cancels_an_earlier_blocked_request() { + for panic in [false, true] { + let journal = journal(Arc::new(InMemory::new()), 0); + let slow = Arc::new(Semaphore::new(0)); + let slow_started = Arc::new(Semaphore::new(0)); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + (0..2) + .map(|_| LaneCounter { + counter: Counter::new(Arc::default()), + slow: slow.clone(), + slow_started: slow_started.clone(), + fast_done: Arc::new(Semaphore::new(0)), + failure: Some((2, panic)), + }) + .collect(), + EmptyHistory, + config(), + ) + .await + .unwrap(); + let first = pipeline.enqueue(lane_request(1, "slow")).await.unwrap(); + timeout(Duration::from_secs(2), slow_started.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let second = pipeline.enqueue(lane_request(2, "fast")).await.unwrap(); + // Never release the first lane: observing only ordered output would hang. + assert!(timeout(Duration::from_secs(2), first.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), second.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), pipeline.shutdown()) + .await + .unwrap() + .is_err()); + timeout(Duration::from_secs(2), async { + while Arc::strong_count(&slow) != 1 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 0); + let observed = Arc::new(Mutex::new(Vec::new())); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + (0..2).map(|_| Counter::new(observed.clone())).collect(), + EmptyHistory, + config(), + ) + .await + .unwrap(); + for (sequence, session) in [(1, "slow"), (2, "fast")] { + pipeline + .enqueue(lane_request(sequence, session)) + .await + .unwrap() + .wait() + .await + .unwrap(); + } + assert_eq!(*observed.lock().unwrap(), vec![(1, 1), (2, 1)]); + pipeline.shutdown().await.unwrap(); + } +} + +struct FailSecondHistory; + +#[async_trait] +impl HistoryLoader for FailSecondHistory { + async fn load(&self, request: &Request, _: usize) -> Result> { + if request.sequence == 2 { + return Err(Error::Stage("history failed".into())); + } + Ok(Vec::new()) + } +} + +#[tokio::test] +async fn history_failure_also_cancels_an_already_blocked_alignment_lane() { + let journal = journal(Arc::new(InMemory::new()), 0); + let slow = Arc::new(Semaphore::new(0)); + let slow_started = Arc::new(Semaphore::new(0)); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + (0..2) + .map(|_| LaneCounter { + counter: Counter::new(Arc::default()), + slow: slow.clone(), + slow_started: slow_started.clone(), + fast_done: Arc::new(Semaphore::new(0)), + failure: None, + }) + .collect(), + FailSecondHistory, + config(), + ) + .await + .unwrap(); + let first = pipeline.enqueue(lane_request(1, "slow")).await.unwrap(); + timeout(Duration::from_secs(2), slow_started.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let second = pipeline.enqueue(lane_request(2, "fast")).await.unwrap(); + assert!(timeout(Duration::from_secs(2), first.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), second.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), pipeline.shutdown()) + .await + .unwrap() + .is_err()); + timeout(Duration::from_secs(2), async { + while Arc::strong_count(&slow) != 1 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 0); +} + +#[tokio::test] +async fn wal_failure_cancels_a_blocked_alignment_lane_without_publishing_a_gap() { + let journal = Journal::new( + Arc::new(InMemory::new()), + Path::from("small-wal"), + binding(0), + 512, + 100, + ) + .unwrap(); + let slow = Arc::new(Semaphore::new(0)); + let slow_started = Arc::new(Semaphore::new(0)); + let release_first = Arc::new(Semaphore::new(0)); + let mut config = config(); + config.wal.max_entries = 1; + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + (0..2) + .map(|_| { + let mut counter = Counter::new(Arc::default()); + counter.align_gate = Some(release_first.clone()); + LaneCounter { + counter, + slow: slow.clone(), + slow_started: slow_started.clone(), + fast_done: Arc::new(Semaphore::new(0)), + failure: None, + } + }) + .collect(), + EmptyHistory, + config, + ) + .await + .unwrap(); + let mut request = lane_request(1, "fast"); + // Fits admission/alignment, but its encoded WAL segment exceeds 512 bytes. + request.payload = vec![1; 256]; + let first = pipeline.enqueue(request).await.unwrap(); + let second = pipeline.enqueue(lane_request(2, "slow")).await.unwrap(); + timeout(Duration::from_secs(2), slow_started.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + release_first.add_permits(1); + assert!(timeout(Duration::from_secs(2), first.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), second.wait()) + .await + .unwrap() + .is_err()); + assert!(timeout(Duration::from_secs(2), pipeline.shutdown()) + .await + .unwrap() + .is_err()); + timeout(Duration::from_secs(2), async { + while Arc::strong_count(&slow) != 1 { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 0); + assert_eq!(journal.acquire().await.unwrap().position().sequence, 0); +} From 6f6f011bb19d47c91f8139d7c0f09621db334b94 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 02:32:48 +0000 Subject: [PATCH 08/16] Scope async trait generated must-use lint on store registry --- crates/lance-context-core/src/registry.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/crates/lance-context-core/src/registry.rs b/crates/lance-context-core/src/registry.rs index ef2c849d..2c74b247 100644 --- a/crates/lance-context-core/src/registry.rs +++ b/crates/lance-context-core/src/registry.rs @@ -46,6 +46,10 @@ pub struct RegistryEntry { /// /// The name says which stores exist; the dataset on object storage is the /// data. Implementations must be safe to share across tasks (`&self`). +#[allow( + clippy::double_must_use, + reason = "async_trait adds must_use to boxed futures" +)] #[async_trait::async_trait] pub trait StoreRegistry: Send + Sync { /// Whether a store named `name` exists. From 2ff5ee07cf2ba9a122a53f1eacfaa84cff01b066 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 02:37:17 +0000 Subject: [PATCH 09/16] Resolve source receipts through independently indexed WAL history --- crates/lance-context-ingestion/README.md | 8 + crates/lance-context-ingestion/src/lib.rs | 2 + crates/lance-context-ingestion/src/receipt.rs | 229 ++++++++++++++++++ .../lance-context-ingestion/tests/receipts.rs | 180 ++++++++++++++ 4 files changed, 419 insertions(+) create mode 100644 crates/lance-context-ingestion/src/receipt.rs create mode 100644 crates/lance-context-ingestion/tests/receipts.rs diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index 52b413d2..7cdf5ebe 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -55,6 +55,14 @@ replayable source / stable partition-local receipts into their own batches. Sink output and input coverage must be committed together; cursor writes can fail after output succeeds, so repeated/regrouped input must be idempotent. The scheduler owns exclusive consumer assignment and sink-side fencing. +- `ReceiptIndex` resolves exact source receipt/session/input-digest identities after + an HTTP retry or restart. A dedicated consumer writes immutable receipt mappings; + `find_many` reads the index concurrently, then reconciles its unindexed WAL suffix + once for the whole requested batch. It handles index writes ahead of the cursor + and rejects a receipt reused at another sequence. A miss covers only the returned + committed position: the admission owner must still check its in-flight map and + serialize sequence assignment. This index does not schedule source fan-out or + replace the requirement to ACK every partition before advancing source progress. - `Writer::with_backlog` optionally limits committed segments outstanding for every required consumer. A consumer that has not started is at zero; table and checkpoint progress are both required when both are configured. The publisher diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs index 5988386c..7b744d06 100644 --- a/crates/lance-context-ingestion/src/lib.rs +++ b/crates/lance-context-ingestion/src/lib.rs @@ -8,11 +8,13 @@ mod journal; #[cfg(feature = "lance")] pub mod lance_sink; mod pipeline; +mod receipt; pub use checkpoint::{CheckpointSink, RecoveredSession, Reducer, SessionCheckpoints, SessionState}; pub use consumer::{Consumer, Sink}; pub use journal::{BacklogPolicy, Binding, Entry, Journal, Position, Transition, Writer}; pub use pipeline::{Ack, Aligner, BatchPolicy, HistoryLoader, Partition, PipelineConfig, Request}; +pub use receipt::{ReceiptBatchLookup, ReceiptIndex, ReceiptLookup, ReceiptSink, SourceReceipt}; pub type Result = std::result::Result; diff --git a/crates/lance-context-ingestion/src/receipt.rs b/crates/lance-context-ingestion/src/receipt.rs new file mode 100644 index 00000000..2e86e856 --- /dev/null +++ b/crates/lance-context-ingestion/src/receipt.rs @@ -0,0 +1,229 @@ +use std::collections::{HashMap, HashSet}; + +use async_trait::async_trait; +use futures::{stream, StreamExt, TryStreamExt}; +use object_store::PutMode; +use serde::{Deserialize, Serialize}; + +use crate::journal::digest; +use crate::{Binding, Entry, Error, Journal, Position, Result, Sink}; + +/// Exact source identity attached to one committed alignment input. This is +/// separate from an output turn ID: a source may retry a call with many turns. +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +pub struct SourceReceipt { + pub binding: Binding, + pub receipt: String, + pub sequence: u64, + pub session: String, + pub input_digest: String, +} + +impl SourceReceipt { + fn from_entry(binding: &Binding, entry: &Entry) -> Result { + if entry.sequence == 0 + || entry.receipt.is_empty() + || entry.session.is_empty() + || entry.input_digest.is_empty() + { + return Err(Error::Invalid("empty source receipt identity".into())); + } + Ok(Self { + binding: binding.clone(), + receipt: entry.receipt.clone(), + sequence: entry.sequence, + session: entry.session.clone(), + input_digest: entry.input_digest.clone(), + }) + } +} + +/// Lookup reconciles index lag through this exact committed WAL position. +/// A miss does not cover later commits or another request admitted in memory. +#[derive(Debug)] +pub struct ReceiptLookup { + pub receipt: Option, + pub through: Position, +} + +/// Only found receipts are present; all requested identities share one captured +/// committed position. Memory is proportional to the caller-bounded request set. +#[derive(Debug)] +pub struct ReceiptBatchLookup { + pub receipts: HashMap, + pub through: Position, +} + +/// Immutable source-receipt lookup with an independently checkpointed index. +/// Use a dedicated Consumer name and the same index for both `find` and `sink`. +/// This is not an admission lock or sequence allocator: the partition's single +/// admission owner must also reconcile its in-flight requests before assigning +/// a new sequence. A lost HTTP response must retry the same receipt and bytes. +#[derive(Clone)] +pub struct ReceiptIndex { + journal: Journal, +} + +impl ReceiptIndex { + pub fn new(journal: Journal) -> Self { + Self { journal } + } + + fn path(&self, receipt: &str) -> object_store::path::Path { + self.journal + .path(&format!("receipts/{}.json", digest(receipt.as_bytes()))) + } + + async fn read(&self, receipt: &str) -> Result> { + let value: SourceReceipt = match self.journal.read(&self.path(receipt)).await { + Ok((value, _)) => value, + Err(Error::Storage(object_store::Error::NotFound { .. })) => return Ok(None), + Err(error) => return Err(error), + }; + if value.binding != *self.journal.binding() + || value.receipt != receipt + || value.sequence == 0 + || value.session.is_empty() + || value.input_digest.is_empty() + { + return Err(Error::Invalid( + "source receipt index binding mismatch".into(), + )); + } + Ok(Some(value)) + } + + /// Read the index cursor before its entry, then capture the committed head. + /// Replay only the unindexed suffix in bounded pages, including when an + /// index write succeeded but its cursor write failed. A duplicate receipt + /// at a different sequence fails closed. No objects are written here. + pub async fn find(&self, consumer: &str, receipt: &str) -> Result { + let mut found = self.find_many(consumer, &[receipt], 1).await?; + Ok(ReceiptLookup { + receipt: found.receipts.remove(receipt), + through: found.through, + }) + } + + /// Resolve one source batch with bounded concurrent index reads and one + /// shared WAL suffix scan, rather than replaying the tail for every call. + /// `receipts` must be nonempty and unique. The admission owner bounds its + /// total count/bytes and retains any newer speculative receipt mappings. + pub async fn find_many( + &self, + consumer: &str, + receipts: &[&str], + concurrency: usize, + ) -> Result { + let wanted = receipts.iter().copied().collect::>(); + if concurrency == 0 + || receipts.is_empty() + || wanted.contains("") + || wanted.len() != receipts.len() + { + return Err(Error::Invalid( + "empty or duplicate receipt lookup identity/budget".into(), + )); + } + let mut after = self.journal.consumer_position(consumer).await?; + let indexed = stream::iter(receipts.iter().copied()) + .map(|receipt| self.read(receipt)) + .buffer_unordered(concurrency) + .try_collect::>() + .await?; + let mut found = indexed + .into_iter() + .flatten() + .map(|value| (value.receipt.clone(), value)) + .collect::>(); + let through = self.journal.position().await?; + if found + .values() + .any(|value| value.sequence > through.sequence) + { + return Err(Error::Invalid("receipt ahead of committed WAL".into())); + } + loop { + let page = self.journal.pending(&after, &through).await?; + if page.is_empty() { + break; + } + for position in page { + for entry in self.journal.entries(&position).await? { + if wanted.contains(entry.receipt.as_str()) { + let actual = SourceReceipt::from_entry(self.journal.binding(), &entry)?; + if found + .get(&entry.receipt) + .is_some_and(|value| value != &actual) + { + return Err(Error::Invalid("source receipt reused or changed".into())); + } + found.insert(entry.receipt, actual); + } + } + after = position; + } + } + Ok(ReceiptBatchLookup { + receipts: found, + through, + }) + } + + pub fn sink(&self, concurrency: usize) -> Result { + if concurrency == 0 { + return Err(Error::Invalid("zero receipt index concurrency".into())); + } + Ok(ReceiptSink { + index: self.clone(), + concurrency, + }) + } + + async fn insert(&self, receipt: SourceReceipt) -> Result<()> { + match self + .journal + .put(&self.path(&receipt.receipt), &receipt, PutMode::Create) + .await + { + Ok(_) => Ok(()), + Err(Error::Storage(object_store::Error::AlreadyExists { .. })) => { + if self.read(&receipt.receipt).await?.as_ref() == Some(&receipt) { + Ok(()) + } else { + Err(Error::Invalid("source receipt reused or changed".into())) + } + } + Err(error) => Err(error), + } + } +} + +pub struct ReceiptSink { + index: ReceiptIndex, + concurrency: usize, +} + +#[async_trait] +impl Sink for ReceiptSink { + async fn apply(&mut self, binding: &Binding, entries: &[Entry]) -> Result<()> { + if binding != self.index.journal.binding() { + return Err(Error::Invalid("receipt sink binding mismatch".into())); + } + let mut unique = HashMap::new(); + for entry in entries { + let receipt = SourceReceipt::from_entry(binding, entry)?; + if let Some(previous) = unique.insert(entry.receipt.as_str(), receipt.clone()) { + if previous != receipt { + return Err(Error::Invalid("source receipt reused or changed".into())); + } + } + } + stream::iter(unique.into_values()) + .map(|receipt| self.index.insert(receipt)) + .buffer_unordered(self.concurrency) + .try_collect::>() + .await?; + Ok(()) + } +} diff --git a/crates/lance-context-ingestion/tests/receipts.rs b/crates/lance-context-ingestion/tests/receipts.rs new file mode 100644 index 00000000..f2671b27 --- /dev/null +++ b/crates/lance-context-ingestion/tests/receipts.rs @@ -0,0 +1,180 @@ +use std::sync::Arc; + +use lance_context_ingestion::{ + Binding, Consumer, Entry, Journal, Position, ReceiptIndex, Sink, Transition, +}; +use object_store::{memory::InMemory, path::Path, ObjectStoreExt}; + +fn journal(store: Arc) -> Journal { + Journal::new( + store, + Path::from("receipt-test"), + Binding { + run: "run".into(), + schema: "schema".into(), + partition: 0, + }, + 1 << 20, + 1, + ) + .unwrap() +} + +fn entry(sequence: u64, receipt: &str) -> Entry { + Entry { + sequence, + session: "same-session".into(), + receipt: receipt.into(), + input_digest: format!("digest-{sequence}"), + transition: Transition { + delta: vec![1], + records: vec![2], + }, + } +} + +#[tokio::test] +async fn lookup_reconciles_partial_index_and_paged_tail_then_uses_only_indexed_metadata() { + let store = Arc::new(InMemory::new()); + let journal = journal(store.clone()); + let index = ReceiptIndex::new(journal.clone()); + let mut writer = journal.acquire().await.unwrap(); + let first = writer + .append(vec![entry(1, "split-0/batch-0/call-0")]) + .await + .unwrap(); + let last = writer + .append(vec![entry(2, "split-0/batch-0/call-1")]) + .await + .unwrap(); + let lookup = index + .find("receipts", "split-0/batch-0/call-1") + .await + .unwrap(); + assert_eq!(lookup.through, last); + assert_eq!(lookup.receipt.unwrap().sequence, 2); + let requested = ["split-0/batch-0/call-0", "split-0/batch-0/call-1", "absent"]; + let batch = index.find_many("receipts", &requested, 2).await.unwrap(); + assert_eq!(batch.through, last); + assert_eq!(batch.receipts.len(), 2); + assert_eq!(batch.receipts[requested[0]].sequence, 1); + assert_eq!(batch.receipts[requested[1]].sequence, 2); + assert!(index + .find("receipts", "absent") + .await + .unwrap() + .receipt + .is_none()); + + // Simulate sink success before the consumer cursor can be saved. + let mut sink = index.sink(2).unwrap(); + sink.apply(journal.binding(), &[entry(1, "split-0/batch-0/call-0")]) + .await + .unwrap(); + assert_eq!( + journal.consumer_position("receipts").await.unwrap(), + Position::default() + ); + assert_eq!( + index + .find("receipts", "split-0/batch-0/call-0") + .await + .unwrap() + .receipt + .unwrap() + .sequence, + 1 + ); + let mut consumer = Consumer::open(journal.clone(), "receipts").await.unwrap(); + while consumer.consume(&mut sink, 4, 1 << 20).await.unwrap() != 0 {} + assert_eq!(consumer.position(), &last); + // Fully indexed reads must not fetch old payloads. The immutable links remain. + for position in [first, last.clone()] { + store + .delete(&Path::from(format!( + "receipt-test/segments/{}.json", + position.segment.unwrap() + ))) + .await + .unwrap(); + } + for (receipt, sequence) in [("split-0/batch-0/call-0", 1), ("split-0/batch-0/call-1", 2)] { + let lookup = index.find("receipts", receipt).await.unwrap(); + assert_eq!(lookup.through, last); + let found = lookup.receipt.unwrap(); + assert_eq!(found.sequence, sequence); + assert_eq!(found.session, "same-session"); + assert_eq!(found.input_digest, format!("digest-{sequence}")); + } + assert!(index + .find("receipts", "absent") + .await + .unwrap() + .receipt + .is_none()); + assert!(index + .find("wrong-cursor", "split-0/batch-0/call-0") + .await + .is_err()); + let indexed_batch = index.find_many("receipts", &requested, 2).await.unwrap(); + assert_eq!(indexed_batch.through, batch.through); + assert_eq!(indexed_batch.receipts, batch.receipts); +} + +#[tokio::test] +async fn reused_receipt_in_committed_tail_fails_lookup_and_does_not_advance_index() { + let journal = journal(Arc::new(InMemory::new())); + let index = ReceiptIndex::new(journal.clone()); + let mut writer = journal.acquire().await.unwrap(); + let first = writer + .append(vec![entry(1, "source-receipt")]) + .await + .unwrap(); + let mut consumer = Consumer::open(journal.clone(), "receipts").await.unwrap(); + let mut sink = index.sink(1).unwrap(); + consumer.consume(&mut sink, 1, 1 << 20).await.unwrap(); + // A faulty admission owner assigned the same source receipt twice. + writer + .append(vec![entry(2, "source-receipt")]) + .await + .unwrap(); + assert!(index.find("receipts", "source-receipt").await.is_err()); + assert!(consumer.consume(&mut sink, 1, 1 << 20).await.is_err()); + assert_eq!(journal.consumer_position("receipts").await.unwrap(), first); + // Sink retry can never overwrite the original immutable source identity. + sink.apply(journal.binding(), &[entry(1, "source-receipt")]) + .await + .unwrap(); +} + +#[tokio::test] +async fn index_rejects_wrong_binding_empty_identity_and_uncommitted_sequence() { + let journal = journal(Arc::new(InMemory::new())); + journal.acquire().await.unwrap(); + let index = ReceiptIndex::new(journal.clone()); + assert!(index.sink(0).is_err()); + assert!(index.find("receipts", "").await.is_err()); + assert!(index.find_many("receipts", &[], 1).await.is_err()); + assert!(index.find_many("receipts", &["one"], 0).await.is_err()); + assert!(index + .find_many("receipts", &["one", "one"], 2) + .await + .is_err()); + let mut sink = index.sink(1).unwrap(); + let mut wrong = journal.binding().clone(); + wrong.partition = 1; + assert!(sink.apply(&wrong, &[entry(1, "receipt")]).await.is_err()); + assert!(sink + .apply(journal.binding(), &[entry(1, "")]) + .await + .is_err()); + assert!(sink + .apply(journal.binding(), &[entry(0, "receipt")]) + .await + .is_err()); + // Misusing Sink directly cannot make lookup attest to an uncommitted input. + sink.apply(journal.binding(), &[entry(1, "receipt")]) + .await + .unwrap(); + assert!(index.find("receipts", "receipt").await.is_err()); +} From d366a73b064fac380eeeb5be173b7536ee689087 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 02:45:57 +0000 Subject: [PATCH 10/16] Admit continuous source requests with stable receipt recovery --- crates/lance-context-ingestion/README.md | 9 + crates/lance-context-ingestion/src/lib.rs | 4 + .../lance-context-ingestion/src/pipeline.rs | 4 + crates/lance-context-ingestion/src/source.rs | 294 +++++++++++++ .../lance-context-ingestion/tests/source.rs | 410 ++++++++++++++++++ 5 files changed, 721 insertions(+) create mode 100644 crates/lance-context-ingestion/src/source.rs create mode 100644 crates/lance-context-ingestion/tests/source.rs diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index 7cdf5ebe..f798601d 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -63,6 +63,15 @@ replayable source / stable partition-local receipts committed position: the admission owner must still check its in-flight map and serialize sequence assignment. This index does not schedule source fan-out or replace the requirement to ACK every partition before advancing source progress. +- `SourcePartition` adds a serialized admission owner for continuous callers that + have stable source receipts but no partition sequence numbers. `enqueue_many` + checks committed and in-flight identities, assigns sequences only to new inputs, + and returns independent ACK waiters. A repeated in-flight receipt watches the + original commit without blocking later alignment dispatch. Dropped ACK waiters + do not cancel work; cancellation during admission requires recovery of the + unknown admitted prefix. Run its dedicated receipt consumer independently and + keep HTTP/source queues bounded. This API does not supply HTTP authentication, + cross-partition fan-out, or a migration from an application's previous WAL format. - `Writer::with_backlog` optionally limits committed segments outstanding for every required consumer. A consumer that has not started is at zero; table and checkpoint progress are both required when both are configured. The publisher diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs index 7b744d06..a9cf8862 100644 --- a/crates/lance-context-ingestion/src/lib.rs +++ b/crates/lance-context-ingestion/src/lib.rs @@ -9,12 +9,16 @@ mod journal; pub mod lance_sink; mod pipeline; mod receipt; +mod source; pub use checkpoint::{CheckpointSink, RecoveredSession, Reducer, SessionCheckpoints, SessionState}; pub use consumer::{Consumer, Sink}; pub use journal::{BacklogPolicy, Binding, Entry, Journal, Position, Transition, Writer}; pub use pipeline::{Ack, Aligner, BatchPolicy, HistoryLoader, Partition, PipelineConfig, Request}; pub use receipt::{ReceiptBatchLookup, ReceiptIndex, ReceiptLookup, ReceiptSink, SourceReceipt}; +pub use source::{ + SourceAck, SourceCommit, SourceConfig, SourcePartition, SourceRequest, SOURCE_RECEIPT_CONSUMER, +}; pub type Result = std::result::Result; diff --git a/crates/lance-context-ingestion/src/pipeline.rs b/crates/lance-context-ingestion/src/pipeline.rs index fc3707fa..4c122712 100644 --- a/crates/lance-context-ingestion/src/pipeline.rs +++ b/crates/lance-context-ingestion/src/pipeline.rs @@ -271,6 +271,10 @@ impl Partition { self.durable.borrow().clone() } + pub(crate) fn subscribe_durable(&self) -> watch::Receiver { + self.durable.clone() + } + /// Stop admission, drain alignment and flush even a partially filled WAL. /// Dropping Partition instead aborts tasks; recovery reconciles any uncertain /// head write. Neither path advances checkpoint or merge consumer cursors. diff --git a/crates/lance-context-ingestion/src/source.rs b/crates/lance-context-ingestion/src/source.rs new file mode 100644 index 00000000..2ff7045a --- /dev/null +++ b/crates/lance-context-ingestion/src/source.rs @@ -0,0 +1,294 @@ +use std::collections::HashMap; + +use tokio::sync::watch; + +use crate::journal::digest; +use crate::{ + Ack, Aligner, Consumer, Error, HistoryLoader, Partition, PipelineConfig, Position, + ReceiptIndex, Request, Result, Writer, +}; + +/// Dedicated cursor for this admission layer's immutable receipt index. +/// Include it in Writer::with_backlog when limiting unindexed source receipts. +pub const SOURCE_RECEIPT_CONSUMER: &str = "source-receipts"; + +#[derive(Clone, Debug)] +pub struct SourceRequest { + pub session: String, + pub receipt: String, + pub payload: Vec, +} + +#[derive(Clone, Debug)] +pub struct SourceConfig { + pub max_batch_requests: usize, + /// Bounds retained input buffer capacities for one serialized admission call. + /// Pipeline reservations, in-flight receipt metadata, and caller/server + /// queues have separate budgets. This is not a whole-process RSS limit. + pub max_batch_bytes: usize, + pub receipt_read_concurrency: usize, +} + +#[derive(Clone, Debug, PartialEq, Eq)] +struct Identity { + session: String, + digest: String, +} + +struct Pending { + identity: Identity, + sequence: u64, +} + +/// Exact input sequence and a durable prefix that contains it. Source fan-out +/// must receive every partition's ACK before advancing its own checkpoint. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct SourceCommit { + pub sequence: u64, + pub through: Position, +} + +enum Wait { + New(Ack), + Pending(watch::Receiver), + Committed(Position), +} + +/// Dropping a waiter never cancels admitted work. In-flight retries wait on the +/// original durable sequence without entering the alignment dispatch queue. +pub struct SourceAck { + sequence: u64, + wait: Wait, +} + +impl SourceAck { + pub async fn wait(self) -> Result { + let through = match self.wait { + Wait::New(ack) => ack.wait().await?, + Wait::Committed(position) => position, + Wait::Pending(mut durable) => loop { + let position = durable.borrow().clone(); + if position.sequence >= self.sequence { + break position; + } + durable.changed().await.map_err(|_| Error::Stopped)?; + }, + }; + Ok(SourceCommit { + sequence: self.sequence, + through, + }) + } +} + +/// One partition's exclusive source admission owner. Stable receipt identities +/// survive restart; sequence assignment for an uncommitted suffix may change. +/// Serializing admission does not wait for WAL ACKs or index/checkpoint work. +/// The caller supplies bounded HTTP/source queues and owns partition routing. +/// Cancellation during admission poisons this handle: drain/drop and recover +/// before retrying, because an unknown prefix may already have been enqueued. +pub struct SourcePartition { + partition: Partition, + index: ReceiptIndex, + config: SourceConfig, + max_input_bytes: usize, + last_assigned: u64, + pending: HashMap, + poisoned: bool, + journal: crate::Journal, +} + +impl SourcePartition { + pub async fn start( + writer: Writer, + aligners: Vec, + loader: L, + pipeline: PipelineConfig, + config: SourceConfig, + ) -> Result { + if config.max_batch_requests == 0 + || config.max_batch_bytes == 0 + || config.receipt_read_concurrency == 0 + { + return Err(Error::Invalid("zero source admission budget".into())); + } + let journal = writer.journal().clone(); + let last_assigned = writer.position().sequence; + let max_input_bytes = pipeline.max_input_bytes; + let partition = Partition::start_with_aligners(writer, aligners, loader, pipeline).await?; + Ok(Self { + partition, + index: ReceiptIndex::new(journal.clone()), + config, + max_input_bytes, + last_assigned, + pending: HashMap::new(), + poisoned: false, + journal, + }) + } + + pub fn receipt_index(&self) -> &ReceiptIndex { + &self.index + } + + /// Scheduler must run only one consumer for this partition/cursor. It may + /// batch independently of admission and WAL; lookup reconciles any lag. + pub async fn open_receipt_consumer(&self) -> Result { + Consumer::open(self.journal.clone(), SOURCE_RECEIPT_CONSUMER).await + } + + pub fn durable_position(&self) -> Position { + self.partition.durable_position() + } + + pub async fn enqueue_many(&mut self, requests: Vec) -> Result> { + if self.poisoned { + return Err(Error::Fenced); + } + if self.partition.subscribe_durable().has_changed().is_err() { + return Err(Error::Stopped); + } + if requests.is_empty() || requests.len() > self.config.max_batch_requests { + return Err(Error::Invalid("empty or oversized source batch".into())); + } + let mut bytes = requests + .capacity() + .checked_mul(std::mem::size_of::()) + .ok_or_else(|| Error::Invalid("source batch size overflow".into()))?; + let mut identities = HashMap::new(); + for request in &requests { + let size = request + .payload + .capacity() + .checked_add(request.session.capacity()) + .and_then(|n| n.checked_add(request.receipt.capacity())) + .ok_or_else(|| Error::Invalid("source request size overflow".into()))?; + bytes = bytes + .checked_add(size) + .ok_or_else(|| Error::Invalid("source batch size overflow".into()))?; + if request.receipt.is_empty() + || request.session.is_empty() + || size > self.max_input_bytes + || bytes > self.config.max_batch_bytes + { + return Err(Error::Invalid( + "source request exceeds input budget or lacks identity".into(), + )); + } + let identity = Identity { + session: request.session.clone(), + digest: digest(&request.payload), + }; + if let Some(previous) = identities.insert(request.receipt.clone(), identity.clone()) { + if previous != identity { + return Err(Error::Invalid("changed source retry in batch".into())); + } + } + } + // Committed mappings can be dropped: lookup below reconciles the index + // and WAL tail. Keep all speculative ones even if their ACK was dropped. + let durable = self.partition.durable_position().sequence; + self.pending.retain(|_, pending| pending.sequence > durable); + let mut unknown = Vec::new(); + for (receipt, identity) in &identities { + if let Some(pending) = self.pending.get(receipt) { + if &pending.identity != identity { + return Err(Error::Invalid("changed in-flight source retry".into())); + } + } else { + unknown.push(receipt.as_str()); + } + } + let lookup = if unknown.is_empty() { + None + } else { + Some( + self.index + .find_many( + SOURCE_RECEIPT_CONSUMER, + &unknown, + self.config.receipt_read_concurrency, + ) + .await?, + ) + }; + if let Some(found) = &lookup { + if found.through.sequence > self.last_assigned { + self.poisoned = true; + return Err(Error::Fenced); + } + for (receipt, original) in &found.receipts { + let identity = &identities[receipt]; + if identity.session != original.session || identity.digest != original.input_digest + { + return Err(Error::Invalid("changed committed source retry".into())); + } + } + } + let new_count = identities + .keys() + .filter(|receipt| { + !self.pending.contains_key(*receipt) + && !lookup + .as_ref() + .is_some_and(|found| found.receipts.contains_key(*receipt)) + }) + .count(); + self.last_assigned + .checked_add(new_count as u64) + .ok_or_else(|| Error::Invalid("source sequence overflow".into()))?; + + self.poisoned = true; + let mut acks = Vec::with_capacity(requests.len()); + for request in requests { + if let Some(pending) = self.pending.get(&request.receipt) { + acks.push(SourceAck { + sequence: pending.sequence, + wait: Wait::Pending(self.partition.subscribe_durable()), + }); + } else if let Some((original, through)) = lookup.as_ref().and_then(|found| { + found + .receipts + .get(&request.receipt) + .map(|receipt| (receipt, &found.through)) + }) { + acks.push(SourceAck { + sequence: original.sequence, + wait: Wait::Committed(through.clone()), + }); + } else { + self.last_assigned += 1; + let sequence = self.last_assigned; + self.pending.insert( + request.receipt.clone(), + Pending { + sequence, + identity: identities[&request.receipt].clone(), + }, + ); + let ack = self + .partition + .enqueue(Request { + sequence, + session: request.session, + receipt: request.receipt, + payload: request.payload, + }) + .await?; + acks.push(SourceAck { + sequence, + wait: Wait::New(ack), + }); + } + } + self.poisoned = false; + Ok(acks) + } + + /// Drain an admitted prefix, including after a cancelled admission call. + /// This handle cannot be reused; recover from the resulting durable head. + pub async fn shutdown(self) -> Result<()> { + self.partition.shutdown().await + } +} diff --git a/crates/lance-context-ingestion/tests/source.rs b/crates/lance-context-ingestion/tests/source.rs new file mode 100644 index 00000000..9d362779 --- /dev/null +++ b/crates/lance-context-ingestion/tests/source.rs @@ -0,0 +1,410 @@ +use std::{ + collections::HashMap, + sync::{Arc, Mutex}, + time::Duration, +}; + +use async_trait::async_trait; +use lance_context_ingestion::{ + Aligner, BatchPolicy, Binding, Entry, HistoryLoader, Journal, PipelineConfig, Position, + Request, Result, SourceConfig, SourcePartition, SourceRequest, Transition, +}; +use object_store::{memory::InMemory, path::Path}; +use tokio::{sync::Semaphore, time::timeout}; + +struct EmptyHistory; +#[async_trait] +impl HistoryLoader for EmptyHistory { + async fn load(&self, _: &Request, _: usize) -> Result> { + Ok(vec![]) + } +} + +struct Counter { + states: HashMap, + observed: Arc>>, + slow: Arc, + fast_done: Arc, +} +#[async_trait] +impl Aligner for Counter { + async fn restore(&mut self, _: &Binding) -> Result { + Ok(Position::default()) + } + async fn replay(&mut self, entry: &Entry) -> Result<()> { + self.states.insert( + entry.session.clone(), + serde_json::from_slice(&entry.transition.delta)?, + ); + Ok(()) + } + async fn align(&mut self, request: &Request, _: &[u8]) -> Result { + if request.session == "slow" { + self.slow.acquire().await.unwrap().forget(); + } + let count = self.states.entry(request.session.clone()).or_default(); + *count += 1; + self.observed.lock().unwrap().push(request.receipt.clone()); + if request.session == "fast" { + self.fast_done.add_permits(1); + } + Ok(Transition { + delta: serde_json::to_vec(count)?, + records: request.receipt.as_bytes().to_vec(), + }) + } +} + +fn journal() -> Journal { + Journal::new( + Arc::new(InMemory::new()), + Path::from("source"), + Binding { + run: "source-run".into(), + schema: "counter".into(), + partition: 0, + }, + 1 << 20, + 2, + ) + .unwrap() +} + +fn config() -> PipelineConfig { + PipelineConfig { + queue_entries: 8, + load_concurrency: 4, + memory_bytes: 1 << 20, + max_input_bytes: 1024, + max_transition_bytes: 1024, + max_history_bytes: 0, + wal: BatchPolicy { + max_entries: 8, + max_bytes: 16 << 10, + max_delay: Duration::from_millis(2), + }, + } +} + +async fn start( + journal: &Journal, + observed: Arc>>, + slow: Arc, + fast_done: Arc, + config: PipelineConfig, +) -> SourcePartition { + SourcePartition::start( + journal.acquire().await.unwrap(), + (0..2) + .map(|_| Counter { + states: HashMap::new(), + observed: observed.clone(), + slow: slow.clone(), + fast_done: fast_done.clone(), + }) + .collect(), + EmptyHistory, + config, + SourceConfig { + max_batch_requests: 16, + max_batch_bytes: 16 << 10, + receipt_read_concurrency: 2, + }, + ) + .await + .unwrap() +} + +fn request(receipt: &str, session: &str) -> SourceRequest { + SourceRequest { + receipt: receipt.into(), + session: session.into(), + payload: receipt.as_bytes().to_vec(), + } +} + +#[tokio::test] +async fn in_flight_source_retries_do_not_realign_or_block_later_dispatch() { + let journal = journal(); + let observed = Arc::new(Mutex::new(Vec::new())); + let slow = Arc::new(Semaphore::new(0)); + let fast_done = Arc::new(Semaphore::new(0)); + let mut source = start( + &journal, + observed.clone(), + slow.clone(), + fast_done.clone(), + config(), + ) + .await; + let a = request("a", "slow"); + let first = source + .enqueue_many(vec![a.clone(), a.clone(), request("b", "fast")]) + .await + .unwrap(); + timeout(Duration::from_secs(2), fast_done.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + let second = source + .enqueue_many(vec![a.clone(), request("c", "fast")]) + .await + .unwrap(); + timeout(Duration::from_secs(2), fast_done.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + assert_eq!(journal.position().await.unwrap().sequence, 0); + let mut changed = a; + changed.payload.push(9); + assert!(source.enqueue_many(vec![changed]).await.is_err()); + slow.add_permits(1); + let mut sequences = Vec::new(); + for ack in first.into_iter().chain(second) { + let commit = timeout(Duration::from_secs(2), ack.wait()) + .await + .unwrap() + .unwrap(); + assert!(commit.through.sequence >= commit.sequence); + sequences.push(commit.sequence); + } + assert_eq!(sequences, [1, 1, 2, 1, 3]); + source.shutdown().await.unwrap(); + let mut aligned = observed.lock().unwrap().clone(); + aligned.sort(); + assert_eq!(aligned, ["a", "b", "c"]); + assert_eq!(journal.position().await.unwrap().sequence, 3); +} + +#[tokio::test] +async fn lost_response_and_restart_recover_receipts_without_realigning_or_resetting_sequences() { + let journal = journal(); + let observed = Arc::new(Mutex::new(Vec::new())); + let slow = Arc::new(Semaphore::new(0)); + let fast_done = Arc::new(Semaphore::new(0)); + let mut source = start( + &journal, + observed.clone(), + slow.clone(), + fast_done.clone(), + config(), + ) + .await; + let original = request("a", "fast"); + drop(source.enqueue_many(vec![original.clone()]).await.unwrap()); + source.shutdown().await.unwrap(); + let before = journal.position().await.unwrap(); + let mut source = start(&journal, observed.clone(), slow, fast_done, config()).await; + let mut changed = original.clone(); + changed.session = "different-session".into(); + assert!(source.enqueue_many(vec![changed]).await.is_err()); + assert_eq!(journal.position().await.unwrap(), before); + let acks = source + .enqueue_many(vec![original, request("b", "fast")]) + .await + .unwrap(); + let mut sequences = Vec::new(); + for ack in acks { + sequences.push(ack.wait().await.unwrap().sequence); + } + assert_eq!(sequences, [1, 2]); + let mut consumer = source.open_receipt_consumer().await.unwrap(); + let mut sink = source.receipt_index().sink(2).unwrap(); + while consumer.consume(&mut sink, 2, 1 << 20).await.unwrap() != 0 {} + assert_eq!(consumer.position().sequence, 2); + let retry = source + .enqueue_many(vec![request("a", "fast")]) + .await + .unwrap(); + assert_eq!( + retry + .into_iter() + .next() + .unwrap() + .wait() + .await + .unwrap() + .sequence, + 1 + ); + source.shutdown().await.unwrap(); + assert_eq!(*observed.lock().unwrap(), ["a", "b"]); + assert_eq!(journal.position().await.unwrap().sequence, 2); +} + +#[tokio::test] +async fn cancelled_partial_admission_requires_recovery_and_retries_exactly_once() { + let journal = journal(); + let observed = Arc::new(Mutex::new(Vec::new())); + let slow = Arc::new(Semaphore::new(0)); + let fast_done = Arc::new(Semaphore::new(0)); + let mut limited = config(); + limited.memory_bytes = 60_000; + let mut source = start( + &journal, + observed.clone(), + slow.clone(), + fast_done.clone(), + limited, + ) + .await; + let first = request("a", "slow"); + drop(source.enqueue_many(vec![first.clone()]).await.unwrap()); + let later = (0..10) + .map(|n| request(&format!("later-{n}"), "fast")) + .collect::>(); + let mut admission = Box::pin(source.enqueue_many(later.clone())); + timeout(Duration::from_secs(2), async { + tokio::select! { + _ = &mut admission => panic!("admission must wait for bounded capacity"), + ready = fast_done.acquire() => ready.unwrap().forget(), + } + }) + .await + .unwrap(); + assert!(timeout(Duration::from_millis(20), &mut admission) + .await + .is_err()); + drop(admission); + assert!(source.enqueue_many(vec![first.clone()]).await.is_err()); + slow.add_permits(1); + timeout(Duration::from_secs(2), source.shutdown()) + .await + .unwrap() + .unwrap(); + let prefix = journal.position().await.unwrap(); + assert!(prefix.sequence > 1 && prefix.sequence < 11); + let mut source = start(&journal, observed.clone(), slow, fast_done, config()).await; + let acks = source + .enqueue_many(std::iter::once(first).chain(later).collect()) + .await + .unwrap(); + for (n, ack) in acks.into_iter().enumerate() { + assert_eq!(ack.wait().await.unwrap().sequence, n as u64 + 1); + } + source.shutdown().await.unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 11); + let observed = observed.lock().unwrap(); + assert_eq!(observed.len(), 11); + let mut unique = observed.clone(); + unique.sort(); + unique.dedup(); + assert_eq!(unique.len(), 11); +} + +#[tokio::test] +async fn invalid_batch_cannot_allocate_a_sequence_or_poison_valid_admission() { + let journal = journal(); + let mut source = start( + &journal, + Arc::default(), + Arc::new(Semaphore::new(0)), + Arc::new(Semaphore::new(0)), + config(), + ) + .await; + let original = request("a", "fast"); + let mut changed = original.clone(); + changed.payload.push(1); + assert!(source + .enqueue_many(vec![original.clone(), changed]) + .await + .is_err()); + let mut oversized = original.clone(); + oversized.payload = vec![0; 1025]; + assert!(source.enqueue_many(vec![oversized]).await.is_err()); + assert!(source.enqueue_many(Vec::new()).await.is_err()); + assert_eq!(journal.position().await.unwrap().sequence, 0); + let ack = source + .enqueue_many(vec![original]) + .await + .unwrap() + .pop() + .unwrap(); + assert_eq!(ack.wait().await.unwrap().sequence, 1); + source.shutdown().await.unwrap(); +} + +struct Failing { + release: Arc, + fail_wal: bool, +} + +#[async_trait] +impl Aligner for Failing { + async fn restore(&mut self, _: &Binding) -> Result { + Ok(Position::default()) + } + async fn replay(&mut self, _: &Entry) -> Result<()> { + Ok(()) + } + async fn align(&mut self, _: &Request, _: &[u8]) -> Result { + self.release.acquire().await.unwrap().forget(); + if self.fail_wal { + // Fits the transition budget but exceeds the journal object budget. + Ok(Transition { + delta: vec![1], + records: vec![1; 256], + }) + } else { + Err(lance_context_ingestion::Error::Stage( + "injected alignment failure".into(), + )) + } + } +} + +#[tokio::test] +async fn original_and_duplicate_waiters_fail_together_on_alignment_or_wal_error() { + for fail_wal in [false, true] { + let journal = Journal::new( + Arc::new(InMemory::new()), + Path::from("failure"), + Binding { + run: "run".into(), + schema: "counter".into(), + partition: 0, + }, + 512, + 2, + ) + .unwrap(); + let release = Arc::new(Semaphore::new(0)); + let mut source = SourcePartition::start( + journal.acquire().await.unwrap(), + vec![Failing { + release: release.clone(), + fail_wal, + }], + EmptyHistory, + config(), + SourceConfig { + max_batch_requests: 4, + max_batch_bytes: 16 << 10, + receipt_read_concurrency: 2, + }, + ) + .await + .unwrap(); + let input = request("original", "session"); + let acks = source + .enqueue_many(vec![input.clone(), input.clone()]) + .await + .unwrap(); + release.add_permits(1); + for ack in acks { + assert!(timeout(Duration::from_secs(2), ack.wait()) + .await + .unwrap() + .is_err()); + } + assert!(source.enqueue_many(vec![input]).await.is_err()); + assert!(timeout(Duration::from_secs(2), source.shutdown()) + .await + .unwrap() + .is_err()); + assert_eq!(journal.position().await.unwrap().sequence, 0); + } +} From 4474e4756eb8b32f35820bfd3e4c8a8704b678a0 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 03:36:28 +0000 Subject: [PATCH 11/16] Keep batched source receipt lookup Send across task boundaries --- crates/lance-context-ingestion/src/receipt.rs | 11 ++++++++--- .../lance-context-ingestion/tests/receipts.rs | 19 +++++++++++++++++++ 2 files changed, 27 insertions(+), 3 deletions(-) diff --git a/crates/lance-context-ingestion/src/receipt.rs b/crates/lance-context-ingestion/src/receipt.rs index 2e86e856..cb001076 100644 --- a/crates/lance-context-ingestion/src/receipt.rs +++ b/crates/lance-context-ingestion/src/receipt.rs @@ -1,7 +1,7 @@ use std::collections::{HashMap, HashSet}; use async_trait::async_trait; -use futures::{stream, StreamExt, TryStreamExt}; +use futures::{future::BoxFuture, stream, FutureExt, StreamExt, TryStreamExt}; use object_store::PutMode; use serde::{Deserialize, Serialize}; @@ -126,8 +126,13 @@ impl ReceiptIndex { )); } let mut after = self.journal.consumer_position(consumer).await?; - let indexed = stream::iter(receipts.iter().copied()) - .map(|receipt| self.read(receipt)) + // Erase borrowed lookup futures before the buffered stream so callers + // can await batch admission inside a Send task on a multithread runtime. + let reads: Vec>>> = receipts + .iter() + .map(|receipt| self.read(receipt).boxed()) + .collect(); + let indexed = stream::iter(reads) .buffer_unordered(concurrency) .try_collect::>() .await?; diff --git a/crates/lance-context-ingestion/tests/receipts.rs b/crates/lance-context-ingestion/tests/receipts.rs index f2671b27..ac7be278 100644 --- a/crates/lance-context-ingestion/tests/receipts.rs +++ b/crates/lance-context-ingestion/tests/receipts.rs @@ -178,3 +178,22 @@ async fn index_rejects_wrong_binding_empty_identity_and_uncommitted_sequence() { .unwrap(); assert!(index.find("receipts", "receipt").await.is_err()); } + +#[tokio::test] +async fn batch_lookup_runs_in_a_send_task_with_borrowed_dynamic_identities() { + let journal = journal(Arc::new(InMemory::new())); + let mut writer = journal.acquire().await.unwrap(); + let last = writer.append(vec![entry(1, "present")]).await.unwrap(); + let index = ReceiptIndex::new(journal); + let found = tokio::spawn(async move { + let owned = ["present".to_owned(), "missing".to_owned()]; + let identities = owned.iter().map(String::as_str).collect::>(); + index.find_many("receipts", &identities, 2).await + }) + .await + .unwrap() + .unwrap(); + assert_eq!(found.through, last); + assert_eq!(found.receipts.len(), 1); + assert_eq!(found.receipts["present"].sequence, 1); +} From 9860fb07414ce66a27d5f3b8378ff8b016039b15 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 04:01:43 +0000 Subject: [PATCH 12/16] Allow pipeline memory pools above four GiB without serializing admission --- .../lance-context-ingestion/src/pipeline.rs | 25 ++++-- .../lance-context-ingestion/tests/pipeline.rs | 85 ++++++++++++++++++- 2 files changed, 101 insertions(+), 9 deletions(-) diff --git a/crates/lance-context-ingestion/src/pipeline.rs b/crates/lance-context-ingestion/src/pipeline.rs index 4c122712..8888f43a 100644 --- a/crates/lance-context-ingestion/src/pipeline.rs +++ b/crates/lance-context-ingestion/src/pipeline.rs @@ -85,13 +85,25 @@ pub struct PipelineConfig { pub load_concurrency: usize, /// Bounds reserved input/output/serialization bytes across both queues and /// active batches. Adapter state, runtime and object-store buffers are extra. - pub memory_bytes: u32, + pub memory_bytes: usize, pub max_input_bytes: usize, pub max_transition_bytes: usize, pub max_history_bytes: usize, pub wal: BatchPolicy, } +impl PipelineConfig { + /// Conservative byte reservation per admitted entry, held through its WAL + /// commit. Total capacity can exceed u32; each semaphore acquisition cannot. + pub fn reservation_bytes(&self) -> Result { + reservation( + self.max_input_bytes, + self.max_transition_bytes, + self.max_history_bytes, + ) + } +} + struct Input { request: Request, ack: oneshot::Sender>, @@ -160,11 +172,7 @@ impl Partition { config: PipelineConfig, ) -> Result { config.wal.validate()?; - let reserve = reservation( - config.max_input_bytes, - config.max_transition_bytes, - config.max_history_bytes, - )?; + let reserve = config.reservation_bytes()?; if aligners.is_empty() || aligners.len() > config.queue_entries { return Err(Error::Invalid( "alignment lanes must fit the nonempty queue budget".into(), @@ -172,7 +180,8 @@ impl Partition { } if config.queue_entries == 0 || config.load_concurrency == 0 - || reserve > config.memory_bytes + || reserve as usize > config.memory_bytes + || config.memory_bytes > Semaphore::MAX_PERMITS { return Err(Error::Invalid( "queue empty or maximum request exceeds memory budget".into(), @@ -219,7 +228,7 @@ impl Partition { Ok(Self { input: Some(input_tx), supervisor: Some(supervisor), - budget: Arc::new(Semaphore::new(config.memory_bytes as usize)), + budget: Arc::new(Semaphore::new(config.memory_bytes)), config, durable: durable_rx, }) diff --git a/crates/lance-context-ingestion/tests/pipeline.rs b/crates/lance-context-ingestion/tests/pipeline.rs index 75de03c4..cf01cec1 100644 --- a/crates/lance-context-ingestion/tests/pipeline.rs +++ b/crates/lance-context-ingestion/tests/pipeline.rs @@ -1128,7 +1128,7 @@ async fn consumer_uncertain_apply_replays_without_skipping_or_duplicating_output async fn byte_admission_is_bounded_and_sparse_wal_flushes_on_timer() { let mut cfg = config(); // Exactly one maximum-sized input/output reservation fits. - cfg.memory_bytes = ((cfg.max_input_bytes + cfg.max_transition_bytes) * 16 + 4096) as u32; + cfg.memory_bytes = (cfg.max_input_bytes + cfg.max_transition_bytes) * 16 + 4096; cfg.wal.max_delay = Duration::from_millis(50); let gate = Arc::new(Semaphore::new(0)); let mut aligner = Counter::new(Arc::default()); @@ -1670,3 +1670,86 @@ async fn wal_failure_cancels_a_blocked_alignment_lane_without_publishing_a_gap() assert_eq!(journal.position().await.unwrap().sequence, 0); assert_eq!(journal.acquire().await.unwrap().position().sequence, 0); } + +#[cfg(target_pointer_width = "64")] +#[tokio::test] +async fn budget_over_four_gib_runs_two_large_reservations_and_bounds_the_third() { + let journal = journal(Arc::new(InMemory::new()), 0); + let observed = Arc::new(Mutex::new(Vec::new())); + let slow = Arc::new(Semaphore::new(0)); + let fast_done = Arc::new(Semaphore::new(0)); + let aligners = (0..2) + .map(|_| LaneCounter { + counter: Counter::new(observed.clone()), + slow: slow.clone(), + fast_done: fast_done.clone(), + failure: None, + slow_started: Arc::new(Semaphore::new(0)), + }) + .collect(); + let mut cfg = config(); + cfg.max_input_bytes = 16 << 20; + cfg.max_transition_bytes = 32 << 20; + cfg.max_history_bytes = 97 << 20; + cfg.memory_bytes = 6144 << 20; + assert!(cfg.reservation_bytes().unwrap() > (2 << 30)); + assert_eq!( + cfg.memory_bytes / cfg.reservation_bytes().unwrap() as usize, + 2 + ); + // Semaphore reservations are accounting, not allocations of these GiBs. + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + aligners, + EmptyHistory, + cfg, + ) + .await + .unwrap(); + let first = pipeline.enqueue(lane_request(1, "slow")).await.unwrap(); + let second = timeout( + Duration::from_secs(2), + pipeline.enqueue(lane_request(2, "fast")), + ) + .await + .unwrap() + .unwrap(); + timeout(Duration::from_secs(2), fast_done.acquire()) + .await + .unwrap() + .unwrap() + .forget(); + assert_eq!(*observed.lock().unwrap(), vec![(2, 1)]); + assert_eq!(journal.position().await.unwrap().sequence, 0); + let mut third = Box::pin(pipeline.enqueue(lane_request(3, "fast"))); + assert!(timeout(Duration::from_millis(30), &mut third) + .await + .is_err()); + slow.add_permits(1); + let third = timeout(Duration::from_secs(2), third) + .await + .unwrap() + .unwrap(); + for ack in [first, second, third] { + timeout(Duration::from_secs(2), ack.wait()) + .await + .unwrap() + .unwrap(); + } + pipeline.shutdown().await.unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 3); +} + +#[tokio::test] +async fn unsupported_semaphore_capacity_is_rejected_without_panicking() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut cfg = config(); + cfg.memory_bytes = usize::MAX; + assert!(Partition::start( + journal.acquire().await.unwrap(), + Counter::new(Arc::default()), + cfg + ) + .await + .is_err()); +} From 669fc9fed301d3769a4ebea3fb8cec4d2627dc28 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 04:32:06 +0000 Subject: [PATCH 13/16] Release aligned input and history reservations before WAL publication --- .../lance-context-ingestion/src/pipeline.rs | 33 +++++- .../lance-context-ingestion/tests/pipeline.rs | 112 +++++++++++++++++- 2 files changed, 135 insertions(+), 10 deletions(-) diff --git a/crates/lance-context-ingestion/src/pipeline.rs b/crates/lance-context-ingestion/src/pipeline.rs index 8888f43a..1c22ceeb 100644 --- a/crates/lance-context-ingestion/src/pipeline.rs +++ b/crates/lance-context-ingestion/src/pipeline.rs @@ -93,8 +93,9 @@ pub struct PipelineConfig { } impl PipelineConfig { - /// Conservative byte reservation per admitted entry, held through its WAL - /// commit. Total capacity can exceed u32; each semaphore acquisition cannot. + /// Maximum byte reservation for loading/alignment. Completed alignment retains + /// only its output reservation through WAL commit. Total capacity can exceed + /// u32; each semaphore acquisition cannot. pub fn reservation_bytes(&self) -> Result { reservation( self.max_input_bytes, @@ -483,10 +484,12 @@ async fn alignment_worker_loop( history, input_digest, ack, - permit, + mut permit, result, } = work; let transition = aligner.align(&request, &history).await; + drop(history); + drop(request.payload); let aligned = transition.and_then(|transition| { if transition .delta @@ -511,11 +514,33 @@ async fn alignment_worker_loop( "single transition exceeds WAL batch limit".into(), )); } + // Input/history are gone. Keep conservative output/serialization + // capacity charged until durable ACK, but let the next alignment run + // while this entry waits for batching or object-store publication. + // The fixed envelope allowance includes the 64-byte input digest. + let retained = reservation( + entry + .session + .capacity() + .checked_add(entry.receipt.capacity()) + .ok_or_else(|| Error::Invalid("retained input size overflow".into()))?, + entry + .transition + .delta + .capacity() + .checked_add(entry.transition.records.capacity()) + .ok_or_else(|| Error::Invalid("retained output size overflow".into()))?, + 0, + )?; + let output_permit = permit.split(retained as usize).ok_or_else(|| { + Error::Invalid("retained output exceeds admission reservation".into()) + })?; + drop(permit); Ok(Aligned { entry, encoded_bytes, ack, - _permit: permit, + _permit: output_permit, }) }); match aligned { diff --git a/crates/lance-context-ingestion/tests/pipeline.rs b/crates/lance-context-ingestion/tests/pipeline.rs index cf01cec1..59d6df15 100644 --- a/crates/lance-context-ingestion/tests/pipeline.rs +++ b/crates/lance-context-ingestion/tests/pipeline.rs @@ -1678,9 +1678,14 @@ async fn budget_over_four_gib_runs_two_large_reservations_and_bounds_the_third() let observed = Arc::new(Mutex::new(Vec::new())); let slow = Arc::new(Semaphore::new(0)); let fast_done = Arc::new(Semaphore::new(0)); + let alignment_gate = Arc::new(Semaphore::new(0)); let aligners = (0..2) .map(|_| LaneCounter { - counter: Counter::new(observed.clone()), + counter: { + let mut counter = Counter::new(observed.clone()); + counter.align_gate = Some(alignment_gate.clone()); + counter + }, slow: slow.clone(), fast_done: fast_done.clone(), failure: None, @@ -1714,6 +1719,12 @@ async fn budget_over_four_gib_runs_two_large_reservations_and_bounds_the_third() .await .unwrap() .unwrap(); + // Both active alignments retain their full reservations: a third cannot enter. + let mut third = Box::pin(pipeline.enqueue(lane_request(3, "fast"))); + assert!(timeout(Duration::from_millis(30), &mut third) + .await + .is_err()); + alignment_gate.add_permits(1); timeout(Duration::from_secs(2), fast_done.acquire()) .await .unwrap() @@ -1721,15 +1732,15 @@ async fn budget_over_four_gib_runs_two_large_reservations_and_bounds_the_third() .forget(); assert_eq!(*observed.lock().unwrap(), vec![(2, 1)]); assert_eq!(journal.position().await.unwrap().sequence, 0); - let mut third = Box::pin(pipeline.enqueue(lane_request(3, "fast"))); - assert!(timeout(Duration::from_millis(30), &mut third) - .await - .is_err()); - slow.add_permits(1); + // Finished out-of-order output keeps its smaller reservation. The third + // alignment can enter even while the first blocks ordered WAL publication. let third = timeout(Duration::from_secs(2), third) .await .unwrap() .unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 0); + slow.add_permits(1); + alignment_gate.add_permits(2); for ack in [first, second, third] { timeout(Duration::from_secs(2), ack.wait()) .await @@ -1753,3 +1764,92 @@ async fn unsupported_semaphore_capacity_is_rejected_without_panicking() { .await .is_err()); } + +#[tokio::test] +async fn completed_alignment_releases_history_budget_before_a_batched_durable_ack() { + let journal = journal(Arc::new(InMemory::new()), 0); + let mut cfg = config(); + cfg.max_history_bytes = 8 << 20; + cfg.memory_bytes = 200 << 20; + assert_eq!( + cfg.memory_bytes / cfg.reservation_bytes().unwrap() as usize, + 1 + ); + cfg.wal.max_entries = 3; + // The test must fill one batch; waiting for a sparse flush cannot pass. + cfg.wal.max_delay = Duration::from_secs(60); + let pipeline = Partition::start_with_aligners( + journal.acquire().await.unwrap(), + vec![Counter::new(Arc::default())], + EmptyHistory, + cfg, + ) + .await + .unwrap(); + let mut acks = Vec::new(); + for sequence in 1..=3 { + acks.push( + timeout(Duration::from_secs(2), pipeline.enqueue(request(sequence))) + .await + .unwrap() + .unwrap(), + ); + } + for ack in acks { + let position = timeout(Duration::from_secs(2), ack.wait()) + .await + .unwrap() + .unwrap(); + assert_eq!(position.sequence, 3); + assert_eq!(position.generation, 1); + } + pipeline.shutdown().await.unwrap(); + let mut sink = Collect::new(); + let mut consumer = Consumer::open(journal.clone(), "table").await.unwrap(); + assert_eq!(consumer.consume(&mut sink, 1, 16384).await.unwrap(), 3); + assert_eq!(sink.entries.lock().unwrap().len(), 3); + assert_eq!(journal.position().await.unwrap().generation, 1); +} + +#[tokio::test] +async fn completed_output_stays_charged_until_wal_durability() { + let journal = journal(Arc::new(InMemory::new()), 0); + let observed = Arc::new(Mutex::new(Vec::new())); + let mut cfg = config(); + cfg.max_transition_bytes = 2048; + cfg.memory_bytes = cfg.reservation_bytes().unwrap() as usize; + cfg.wal.max_delay = Duration::from_secs(60); + let pipeline = Partition::start( + journal.acquire().await.unwrap(), + Counter::new(observed.clone()), + cfg, + ) + .await + .unwrap(); + let mut first_request = request(1); + first_request.payload = vec![b'x'; 900]; + let mut second_request = request(2); + second_request.payload = vec![b'y'; 900]; + let first = pipeline.enqueue(first_request).await.unwrap(); + timeout(Duration::from_secs(2), async { + while observed.lock().unwrap().is_empty() { + tokio::task::yield_now().await; + } + }) + .await + .unwrap(); + // The first transition's retained capacity prevents another large input + // from entering, even though its history/input loading reservation is gone. + assert!( + timeout(Duration::from_millis(30), pipeline.enqueue(second_request)) + .await + .is_err() + ); + assert_eq!(journal.position().await.unwrap().sequence, 0); + timeout(Duration::from_secs(2), pipeline.shutdown()) + .await + .unwrap() + .unwrap(); + assert_eq!(first.wait().await.unwrap().sequence, 1); + assert_eq!(journal.position().await.unwrap().sequence, 1); +} From 9d61c1f6ed8153dc87580e18c2037d32d23ddc57 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 06:19:05 +0000 Subject: [PATCH 14/16] Preserve bulk source batches through WAL publication --- crates/lance-context-ingestion/README.md | 44 ++- crates/lance-context-ingestion/src/lib.rs | 4 +- .../lance-context-ingestion/src/pipeline.rs | 133 ++++++++- crates/lance-context-ingestion/src/source.rs | 43 ++- .../lance-context-ingestion/tests/source.rs | 276 ++++++++++++++++++ 5 files changed, 489 insertions(+), 11 deletions(-) diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index f798601d..e7edac07 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -27,22 +27,28 @@ replayable source / stable partition-local receipts publication. Different virtual partitions run independently. Within a partition, load completion may reorder, while alignment and durable publication stay ordered. Do not change session-to-virtual-partition routing when changing worker count. + - `enqueue` reserves bytes before admission. A reservation remains held through the durable ACK, covering input, history, bounded output and serialization. Adapter caches, transient adapter allocation, executor and storage-client memory need their own budgets. Sources must also bound concurrent requests waiting for admission. + - `Aligner` owns the session cache and speculative state. Prefetched checkpoints may be stale: reconcile their revisions with pending deltas. On failure, discard the adapter and recover; speculative changes must never become checkpoints directly. + - `Journal` stores immutable segments binding records, state deltas, exact input digests, receipt identities, run/schema and predecessor sequence. Conditional head updates fence stale writers. Failed or cancelled commits poison a writer. + - ACK follows head publication. An uploaded orphan is not committed. Retry the same partition sequence, receipt, session and input bytes after an uncertain result. A gap is rejected; older receipts are compared with the committed journal. + - Immutable skip links support bounded chronological recovery pages and historical receipt lookup without reading unrelated record payloads. Recovery replays all pages after the adapter's checkpoint, not just one next WAL segment. + - `Partition::start_with_aligners` can run several session alignment lanes inside one stable durable partition, independently of history-loader concurrency and WAL batch size. Same-session calls stay ordered; completed lanes rejoin the original @@ -51,10 +57,12 @@ replayable source / stable partition-local receipts stops all speculative lanes. Lane count must fit the queue-entry budget; adapter caches are additional to the shared input/output byte budget. This does not schedule workers on other machines or remove the WAL ordering barrier. + - Named `Consumer`s have independent durable cursors and coalesce producer segments into their own batches. Sink output and input coverage must be committed together; cursor writes can fail after output succeeds, so repeated/regrouped input must be idempotent. The scheduler owns exclusive consumer assignment and sink-side fencing. + - `ReceiptIndex` resolves exact source receipt/session/input-digest identities after an HTTP retry or restart. A dedicated consumer writes immutable receipt mappings; `find_many` reads the index concurrently, then reconciles its unindexed WAL suffix @@ -63,6 +71,7 @@ replayable source / stable partition-local receipts committed position: the admission owner must still check its in-flight map and serialize sequence assignment. This index does not schedule source fan-out or replace the requirement to ACK every partition before advancing source progress. + - `SourcePartition` adds a serialized admission owner for continuous callers that have stable source receipts but no partition sequence numbers. `enqueue_many` checks committed and in-flight identities, assigns sequences only to new inputs, @@ -72,6 +81,7 @@ replayable source / stable partition-local receipts unknown admitted prefix. Run its dedicated receipt consumer independently and keep HTTP/source queues bounded. This API does not supply HTTP authentication, cross-partition fan-out, or a migration from an application's previous WAL format. + - `Writer::with_backlog` optionally limits committed segments outstanding for every required consumer. A consumer that has not started is at zero; table and checkpoint progress are both required when both are configured. The publisher @@ -83,6 +93,7 @@ replayable source / stable partition-local receipts not itself a throughput optimization. It bounds unconsumed payload bytes by `max_segments * max_segment_bytes`, not retained history, orphan uploads or total storage. No WAL garbage collection or scheduling is implied. + - `SessionCheckpoints` provides an actual object-store checkpoint sink: group by session, reduce ordered deltas, then write each session once with a conditional put. A partially successful checkpoint batch can leave some session states ahead of the @@ -119,19 +130,44 @@ Tests use the real `object_store::memory::InMemory` implementation for CAS behav with fault injection around writes. These tests do not establish cloud durability, process-crash behavior on a real durable service, or production throughput. +## Bulk sources + +For replayable batch sources, use `SourcePartition::start_batched` with a fresh +`BatchFlush` shared by the partition's alignment adapters. Pass the existing source +batch to `enqueue_many`; do not replace its stable record receipts when combining +batches for transport or WAL publication. The admission owner validates the entire +batch, preserves in-flight deduplication, and requests a flush through its final +assigned sequence. An all-duplicate batch does not manufacture a WAL entry. + +This mode does not use `BatchPolicy::max_delay`. WAL collection continues until a +source batch boundary, the byte/count limit, output memory headroom, or shutdown. +A later already-admitted batch boundary can coalesce available batches. Limits may +split a large batch into committed prefixes: rows, state deltas and source receipts +remain together in each segment, and callers await **all** returned ACKs before +advancing source progress. Table/checkpoint consumers still group segments +independently. The bulk constructor changes scheduling, not the WAL format. + +Same-session alignment sees its speculative state throughout the batch. If an +adapter evicts uncommitted state and needs to recover it, call +`BatchFlush::request_prefix(sequence)` before waiting for that prefix's durability. +This flushes available ordered outputs even when a later batch boundary is pending; +otherwise the blocked alignment could prevent the batch itself from completing. +Do not treat a stale checkpoint as the current batch's state. Cache/input/output +budgets still apply; increasing a batch target does not permit unbounded memory. + ## Remaining integration 1. Adapt the existing revisioned session history/cache and cross-call alignment implementation. Preserve its compaction/branch identity rules and source ordering; the generic pipeline deliberately does not invent a new turn-ID algorithm. -2. Wire table ownership and independently scheduled session ZoneMap maintenance. +1. Wire table ownership and independently scheduled session ZoneMap maintenance. Staging is independent of publication; a distributed worker transport must carry validated staged results and retain ownership through manifest commit. -3. Add source fan-out receipts and contiguous source progress. A source call spanning +1. Add source fan-out receipts and contiguous source progress. A source call spanning partitions is complete only after every required partition ACK. -4. Add worker ownership orchestration, stage timing/queue telemetry, consumer run loops, +1. Add worker ownership orchestration, stage timing/queue telemetry, consumer run loops, deployment of the backlog policy and safe WAL reclamation. No WAL files are deleted here. -5. Verify real compacted sessions, process crash/restart with durable storage, Lance +1. Verify real compacted sessions, process crash/restart with durable storage, Lance uncertain commits and source retry integration before a guarded production handoff. Existing source-reader local audit history must survive that handoff. diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs index a9cf8862..dcbd6776 100644 --- a/crates/lance-context-ingestion/src/lib.rs +++ b/crates/lance-context-ingestion/src/lib.rs @@ -14,7 +14,9 @@ mod source; pub use checkpoint::{CheckpointSink, RecoveredSession, Reducer, SessionCheckpoints, SessionState}; pub use consumer::{Consumer, Sink}; pub use journal::{BacklogPolicy, Binding, Entry, Journal, Position, Transition, Writer}; -pub use pipeline::{Ack, Aligner, BatchPolicy, HistoryLoader, Partition, PipelineConfig, Request}; +pub use pipeline::{ + Ack, Aligner, BatchFlush, BatchPolicy, HistoryLoader, Partition, PipelineConfig, Request, +}; pub use receipt::{ReceiptBatchLookup, ReceiptIndex, ReceiptLookup, ReceiptSink, SourceReceipt}; pub use source::{ SourceAck, SourceCommit, SourceConfig, SourcePartition, SourceRequest, SOURCE_RECEIPT_CONSUMER, diff --git a/crates/lance-context-ingestion/src/pipeline.rs b/crates/lance-context-ingestion/src/pipeline.rs index 1c22ceeb..be802799 100644 --- a/crates/lance-context-ingestion/src/pipeline.rs +++ b/crates/lance-context-ingestion/src/pipeline.rs @@ -79,6 +79,60 @@ impl BatchPolicy { } } +/// Flush requests for one batch partition. A request commits the ordered prefix +/// through `sequence`; it never acknowledges uncommitted data. Create a fresh +/// handle per partition process. Adapters use `request_prefix` before awaiting +/// recovery of evicted speculative state, avoiding a batch-boundary deadlock. +#[derive(Clone)] +pub struct BatchFlush { + through: watch::Sender, +} + +#[derive(Clone, Copy, Default)] +struct FlushTargets { + batch_end: u64, + required_prefix: u64, +} + +impl Default for BatchFlush { + fn default() -> Self { + Self::new() + } +} + +impl BatchFlush { + pub fn new() -> Self { + Self { + through: watch::channel(FlushTargets::default()).0, + } + } + + pub fn request(&self, sequence: u64) { + self.through.send_if_modified(|through| { + if sequence > through.batch_end { + through.batch_end = sequence; + true + } else { + false + } + }); + } + + /// Commit available outputs until this prefix is durable. Use before waiting + /// for evicted speculative state. Unlike an ordinary batch boundary, this + /// must not wait for later alignment that depends on the requested prefix. + pub fn request_prefix(&self, sequence: u64) { + self.through.send_if_modified(|through| { + if sequence > through.required_prefix { + through.required_prefix = sequence; + true + } else { + false + } + }); + } +} + #[derive(Clone, Debug)] pub struct PipelineConfig { pub queue_entries: usize, @@ -141,6 +195,7 @@ pub struct Partition { budget: Arc, config: PipelineConfig, durable: watch::Receiver, + _batch_flush: Option, } impl Partition { @@ -167,10 +222,32 @@ impl Partition { /// recovery. Their state/cache allocations are additional to `memory_bytes`. /// WAL output remains globally ordered even when later sessions finish first. pub async fn start_with_aligners( + writer: Writer, + aligners: Vec, + loader: L, + config: PipelineConfig, + ) -> Result { + Self::start_inner(writer, aligners, loader, config, None).await + } + + /// Bulk input flushes at caller batch boundaries, size/count limits, memory + /// headroom, or shutdown. `wal.max_delay` is not used in this mode. + pub async fn start_batched_with_aligners( + writer: Writer, + aligners: Vec, + loader: L, + config: PipelineConfig, + flush: BatchFlush, + ) -> Result { + Self::start_inner(writer, aligners, loader, config, Some(flush)).await + } + + async fn start_inner( writer: Writer, mut aligners: Vec, loader: L, config: PipelineConfig, + flush: Option, ) -> Result { config.wal.validate()?; let reserve = config.reservation_bytes()?; @@ -217,7 +294,14 @@ impl Partition { durable_rx.clone(), config.clone(), )); - stages.spawn(wal_loop(writer, wal_rx, durable_tx, config.wal.clone())); + stages.spawn(wal_loop( + writer, + wal_rx, + durable_tx, + config.wal.clone(), + flush.as_ref().map(|f| f.through.subscribe()), + config.memory_bytes - reserve as usize, + )); let supervisor = tokio::spawn(async move { // Observe every stage independently. On error, dropping this JoinSet // cancels even a loader or alignment lane blocked on unrelated work. @@ -232,6 +316,7 @@ impl Partition { budget: Arc::new(Semaphore::new(config.memory_bytes)), config, durable: durable_rx, + _batch_flush: flush, }) } @@ -561,6 +646,8 @@ async fn wal_loop( mut input: mpsc::Receiver, durable: watch::Sender, policy: BatchPolicy, + mut flush: Option>, + output_headroom: usize, ) -> Result<()> { let mut carry: Option = None; loop { @@ -573,18 +660,54 @@ async fn wal_loop( }; let deadline = Instant::now() + policy.max_delay; let mut bytes = first.encoded_bytes; + let mut reserved = first._permit.num_permits(); let mut batch = vec![first]; while batch.len() < policy.max_entries && bytes < policy.max_bytes { - match timeout_at(deadline, input.recv()).await { - Ok(Some(next)) if next.encoded_bytes <= policy.max_bytes - bytes => { + let next = if let Some(flush) = flush.as_mut() { + // Keep room to admit a maximum-sized request while collecting + // outputs. Otherwise a partial batch could hold all permits and + // prevent the source from ever reaching its explicit boundary. + let requested = *flush.borrow_and_update(); + if reserved >= output_headroom + || (requested.batch_end > writer.position().sequence + && batch.last().unwrap().entry.sequence >= requested.batch_end) + { + break; + } + if requested.required_prefix > writer.position().sequence { + // An evicted state can block an earlier alignment result. + // Drain outputs already available, then commit that prefix + // instead of waiting for the whole source batch to finish. + match input.try_recv() { + Ok(item) => Some(item), + Err(_) => break, + } + } else { + tokio::select! { + item = input.recv() => item, + changed = flush.changed() => { + if changed.is_err() { return Err(Error::Stopped); } + continue; + } + } + } + } else { + match timeout_at(deadline, input.recv()).await { + Ok(next) => next, + Err(_) => break, + } + }; + match next { + Some(next) if next.encoded_bytes <= policy.max_bytes - bytes => { bytes += next.encoded_bytes; + reserved += next._permit.num_permits(); batch.push(next); } - Ok(Some(next)) => { + Some(next) => { carry = Some(next); break; } - Ok(None) | Err(_) => break, + None => break, } } let entries = batch diff --git a/crates/lance-context-ingestion/src/source.rs b/crates/lance-context-ingestion/src/source.rs index 2ff7045a..97ea2fdd 100644 --- a/crates/lance-context-ingestion/src/source.rs +++ b/crates/lance-context-ingestion/src/source.rs @@ -95,6 +95,7 @@ pub struct SourcePartition { last_assigned: u64, pending: HashMap, poisoned: bool, + batch_flush: Option, journal: crate::Journal, } @@ -105,6 +106,31 @@ impl SourcePartition { loader: L, pipeline: PipelineConfig, config: SourceConfig, + ) -> Result { + Self::start_inner(writer, aligners, loader, pipeline, config, None).await + } + + /// Preserve a source batch through alignment and WAL collection. Multiple + /// calls from one session see speculative state in admission order. A batch + /// may split at byte/count/memory limits; every returned ACK remains durable. + pub async fn start_batched( + writer: Writer, + aligners: Vec, + loader: L, + pipeline: PipelineConfig, + config: SourceConfig, + flush: crate::BatchFlush, + ) -> Result { + Self::start_inner(writer, aligners, loader, pipeline, config, Some(flush)).await + } + + async fn start_inner( + writer: Writer, + aligners: Vec, + loader: L, + pipeline: PipelineConfig, + config: SourceConfig, + batch_flush: Option, ) -> Result { if config.max_batch_requests == 0 || config.max_batch_bytes == 0 @@ -115,7 +141,18 @@ impl SourcePartition { let journal = writer.journal().clone(); let last_assigned = writer.position().sequence; let max_input_bytes = pipeline.max_input_bytes; - let partition = Partition::start_with_aligners(writer, aligners, loader, pipeline).await?; + let partition = if let Some(flush) = &batch_flush { + Partition::start_batched_with_aligners( + writer, + aligners, + loader, + pipeline, + flush.clone(), + ) + .await? + } else { + Partition::start_with_aligners(writer, aligners, loader, pipeline).await? + }; Ok(Self { partition, index: ReceiptIndex::new(journal.clone()), @@ -124,6 +161,7 @@ impl SourcePartition { last_assigned, pending: HashMap::new(), poisoned: false, + batch_flush, journal, }) } @@ -282,6 +320,9 @@ impl SourcePartition { }); } } + if let Some(flush) = &self.batch_flush { + flush.request(self.last_assigned); + } self.poisoned = false; Ok(acks) } diff --git a/crates/lance-context-ingestion/tests/source.rs b/crates/lance-context-ingestion/tests/source.rs index 9d362779..e13601cb 100644 --- a/crates/lance-context-ingestion/tests/source.rs +++ b/crates/lance-context-ingestion/tests/source.rs @@ -408,3 +408,279 @@ async fn original_and_duplicate_waiters_fail_together_on_alignment_or_wal_error( assert_eq!(journal.position().await.unwrap().sequence, 0); } } + +async fn start_bulk( + journal: &Journal, + observed: Arc>>, + slow: Arc, + fast_done: Arc, + pipeline: PipelineConfig, +) -> SourcePartition { + SourcePartition::start_batched( + journal.acquire().await.unwrap(), + (0..2) + .map(|_| Counter { + states: HashMap::new(), + observed: observed.clone(), + slow: slow.clone(), + fast_done: fast_done.clone(), + }) + .collect(), + EmptyHistory, + pipeline, + SourceConfig { + max_batch_requests: 256, + max_batch_bytes: 64 << 10, + receipt_read_concurrency: 2, + }, + lance_context_ingestion::BatchFlush::new(), + ) + .await + .unwrap() +} + +#[tokio::test] +async fn bulk_batch_preserves_pending_dedup_and_restarts_without_checkpoint_or_realigning() { + let journal = journal(); + let observed = Arc::new(Mutex::new(Vec::new())); + let slow = Arc::new(Semaphore::new(0)); + let fast = Arc::new(Semaphore::new(0)); + let mut pipeline = config(); + pipeline.wal.max_entries = 128; + let mut source = start_bulk( + &journal, + observed.clone(), + slow.clone(), + fast.clone(), + pipeline.clone(), + ) + .await; + let mut requests: Vec<_> = (0..39) + .map(|i| request(&format!("batch-{i}"), "fast")) + .collect(); + requests.push(request("batch-last", "slow")); + requests.push(request("batch-0", "fast")); // Duplicate within the same uncommitted batch. + let mut acks = source.enqueue_many(requests.clone()).await.unwrap(); + timeout(Duration::from_secs(5), fast.acquire_many(39)) + .await + .unwrap() + .unwrap() + .forget(); + let first = acks.remove(0).wait(); + tokio::pin!(first); + // The 2ms streaming timer must not split this batch while its final call is + // still aligning. All earlier calls have actually finished alignment. + assert!(timeout(Duration::from_millis(30), &mut first) + .await + .is_err()); + assert_eq!(journal.position().await.unwrap(), Position::default()); + slow.add_permits(1); + timeout(Duration::from_secs(5), &mut first) + .await + .unwrap() + .unwrap(); + for ack in acks { + ack.wait().await.unwrap(); + } + let head = journal.position().await.unwrap(); + assert_eq!((head.sequence, head.generation), (40, 1)); + let entries = journal.entries(&head).await.unwrap(); + assert_eq!(entries.len(), 40); + for (index, entry) in entries[..39].iter().enumerate() { + assert_eq!( + serde_json::from_slice::(&entry.transition.delta).unwrap(), + index as u64 + 1 + ); + } + // Stop after a durable ACK without checkpoint/receipt consumers. Recovery + // must use the committed WAL, including same-batch source identity mappings. + drop(source); + let mut source = start_bulk(&journal, observed.clone(), slow, fast, pipeline).await; + for ack in source.enqueue_many(requests).await.unwrap() { + ack.wait().await.unwrap(); + } + assert_eq!(journal.position().await.unwrap(), head); + assert_eq!(observed.lock().unwrap().len(), 40); + let mut changed = request("batch-0", "fast"); + changed.payload = b"different contents".to_vec(); + assert!(source.enqueue_many(vec![changed]).await.is_err()); + for ack in source + .enqueue_many(vec![request("next", "fast")]) + .await + .unwrap() + { + ack.wait().await.unwrap(); + } + let last = journal.position().await.unwrap(); + assert_eq!(last.sequence, 41); + assert_eq!( + serde_json::from_slice::(&journal.entries(&last).await.unwrap()[0].transition.delta) + .unwrap(), + 40 + ); + source.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn bulk_batch_splits_at_size_or_memory_headroom_without_waiting_for_a_timer() { + for limit in ["entries", "bytes", "memory"] { + let journal = journal(); + let mut pipeline = config(); + pipeline.wal.max_delay = Duration::from_secs(3600); + pipeline.wal.max_entries = if limit == "entries" { 7 } else { 128 }; + if limit == "memory" { + pipeline.memory_bytes = pipeline.reservation_bytes().unwrap() as usize; + } + if limit == "bytes" { + pipeline.wal.max_bytes = 1024; + } + let mut source = start_bulk( + &journal, + Arc::new(Mutex::new(Vec::new())), + Arc::new(Semaphore::new(0)), + Arc::new(Semaphore::new(0)), + pipeline, + ) + .await; + let requests = (0..40) + .map(|i| request(&format!("row-{i}"), "fast")) + .collect(); + timeout(Duration::from_secs(5), async { + for ack in source.enqueue_many(requests).await.unwrap() { + ack.wait().await.unwrap(); + } + }) + .await + .unwrap(); + let head = journal.position().await.unwrap(); + assert_eq!(head.sequence, 40); + assert!(head.generation > 1); + if limit == "entries" { + assert_eq!(head.generation, 6); + } + let mut cursor = Position::default(); + let mut entries = Vec::new(); + while cursor != head { + for position in journal.pending(&cursor, &head).await.unwrap() { + entries.extend(journal.entries(&position).await.unwrap()); + cursor = position; + } + } + assert_eq!(entries.len(), 40); + assert!(entries + .iter() + .enumerate() + .all(|(i, e)| e.sequence == i as u64 + 1)); + source.shutdown().await.unwrap(); + } +} + +#[tokio::test] +async fn bulk_recovery_prefix_is_not_hidden_by_a_later_batch_boundary() { + let journal = journal(); + let slow = Arc::new(Semaphore::new(0)); + let fast = Arc::new(Semaphore::new(0)); + let flush = lance_context_ingestion::BatchFlush::new(); + let mut source = SourcePartition::start_batched( + journal.acquire().await.unwrap(), + vec![Counter { + states: HashMap::new(), + observed: Arc::new(Mutex::new(Vec::new())), + slow: slow.clone(), + fast_done: fast.clone(), + }], + EmptyHistory, + config(), + SourceConfig { + max_batch_requests: 16, + max_batch_bytes: 16384, + receipt_read_concurrency: 2, + }, + flush.clone(), + ) + .await + .unwrap(); + let mut acks = source + .enqueue_many(vec![request("first", "fast"), request("last", "slow")]) + .await + .unwrap(); + fast.acquire().await.unwrap().forget(); + // enqueue_many already requested the full batch (sequence 2). The pending + // alignment needs sequence 1 committed before it can recover evicted state. + flush.request_prefix(1); + timeout(Duration::from_secs(2), acks.remove(0).wait()) + .await + .unwrap() + .unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 1); + let last = acks.remove(0).wait(); + tokio::pin!(last); + assert!(timeout(Duration::from_millis(20), &mut last).await.is_err()); + slow.add_permits(1); + timeout(Duration::from_secs(2), &mut last) + .await + .unwrap() + .unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 2); + source.shutdown().await.unwrap(); +} + +#[tokio::test] +async fn bulk_cancelled_partial_admission_recovers_without_duplicate_rows() { + let journal = journal(); + let observed = Arc::new(Mutex::new(Vec::new())); + let slow = Arc::new(Semaphore::new(0)); + let fast_done = Arc::new(Semaphore::new(0)); + let mut limited = config(); + limited.memory_bytes = 60_000; + let mut source = start_bulk( + &journal, + observed.clone(), + slow.clone(), + fast_done.clone(), + limited, + ) + .await; + let first = request("a", "slow"); + drop(source.enqueue_many(vec![first.clone()]).await.unwrap()); + let later = (0..10) + .map(|n| request(&format!("later-{n}"), "fast")) + .collect::>(); + let mut admission = Box::pin(source.enqueue_many(later.clone())); + timeout(Duration::from_secs(2), async { + tokio::select! { + _ = &mut admission => panic!("admission must wait for bounded capacity"), + ready = fast_done.acquire() => ready.unwrap().forget(), + } + }) + .await + .unwrap(); + assert!(timeout(Duration::from_millis(20), &mut admission) + .await + .is_err()); + drop(admission); + assert!(source.enqueue_many(vec![first.clone()]).await.is_err()); + slow.add_permits(1); + timeout(Duration::from_secs(2), source.shutdown()) + .await + .unwrap() + .unwrap(); + let prefix = journal.position().await.unwrap(); + assert!(prefix.sequence > 1 && prefix.sequence < 11); + let mut source = start_bulk(&journal, observed.clone(), slow, fast_done, config()).await; + let acks = source + .enqueue_many(std::iter::once(first).chain(later).collect()) + .await + .unwrap(); + for (n, ack) in acks.into_iter().enumerate() { + assert_eq!(ack.wait().await.unwrap().sequence, n as u64 + 1); + } + source.shutdown().await.unwrap(); + assert_eq!(journal.position().await.unwrap().sequence, 11); + let observed = observed.lock().unwrap(); + assert_eq!(observed.len(), 11); + let mut unique = observed.clone(); + unique.sort(); + unique.dedup(); + assert_eq!(unique.len(), 11); +} From 7e3112ddc162196c2bd29d7fa12979c62ca0489f Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 20:24:38 +0000 Subject: [PATCH 15/16] Persist local alignment cells and typed output in atomic Lance batches --- Cargo.lock | 1 + crates/lance-context-ingestion/Cargo.toml | 4 +- crates/lance-context-ingestion/README.md | 27 + .../lance-context-ingestion/src/lance_sink.rs | 60 +- crates/lance-context-ingestion/src/lib.rs | 2 + .../src/local_lance.rs | 1043 +++++++++++++++++ .../tests/local_lance.rs | 545 +++++++++ 7 files changed, 1680 insertions(+), 2 deletions(-) create mode 100644 crates/lance-context-ingestion/src/local_lance.rs create mode 100644 crates/lance-context-ingestion/tests/local_lance.rs diff --git a/Cargo.lock b/Cargo.lock index 968ded52..a597c136 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3922,6 +3922,7 @@ dependencies = [ "lance-io", "lance-table", "object_store", + "redb", "serde", "serde_json", "sha2 0.10.9", diff --git a/crates/lance-context-ingestion/Cargo.toml b/crates/lance-context-ingestion/Cargo.toml index 46c30962..f037351d 100644 --- a/crates/lance-context-ingestion/Cargo.toml +++ b/crates/lance-context-ingestion/Cargo.toml @@ -7,7 +7,7 @@ description = "Bounded session-ordered ingestion with recoverable WAL and indepe [features] default = [] -lance = ["dep:arrow-array", "dep:arrow-ipc", "dep:arrow-schema", "dep:lance", "dep:lance-file", "dep:lance-index", "dep:lance-io", "dep:lance-table"] +lance = ["dep:arrow-array", "dep:arrow-ipc", "dep:arrow-schema", "dep:lance", "dep:lance-file", "dep:lance-index", "dep:lance-io", "dep:lance-table", "dep:redb", "dep:tempfile"] [dependencies] async-trait = "0.1" @@ -21,9 +21,11 @@ lance-io = { version = "9.0.0", optional = true } lance-table = { version = "9.0.0", optional = true } futures = "0.3" object_store = "0.13.2" +redb = { version = "3.1.3", optional = true } serde = { version = "1", features = ["derive"] } serde_json = "1" sha2 = "0.10" +tempfile = { version = "3", optional = true } thiserror = "2" tokio = { version = "1", features = ["macros", "rt-multi-thread", "sync", "time"] } uuid = { version = "1", features = ["v4", "v5"] } diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index e7edac07..85959f5c 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -178,3 +178,30 @@ CARGO_TARGET_DIR=/tmp/trace-streaming-target cargo test -p lance-context-ingesti CARGO_TARGET_DIR=/tmp/trace-streaming-target cargo clippy -p lance-context-ingestion --features lance --all-targets --offline -- -D warnings cargo fmt -p lance-context-ingestion --check ``` + +# Local state and a Lance recovery log + +The `local_lance` module (`lance` feature) stores individual binary state cells +and exact receipts in a local redb database. This is a disposable local index; +it does not require a remote KV request for each message. Session routing and +the durable partition lease remain the caller's responsibility. + +An `AlignedCall` contains typed output rows and the matching state mutations. +`LocalLancePartition::commit` coalesces calls into one Lance 2.2 commit with Zstd +requested on its columns, then applies one local transaction. A cancelled or +failed commit fences the instance. Reopen and check the original receipts before +realigning or acknowledging a retry. The exact-version commit fence prevents a +stale writer from acknowledging another writer's output at the same sequence. + +`checkpoint` streams the local tables in bounded batches to a separate immutable +Lance dataset. `publish_checkpoint` pins its exact URI/version in the WAL's +manifest under the same lease/version fence. Do this on a byte/time threshold, +not once per source call. Cold restart discovers that pointer, restores state, +and reads only new immutable WAL fragments. Recovery projects state and receipt +columns, excluding output bodies. No method deletes WAL or checkpoint history. + +The first output schema supports flat Arrow columns. State cell encoding and +legacy migration are adapter contracts; this API does not convert JSON payloads +supplied by a caller. This is an opt-in backend: existing `Journal` and +`SessionCheckpoints` adapters keep their existing storage and recovery behavior. +Migrate their committed suffix and receipts before changing the intake log. diff --git a/crates/lance-context-ingestion/src/lance_sink.rs b/crates/lance-context-ingestion/src/lance_sink.rs index dc71677b..904a0d63 100644 --- a/crates/lance-context-ingestion/src/lance_sink.rs +++ b/crates/lance-context-ingestion/src/lance_sink.rs @@ -11,7 +11,7 @@ use arrow_schema::Schema; use async_trait::async_trait; use futures::stream::BoxStream; use lance::dataset::{ - transaction::{Operation, Transaction}, + transaction::{Operation, Transaction, UpdateMap}, CommitBuilder, Dataset, InsertBuilder, WriteMode, WriteParams, }; use lance::index::DatasetIndexExt; @@ -267,6 +267,46 @@ impl LanceTableSink { &self.dataset } + /// Publish control pointers under the same lease and exact-version fence as + /// data. This merges table metadata without changing record schema or rows. + pub async fn update_metadata_at( + &mut self, + updates: HashMap, + base: u64, + ) -> Result<()> { + if self.poisoned { + return Err(Error::Fenced); + } + self.poisoned = true; + self.dataset.checkout_latest().await.map_err(failure)?; + if self.dataset.version().version != base { + return Err(Error::Fenced); + } + let operation = Operation::UpdateConfig { + config_updates: None, + table_metadata_updates: Some(UpdateMap { + update_entries: updates.into_iter().map(Into::into).collect(), + replace: false, + }), + schema_metadata_updates: None, + field_metadata_updates: HashMap::new(), + }; + let handler = Arc::new(PinnedCommit { + delegate: self.handler.clone(), + next_version: base + .checked_add(1) + .ok_or_else(|| Error::Invalid("table version overflow".into()))?, + }); + self.dataset = CommitBuilder::new(Arc::new(self.dataset.clone())) + .with_commit_handler(handler) + .with_max_retries(0) + .execute(Transaction::new(base, operation, None)) + .await + .map_err(failure)?; + self.poisoned = false; + Ok(()) + } + pub async fn covered_sequence(&mut self, binding: &Binding) -> Result { self.dataset.checkout_latest().await.map_err(failure)?; validate_dataset(&self.dataset, binding)?; @@ -278,10 +318,28 @@ impl LanceTableSink { } pub async fn commit_staged(&mut self, staged: Vec) -> Result { + self.commit_staged_inner(staged, None).await + } + + /// Commit only against the version used to compute local state changes. + /// A changed base requires recovery and realignment, never an append retry. + pub async fn commit_staged_at(&mut self, staged: Vec, version: u64) -> Result { + self.commit_staged_inner(staged, Some(version)).await + } + + async fn commit_staged_inner( + &mut self, + staged: Vec, + expected_version: Option, + ) -> Result { if self.poisoned { return Err(Error::Fenced); } self.dataset.checkout_latest().await.map_err(failure)?; + if expected_version.is_some_and(|version| version != self.dataset.version().version) { + self.poisoned = true; + return Err(Error::Fenced); + } let marks = watermarks(&self.dataset).await?; let mut seen = HashSet::new(); let mut fragments = Vec::new(); diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs index dcbd6776..3ffbd9b4 100644 --- a/crates/lance-context-ingestion/src/lib.rs +++ b/crates/lance-context-ingestion/src/lib.rs @@ -7,6 +7,8 @@ mod consumer; mod journal; #[cfg(feature = "lance")] pub mod lance_sink; +#[cfg(feature = "lance")] +pub mod local_lance; mod pipeline; mod receipt; mod source; diff --git a/crates/lance-context-ingestion/src/local_lance.rs b/crates/lance-context-ingestion/src/local_lance.rs new file mode 100644 index 00000000..23cc2149 --- /dev/null +++ b/crates/lance-context-ingestion/src/local_lance.rs @@ -0,0 +1,1043 @@ +//! Local alignment state backed by a partition's append-only Lance log. +//! The caller owns session routing and the durable partition lease. Local files +//! are disposable: only a successful Lance commit establishes durability. Supply +//! the lease-checking commit handler when creating/opening the log and this API. +use std::{path::Path, sync::Arc}; + +use arrow_array::{ + new_null_array, Array, ArrayRef, LargeBinaryArray, RecordBatch, RecordBatchIterator, + StringArray, StructArray, UInt64Array, UInt8Array, +}; +use arrow_schema::{DataType, Field, Schema}; +use futures::TryStreamExt; +use lance::{dataset::WriteParams, Dataset}; +use lance_file::version::LanceFileVersion; +use lance_table::io::commit::CommitHandler; +use redb::{Database, ReadableDatabase, ReadableTable, TableDefinition}; + +use crate::{ + lance_sink::{ + encode_records, stage, table_schema, LanceTableSink, RUN_METADATA, SCHEMA_METADATA, + }, + Binding, Entry, Error, Result, Transition, +}; + +const STATE: TableDefinition<&[u8], &[u8]> = TableDefinition::new("state"); +const RECEIPTS: TableDefinition<&[u8], &[u8]> = TableDefinition::new("receipts"); +const META: TableDefinition<&str, &[u8]> = TableDefinition::new("meta"); +const PARTITION: &str = "lance-context.ingestion.partition"; +const FORMAT: &str = "lance-context.ingestion.local-state-format"; +const CHECKPOINT_FORMAT: &str = "lance-context.ingestion.checkpoint-format"; +const CHECKPOINT_URI: &str = "lance-context.ingestion.checkpoint-wal-uri"; +const CHECKPOINT_VERSION: &str = "lance-context.ingestion.checkpoint-wal-version"; +const CHECKPOINT_SEQUENCE: &str = "lance-context.ingestion.checkpoint-sequence"; +const PUBLISHED_CHECKPOINT_URI: &str = "lance-context.ingestion.checkpoint-uri"; +const PUBLISHED_CHECKPOINT_VERSION: &str = "lance-context.ingestion.checkpoint-version"; +const COLUMNS: [&str; 8] = [ + "sequence", + "ordinal", + "kind", + "session", + "receipt", + "input_digest", + "key", + "value", +]; +const RECEIPT: u8 = 0; +const PUT: u8 = 1; +const DELETE: u8 = 2; +const RECORD: u8 = 3; + +fn fail(error: impl std::fmt::Display) -> Error { + Error::Stage(error.to_string()) +} + +/// Values are individual binary state cells, not serialized session objects. +/// The adapter defines key/value encoding and must version it in Binding.schema. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct Mutation { + pub key: Vec, + pub value: Option>, +} + +/// One already aligned call. Mutations, typed output rows, and the receipt are +/// one durable unit. The final receipt row seals this sequence during recovery. +pub struct AlignedCall { + pub sequence: u64, + pub session: String, + pub receipt: String, + pub input_digest: String, + pub mutations: Vec, + pub records: RecordBatch, +} + +/// Binary tuple framing preserves arbitrary session/key bytes without delimiters +/// or changing the adapter's existing identity hashes. +fn local_key(session: &str, key: &[u8]) -> Vec { + let mut result = Vec::with_capacity(8 + session.len() + key.len()); + result.extend_from_slice(&(session.len() as u64).to_be_bytes()); + result.extend_from_slice(session.as_bytes()); + result.extend_from_slice(key); + result +} + +fn receipt_value(sequence: u64, input_digest: &str) -> Vec { + let mut value = sequence.to_le_bytes().to_vec(); + value.extend_from_slice(input_digest.as_bytes()); + value +} + +/// Typed record columns remain columnar inside the record struct. Recovery +/// projects only the eight state/receipt columns, excluding message bodies. +pub fn log_schema(records: &Schema, binding: &Binding) -> Schema { + let mut fields = vec![ + Field::new("sequence", DataType::UInt64, false), + Field::new("ordinal", DataType::UInt64, false), + Field::new("kind", DataType::UInt8, false), + Field::new("session", DataType::Utf8, false), + Field::new("receipt", DataType::Utf8, false), + Field::new("input_digest", DataType::Utf8, false), + Field::new("key", DataType::LargeBinary, false), + Field::new("value", DataType::LargeBinary, false), + ]; + let records = table_schema(records, &binding.run, &binding.schema); + let record_fields = records + .fields + .iter() + .map(|field| { + let mut metadata = field.metadata().clone(); + metadata.insert( + "lance-context.ingestion.output-nullable".into(), + field.is_nullable().to_string(), + ); + Arc::new( + field + .as_ref() + .clone() + .with_nullable(true) + .with_metadata(metadata), + ) + }) + .collect::>(); + fields.push(Field::new( + "record", + DataType::Struct(record_fields.into()), + true, + )); + let mut schema = table_schema(&Schema::new(fields), &binding.run, &binding.schema); + schema + .metadata + .insert(PARTITION.into(), binding.partition.to_string()); + schema.metadata.insert(FORMAT.into(), "1".into()); + schema +} + +fn event_batch( + schema: Arc, + call: &AlignedCall, + first_ordinal: u64, + kinds: Vec, + keys: Vec<&[u8]>, + values: Vec<&[u8]>, + records: Option, +) -> Result { + let n = kinds.len(); + let record = records.map_or_else( + || new_null_array(schema.field(8).data_type(), n), + |batch| Arc::new(StructArray::from(batch)) as ArrayRef, + ); + RecordBatch::try_new( + schema, + vec![ + Arc::new(UInt64Array::from(vec![call.sequence; n])), + Arc::new(UInt64Array::from_iter_values( + first_ordinal..first_ordinal + n as u64, + )), + Arc::new(UInt8Array::from(kinds)), + Arc::new(StringArray::from(vec![call.session.as_str(); n])), + Arc::new(StringArray::from(vec![call.receipt.as_str(); n])), + Arc::new(StringArray::from(vec![call.input_digest.as_str(); n])), + Arc::new(LargeBinaryArray::from(keys)), + Arc::new(LargeBinaryArray::from(values)), + record, + ], + ) + .map_err(fail) +} + +fn encode_call(schema: Arc, call: &AlignedCall, batch_rows: usize) -> Result> { + let mut batches = Vec::new(); + let mut ordinal = 0; + for mutations in call.mutations.chunks(batch_rows) { + batches.push(event_batch( + schema.clone(), + call, + ordinal, + mutations + .iter() + .map(|m| if m.value.is_some() { PUT } else { DELETE }) + .collect(), + mutations.iter().map(|m| m.key.as_slice()).collect(), + mutations + .iter() + .map(|m| m.value.as_deref().unwrap_or_default()) + .collect(), + None, + )?); + ordinal += mutations.len() as u64; + } + for start in (0..call.records.num_rows()).step_by(batch_rows) { + let n = batch_rows.min(call.records.num_rows() - start); + // Match compression metadata on the durable child fields, not just types. + let DataType::Struct(fields) = schema.field(8).data_type() else { + unreachable!() + }; + let records = RecordBatch::try_new( + Arc::new(Schema::new(fields.clone())), + call.records.slice(start, n).columns().to_vec(), + ) + .map_err(fail)?; + batches.push(event_batch( + schema.clone(), + call, + ordinal, + vec![RECORD; n], + vec![&[]; n], + vec![&[]; n], + Some(records), + )?); + ordinal += n as u64; + } + batches.push(event_batch( + schema, + call, + ordinal, + vec![RECEIPT], + vec![&[]], + vec![&[]], + None, + )?); + encode_records(&batches) +} + +/// Open only on a blocking worker/runtime suitable for local database I/O. +/// The cache limit bounds redb's page cache, not Arrow or alignment allocations. +/// A failed/cancelled commit fences this instance, including uncertain success. +pub struct LocalLancePartition { + db: Database, + sink: LanceTableSink, + binding: Binding, + through: u64, + version: u64, + max_batch_bytes: usize, + batch_rows: usize, + poisoned: bool, +} + +impl LocalLancePartition { + /// Rebuild missing local state or replay its suffix from the supplied latest + /// committed log. The caller must hold the partition lease; this never steals + /// ownership or accepts a dataset whose history was compacted/deleted. + pub async fn open( + local_file: &Path, + dataset: Dataset, + binding: Binding, + handler: Arc, + cache_bytes: usize, + max_batch_bytes: usize, + batch_rows: usize, + ) -> Result { + let schema = Schema::from(dataset.schema()); + if cache_bytes == 0 + || max_batch_bytes == 0 + || batch_rows == 0 + || schema.metadata.get(PARTITION) != Some(&binding.partition.to_string()) + || schema.metadata.get(FORMAT).map(String::as_str) != Some("1") + || schema.fields.len() != 9 + { + return Err(Error::Invalid( + "invalid local Lance state configuration".into(), + )); + } + let mut sink = LanceTableSink::new(dataset, handler, max_batch_bytes)?; + let head = sink.covered_sequence(&binding).await?; + if !local_file.exists() { + let metadata = &sink.dataset().manifest().table_metadata; + match ( + metadata.get(PUBLISHED_CHECKPOINT_URI), + metadata.get(PUBLISHED_CHECKPOINT_VERSION), + ) { + (Some(uri), Some(version)) => { + let version: u64 = version.parse().map_err(fail)?; + let mut builder = lance::dataset::builder::DatasetBuilder::from_uri(uri) + .with_version(version); + if let Some(params) = sink.dataset().store_params() { + builder = builder.with_store_params(params.clone()); + } + let checkpoint = builder.load().await.map_err(fail)?; + restore_checkpoint( + local_file, + &checkpoint, + sink.dataset(), + &binding, + cache_bytes, + batch_rows, + ) + .await?; + } + (None, None) => {} + _ => { + return Err(Error::Invalid( + "incomplete published checkpoint pointer".into(), + )) + } + } + } + let db = Database::builder() + .set_cache_size(cache_bytes) + .create(local_file) + .map_err(fail)?; + let tx = db.begin_write().map_err(fail)?; + let mut metadata = tx.open_table(META).map_err(fail)?; + for (key, value) in [ + ("run", binding.run.clone()), + ("schema", binding.schema.clone()), + ("partition", binding.partition.to_string()), + ("uri", sink.dataset().uri().to_owned()), + ] { + let old = metadata.get(key).map_err(fail)?.map(|v| v.value().to_vec()); + if old.as_ref().is_some_and(|old| old != value.as_bytes()) { + return Err(Error::Invalid("local state binding mismatch".into())); + } + metadata.insert(key, value.as_bytes()).map_err(fail)?; + } + let through = metadata + .get("through") + .map_err(fail)? + .map(|value| { + value + .value() + .try_into() + .map(u64::from_le_bytes) + .map_err(|_| Error::Invalid("invalid local state sequence".into())) + }) + .transpose()? + .unwrap_or(0); + let version = metadata + .get("version") + .map_err(fail)? + .map(|value| { + value + .value() + .try_into() + .map(u64::from_le_bytes) + .map_err(|_| Error::Invalid("invalid local state version".into())) + }) + .transpose()? + .unwrap_or(0); + if through > head || version > sink.dataset().version().version { + return Err(Error::Invalid("local state ahead of supplied log".into())); + } + drop(metadata); + tx.open_table(STATE).map_err(fail)?; + tx.open_table(RECEIPTS).map_err(fail)?; + tx.commit().map_err(fail)?; + let mut value = Self { + db, + sink, + binding, + through, + version, + max_batch_bytes, + batch_rows, + poisoned: true, + }; + value.replay(head).await?; + value.poisoned = false; + Ok(value) + } + + pub fn through_sequence(&self) -> u64 { + self.through + } + pub fn dataset(&self) -> &Dataset { + self.sink.dataset() + } + + /// Write an immutable full checkpoint in bounded Arrow batches. Invoke on a + /// byte/time threshold, not after each call. The returned exact dataset must + /// be published by the lease owner before relying on it for recovery. This + /// method performs no WAL deletion or automatic retention change. + pub async fn checkpoint(&self, uri: &str) -> Result { + if self.poisoned { + return Err(Error::Fenced); + } + let tx = self.db.begin_read().map_err(fail)?; + let state = tx.open_table(STATE).map_err(fail)?; + let receipts = tx.open_table(RECEIPTS).map_err(fail)?; + let mut rows = state + .range::<&[u8]>(..) + .map_err(fail)? + .map(|item| (PUT, item)) + .chain( + receipts + .range::<&[u8]>(..) + .map_err(fail)? + .map(|item| (RECEIPT, item)), + ) + .peekable(); + let mut schema = table_schema( + &Schema::new(vec![ + Field::new("kind", DataType::UInt8, false), + Field::new("key", DataType::LargeBinary, false), + Field::new("value", DataType::LargeBinary, false), + ]), + &self.binding.run, + &self.binding.schema, + ); + for (key, value) in [ + (PARTITION, self.binding.partition.to_string()), + (CHECKPOINT_FORMAT, "1".into()), + (CHECKPOINT_URI, self.sink.dataset().uri().to_owned()), + (CHECKPOINT_VERSION, self.version.to_string()), + (CHECKPOINT_SEQUENCE, self.through.to_string()), + ] { + schema.metadata.insert(key.into(), value); + } + let schema = Arc::new(schema); + let output_schema = schema.clone(); + let batch_rows = self.batch_rows; + let batch_bytes = self.max_batch_bytes; + let mut ended = false; + let batches = std::iter::from_fn(move || { + if ended { + return None; + } + let mut kinds = Vec::new(); + let mut keys = Vec::new(); + let mut values = Vec::new(); + let mut bytes = 0usize; + for _ in 0..batch_rows { + if let Some((_, Ok((key, value)))) = rows.peek() { + let next_bytes = key.value().len().saturating_add(value.value().len()); + if !kinds.is_empty() && bytes.saturating_add(next_bytes) > batch_bytes { + break; + } + } + let Some((kind, result)) = rows.next() else { + ended = true; + break; + }; + let (key, value) = match result { + Ok(value) => value, + Err(error) => { + ended = true; + return Some(Err(arrow_schema::ArrowError::ExternalError(Box::new( + error, + )))); + } + }; + bytes += key.value().len() + value.value().len(); + if bytes > batch_bytes { + ended = true; + return Some(Err(arrow_schema::ArrowError::InvalidArgumentError( + "checkpoint batch exceeds budget".into(), + ))); + } + kinds.push(kind); + keys.push(key.value().to_vec()); + values.push(value.value().to_vec()); + } + if kinds.is_empty() { + return None; + } + Some(RecordBatch::try_new( + output_schema.clone(), + vec![ + Arc::new(UInt8Array::from(kinds)), + Arc::new(LargeBinaryArray::from_iter_values(keys)), + Arc::new(LargeBinaryArray::from_iter_values(values)), + ], + )) + }); + Dataset::write( + RecordBatchIterator::new(batches, schema), + uri, + Some(WriteParams { + data_storage_version: Some(LanceFileVersion::V2_2), + max_bytes_per_file: self.max_batch_bytes, + store_params: self.dataset().store_params().cloned(), + ..Default::default() + }), + ) + .await + .map_err(fail) + } + + /// Make an exact completed checkpoint discoverable on cold restart. If the + /// log advanced since capture, reject it; never point at a partial snapshot. + pub async fn publish_checkpoint(&mut self, checkpoint: &Dataset) -> Result<()> { + if self.poisoned { + return Err(Error::Fenced); + } + let metadata = &checkpoint.schema().metadata; + if metadata.get(CHECKPOINT_URI).map(String::as_str) != Some(self.dataset().uri()) + || metadata.get(RUN_METADATA) != Some(&self.binding.run) + || metadata.get(SCHEMA_METADATA) != Some(&self.binding.schema) + || metadata.get(CHECKPOINT_VERSION) != Some(&self.version.to_string()) + || metadata.get(CHECKPOINT_SEQUENCE) != Some(&self.through.to_string()) + || metadata.get(PARTITION) != Some(&self.binding.partition.to_string()) + || metadata.get(CHECKPOINT_FORMAT).map(String::as_str) != Some("1") + { + return Err(Error::Invalid( + "checkpoint differs from current local state".into(), + )); + } + self.poisoned = true; + self.sink + .update_metadata_at( + std::collections::HashMap::from([ + (PUBLISHED_CHECKPOINT_URI.into(), checkpoint.uri().to_owned()), + ( + PUBLISHED_CHECKPOINT_VERSION.into(), + checkpoint.version().version.to_string(), + ), + ]), + self.version, + ) + .await?; + let tx = self.db.begin_write().map_err(fail)?; + self.finish_local(tx, self.through)?; + self.poisoned = false; + Ok(()) + } + + pub fn get(&self, session: &str, key: &[u8]) -> Result>> { + if self.poisoned { + return Err(Error::Fenced); + } + let tx = self.db.begin_read().map_err(fail)?; + let table = tx.open_table(STATE).map_err(fail)?; + let value = table + .get(local_key(session, key).as_slice()) + .map_err(fail)?; + Ok(value.map(|value| value.value().to_vec())) + } + + /// Bounded local prefix scan. `after` is an exclusive state key, not an + /// opaque remote cursor. This performs no network request or full DB scan. + pub fn scan( + &self, + session: &str, + prefix: &[u8], + after: Option<&[u8]>, + limit: usize, + ) -> Result> { + if self.poisoned { + return Err(Error::Fenced); + } + if session.is_empty() + || limit == 0 + || limit > self.batch_rows + || after.is_some_and(|key| !key.starts_with(prefix)) + { + return Err(Error::Invalid("invalid local state scan bounds".into())); + } + let tx = self.db.begin_read().map_err(fail)?; + let table = tx.open_table(STATE).map_err(fail)?; + let encoded_prefix = local_key(session, prefix); + let start = local_key(session, after.unwrap_or(prefix)); + let mut output = Vec::new(); + for item in table.range(start.as_slice()..).map_err(fail)? { + let (key, value) = item.map_err(fail)?; + if !key.value().starts_with(&encoded_prefix) { + break; + } + let suffix = &key.value()[8 + session.len()..]; + if after.is_some_and(|after| suffix <= after) { + continue; + } + output.push(Mutation { + key: suffix.to_vec(), + value: Some(value.value().to_vec()), + }); + if output.len() == limit { + break; + } + } + Ok(output) + } + + /// Check before alignment. Same receipt with different input is an error, + /// even when no output rows were produced by the original call. + pub fn receipt(&self, session: &str, receipt: &str, input_digest: &str) -> Result> { + if self.poisoned { + return Err(Error::Fenced); + } + let tx = self.db.begin_read().map_err(fail)?; + let table = tx.open_table(RECEIPTS).map_err(fail)?; + let value = table + .get(local_key(session, receipt.as_bytes()).as_slice()) + .map_err(fail)?; + let Some(value) = value else { + return Ok(None); + }; + let bytes = value.value(); + if bytes.len() < 8 || &bytes[8..] != input_digest.as_bytes() { + return Err(Error::Invalid("changed committed receipt".into())); + } + Ok(Some(u64::from_le_bytes(bytes[..8].try_into().unwrap()))) + } + + /// Publish one coalesced batch, then apply its local transaction. Success + /// means both output and state deltas are durable in the SAME Lance version. + /// After uncertain success reopen and check receipts before realigning. + pub async fn commit(&mut self, calls: &[AlignedCall]) -> Result { + if self.poisoned { + return Err(Error::Fenced); + } + if calls.is_empty() { + return Ok(self.through); + } + let mut expected = self.through; + let mut seen = std::collections::HashSet::new(); + let schema = Arc::new(Schema::from(self.sink.dataset().schema())); + let mut entries = Vec::with_capacity(calls.len()); + let mut bytes = 0usize; + let mut input_bytes = 0usize; + for call in calls { + expected = expected + .checked_add(1) + .ok_or_else(|| Error::Invalid("sequence overflow".into()))?; + if call.sequence != expected + || call.session.is_empty() + || call.receipt.is_empty() + || call.input_digest.is_empty() + || call.mutations.iter().any(|m| m.key.is_empty()) + || !seen.insert((&call.session, &call.receipt)) + || self + .receipt(&call.session, &call.receipt, &call.input_digest)? + .is_some() + { + return Err(Error::Invalid( + "invalid local alignment batch or committed retry".into(), + )); + } + let expected_schema = log_schema(call.records.schema().as_ref(), &self.binding); + if expected_schema != *schema + || call + .records + .schema() + .fields + .iter() + .zip(call.records.columns()) + .any(|(field, column)| { + field.data_type().is_nested() + || (!field.is_nullable() && column.null_count() != 0) + }) + { + return Err(Error::Invalid( + "output schema differs from durable log".into(), + )); + } + let row_count = call + .mutations + .len() + .saturating_add(call.records.num_rows()) + .saturating_add(1); + let row_overhead = call + .session + .len() + .saturating_add(call.receipt.len()) + .saturating_add(call.input_digest.len()) + .saturating_add(64); + input_bytes = input_bytes + .saturating_add(row_count.saturating_mul(row_overhead)) + .saturating_add(call.records.get_array_memory_size()); + for mutation in &call.mutations { + let cell_bytes = mutation + .key + .len() + .saturating_add(mutation.value.as_ref().map_or(0, Vec::len)); + if cell_bytes > 1 << 20 { + return Err(Error::Invalid( + "state cell exceeds 1 MiB; split the state into smaller keys".into(), + )); + } + input_bytes = input_bytes.saturating_add(cell_bytes); + } + if input_bytes > self.max_batch_bytes { + return Err(Error::Invalid( + "local Lance input exceeds byte budget".into(), + )); + } + let records = encode_call(schema.clone(), call, self.batch_rows)?; + bytes = bytes + .checked_add(records.len()) + .ok_or_else(|| Error::Invalid("batch size overflow".into()))?; + if bytes > self.max_batch_bytes { + return Err(Error::Invalid( + "local Lance batch exceeds byte budget".into(), + )); + } + entries.push(Entry { + sequence: call.sequence, + session: call.session.clone(), + receipt: call.receipt.clone(), + input_digest: call.input_digest.clone(), + transition: Transition { + delta: Vec::new(), + records, + }, + }); + } + let base = self.version; + self.poisoned = true; + let staged = stage( + self.sink.dataset(), + &self.binding, + &entries, + self.max_batch_bytes, + ) + .await?; + self.sink.commit_staged_at(vec![staged], base).await?; + let tx = self.db.begin_write().map_err(fail)?; + { + let mut state = tx.open_table(STATE).map_err(fail)?; + let mut receipts = tx.open_table(RECEIPTS).map_err(fail)?; + for call in calls { + for mutation in &call.mutations { + let key = local_key(&call.session, &mutation.key); + if let Some(value) = &mutation.value { + state + .insert(key.as_slice(), value.as_slice()) + .map_err(fail)?; + } else { + state.remove(key.as_slice()).map_err(fail)?; + } + } + receipts + .insert( + local_key(&call.session, call.receipt.as_bytes()).as_slice(), + receipt_value(call.sequence, &call.input_digest).as_slice(), + ) + .map_err(fail)?; + } + } + self.finish_local(tx, expected)?; + self.poisoned = false; + Ok(expected) + } + + fn finish_local(&mut self, tx: redb::WriteTransaction, sequence: u64) -> Result<()> { + let version = self.sink.dataset().version().version; + { + let mut meta = tx.open_table(META).map_err(fail)?; + meta.insert("through", sequence.to_le_bytes().as_slice()) + .map_err(fail)?; + meta.insert("version", version.to_le_bytes().as_slice()) + .map_err(fail)?; + } + tx.commit().map_err(fail)?; + self.through = sequence; + self.version = version; + Ok(()) + } + + async fn replay(&mut self, head: u64) -> Result<()> { + // Old immutable fragments need not be opened to replay a new suffix. + // Metadata equality is required; compaction/deletions are not an append. + let mut fragments = self.sink.dataset().get_fragments(); + if self.version > 0 { + let base = self + .sink + .dataset() + .checkout_version(self.version) + .await + .map_err(fail)?; + let previous = base.get_fragments(); + for old in &previous { + if !fragments.iter().any(|new| new.metadata() == old.metadata()) { + return Err(Error::Invalid( + "state log is not an immutable append descendant".into(), + )); + } + } + fragments.retain(|fragment| !previous.iter().any(|old| old.id() == fragment.id())); + } + if fragments.is_empty() { + if self.through != head { + return Err(Error::Invalid("state log watermark without data".into())); + } + let tx = self.db.begin_write().map_err(fail)?; + return self.finish_local(tx, head); + } + let mut scan = self.sink.dataset().scan(); + scan.with_fragments( + fragments + .into_iter() + .map(|fragment| fragment.metadata().clone()) + .collect(), + ); + scan.project(&COLUMNS).map_err(fail)?; + scan.filter(&format!("sequence > {}", self.through)) + .map_err(fail)?; + scan.batch_size(self.batch_rows); + scan.scan_in_order(true); + let mut stream = scan.try_into_stream().await.map_err(fail)?; + let tx = self.db.begin_write().map_err(fail)?; + let mut state = tx.open_table(STATE).map_err(fail)?; + let mut receipts = tx.open_table(RECEIPTS).map_err(fail)?; + let mut through = self.through; + let mut ordinal = 0; + let mut identity: Option<(String, String, String)> = None; + while let Some(batch) = stream.try_next().await.map_err(fail)? { + let sequences = batch + .column(0) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid sequence column"))?; + let ordinals = batch + .column(1) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid ordinal column"))?; + let kinds = batch + .column(2) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid kind column"))?; + let sessions = batch + .column(3) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid session column"))?; + let ids = batch + .column(4) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid receipt column"))?; + let digests = batch + .column(5) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid digest column"))?; + let keys = batch + .column(6) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid key column"))?; + let values = batch + .column(7) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid value column"))?; + if batch + .columns() + .iter() + .any(|column| column.null_count() != 0) + { + return Err(Error::Invalid("null local state log cell".into())); + } + for i in 0..batch.num_rows() { + let sequence = sequences.value(i); + if through.checked_add(1) != Some(sequence) + || sequence > head + || ordinals.value(i) != ordinal + { + return Err(Error::Invalid( + "local state log sequence or ordinal gap".into(), + )); + } + let session = sessions.value(i); + let receipt = ids.value(i); + let digest = digests.value(i); + if session.is_empty() || receipt.is_empty() || digest.is_empty() { + return Err(Error::Invalid("empty local state log identity".into())); + } + let expected = + identity.get_or_insert_with(|| (session.into(), receipt.into(), digest.into())); + if expected.0 != session || expected.1 != receipt || expected.2 != digest { + return Err(Error::Invalid( + "mixed identities in local state sequence".into(), + )); + } + let key = local_key(session, keys.value(i)); + match kinds.value(i) { + PUT if !keys.value(i).is_empty() => { + state + .insert(key.as_slice(), values.value(i)) + .map_err(fail)?; + } + DELETE if !keys.value(i).is_empty() && values.value(i).is_empty() => { + state.remove(key.as_slice()).map_err(fail)?; + } + RECORD if keys.value(i).is_empty() && values.value(i).is_empty() => {} + RECEIPT if keys.value(i).is_empty() && values.value(i).is_empty() => { + let key = local_key(session, receipt.as_bytes()); + if receipts.get(key.as_slice()).map_err(fail)?.is_some() { + return Err(Error::Invalid( + "duplicate receipt in committed log".into(), + )); + } + receipts + .insert(key.as_slice(), receipt_value(sequence, digest).as_slice()) + .map_err(fail)?; + through = sequence; + ordinal = 0; + identity = None; + continue; + } + _ => return Err(Error::Invalid("invalid local state log event".into())), + } + ordinal += 1; + } + } + if through != head || ordinal != 0 || identity.is_some() { + return Err(Error::Invalid("incomplete local state log prefix".into())); + } + drop(state); + drop(receipts); + self.finish_local(tx, head) + } +} + +/// Seed a missing local cache from an owner-published exact checkpoint. Call +/// `LocalLancePartition::open` afterwards to verify/replay the committed suffix. +/// Existing local files are never overwritten. WAL/checkpoint objects remain +/// untouched, including when a read fails halfway through this transaction. +pub async fn restore_checkpoint( + local_file: &Path, + checkpoint: &Dataset, + wal: &Dataset, + binding: &Binding, + cache_bytes: usize, + batch_rows: usize, +) -> Result<()> { + let schema = Schema::from(checkpoint.schema()); + let metadata = &schema.metadata; + let number = |key: &str| -> Result { + metadata + .get(key) + .ok_or_else(|| fail("missing checkpoint binding"))? + .parse() + .map_err(fail) + }; + let sequence = number(CHECKPOINT_SEQUENCE)?; + let version = number(CHECKPOINT_VERSION)?; + let wal_schema = Schema::from(wal.schema()); + if local_file.exists() + || cache_bytes == 0 + || batch_rows == 0 + || metadata.get(CHECKPOINT_FORMAT).map(String::as_str) != Some("1") + || metadata.get(CHECKPOINT_URI).map(String::as_str) != Some(wal.uri()) + || version > wal.version().version + || version == 0 + || [metadata, &wal_schema.metadata].iter().any(|meta| { + meta.get(RUN_METADATA) != Some(&binding.run) + || meta.get(SCHEMA_METADATA) != Some(&binding.schema) + || meta.get(PARTITION) != Some(&binding.partition.to_string()) + }) + || checkpoint.manifest().data_storage_format.version != "2.2" + { + return Err(Error::Invalid( + "checkpoint binding mismatch or existing local cache".into(), + )); + } + // Ensure the captured ordinary version still belongs to this durable log. + wal.checkout_version(version).await.map_err(fail)?; + // Install only a complete, committed local database. A failed read or + // cancelled future drops the temporary file; the next open can retry the + // published checkpoint instead of accidentally falling back to old WAL. + let temporary = tempfile::NamedTempFile::new_in( + local_file + .parent() + .filter(|parent| !parent.as_os_str().is_empty()) + .unwrap_or(Path::new(".")), + ) + .map_err(fail)?; + let db = Database::builder() + .set_cache_size(cache_bytes) + .create(temporary.path()) + .map_err(fail)?; + let tx = db.begin_write().map_err(fail)?; + let mut state = tx.open_table(STATE).map_err(fail)?; + let mut receipts = tx.open_table(RECEIPTS).map_err(fail)?; + let mut scan = checkpoint.scan(); + scan.project(&["kind", "key", "value"]) + .map_err(fail)? + .batch_size(batch_rows); + let mut stream = scan.try_into_stream().await.map_err(fail)?; + while let Some(batch) = stream.try_next().await.map_err(fail)? { + let kinds = batch + .column(0) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid checkpoint kind"))?; + let keys = batch + .column(1) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid checkpoint key"))?; + let values = batch + .column(2) + .as_any() + .downcast_ref::() + .ok_or_else(|| fail("invalid checkpoint value"))?; + if batch + .columns() + .iter() + .any(|column| column.null_count() != 0) + { + return Err(Error::Invalid("null checkpoint cell".into())); + } + for i in 0..batch.num_rows() { + let key = keys.value(i); + let value = values.value(i); + if key.len() < 9 { + return Err(Error::Invalid("invalid checkpoint state key".into())); + } + let session_len = u64::from_be_bytes(key[..8].try_into().unwrap()); + if session_len == 0 || session_len >= (key.len() - 8) as u64 { + return Err(Error::Invalid("invalid checkpoint session framing".into())); + } + let table = match kinds.value(i) { + PUT => &mut state, + RECEIPT + if value.len() > 8 + && u64::from_le_bytes(value[..8].try_into().unwrap()) > 0 + && u64::from_le_bytes(value[..8].try_into().unwrap()) <= sequence => + { + &mut receipts + } + _ => return Err(Error::Invalid("invalid checkpoint state event".into())), + }; + if table.get(key).map_err(fail)?.is_some() { + return Err(Error::Invalid("duplicate checkpoint state key".into())); + } + table.insert(key, value).map_err(fail)?; + } + } + drop(state); + drop(receipts); + { + let mut meta = tx.open_table(META).map_err(fail)?; + for (key, value) in [ + ("run", binding.run.clone()), + ("schema", binding.schema.clone()), + ("partition", binding.partition.to_string()), + ("uri", wal.uri().to_owned()), + ] { + meta.insert(key, value.as_bytes()).map_err(fail)?; + } + meta.insert("through", sequence.to_le_bytes().as_slice()) + .map_err(fail)?; + meta.insert("version", version.to_le_bytes().as_slice()) + .map_err(fail)?; + } + tx.commit().map_err(fail)?; + drop(db); + temporary.as_file().sync_all().map_err(fail)?; + temporary.persist_noclobber(local_file).map_err(fail)?; + Ok(()) +} diff --git a/crates/lance-context-ingestion/tests/local_lance.rs b/crates/lance-context-ingestion/tests/local_lance.rs new file mode 100644 index 00000000..03da8b9c --- /dev/null +++ b/crates/lance-context-ingestion/tests/local_lance.rs @@ -0,0 +1,545 @@ +#![cfg(feature = "lance")] + +use std::sync::Arc; + +use arrow_array::{RecordBatch, RecordBatchIterator, StringArray}; +use arrow_schema::{DataType, Field, Schema}; +use futures::TryStreamExt; +use lance::{dataset::WriteParams, Dataset}; +use lance_context_ingestion::{ + local_lance::{log_schema, restore_checkpoint, AlignedCall, LocalLancePartition, Mutation}, + Binding, +}; +use lance_file::version::LanceFileVersion; +use lance_table::io::commit::commit_handler_from_url; + +fn binding() -> Binding { + Binding { + run: "test-run".into(), + schema: "typed-state-v1".into(), + partition: 7, + } +} +fn records(values: &[&str]) -> RecordBatch { + RecordBatch::try_new( + Arc::new(Schema::new(vec![Field::new( + "content", + DataType::Utf8, + false, + )])), + vec![Arc::new(StringArray::from(values.to_vec()))], + ) + .unwrap() +} +async fn table(uri: &str) -> Dataset { + let schema = Arc::new(log_schema(records(&[]).schema().as_ref(), &binding())); + Dataset::write( + RecordBatchIterator::new(vec![Ok(RecordBatch::new_empty(schema.clone()))], schema), + uri, + Some(WriteParams { + data_storage_version: Some(LanceFileVersion::V2_2), + ..Default::default() + }), + ) + .await + .unwrap() +} +async fn open(path: &std::path::Path, dataset: Dataset) -> LocalLancePartition { + let handler = commit_handler_from_url(dataset.uri(), &None).await.unwrap(); + LocalLancePartition::open(path, dataset, binding(), handler, 1 << 20, 16 << 20, 2) + .await + .unwrap() +} +fn call(sequence: u64, session: &str, value: Option<&[u8]>, output: &[&str]) -> AlignedCall { + AlignedCall { + sequence, + session: session.into(), + receipt: format!("call-{sequence}"), + input_digest: format!("input-{sequence}"), + mutations: vec![Mutation { + key: b"node".to_vec(), + value: value.map(<[u8]>::to_vec), + }], + records: records(output), + } +} + +#[tokio::test] +async fn batch_commit_restart_and_empty_disk_restore_preserve_binary_state_and_typed_output() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let local = dir.path().join("local.redb"); + let mut writer = open(&local, dataset).await; + let base = writer.dataset().version().version; + let calls = [ + call( + 1, + "session-a", + Some(&[0, 255, 13]), + &["α", "escaped\nmessage"], + ), + call(2, "session-b", Some(b"old"), &["other"]), + call(3, "session-b", None, &[]), + ]; + assert_eq!(writer.commit(&calls).await.unwrap(), 3); + assert_eq!(writer.dataset().version().version, base + 1); + assert_eq!( + writer.get("session-a", b"node").unwrap(), + Some(vec![0, 255, 13]) + ); + assert_eq!(writer.get("session-b", b"node").unwrap(), None); + assert_eq!( + writer.receipt("session-b", "call-3", "input-3").unwrap(), + Some(3) + ); + assert!(writer.receipt("session-b", "call-3", "changed").is_err()); + assert!(writer.commit(&calls).await.is_err()); + let mut scan = writer.dataset().scan(); + scan.filter("kind = 3") + .unwrap() + .project(&["record.content"]) + .unwrap(); + let rows = scan + .try_into_stream() + .await + .unwrap() + .try_collect::>() + .await + .unwrap(); + assert_eq!(rows.iter().map(RecordBatch::num_rows).sum::(), 3); + assert_eq!( + writer.dataset().manifest().data_storage_format.version, + "2.2" + ); + drop(writer); + let writer = open(&local, Dataset::open(uri.to_str().unwrap()).await.unwrap()).await; + assert_eq!(writer.through_sequence(), 3); + drop(writer); + std::fs::remove_file(&local).unwrap(); + let writer = open(&local, Dataset::open(uri.to_str().unwrap()).await.unwrap()).await; + assert_eq!( + writer.get("session-a", b"node").unwrap(), + Some(vec![0, 255, 13]) + ); + assert_eq!(writer.get("session-b", b"node").unwrap(), None); + assert_eq!( + writer.receipt("session-b", "call-3", "input-3").unwrap(), + Some(3) + ); +} + +#[tokio::test] +async fn stale_aligner_cannot_commit_or_ack_different_output_at_the_same_sequence() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let mut a = open(&dir.path().join("a.redb"), dataset.clone()).await; + let mut b = open(&dir.path().join("b.redb"), dataset).await; + a.commit(&[call(1, "s", Some(b"winner"), &["winner"])]) + .await + .unwrap(); + assert!(b + .commit(&[call(1, "s", Some(b"loser"), &["loser"])]) + .await + .is_err()); + assert!(b.get("s", b"node").is_err()); + drop(b); + let recovered = open( + &dir.path().join("b.redb"), + Dataset::open(uri.to_str().unwrap()).await.unwrap(), + ) + .await; + assert_eq!( + recovered.get("s", b"node").unwrap(), + Some(b"winner".to_vec()) + ); + assert_eq!( + recovered.receipt("s", "call-1", "input-1").unwrap(), + Some(1) + ); +} + +#[tokio::test] +async fn acknowledged_state_reads_are_local_and_session_keys_do_not_collide() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let mut writer = open(&dir.path().join("cache.redb"), dataset).await; + writer + .commit(&[ + call(1, "a\0b", Some(b"one"), &["one"]), + call(2, "a", Some(b"two"), &["two"]), + ]) + .await + .unwrap(); + std::fs::rename(&uri, dir.path().join("offline.lance")).unwrap(); + assert_eq!(writer.get("a\0b", b"node").unwrap(), Some(b"one".to_vec())); + assert_eq!(writer.get("a", b"node").unwrap(), Some(b"two".to_vec())); + assert_eq!(writer.receipt("a", "call-2", "input-2").unwrap(), Some(2)); +} + +#[tokio::test] +async fn missing_committed_file_fails_recovery_without_advancing_local_state() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let mut writer = open(&dir.path().join("writer.redb"), dataset).await; + writer + .commit(&[call(1, "s", Some(b"v"), &["r"])]) + .await + .unwrap(); + let dataset = Dataset::open(uri.to_str().unwrap()).await.unwrap(); + let file = dataset.get_fragments()[0].metadata().files[0].path.clone(); + let path = uri.join("data").join(file); + let hidden = path.with_extension("hidden"); + std::fs::rename(&path, &hidden).unwrap(); + let handler = commit_handler_from_url(dataset.uri(), &None).await.unwrap(); + let local = dir.path().join("fresh.redb"); + assert!( + LocalLancePartition::open(&local, dataset, binding(), handler, 1 << 20, 16 << 20, 2) + .await + .is_err() + ); + std::fs::rename(hidden, path).unwrap(); + let recovered = open(&local, Dataset::open(uri.to_str().unwrap()).await.unwrap()).await; + assert_eq!(recovered.through_sequence(), 1); + assert_eq!(recovered.get("s", b"node").unwrap(), Some(b"v".to_vec())); +} + +#[tokio::test] +async fn checkpoint_restores_then_reads_only_new_fragments_and_replays_deletes() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let mut writer = open(&dir.path().join("writer.redb"), dataset).await; + writer + .commit(&[ + call(1, "a", Some(&[0, 255]), &["old"]), + call(2, "b", Some(b"old"), &[]), + ]) + .await + .unwrap(); + let checkpoint = writer + .checkpoint(dir.path().join("checkpoint.lance").to_str().unwrap()) + .await + .unwrap(); + let old_files = writer + .dataset() + .get_fragments() + .iter() + .flat_map(|f| { + f.metadata() + .files + .iter() + .map(|f| f.path.clone()) + .collect::>() + }) + .collect::>(); + writer + .commit(&[ + call(3, "b", None, &[]), + call(4, "c", Some(b"new"), &["new"]), + ]) + .await + .unwrap(); + let wal = Dataset::open(uri.to_str().unwrap()).await.unwrap(); + let local = dir.path().join("recovered.redb"); + restore_checkpoint(&local, &checkpoint, &wal, &binding(), 1 << 20, 1) + .await + .unwrap(); + for name in old_files { + let path = uri.join("data").join(name); + std::fs::rename(&path, path.with_extension("hidden")).unwrap(); + } + let recovered = open(&local, wal).await; + assert_eq!(recovered.through_sequence(), 4); + assert_eq!(recovered.get("a", b"node").unwrap(), Some(vec![0, 255])); + assert_eq!(recovered.get("b", b"node").unwrap(), None); + assert_eq!(recovered.get("c", b"node").unwrap(), Some(b"new".to_vec())); + assert_eq!( + recovered.receipt("a", "call-1", "input-1").unwrap(), + Some(1) + ); + assert_eq!( + recovered.receipt("b", "call-3", "input-3").unwrap(), + Some(3) + ); + assert!(restore_checkpoint( + &local, + &checkpoint, + recovered.dataset(), + &binding(), + 1 << 20, + 1 + ) + .await + .is_err()); +} + +#[tokio::test] +async fn published_checkpoint_is_discovered_automatically_without_old_data_files() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let mut writer = open(&dir.path().join("writer.redb"), dataset).await; + writer + .commit(&[call(1, "s", Some(b"before"), &["old"])]) + .await + .unwrap(); + let old_files = writer + .dataset() + .get_fragments() + .iter() + .flat_map(|f| { + f.metadata() + .files + .iter() + .map(|f| f.path.clone()) + .collect::>() + }) + .collect::>(); + let checkpoint = writer + .checkpoint(dir.path().join("checkpoint.lance").to_str().unwrap()) + .await + .unwrap(); + writer.publish_checkpoint(&checkpoint).await.unwrap(); + writer + .commit(&[call(2, "s", Some(b"after"), &["new"])]) + .await + .unwrap(); + assert!(writer.publish_checkpoint(&checkpoint).await.is_err()); + for name in old_files { + let path = uri.join("data").join(name); + std::fs::rename(&path, path.with_extension("hidden")).unwrap(); + } + let recovered = open( + &dir.path().join("fresh.redb"), + Dataset::open(uri.to_str().unwrap()).await.unwrap(), + ) + .await; + assert_eq!(recovered.through_sequence(), 2); + assert_eq!( + recovered.get("s", b"node").unwrap(), + Some(b"after".to_vec()) + ); + assert_eq!( + recovered.receipt("s", "call-1", "input-1").unwrap(), + Some(1) + ); + assert_eq!( + recovered.scan("s", b"n", None, 1).unwrap(), + vec![Mutation { + key: b"node".to_vec(), + value: Some(b"after".to_vec()) + }] + ); + assert!(recovered + .scan("s", b"n", Some(b"node"), 1) + .unwrap() + .is_empty()); +} + +#[tokio::test] +async fn oversized_state_cells_fail_before_publication_and_do_not_poison_valid_retry() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let mut writer = open(&dir.path().join("writer.redb"), dataset).await; + let version = writer.dataset().version().version; + let huge = vec![1u8; (1 << 20) + 1]; + assert!(writer + .commit(&[call(1, "s", Some(&huge), &[])]) + .await + .is_err()); + assert_eq!(writer.through_sequence(), 0); + assert_eq!( + Dataset::open(uri.to_str().unwrap()) + .await + .unwrap() + .version() + .version, + version + ); + writer + .commit(&[call(1, "s", Some(b"small"), &[])]) + .await + .unwrap(); + assert_eq!(writer.get("s", b"node").unwrap(), Some(b"small".to_vec())); +} + +#[tokio::test] +async fn failed_checkpoint_restore_leaves_no_cache_and_retries_without_covered_wal() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let mut writer = open(&dir.path().join("writer.redb"), dataset).await; + writer + .commit(&[call(1, "s", Some(b"v"), &["r"])]) + .await + .unwrap(); + let checkpoint_uri = dir.path().join("checkpoint.lance"); + let checkpoint = writer + .checkpoint(checkpoint_uri.to_str().unwrap()) + .await + .unwrap(); + writer.publish_checkpoint(&checkpoint).await.unwrap(); + for fragment in writer.dataset().get_fragments() { + for file in &fragment.metadata().files { + let path = uri.join("data").join(&file.path); + std::fs::rename(&path, path.with_extension("hidden")).unwrap(); + } + } + let file = checkpoint.get_fragments()[0].metadata().files[0] + .path + .clone(); + let path = checkpoint_uri.join("data").join(file); + let hidden = path.with_extension("hidden"); + std::fs::rename(&path, &hidden).unwrap(); + let local = dir.path().join("recovered.redb"); + let wal = Dataset::open(uri.to_str().unwrap()).await.unwrap(); + let handler = commit_handler_from_url(wal.uri(), &None).await.unwrap(); + assert!( + LocalLancePartition::open(&local, wal, binding(), handler, 1 << 20, 16 << 20, 2) + .await + .is_err() + ); + assert!(!local.exists()); + std::fs::rename(hidden, path).unwrap(); + let recovered = open(&local, Dataset::open(uri.to_str().unwrap()).await.unwrap()).await; + assert_eq!(recovered.through_sequence(), 1); + assert_eq!(recovered.get("s", b"node").unwrap(), Some(b"v".to_vec())); + assert_eq!( + recovered.receipt("s", "call-1", "input-1").unwrap(), + Some(1) + ); +} + +#[tokio::test] +async fn checkpoint_byte_batches_and_empty_checkpoint_restore() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let local = dir.path().join("writer.redb"); + let mut writer = open(&local, dataset).await; + let empty = writer + .checkpoint(dir.path().join("empty.lance").to_str().unwrap()) + .await + .unwrap(); + writer.publish_checkpoint(&empty).await.unwrap(); + let restored = open( + &dir.path().join("empty.redb"), + Dataset::open(uri.to_str().unwrap()).await.unwrap(), + ) + .await; + assert_eq!(restored.through_sequence(), 0); + drop(restored); + // Individual calls fit, but their combined checkpoint exceeds the writer + // byte budget. Checkpoint must split before reaching its row-count limit. + let value = vec![7u8; 512 << 10]; + for sequence in 1..=3 { + writer + .commit(&[call(sequence, &format!("s{sequence}"), Some(&value), &[])]) + .await + .unwrap(); + } + drop(writer); + let wal = Dataset::open(uri.to_str().unwrap()).await.unwrap(); + let handler = commit_handler_from_url(wal.uri(), &None).await.unwrap(); + let writer = + LocalLancePartition::open(&local, wal, binding(), handler, 1 << 20, 768 << 10, 100) + .await + .unwrap(); + let checkpoint = writer + .checkpoint(dir.path().join("bounded.lance").to_str().unwrap()) + .await + .unwrap(); + let restored = dir.path().join("bounded.redb"); + restore_checkpoint( + &restored, + &checkpoint, + writer.dataset(), + &binding(), + 1 << 20, + 100, + ) + .await + .unwrap(); + let restored = open( + &restored, + Dataset::open(uri.to_str().unwrap()).await.unwrap(), + ) + .await; + for sequence in 1..=3 { + assert_eq!( + restored.get(&format!("s{sequence}"), b"node").unwrap(), + Some(value.clone()) + ); + } +} + +#[derive(Debug)] +struct CommitThenFail { + delegate: Arc, +} + +#[async_trait::async_trait] +impl lance_table::io::commit::CommitHandler for CommitThenFail { + async fn commit( + &self, + manifest: &mut lance_table::format::Manifest, + indices: Option>, + base: &object_store::path::Path, + store: &lance_io::object_store::ObjectStore, + writer: lance_table::io::commit::ManifestWriter, + scheme: lance_table::io::commit::ManifestNamingScheme, + transaction: Option, + ) -> std::result::Result< + lance_table::io::commit::ManifestLocation, + lance_table::io::commit::CommitError, + > { + self.delegate + .commit(manifest, indices, base, store, writer, scheme, transaction) + .await?; + Err(lance_table::io::commit::CommitError::OtherError( + lance::Error::invalid_input("injected lost successful commit response"), + )) + } +} + +#[tokio::test] +async fn lost_manifest_ack_recovers_both_local_state_and_receipts_without_new_version() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let local = dir.path().join("writer.redb"); + let handler = Arc::new(CommitThenFail { + delegate: commit_handler_from_url(dataset.uri(), &None).await.unwrap(), + }); + let mut writer = + LocalLancePartition::open(&local, dataset, binding(), handler, 1 << 20, 16 << 20, 2) + .await + .unwrap(); + let calls = [call(1, "s", Some(b"once"), &["once"])]; + assert!(writer.commit(&calls).await.is_err()); + assert!(writer.receipt("s", "call-1", "input-1").is_err()); + drop(writer); + let committed = Dataset::open(uri.to_str().unwrap()).await.unwrap(); + let version = committed.version().version; + let mut recovered = open(&local, committed).await; + assert_eq!(recovered.through_sequence(), 1); + assert_eq!(recovered.get("s", b"node").unwrap(), Some(b"once".to_vec())); + assert_eq!( + recovered.receipt("s", "call-1", "input-1").unwrap(), + Some(1) + ); + assert!(recovered.commit(&calls).await.is_err()); + assert_eq!( + Dataset::open(uri.to_str().unwrap()) + .await + .unwrap() + .version() + .version, + version + ); +} From 8f4583fee8ef1737288677e334a305575f8080c6 Mon Sep 17 00:00:00 2001 From: Beinan Wang Date: Wed, 7 Oct 2026 21:18:21 +0000 Subject: [PATCH 16/16] Validate immutable Lance output ranges and bounded local recovery batches --- Cargo.lock | 29 +- crates/lance-context-ingestion/README.md | 13 + .../lance-context-ingestion/src/lance_sink.rs | 59 +++- crates/lance-context-ingestion/src/lib.rs | 2 + .../src/local_lance.rs | 109 +++++- .../src/local_lance_reader.rs | 312 ++++++++++++++++++ .../tests/local_lance.rs | 214 ++++++++++++ 7 files changed, 701 insertions(+), 37 deletions(-) create mode 100644 crates/lance-context-ingestion/src/local_lance_reader.rs diff --git a/Cargo.lock b/Cargo.lock index a597c136..f7feae68 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -501,9 +501,9 @@ dependencies = [ [[package]] name = "aws-lc-rs" -version = "1.17.0" +version = "1.18.1" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5ec2f1fc3ec205783a5da9a7e6c1509cc69dedf09a1949e412c1e18469326d00" +checksum = "b281d307588d634de920874890732659e2e7672f72b5e10e81badc1a8a83621e" dependencies = [ "aws-lc-sys", "zeroize", @@ -511,14 +511,15 @@ dependencies = [ [[package]] name = "aws-lc-sys" -version = "0.41.0" +version = "0.45.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1a2f9779ce85b93ab6170dd940ad0169b5766ff848247aff13bb788b832fe3f4" +checksum = "9bff6c3b54fad79a2e60b8102caf565819711497c1f5f092f49508e2f5c31b27" dependencies = [ "cc", "cmake", "dunce", "fs_extra", + "pkg-config", ] [[package]] @@ -6026,9 +6027,9 @@ dependencies = [ [[package]] name = "quinn-proto" -version = "0.11.14" +version = "0.11.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "434b42fec591c96ef50e21e886936e66d3cc3f737104fdb9b737c40ffb94c098" +checksum = "4fcb935c5bec503c2f0e306bdd3e58bb9029dcb14fa8d9ac76e3a5256ac0763e" dependencies = [ "aws-lc-rs", "bytes", @@ -6678,9 +6679,9 @@ dependencies = [ [[package]] name = "rustls" -version = "0.23.40" +version = "0.23.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ef86cd5876211988985292b91c96a8f2d298df24e75989a43a3c73f2d4d8168b" +checksum = "0d41d731c7d2f962d1ccc364cec258de3c0e93b38c2fb3ba97ac74513048d634" dependencies = [ "aws-lc-rs", "log", @@ -6743,9 +6744,9 @@ checksum = "f87165f0995f63a9fbeea62b64d10b4d9d8e78ec6d7d51fb2125fda7bb36788f" [[package]] name = "rustls-webpki" -version = "0.103.13" +version = "0.103.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "61c429a8649f110dddef65e2a5ad240f747e85f7758a6bccc7e5777bd33f756e" +checksum = "f3c3cf1d8b1e7d4927e2d154c3fcb02979afb9939629c62cd9048d4f07b60ac2" dependencies = [ "aws-lc-rs", "ring", @@ -6959,9 +6960,9 @@ dependencies = [ [[package]] name = "serde_with" -version = "3.20.0" +version = "3.21.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e72c1c2cb7b223fafb600a619537a871c2818583d619401b785e7c0b746ccde2" +checksum = "76a5c54c7310e7b8b9577c286d7e399ddd876c3e12b3ed917a8aabc4b96e9e8c" dependencies = [ "base64", "bs58", @@ -6979,9 +6980,9 @@ dependencies = [ [[package]] name = "serde_with_macros" -version = "3.20.0" +version = "3.21.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b90c488738ecb4fb0262f41f43bc40efc5868d9fb744319ddf5f5317f417bfac" +checksum = "84d57bc0c8b9a17920c178daa6bb924850d54a9c97ab45194bb8c17ad66bb660" dependencies = [ "darling", "proc-macro2", diff --git a/crates/lance-context-ingestion/README.md b/crates/lance-context-ingestion/README.md index 85959f5c..55e4de63 100644 --- a/crates/lance-context-ingestion/README.md +++ b/crates/lance-context-ingestion/README.md @@ -205,3 +205,16 @@ legacy migration are adapter contracts; this API does not convert JSON payloads supplied by a caller. This is an opt-in backend: existing `Journal` and `SessionCheckpoints` adapters keep their existing storage and recovery behavior. Migrate their committed suffix and receipts before changing the intake log. + +`local_lance_reader::OutputRange` reads only typed output from an immutable +append range after validating both version watermarks, run/schema/partition and +full fragment ancestry. It enforces physical and decoded byte limits separately. +`BatchReference` is a bounded, checksummed binary descriptor binding that range, +its generation/predecessor and application metadata. A consumer must budget the +referenced payload and its own decoded representation, never the descriptor size. +No parsing failure permits falling back to a different log format. + +`batch_charge` checks the input, IPC encoding and retained Arrow allocation charge +before admission. Shared IPC buffers are charged once per allocation while their +batches remain alive. Charge computation currently encodes and decodes a call; +commit encodes again. Account for that CPU work when profiling the adapter. diff --git a/crates/lance-context-ingestion/src/lance_sink.rs b/crates/lance-context-ingestion/src/lance_sink.rs index 904a0d63..1bce3ee2 100644 --- a/crates/lance-context-ingestion/src/lance_sink.rs +++ b/crates/lance-context-ingestion/src/lance_sink.rs @@ -31,6 +31,47 @@ pub const RUN_METADATA: &str = "lance-context.ingestion.run"; pub const SCHEMA_METADATA: &str = "lance-context.ingestion.schema"; const SHARD_NAMESPACE: Uuid = Uuid::from_u128(0x8370a74b_3aaf_4083_8ed0_610b9bd6e8da); +// IPC arrays may share an entire message allocation. Charging its capacity once +// per column can shrink a batch by the number of columns, despite retaining only +// one allocation. Callers must keep charged batches alive for this budget's life. +#[derive(Default)] +pub(crate) struct RetainedArrowBudget { + allocations: HashMap, + bytes: usize, +} +impl RetainedArrowBudget { + pub(crate) fn add(&mut self, batch: &RecordBatch) -> usize { + let mut arrays = batch + .columns() + .iter() + .map(|array| array.to_data()) + .collect::>(); + self.bytes = self + .bytes + .saturating_add(std::mem::size_of::()); + while let Some(data) = arrays.pop() { + self.bytes = self.bytes.saturating_add(std::mem::size_of_val(&data)); + for buffer in data + .buffers() + .iter() + .chain(data.nulls().map(|nulls| nulls.buffer())) + { + let size = buffer + .capacity() + .max(buffer.ptr_offset().saturating_add(buffer.len())); + let charged = self + .allocations + .entry(buffer.data_ptr().as_ptr() as usize) + .or_default(); + self.bytes = self.bytes.saturating_add(size.saturating_sub(*charged)); + *charged = (*charged).max(size); + } + arrays.extend(data.child_data().iter().cloned()); + } + self.bytes + } +} + fn failure(error: impl std::fmt::Display) -> Error { Error::Stage(error.to_string()) } @@ -157,6 +198,17 @@ async fn watermarks(dataset: &Dataset) -> Result> { .collect()) } +/// Read the covered entry sequence of this exact supplied snapshot. This never +/// refreshes the dataset, so immutable batch references cannot drift to latest. +pub async fn covered_sequence_at(dataset: &Dataset, binding: &Binding) -> Result { + validate_dataset(dataset, binding)?; + Ok(watermarks(dataset) + .await? + .get(&shard(binding)?) + .copied() + .unwrap_or(0)) +} + /// Decode a bounded WAL range and prepare immutable Lance 2.2 files only. This /// is safe to distribute across workers: no table manifest or WAL cursor changes. /// `max_bytes` bounds decoded Arrow data retained by this stage; input WAL and @@ -179,7 +231,7 @@ pub async fn stage( } let schema = Schema::from(dataset.schema()); let mut batches = Vec::new(); - let mut bytes = 0_usize; + let mut retained = RetainedArrowBudget::default(); for entry in entries { if entry.transition.records.is_empty() { continue; @@ -193,10 +245,7 @@ pub async fn stage( } for batch in reader { let batch = batch.map_err(failure)?; - bytes = bytes - .checked_add(batch.get_array_memory_size()) - .ok_or_else(|| Error::Invalid("Arrow byte count overflow".into()))?; - if bytes > max_bytes { + if retained.add(&batch) > max_bytes { return Err(Error::Invalid( "staging decoded Arrow budget exceeded".into(), )); diff --git a/crates/lance-context-ingestion/src/lib.rs b/crates/lance-context-ingestion/src/lib.rs index 3ffbd9b4..236f79b5 100644 --- a/crates/lance-context-ingestion/src/lib.rs +++ b/crates/lance-context-ingestion/src/lib.rs @@ -9,6 +9,8 @@ mod journal; pub mod lance_sink; #[cfg(feature = "lance")] pub mod local_lance; +#[cfg(feature = "lance")] +pub mod local_lance_reader; mod pipeline; mod receipt; mod source; diff --git a/crates/lance-context-ingestion/src/local_lance.rs b/crates/lance-context-ingestion/src/local_lance.rs index 23cc2149..ee8d3e0d 100644 --- a/crates/lance-context-ingestion/src/local_lance.rs +++ b/crates/lance-context-ingestion/src/local_lance.rs @@ -375,17 +375,31 @@ impl LocalLancePartition { let tx = self.db.begin_read().map_err(fail)?; let state = tx.open_table(STATE).map_err(fail)?; let receipts = tx.open_table(RECEIPTS).map_err(fail)?; - let mut rows = state - .range::<&[u8]>(..) - .map_err(fail)? - .map(|item| (PUT, item)) - .chain( - receipts - .range::<&[u8]>(..) - .map_err(fail)? - .map(|item| (RECEIPT, item)), - ) - .peekable(); + let mut state_rows = state.range::<&[u8]>(..).map_err(fail)?; + let mut receipt_rows = receipts.range::<&[u8]>(..).map_err(fail)?; + let mut reading_receipts = false; + // Erase borrowed redb guards before crossing the async writer boundary. + // Only one pending owned cell plus the bounded Arrow batch is retained. + type CheckpointRow = std::result::Result<(u8, Vec, Vec), redb::StorageError>; + let rows: Box + Send> = + Box::new(std::iter::from_fn(move || { + let (kind, item) = if !reading_receipts { + match state_rows.next() { + Some(item) => (PUT, item), + None => { + reading_receipts = true; + (RECEIPT, receipt_rows.next()?) + } + } + } else { + (RECEIPT, receipt_rows.next()?) + }; + Some(match item { + Ok((key, value)) => Ok((kind, key.value().to_vec(), value.value().to_vec())), + Err(error) => Err(error), + }) + })); + let mut rows = rows.peekable(); let mut schema = table_schema( &Schema::new(vec![ Field::new("kind", DataType::UInt8, false), @@ -418,17 +432,17 @@ impl LocalLancePartition { let mut values = Vec::new(); let mut bytes = 0usize; for _ in 0..batch_rows { - if let Some((_, Ok((key, value)))) = rows.peek() { - let next_bytes = key.value().len().saturating_add(value.value().len()); + if let Some(Ok((_, key, value))) = rows.peek() { + let next_bytes = key.len().saturating_add(value.len()); if !kinds.is_empty() && bytes.saturating_add(next_bytes) > batch_bytes { break; } } - let Some((kind, result)) = rows.next() else { + let Some(result) = rows.next() else { ended = true; break; }; - let (key, value) = match result { + let (kind, key, value) = match result { Ok(value) => value, Err(error) => { ended = true; @@ -437,7 +451,7 @@ impl LocalLancePartition { )))); } }; - bytes += key.value().len() + value.value().len(); + bytes += key.len() + value.len(); if bytes > batch_bytes { ended = true; return Some(Err(arrow_schema::ArrowError::InvalidArgumentError( @@ -445,8 +459,8 @@ impl LocalLancePartition { ))); } kinds.push(kind); - keys.push(key.value().to_vec()); - values.push(value.value().to_vec()); + keys.push(key); + values.push(value); } if kinds.is_empty() { return None; @@ -589,6 +603,65 @@ impl LocalLancePartition { Ok(Some(u64::from_le_bytes(bytes[..8].try_into().unwrap()))) } + /// Conservative admission charge for a call, including repeated receipt + /// columns, IPC framing and decoded Arrow buffers. Summed charges bound all + /// three limits used by `commit` and staging. This does not publish data. + pub fn batch_charge(&self, call: &AlignedCall) -> Result { + let rows = call + .mutations + .len() + .saturating_add(call.records.num_rows()) + .saturating_add(1); + let overhead = call + .session + .len() + .saturating_add(call.receipt.len()) + .saturating_add(call.input_digest.len()) + .saturating_add(64); + let mut input = rows + .saturating_mul(overhead) + .saturating_add(call.records.get_array_memory_size()); + for cell in &call.mutations { + let size = cell + .key + .len() + .saturating_add(cell.value.as_ref().map_or(0, Vec::len)); + if size > 1 << 20 { + return Err(Error::Invalid( + "state cell exceeds 1 MiB; split the state into smaller keys".into(), + )); + } + input = input.saturating_add(size); + } + if input > self.max_batch_bytes { + return Err(Error::Invalid( + "local Lance input exceeds byte budget".into(), + )); + } + let schema = Arc::new(Schema::from(self.sink.dataset().schema())); + if log_schema(call.records.schema().as_ref(), &self.binding) != *schema { + return Err(Error::Invalid( + "output schema differs from durable log".into(), + )); + } + let encoded = encode_call(schema, call, self.batch_rows)?; + let mut retained = crate::lance_sink::RetainedArrowBudget::default(); + let mut batches = Vec::new(); + let mut decoded = 0usize; + for batch in arrow_ipc::reader::StreamReader::try_new(std::io::Cursor::new(&encoded), None) + .map_err(fail)? + { + let batch = batch.map_err(fail)?; + decoded = retained.add(&batch); + batches.push(batch); + } + Ok(input.max(encoded.len()).max(decoded)) + } + + pub fn max_batch_bytes(&self) -> usize { + self.max_batch_bytes + } + /// Publish one coalesced batch, then apply its local transaction. Success /// means both output and state deltas are durable in the SAME Lance version. /// After uncertain success reopen and check receipts before realigning. diff --git a/crates/lance-context-ingestion/src/local_lance_reader.rs b/crates/lance-context-ingestion/src/local_lance_reader.rs new file mode 100644 index 00000000..bd276e20 --- /dev/null +++ b/crates/lance-context-ingestion/src/local_lance_reader.rs @@ -0,0 +1,312 @@ +//! Version-pinned reads for an independent consumer of a local-state Lance log. +use crate::{lance_sink::covered_sequence_at, Binding, Error, Result}; +use arrow_array::{Array, RecordBatch, StructArray}; +use futures::TryStreamExt; +use lance::Dataset; + +fn fail(error: impl std::fmt::Display) -> Error { + Error::Stage(error.to_string()) +} + +/// A complete immutable range. Versions identify manifests; sequences identify +/// entries, and neither is the downstream consumer's batch/generation number. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct LogRange { + pub base_version: u64, + pub version: u64, + pub first_sequence: u64, + pub last_sequence: u64, +} + +/// Validates both watermarks and full fragment ancestry before reading. The +/// physical byte count is the actual newly referenced files, not a descriptor's +/// small serialized size. State/receipt rows never appear in returned output. +pub struct OutputRange { + dataset: Dataset, + fragments: Vec, + pub physical_bytes: u64, + range: LogRange, +} + +impl OutputRange { + pub async fn open(dataset: Dataset, binding: &Binding, range: LogRange) -> Result { + if range.base_version == 0 + || range.version <= range.base_version + || dataset.version().version != range.version + || range.first_sequence == 0 + || range.last_sequence < range.first_sequence + { + return Err(Error::Invalid("invalid immutable Lance log range".into())); + } + let schema = &dataset.schema().metadata; + if schema.get("lance-context.ingestion.partition") != Some(&binding.partition.to_string()) + || schema + .get("lance-context.ingestion.local-state-format") + .map(String::as_str) + != Some("1") + { + return Err(Error::Invalid("unexpected local Lance log binding".into())); + } + let base = dataset + .checkout_version(range.base_version) + .await + .map_err(fail)?; + if covered_sequence_at(&base, binding).await?.checked_add(1) != Some(range.first_sequence) + || covered_sequence_at(&dataset, binding).await? != range.last_sequence + || base.schema() != dataset.schema() + { + return Err(Error::Invalid( + "Lance log range watermark/schema mismatch".into(), + )); + } + let previous = base.get_fragments(); + let current = dataset.get_fragments(); + let by_id = current + .iter() + .map(|f| (f.id(), f.metadata())) + .collect::>(); + if previous + .iter() + .any(|old| by_id.get(&old.id()).copied() != Some(old.metadata())) + { + return Err(Error::Invalid("Lance log range is not append-only".into())); + } + let previous_ids = previous + .iter() + .map(|f| f.id()) + .collect::>(); + let fragments = current + .into_iter() + .filter(|new| !previous_ids.contains(&new.id())) + .map(|f| f.metadata().clone()) + .collect::>(); + let mut physical_bytes = 0u64; + for fragment in &fragments { + for file in &fragment.files { + let size = file + .file_size_bytes + .get() + .ok_or_else(|| Error::Invalid("Lance log file size missing".into()))? + .get(); + physical_bytes = physical_bytes + .checked_add(size) + .ok_or_else(|| Error::Invalid("Lance log file sizes overflow".into()))?; + } + } + if fragments.is_empty() { + return Err(Error::Invalid( + "Lance log range lacks sealed receipt rows".into(), + )); + } + Ok(Self { + dataset, + fragments, + physical_bytes, + range, + }) + } + + /// Bound both referenced physical data and retained decoded output. The + /// caller separately budgets scanner working memory and its downstream + /// representation. Errors return no partial success or advancement cursor. + pub async fn read( + self, + max_physical_bytes: u64, + max_decoded_bytes: usize, + batch_rows: usize, + ) -> Result> { + if self.physical_bytes > max_physical_bytes || max_decoded_bytes == 0 || batch_rows == 0 { + return Err(Error::Invalid("Lance log range exceeds read budget".into())); + } + let mut scan = self.dataset.scan(); + scan.with_fragments(self.fragments) + .batch_size(batch_rows) + .scan_in_order(true); + scan.project(&["record"]).map_err(fail)?; + scan.filter(&format!( + "kind = 3 AND sequence >= {} AND sequence <= {}", + self.range.first_sequence, self.range.last_sequence + )) + .map_err(fail)?; + let mut stream = scan.try_into_stream().await.map_err(fail)?; + let mut bytes = 0usize; + let mut output = Vec::new(); + while let Some(batch) = stream.try_next().await.map_err(fail)? { + bytes = bytes + .checked_add(batch.get_array_memory_size()) + .ok_or_else(|| Error::Invalid("decoded Lance log size overflow".into()))?; + if bytes > max_decoded_bytes { + return Err(Error::Invalid( + "decoded Lance log exceeds read budget".into(), + )); + } + let records = batch + .column(0) + .as_any() + .downcast_ref::() + .ok_or_else(|| Error::Invalid("Lance log output is not typed records".into()))?; + if records.null_count() != 0 { + return Err(Error::Invalid("null Lance log output row".into())); + } + output.push(RecordBatch::from(records.clone())); + } + Ok(output) + } +} + +/// Small immutable object-store descriptor for a downstream batch publisher. +/// Application metadata is explicitly binary and bounded, e.g. native cumulative +/// counters and migration provenance. OutputRange independently checks the +/// referenced files; a descriptor is never accepted as proof of payload size. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct BatchReference { + pub binding: Binding, + pub uri: String, + pub generation: u64, + pub previous_generation: u64, + pub range: LogRange, + pub physical_bytes: u64, + pub decoded_byte_limit: u64, + pub output_rows: u64, + pub application_metadata: Vec, +} + +impl BatchReference { + pub fn encode(&self) -> Result> { + use sha2::{Digest, Sha256}; + if self.previous_generation.checked_add(1) != Some(self.generation) + || self.range.base_version == 0 + || self.range.version <= self.range.base_version + || self.range.first_sequence == 0 + || self.range.last_sequence < self.range.first_sequence + || self.decoded_byte_limit == 0 + || self.physical_bytes == 0 + || [&self.binding.run, &self.binding.schema, &self.uri] + .iter() + .any(|s| s.is_empty() || s.len() > 4096) + || self.application_metadata.len() > 64 * 1024 + { + return Err(Error::Invalid( + "invalid binary Lance batch reference".into(), + )); + } + let mut bytes = b"LCBATCH1".to_vec(); + for n in [ + self.generation, + self.previous_generation, + self.range.base_version, + self.range.version, + self.range.first_sequence, + self.range.last_sequence, + self.physical_bytes, + self.decoded_byte_limit, + self.output_rows, + ] { + bytes.extend_from_slice(&n.to_le_bytes()); + } + bytes.extend_from_slice(&self.binding.partition.to_le_bytes()); + for value in [ + self.binding.run.as_bytes(), + self.binding.schema.as_bytes(), + self.uri.as_bytes(), + self.application_metadata.as_slice(), + ] { + bytes.extend_from_slice(&(value.len() as u32).to_le_bytes()); + bytes.extend_from_slice(value); + } + let digest = Sha256::digest(&bytes); + bytes.extend_from_slice(&digest); + Ok(bytes) + } + + pub fn decode(bytes: &[u8]) -> Result { + use sha2::{Digest, Sha256}; + use std::io::{Cursor, Read}; + if bytes.len() < 8 + 9 * 8 + 4 + 4 * 4 + 32 + || bytes.len() > 128 * 1024 + || &bytes[..8] != b"LCBATCH1" + { + return Err(Error::Invalid( + "invalid binary Lance batch reference size/format".into(), + )); + } + let (payload, digest) = bytes.split_at(bytes.len() - 32); + if Sha256::digest(payload).as_slice() != digest { + return Err(Error::Invalid( + "binary Lance batch reference checksum mismatch".into(), + )); + } + let mut cursor = Cursor::new(&payload[8..]); + let mut number = || -> Result { + let mut b = [0u8; 8]; + cursor.read_exact(&mut b).map_err(fail)?; + Ok(u64::from_le_bytes(b)) + }; + let generation = number()?; + let previous_generation = number()?; + let base_version = number()?; + let version = number()?; + let first_sequence = number()?; + let last_sequence = number()?; + let physical_bytes = number()?; + let decoded_byte_limit = number()?; + let output_rows = number()?; + let mut p = [0u8; 4]; + cursor.read_exact(&mut p).map_err(fail)?; + let partition = u32::from_le_bytes(p); + let mut field = || -> Result> { + let mut b = [0u8; 4]; + cursor.read_exact(&mut b).map_err(fail)?; + let len = u32::from_le_bytes(b) as usize; + if len > 64 * 1024 + || len + > cursor + .get_ref() + .len() + .saturating_sub(cursor.position() as usize) + { + return Err(Error::Invalid( + "binary Lance batch reference field size".into(), + )); + } + let mut value = vec![0; len]; + cursor.read_exact(&mut value).map_err(fail)?; + Ok(value) + }; + let run = String::from_utf8(field()?).map_err(fail)?; + let schema = String::from_utf8(field()?).map_err(fail)?; + let uri = String::from_utf8(field()?).map_err(fail)?; + let application_metadata = field()?; + if cursor.position() as usize != payload.len() - 8 { + return Err(Error::Invalid( + "trailing binary Lance batch reference data".into(), + )); + } + let value = Self { + binding: Binding { + run, + schema, + partition, + }, + uri, + generation, + previous_generation, + range: LogRange { + base_version, + version, + first_sequence, + last_sequence, + }, + physical_bytes, + decoded_byte_limit, + output_rows, + application_metadata, + }; + if value.encode()?.as_slice() != bytes { + return Err(Error::Invalid( + "noncanonical binary Lance batch reference".into(), + )); + } + Ok(value) + } +} diff --git a/crates/lance-context-ingestion/tests/local_lance.rs b/crates/lance-context-ingestion/tests/local_lance.rs index 03da8b9c..19cf20dd 100644 --- a/crates/lance-context-ingestion/tests/local_lance.rs +++ b/crates/lance-context-ingestion/tests/local_lance.rs @@ -543,3 +543,217 @@ async fn lost_manifest_ack_recovers_both_local_state_and_receipts_without_new_ve version ); } + +#[tokio::test] +async fn immutable_output_range_reads_only_its_payload_and_enforces_real_byte_budgets() { + use lance_context_ingestion::local_lance_reader::{LogRange, OutputRange}; + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wal.lance"); + let dataset = table(uri.to_str().unwrap()).await; + let mut writer = open(&dir.path().join("writer.redb"), dataset).await; + writer + .commit(&[call(1, "old", Some(b"old"), &["old"])]) + .await + .unwrap(); + let base = writer.dataset().version().version; + let files = writer + .dataset() + .get_fragments() + .iter() + .flat_map(|f| { + f.metadata() + .files + .iter() + .map(|f| f.path.clone()) + .collect::>() + }) + .collect::>(); + writer + .commit(&[ + call(2, "s", Some(b"new-state"), &["new-output"]), + call(3, "s", None, &[]), + ]) + .await + .unwrap(); + let version = writer.dataset().version().version; + let range = LogRange { + base_version: base, + version, + first_sequence: 2, + last_sequence: 3, + }; + for file in files { + let path = uri.join("data").join(file); + std::fs::rename(&path, path.with_extension("hidden")).unwrap(); + } + let snapshot = writer.dataset().clone(); + let input = OutputRange::open(snapshot.clone(), &binding(), range.clone()) + .await + .unwrap(); + assert!(input.physical_bytes > 0); + assert!(input.read(1, 1 << 20, 2).await.is_err()); + let input = OutputRange::open(snapshot.clone(), &binding(), range.clone()) + .await + .unwrap(); + assert!(input.read(1 << 20, 1, 2).await.is_err()); + let input = OutputRange::open(snapshot.clone(), &binding(), range.clone()) + .await + .unwrap(); + let rows = input.read(1 << 20, 1 << 20, 2).await.unwrap(); + assert_eq!(rows.iter().map(RecordBatch::num_rows).sum::(), 1); + assert_eq!( + rows[0] + .column(0) + .as_any() + .downcast_ref::() + .unwrap() + .value(0), + "new-output" + ); + let wrong = LogRange { + first_sequence: 1, + ..range.clone() + }; + assert!(OutputRange::open(snapshot.clone(), &binding(), wrong) + .await + .is_err()); + writer + .commit(&[call(4, "s", Some(b"later"), &["later"])]) + .await + .unwrap(); + assert!( + OutputRange::open(writer.dataset().clone(), &binding(), range.clone()) + .await + .is_err() + ); + assert_eq!( + OutputRange::open(snapshot, &binding(), range) + .await + .unwrap() + .read(1 << 20, 1 << 20, 2) + .await + .unwrap() + .iter() + .map(RecordBatch::num_rows) + .sum::(), + 1 + ); + let before = writer.dataset().version().version; + writer.commit(&[call(5, "s", None, &[])]).await.unwrap(); + let empty = LogRange { + base_version: before, + version: writer.dataset().version().version, + first_sequence: 5, + last_sequence: 5, + }; + assert!( + OutputRange::open(writer.dataset().clone(), &binding(), empty) + .await + .unwrap() + .read(1 << 20, 1 << 20, 2) + .await + .unwrap() + .is_empty() + ); +} + +#[test] +fn binary_batch_reference_rejects_corruption_truncation_and_invalid_continuity() { + use lance_context_ingestion::local_lance_reader::{BatchReference, LogRange}; + use sha2::{Digest, Sha256}; + let mut reference = BatchReference { + binding: binding(), + uri: "az://bucket/session-log.lance".into(), + generation: 7, + previous_generation: 6, + range: LogRange { + base_version: 11, + version: 13, + first_sequence: 9, + last_sequence: 21, + }, + physical_bytes: 105_000, + decoded_byte_limit: 1 << 20, + output_rows: 30, + application_metadata: vec![0, 255, 13, 10], + }; + let bytes = reference.encode().unwrap(); + assert_eq!(BatchReference::decode(&bytes).unwrap(), reference); + for len in 0..bytes.len() { + assert!(BatchReference::decode(&bytes[..len]).is_err()); + } + for i in 0..bytes.len() { + let mut changed = bytes.clone(); + changed[i] ^= 1; + assert!(BatchReference::decode(&changed).is_err()); + } + // Recompute checksum so framing/continuity validation is independently tested. + let mut changed = bytes[..bytes.len() - 32].to_vec(); + changed[8..16].copy_from_slice(&9u64.to_le_bytes()); + changed.extend_from_slice(&Sha256::digest(&changed)); + assert!(BatchReference::decode(&changed).is_err()); + let mut trailing = bytes[..bytes.len() - 32].to_vec(); + trailing.push(0); + trailing.extend_from_slice(&Sha256::digest(&trailing)); + assert!(BatchReference::decode(&trailing).is_err()); + reference.application_metadata = vec![0; 65537]; + assert!(reference.encode().is_err()); + reference.application_metadata.clear(); + reference.uri = "x".repeat(4097); + assert!(reference.encode().is_err()); +} + +#[tokio::test] +async fn wide_shared_ipc_buffers_keep_small_calls_in_one_bounded_commit() { + let dir = tempfile::tempdir().unwrap(); + let uri = dir.path().join("wide.lance"); + let output_schema = Arc::new(Schema::new( + (0..24) + .map(|i| Field::new(format!("field{i}"), DataType::Utf8, false)) + .collect::>(), + )); + let output = RecordBatch::try_new( + output_schema.clone(), + (0..24) + .map(|_| { + Arc::new(StringArray::from(vec!["small escaped 世界\nvalue"])) + as arrow_array::ArrayRef + }) + .collect(), + ) + .unwrap(); + let schema = Arc::new(log_schema(&output_schema, &binding())); + let dataset = Dataset::write( + RecordBatchIterator::new(vec![Ok(RecordBatch::new_empty(schema.clone()))], schema), + uri.to_str().unwrap(), + Some(WriteParams { + data_storage_version: Some(LanceFileVersion::V2_2), + ..Default::default() + }), + ) + .await + .unwrap(); + let mut writer = open(&dir.path().join("wide.redb"), dataset).await; + let calls = (1..=128) + .map(|sequence| AlignedCall { + sequence, + session: "session".into(), + receipt: format!("r{sequence}"), + input_digest: format!("d{sequence}"), + mutations: vec![], + records: output.clone(), + }) + .collect::>(); + let charge = calls + .iter() + .map(|call| writer.batch_charge(call).unwrap()) + .sum::(); + assert!( + charge <= 4 << 20, + "shared allocations overcharged: {charge}" + ); + let base = writer.dataset().version().version; + writer.commit(&calls).await.unwrap(); + assert_eq!(writer.dataset().version().version, base + 1); + assert_eq!(writer.through_sequence(), 128); +}