Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,7 @@ jobs:
test_matrix_os: '["ubuntu-latest", "macos-latest"]'
test_command: >-
cargo test --locked --workspace --no-fail-fast
&& cargo test --locked --target-dir target --manifest-path vendor/noq-proto/Cargo.toml --lib --no-fail-fast
&& cargo test --locked -p rds-net -p rds-agent -p rds-cli -p rds-relay -p rds-server --features rds-net/transport-noq,rds-agent/transport-noq,rds-cli/transport-noq,rds-relay/owned-relay,rds-server/owned-relay --no-fail-fast
&& cargo test --locked -p rds-net --features metrics --test metrics --no-fail-fast
&& cargo test --locked -p rds-desktop -p rds-cli --features rds-cli/desktop --no-fail-fast
Expand Down
85 changes: 85 additions & 0 deletions crates/rds-net/src/backends/iroh/latency_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -451,6 +451,91 @@ async fn actual_iroh_actor_reopens_a_retired_standby_before_the_next_failure() {
.expect("bounded standby-restoration fixture did not terminate");
}

#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn actual_iroh_actor_establishes_a_standby_that_was_blocked_at_startup() {
tokio::time::timeout(Duration::from_secs(25), async {
let fabric = Fabric::default();
let blocked = Arc::new(AtomicBool::new(true));
let dropped = Arc::new(AtomicU64::new(0));
let mut endpoints = Vec::new();
for name in [b"a", b"b"] {
endpoints.push(
endpoint_with_selector(
Link {
addr: CustomAddr::from_parts(0x726473636f6c6431, name),
fabric: fabric.clone(),
drop_packets: blocked.clone(),
dropped: dropped.clone(),
},
Arc::new(super::latency::LatencySelector),
)
.await,
);
}
let a = &endpoints[0];
let b = &endpoints[1];
// Include the configured standby even before its local watcher has
// published the first address snapshot, like a complete bootstrap ticket.
let mut target = b.addr();
target
.addrs
.insert(TransportAddr::Custom(CustomAddr::from_parts(
0x726473636f6c6431,
b"b",
)));
let (client, server) = tokio::time::timeout(Duration::from_secs(5), async {
tokio::join!(a.connect(target, rds_core::ALPN), async {
b.accept().await.unwrap().await
})
})
.await
.expect("the blocked standby delayed the healthy initial route");
let (client, server) = (client.unwrap(), server.unwrap());
let (mut request, mut response) = client.open_bi().await.unwrap();
request.write_all(b"before").await.unwrap();
let (mut reply, mut input) = server.accept_bi().await.unwrap();
let mut body = [0; 6];
input.read_exact(&mut body).await.unwrap();
assert_eq!(&body, b"before");
reply.write_all(b"ok").await.unwrap();
let mut ack = [0; 2];
response.read_exact(&mut ack).await.unwrap();
assert_eq!(&ack, b"ok");

// Miss several initial validation probes while the live IP route keeps
// this same connection and stream usable.
tokio::time::sleep(Duration::from_secs(8)).await;
assert!(dropped.load(Ordering::Relaxed) > 0);
assert!(client.paths().iter().all(|path| path.is_ip()));
blocked.store(false, Ordering::Release);
let restored_at = Instant::now();
tokio::time::timeout(Duration::from_secs(5), async {
while !client.paths().iter().any(|path| !path.is_ip())
|| !server.paths().iter().any(|path| !path.is_ip())
{
tokio::time::sleep(Duration::from_millis(20)).await;
}
})
.await
.expect("the restored initial standby did not validate within five seconds");
eprintln!(
"initial standby validation after restore: {} ms",
restored_at.elapsed().as_millis()
);
request.write_all(b"after!").await.unwrap();
input.read_exact(&mut body).await.unwrap();
assert_eq!(&body, b"after!");
reply.write_all(b"ok").await.unwrap();
response.read_exact(&mut ack).await.unwrap();
assert_eq!(&ack, b"ok");
assert!(client.close_reason().is_none());
assert!(server.close_reason().is_none());
tokio::join!(a.close(), b.close());
})
.await
.unwrap();
}

#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn actual_iroh_actor_retires_blackholed_preferred_link_and_delivers_pending_stream_bytes() {
tokio::time::timeout(Duration::from_secs(15),async {
Expand Down
140 changes: 140 additions & 0 deletions crates/rds-net/tests/relay_registrations.rs
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,146 @@ async fn relay() -> (iroh_relay::server::Server, String) {
(server, url)
}

struct RelayGate {
url: String,
ready: tokio::sync::watch::Sender<bool>,
task: tokio::task::JoinHandle<()>,
}

impl Drop for RelayGate {
fn drop(&mut self) {
self.task.abort();
}
}

async fn relay_gate(destination: std::net::SocketAddr) -> RelayGate {
let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
let url = format!("http://{}", listener.local_addr().unwrap());
let (ready, readiness) = tokio::sync::watch::channel(false);
let task = tokio::spawn(async move {
let mut clients = tokio::task::JoinSet::new();
let mut first_connection = true;
loop {
tokio::select! {
accepted = listener.accept(), if clients.len() < 16 => {
let (mut downstream, _) = accepted.unwrap();
let mut ready = readiness.clone();
let already_registered_side = first_connection;
first_connection = false;
clients.spawn(async move {
while !already_registered_side && !*ready.borrow() {
if ready.changed().await.is_err() {
return;
}
}
if let Ok(mut upstream) = tokio::net::TcpStream::connect(destination).await {
let _ = tokio::io::copy_bidirectional(&mut downstream, &mut upstream).await;
}
});
}
Some(_) = clients.join_next(), if !clients.is_empty() => {}
}
}
});
RelayGate { url, ready, task }
}

#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn late_relay_registration_validates_the_original_connection_without_replay() {
let (primary, primary_url) = relay().await;
let (standby, _) = relay().await;
let gate = relay_gate(standby.http_addr().unwrap()).await;
let config = || EndpointConfig {
keep_relays_connected: true,
prefer_relay_order: true,
transports: Transports::RelayOnly,
path_preference: PathPreference::Latency,
..EndpointConfig::default()
.with_relays([&primary_url, &gate.url])
.unwrap()
};
let server = rds_net::backends::iroh::bind_endpoint(config())
.await
.unwrap();
// The receiver is already registered on both relays, as with a healthy
// server and a client whose external route is initially filtered.
timeout(Duration::from_secs(10), async {
while server.addr().addrs.len() != 2 {
sleep(Duration::from_millis(20)).await;
}
})
.await
.expect("receiver registrations must be ready before the client starts");
let client = rds_net::backends::iroh::bind_endpoint(config())
.await
.unwrap();
timeout(Duration::from_secs(10), async {
while client.addr().addrs.len() != 1 {
sleep(Duration::from_millis(20)).await;
}
})
.await
.expect("one authenticated registration, despite accepted standby TCP sockets");
let mut ticket = server.addr();
ticket
.addrs
.insert(iroh::TransportAddr::Relay(gate.url.parse().unwrap()));
let (outgoing, incoming) = timeout(Duration::from_secs(5), async {
tokio::join!(client.connect(ticket, rds_core::ALPN), async {
server.accept().await.unwrap().await
})
})
.await
.unwrap();
let (outgoing, incoming) = (outgoing.unwrap(), incoming.unwrap());
let (mut tx, mut rx) = outgoing.open_bi().await.unwrap();
tx.write_all(b"before").await.unwrap();
let (mut reply, mut input) = incoming.accept_bi().await.unwrap();
let mut body = [0; 6];
input.read_exact(&mut body).await.unwrap();
assert_eq!(&body, b"before");
reply.write_all(b"ok").await.unwrap();
let mut ack = [0; 2];
rx.read_exact(&mut ack).await.unwrap();

// A successful local TCP connect is deliberately insufficient here: the
// WebSocket/authenticated relay protocol cannot complete yet.
sleep(Duration::from_secs(16)).await;
assert_eq!(client.addr().addrs.len(), 1);
assert_eq!(outgoing.paths().iter().count(), 1);
gate.ready.send(true).unwrap();
timeout(Duration::from_secs(15), async {
while server.addr().addrs.len() != 2 || client.addr().addrs.len() != 2 {
sleep(Duration::from_millis(20)).await;
}
})
.await
.expect("authenticated relay readiness did not recover");
let registered_at = Instant::now();
timeout(Duration::from_secs(5), async {
while outgoing.paths().iter().count() != 2 || incoming.paths().iter().count() != 2 {
sleep(Duration::from_millis(20)).await;
}
})
.await
.expect("late authenticated relay did not validate the existing connection's path");
eprintln!(
"late relay path validation after registration: {} ms",
registered_at.elapsed().as_millis()
);
tx.write_all(b"after!").await.unwrap();
input.read_exact(&mut body).await.unwrap();
assert_eq!(&body, b"after!");
reply.write_all(b"ok").await.unwrap();
rx.read_exact(&mut ack).await.unwrap();
assert_eq!(&ack, b"ok");
assert!(outgoing.close_reason().is_none());
assert!(incoming.close_reason().is_none());
tokio::join!(client.close(), server.close());
primary.shutdown().await.unwrap();
standby.shutdown().await.unwrap();
}

#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn three_idle_registrations_survive_two_relay_failures_on_the_same_stream() {
let (first, first_url) = relay().await;
Expand Down
7 changes: 7 additions & 0 deletions docs/architecture.md
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,13 @@ idle ACK cannot prove replacement progress during a common outage. The existing
stream transfers to that sibling through Noq's ordinary path-abandon machinery;
the last open path is never retired. See [the recovery evidence](reports/rds-path-ack-progress-20261006.md).

Initial multipath validation has its own engine deadline, independent of idle
policy for established paths. The budget uses three times the larger initial
PTO and PTO of live validated paths. An unsuccessful attempt closes through
existing PATH_ABANDON/CID handling; its ID is not reused. Validation success
cancels this timer. RFC 9000 address migration keeps its separate timeout and
previous-route fallback, and last-path protection remains in the engine.

Tagged sync completion now transfers connection exclusion and the data-lane
admission permit into the engine. Filesystem/control work and its reader close before that guard releases,
and only then is the terminal server FIN exposed to the peer. This makes an
Expand Down
1 change: 1 addition & 0 deletions docs/continuation-plan.md
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@ older revisions independently; neither is silently assumed to equal main.
| S6 | Optional platform/media modules remain real stubs. Directory sync, native serving on macOS/Wayland, audio and automatic GDS grant issuance are not implemented. | Backend probe files, audio crate, sync engine, GDS policy adapters | Explicit later implementation/qualification tasks; never mark these complete from X11 or loopback results |
| S7 | Owned relay drain priority was reused as RTT, making progress budgets effectively infinite. | Noq path policy | Keep real RTT separate from ranking; negative deadline regression and positive engine/drain qualification |
| S8 | A compiled architecture test retained an absolute source-checkout filename, making shared-cache verification depend on a removed checkout. | Core layering test; infrastructure fixture path resolution | Embed source-only assertions; resolve required live fixture files from the runtime checkout; preserve the original assertions |
| S9 | Initial path validation without a finite path idle policy can retain a failed standby attempt and its long probe backoff. | Shared Noq-proto engine used by both backends | Independent validation deadline, explicit abandonment/fresh-ID recovery, last-path protection, full engine suite and real actor restoration |

## Wave A — interactive recovery correctness (W3.6, W6.8)

Expand Down
56 changes: 56 additions & 0 deletions docs/reports/rds-initial-path-validation-20261007.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
# Initial multipath validation — 2026-10-07

A newly allocated multipath path could remain unvalidated indefinitely when
its established-path idle policy was disabled. `open_path_ensure` continued to
return that same unvalidated attempt, retaining its challenge backoff. Initial
path creation did not arm an independent validation deadline; the existing
validation timer belonged to RFC 9000 address migration.

## Correction and boundaries

Add a dedicated initial validation timer using three times the larger initial
PTO and PTO of live validated paths. Successful validation cancels it. An expired
unvalidated attempt closes through existing PATH_ABANDON, CID retirement and
fresh-ID allocation; the last remaining path is protected. Migration keeps its
separate timer and previous-route fallback. The shared vendored Noq-proto engine
serves both Iroh and owned Noq. Qlog uses the path-validation timer category.
No wire frame, application replay or congestion controller is added.

[RFC 9000 §8.2.4](https://www.rfc-editor.org/rfc/rfc9000.html#section-8.2.4)
recommends a validation timeout based on three PTOs and the larger new/current
estimate. Using all live validated estimates is this implementation's
conservative extension for multiple known paths, not a mandated exact formula.
[Multipath draft-21 §3.1](https://www.ietf.org/archive/id/draft-ietf-quic-multipath-21.html#section-3.1)
requires explicit closure of a failed initiation;
[§3.4](https://www.ietf.org/archive/id/draft-ietf-quic-multipath-21.html#section-3.4)
prohibits reusing its consumed ID. These standards predate the requested
2026-09-26 research boundary; source and verification observations are dated
2026-10-07.

## Retained verification

- The new deterministic no-idle-policy regression failed on the original engine:
no abandonment event arrived after the validation interval. It passes with the
correction, including fresh-ID restoration and surviving established sibling.
- A first full engine candidate passed 427 tests and failed one historical
client/server event-order expectation. Separate initial/migration timers and
an assertion accepting either independently expired or peer-abandoned closure
of the same failed ID retain the actual invariant. The final engine suite has
429 passing tests, including last-path protection.
- macOS: qlog compilation, the complete default/Noq network tests and strict
all-target network Clippy passed. A real Iroh actor with an initially blocked
custom standby restored it in 1006ms and continued the original stream.
- A bounded loopback TCP gate accepts the standby socket while delaying the
authenticated relay protocol on one side. The other endpoint is already
registered. After restoring the gate, the existing connection validates the
standby in 987ms and retains the original bidirectional stream. Gate tasks and
sockets remain owned by the fixture; there are no external hosts.
- Both real actor/gated-relay fixtures also pass on the original production
source. They establish preserved behavior and an acceptance bound, **not** a
causal performance improvement from this correction. An observed installed
recovery delay is not explained by these loopback results.

The full engine suite now runs in both supported CI test lanes. Linux/X11,
current-head CI, immutable release builds and installed qualification remain
required before rollout. This report does not close W3/W6/W10 or claim physical
interface, suspend, power-loss or causal input-to-pixel acceptance.
23 changes: 23 additions & 0 deletions vendor/noq-proto/RDS-PATCH.md
Original file line number Diff line number Diff line change
Expand Up @@ -32,3 +32,26 @@ The MTU constructor's debug assertion keeps its numeric invariant but uses a
static message. CodeQL treated the formatted minimum-MTU value as configuration
data derived from test certificate setup. Removing the interpolation avoids
that diagnostic data flow without changing the MTU decision or disabling scans.

## Initial multipath validation deadline

A new unvalidated path now has a dedicated deadline independent of its idle
policy. The budget is three times the larger initial PTO and PTO of the live
validated paths. This conservative choice follows RFC 9000 §8.2.4 guidance;
using all live validated estimates avoids judging an unknown route solely from
an unrelated faster path. RFC 9000 migration retains its separate timer and
previous-route fallback. Validation success cancels the initial timer. Failure
closes the attempt with the existing TimedOut reason and PATH_ABANDON machinery;
last-path protection, CID retirement, ID nonreuse and retransmission remain in
the existing engine boundary. Qlog records the standard path-validation timer.

No new wire frame, credential, congestion controller or application replay is
introduced. Multipath draft-21 §3.1 requires explicitly closing a failed path
initiation; §3.4 prohibits reusing an abandoned path ID. Deterministic engine
regressions cover no-idle-policy timeout, fresh-ID recovery and last-path
protection. The complete engine unit suite runs on both supported CI targets.
A real Iroh actor fixture blocks an initially configured standby, restores it
and continues the original bidirectional stream through the same connection.

Sources: https://www.rfc-editor.org/rfc/rfc9000.html#section-8.2.4
and https://www.ietf.org/archive/id/draft-ietf-quic-multipath-21.html#section-3.1
Loading
Loading