From 3f37969684a1e2a158c7f2637596b3b2b53be5ed Mon Sep 17 00:00:00 2001 From: rldyourmnd Date: Sun, 27 Sep 2026 23:34:50 +0500 Subject: [PATCH 1/2] test(rds-net): widen mux_isolation propagation budgets for loaded macOS CI MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The sibling-isolation fixture polls distributed state: path-set propagation and NAT-traversal advertisement receipt/removal on the peer. On loaded macos-latest runners 2s was marginal — CI observed the client holding a validated IPv6 sibling while the server's path set had not yet observed it, failing terminal_child_receive_does_not_stop_healthy_path at the pre-fault observation wait. Raise the peer-observation waits (2s -> 8s sibling, 2s -> 5s advertisement) within the existing 15s outer budget. Local single-socket wait and functional-phase budgets are unchanged. --- crates/rds-net/tests/mux_isolation.rs | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/crates/rds-net/tests/mux_isolation.rs b/crates/rds-net/tests/mux_isolation.rs index 979490e..f313adf 100644 --- a/crates/rds-net/tests/mux_isolation.rs +++ b/crates/rds-net/tests/mux_isolation.rs @@ -246,7 +246,7 @@ async fn exercise(receive: bool) { // PathId. Both policies must observe a validated IPv6 sibling, not // necessarily the particular path returned by our explicit open. let facade = rds_net::Connection::from(b.clone()); - tokio::time::timeout(Duration::from_secs(2), async { + tokio::time::timeout(Duration::from_secs(8), async { while observed_path_to(&a, secondary_addr, false).is_none() || observed_path_to(&b, client_secondary_addr, false).is_none() { @@ -269,7 +269,7 @@ async fn exercise(receive: bool) { let mut prefix = [0; 6]; request.read_exact(&mut prefix).await.unwrap(); assert_eq!(&prefix, b"before"); - tokio::time::timeout(Duration::from_secs(2), async { + tokio::time::timeout(Duration::from_secs(5), async { while !a .inner() .get_remote_nat_traversal_addresses() @@ -337,7 +337,7 @@ async fn exercise(receive: bool) { assert_eq!(health.snapshot()[0].failed, Some(io::ErrorKind::BrokenPipe)); assert!(health.snapshot()[1].failed.is_none()); assert!(observed_path_to(&b, client_secondary_addr, true).is_some()); - tokio::time::timeout(Duration::from_secs(2), async { + tokio::time::timeout(Duration::from_secs(5), async { while a .inner() .get_remote_nat_traversal_addresses() From f99a3b88a36a87ef7afb5fe567555b2282a837de Mon Sep 17 00:00:00 2001 From: rldyourmnd Date: Sun, 27 Sep 2026 23:51:23 +0500 Subject: [PATCH 2/2] fix(rds-net): reconcile policy path set against lagged event delivery MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The connection driver learned validated paths only through the PathEvent::Established broadcast (capacity 32). A lagged receiver loses events permanently: a validated sibling path stayed invisible to path_stats, telemetry and reselection, which macOS CI exposed as mux_isolation timing out with the server never observing the client's IPv6 sibling. Rescan the bounded negotiated PathId space each driver iteration and adopt live paths whose PATH_RESPONSE was received — the same condition that emits Established — so event loss self-heals on the next loop turn. Unvalidated candidates still stay out of selection. Dead-path removal was already handled by reselect's retain. --- crates/rds-net/src/backends/noq/policy.rs | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/crates/rds-net/src/backends/noq/policy.rs b/crates/rds-net/src/backends/noq/policy.rs index 9607153..425c7be 100644 --- a/crates/rds-net/src/backends/noq/policy.rs +++ b/crates/rds-net/src/backends/noq/policy.rs @@ -388,6 +388,24 @@ pub(super) async fn connection_driver_observed( } // Lost/lagged path events cannot accumulate stale QNT history. qnt_paths.retain(|id| owner.path(*id).is_some()); + // Established/Abandoned delivery is a bounded broadcast: a lagged + // event must not hide a live path from selection nor keep a dead + // one selected. The negotiated PathId space is bounded, so + // reconcile by rescan rather than trusting event completeness. + // Adoption keeps the Established boundary: a path joins selection + // only after our challenge was answered, the same condition that + // emits the event — unvalidated candidates stay out. + for raw in 0..super::MAX_MULTIPATH_PATHS { + let id = noq::PathId::from(raw); + if paths.contains_key(&id) { + continue; + } + if let Some(path) = owner.path(id) + && path.stats().frame_rx.path_response > 0 + { + paths.insert(id, path.weak_handle()); + } + } drop(owner); reselect( &conn,