From b2e957773dd6403b12e367b97d9b4dbfb457422d Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 19:56:21 +0500 Subject: [PATCH 01/15] wip: ensemble fault model, inbox race fix, iroh routable hints Salvage checkpoint before restoring sim link-control wrappers. --- ensemble/src/lib.rs | 696 ++++++++++++++++++++++++++++--- humd/src/peer_transport/iroh.rs | 32 +- sim/src/lib.rs | 108 ++++- sim/tests/lossy_link.rs | 175 ++++++++ sim/tests/partition_and_heal.rs | 4 + sim/tests/two_humds_ping_pong.rs | 15 +- 6 files changed, 960 insertions(+), 70 deletions(-) create mode 100644 sim/tests/lossy_link.rs diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index 115f7e42..3137c375 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -27,13 +27,17 @@ use std::collections::{HashMap, VecDeque}; use std::fmt; +use std::ops::Deref; +use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; use std::sync::Arc; use std::time::Duration; use anyhow::Result; use async_trait::async_trait; use ed25519_dalek::{Signature, Signer, SigningKey, Verifier, VerifyingKey}; -use parking_lot::RwLock; +use parking_lot::{Mutex, RwLock}; +use rand::rngs::StdRng; +use rand::{Rng, SeedableRng}; use serde::{Deserialize, Serialize}; use tokio::sync::{broadcast, mpsc}; use tokio::task::JoinSet; @@ -295,15 +299,146 @@ pub trait Transport: Send + Sync { async fn connect(&self, addr: &HumdAddr) -> Result>; } +// ── Link fault model (sim) ───────────────────────────────────────────────── + +/// Deterministic faults, counted in offered tones. Takes precedence over +/// [`Noise`] so a scripted scenario lands exactly where it says. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct Script { + /// Decremented per offered tone. + pub drop_next: usize, + pub dup_next: usize, + /// Decremented per offered pair; reverses its delivery. + pub reorder_next: usize, + pub drop_every: usize, + pub dup_every: usize, +} + +impl Script { + /// `counted` wins over `every`: a scripted `drop_next` of 3 drops + /// exactly the next 3, whatever the periodic rule would have said. + fn take(counted: &mut usize, every: usize, offered: u64) -> bool { + match *counted { + 0 => every > 0 && offered % every as u64 == 0, + n => { + *counted = n - 1; + true + } + } + } + + fn verdict(&mut self, offered: u64) -> Option { + let drop = Self::take(&mut self.drop_next, self.drop_every, offered); + let duplicate = Self::take(&mut self.dup_next, self.dup_every, offered); + (drop || duplicate).then_some(Verdict { drop, duplicate }) + } +} + +/// Probabilistic faults, drawn from the link's seeded PRNG. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct Noise { + pub drop_pct: u8, + pub dup_pct: u8, +} + +impl Noise { + fn verdict(&self, rng: &mut impl Rng) -> Verdict { + Verdict { + drop: self.drop_pct > 0 && rng.gen_range(0..100u8) < self.drop_pct, + duplicate: self.dup_pct > 0 && rng.gen_range(0..100u8) < self.dup_pct, + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Verdict { + pub drop: bool, + pub duplicate: bool, +} + +#[derive(Debug, Clone)] +pub struct LinkFaults { + pub script: Script, + pub noise: Noise, + seed: u64, + offered: u64, +} + +impl Default for LinkFaults { + fn default() -> Self { + Self { script: Script::default(), noise: Noise::default(), seed: 0x5EED_C0DE, offered: 0 } + } +} + +impl LinkFaults { + pub fn script(mut self, script: Script) -> Self { + self.script = script; + self + } + + pub fn noise(mut self, noise: Noise) -> Self { + self.noise = noise; + self + } + + pub fn drop_next(mut self, n: usize) -> Self { + self.script.drop_next = n; + self + } + + pub fn dup_next(mut self, n: usize) -> Self { + self.script.dup_next = n; + self + } + + pub fn reorder_next(mut self, n: usize) -> Self { + self.script.reorder_next = n; + self + } + + pub fn drop_pct(mut self, pct: u8, seed: u64) -> Self { + self.noise.drop_pct = pct.min(100); + self.seed = seed; + self + } + + pub fn seed(mut self, seed: u64) -> Self { + self.seed = seed; + self + } + + /// `offered` advances for every tone the link accepts, partitioned + /// tones included, so a scenario can be written against a global + /// position rather than a per-state one. + fn verdict(&mut self, rng: &mut impl Rng) -> Verdict { + self.offered += 1; + self.script.verdict(self.offered).unwrap_or_else(|| self.noise.verdict(rng)) + } +} + +/// Ground truth for a link: what it was handed versus what arrived. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct LinkCounters { + pub offered: u64, + pub delivered: u64, + pub dropped: u64, + pub duplicated: u64, + pub reordered: u64, + pub buffered: u64, + pub lost_on_heal: u64, + pub evicted: u64, +} + // ── In-memory transport (sim) ────────────────────────────────────────────── /// Two `InMemoryEndpoint`s wired together with `mpsc` channels. Lets /// the sim build a ring/mesh/star of fake-networked humds inside one /// process with deterministic, low-latency delivery. /// -/// Latency / drop / partition behaviour is a follow-up — for v0 the -/// channels deliver instantly and never drop. The sim layer wraps -/// these with controllable middleware. +/// Delivery is fault-injectable via [`LinkFaults`]: loss, duplication, +/// reordering, and partition all behave the way a real link does, and +/// [`LinkCounters`] records what actually happened. Nothing here runs in +/// production — the daemon only ever sees the [`PeerConnection`] trait. /// Max tones held while partitioned. Realistic enough for sim narratives /// (a few dozen petals during a partition window); large enough not to /// fall behind in the tests we run. If the queue fills, oldest tones drop @@ -315,16 +450,30 @@ pub struct InMemoryEndpoint { peer: HumdAddr, caps: PeerCapabilities, tx: mpsc::Sender, - rx: parking_lot::Mutex>>, - /// Sim-controlled link state. When `dropped == true`, `send()` accepts - /// the tone and buffers it (bounded VecDeque) instead of pushing it - /// to the peer's receiver. On `set_partitioned(false)`, the buffered - /// tones flush to the peer in order before normal operation resumes. - partition: parking_lot::Mutex, + rx: Mutex>>, + /// Sim-controlled link state. When `partitioned == true`, `send()` + /// accepts the tone and buffers it (bounded VecDeque) instead of + /// pushing it to the peer's receiver. On `set_partitioned(false)` + /// the buffer is flushed — but *through the fault model*, not + /// replayed intact. A healing link is still a lossy link; that + /// asymmetry is the whole point, and it is what a real mesh does. + partition: Mutex, + /// Fault profile for this link. Default = perfect, so existing tests + /// keep their current semantics until they opt in. + faults: Mutex, + /// Seeded PRNG for the statistical knobs. Held separately from + /// `faults` so `LinkFaults` stays `Clone` and comparable. + rng: Mutex, + /// Ground truth. Read by tests to assert the ensemble's own loss + /// accounting against what the link actually did. + counters: Mutex, + /// Held tone during a reorder pair, awaiting its partner so the pair + /// can be delivered in reverse. + reorder_hold: Mutex>, } struct PartitionState { - dropped: bool, + partitioned: bool, buffer: VecDeque, } @@ -352,25 +501,39 @@ impl InMemoryEndpoint { ) -> (Arc, Arc) { let (tx_ab, rx_ab) = mpsc::channel::(256); let (tx_ba, rx_ba) = mpsc::channel::(256); + // Each direction of the link gets its own PRNG stream. Sharing one + // seed would make both halves fail identically at the same + // points, which is a correlated outage — the opposite of what a + // two-way link actually does. + let seed_a = LinkFaults::default().seed; + let seed_b = seed_a ^ 0x9E37_79B9_7F4A_7C15; let a = Arc::new(InMemoryEndpoint { peer: HumdAddr::new(b_id), caps: b_caps.clone(), tx: tx_ab, - rx: parking_lot::Mutex::new(Some(rx_ba)), - partition: parking_lot::Mutex::new(PartitionState { - dropped: false, + rx: Mutex::new(Some(rx_ba)), + partition: Mutex::new(PartitionState { + partitioned: false, buffer: VecDeque::new(), }), + faults: Mutex::new(LinkFaults { seed: seed_a, ..Default::default() }), + rng: Mutex::new(StdRng::seed_from_u64(seed_a)), + counters: Mutex::new(LinkCounters::default()), + reorder_hold: Mutex::new(None), }); let b = Arc::new(InMemoryEndpoint { peer: HumdAddr::new(a_id), caps: a_caps, tx: tx_ba, - rx: parking_lot::Mutex::new(Some(rx_ab)), - partition: parking_lot::Mutex::new(PartitionState { - dropped: false, + rx: Mutex::new(Some(rx_ab)), + partition: Mutex::new(PartitionState { + partitioned: false, buffer: VecDeque::new(), }), + faults: Mutex::new(LinkFaults { seed: seed_b, ..Default::default() }), + rng: Mutex::new(StdRng::seed_from_u64(seed_b)), + counters: Mutex::new(LinkCounters::default()), + reorder_hold: Mutex::new(None), }); (a, b) } @@ -380,32 +543,79 @@ impl InMemoryEndpoint { /// when full) instead of delivering them. Flipping back to `false` /// flushes the buffer to the peer in original order. /// - /// Sim-facing knob — production transports never call this. Note: - /// partition is one-directional per endpoint. To fully isolate two - /// peers, the caller flips both endpoints in the pair. - pub fn set_partitioned(&self, dropped: bool) { - // Drain the buffer under the lock if we're healing — but issue - // the actual sends *after* releasing it so the await doesn't - // happen while holding a sync mutex. - let drained: Vec = { + /// Partition is per-endpoint and per-direction; isolate a link by + /// flipping both of its endpoints. Healing routes the buffer back + /// through the fault model, so a recovered link is lossy like any + /// other — never a perfect replay. + pub fn set_partitioned(&self, partitioned: bool) { + let drained = { let mut p = self.partition.lock(); - p.dropped = dropped; - if !dropped { - p.buffer.drain(..).collect() - } else { - Vec::new() + p.partitioned = partitioned; + match partitioned { + true => Vec::new(), + false => p.buffer.drain(..).collect(), } }; - if !drained.is_empty() { - // try_send for the flush — if the receiver is closed or full - // we silently drop, which is the realistic "buffer overrun - // during long partition" semantic. Tests use a short window - // so this branch should never fire in practice. - for tone in drained { - let _ = self.tx.try_send(tone); + for tone in drained { + let verdict = self.take_verdict(); + let lost = verdict.drop || self.push(tone, verdict.duplicate).is_err(); + if lost { + self.counters.lock().lost_on_heal += 1; } } } + + /// One tone onto the wire, optionally twice. `try_send` — a full or + /// dead receiver during a heal flush is a lost tone, not an error the + /// caller can act on. + fn push(&self, tone: Tone, duplicate: bool) -> Result<()> { + self.tx.try_send(tone.clone()).map_err(|e| anyhow::anyhow!("push: {e}"))?; + self.counters.lock().delivered += 1; + if duplicate && self.tx.try_send(tone).is_ok() { + let mut c = self.counters.lock(); + c.delivered += 1; + c.duplicated += 1; + } + Ok(()) + } + + fn take_verdict(&self) -> Verdict { + let mut rng = self.rng.lock(); + self.faults.lock().verdict(&mut *rng) + } + + /// `reorder_next` counts *pairs*. A tone held back always pairs with + /// its successor, so an outstanding hold outranks the budget. + fn take_reorder(&self) -> bool { + if self.reorder_hold.lock().is_some() { + let mut f = self.faults.lock(); + f.script.reorder_next = f.script.reorder_next.saturating_sub(1); + return true; + } + self.faults.lock().script.reorder_next > 0 + } + + pub fn set_faults(&self, faults: LinkFaults) { + let seed = faults.seed; + *self.faults.lock() = faults; + *self.rng.lock() = StdRng::seed_from_u64(seed); + } + + pub fn faults(&self) -> LinkFaults { self.faults.lock().clone() } + + pub fn counters(&self) -> LinkCounters { *self.counters.lock() } + + pub fn buffered(&self) -> usize { self.partition.lock().buffer.len() } + + pub fn is_partitioned(&self) -> bool { self.partition.lock().partitioned } + + /// Frees a tone held for reorder when no partner arrives. + pub fn flush_reorder(&self) -> bool { + match self.reorder_hold.lock().take() { + Some(tone) => self.tx.try_send(tone).is_ok(), + None => false, + } + } } #[async_trait] @@ -414,20 +624,50 @@ impl PeerConnection for InMemoryEndpoint { fn capabilities(&self) -> &PeerCapabilities { &self.caps } async fn send(&self, tone: Tone) -> Result<()> { - // Partitioned: buffer with a bounded queue (oldest evicted when - // capacity is hit). Tone is "accepted" from the caller's point - // of view — the wire just hasn't delivered yet. + self.counters.lock().offered += 1; + { let mut p = self.partition.lock(); - if p.dropped { + if p.partitioned { if p.buffer.len() >= PARTITION_BUFFER_CAP { p.buffer.pop_front(); + self.counters.lock().evicted += 1; } p.buffer.push_back(tone); + self.counters.lock().buffered += 1; return Ok(()); } } - self.tx.send(tone).await.map_err(|e| anyhow::anyhow!("send: {e}")) + + let verdict = self.take_verdict(); + if verdict.drop { + self.counters.lock().dropped += 1; + return Ok(()); + } + + if self.take_reorder() { + let held = self.reorder_hold.lock().take(); + if let Some(held) = held { + self.counters.lock().reordered += 1; + self.tx.send(tone).await.map_err(|e| anyhow::anyhow!("send: {e}"))?; + self.counters.lock().delivered += 1; + self.tx.send(held).await.map_err(|e| anyhow::anyhow!("send: {e}"))?; + self.counters.lock().delivered += 1; + return Ok(()); + } + *self.reorder_hold.lock() = Some(tone); + return Ok(()); + } + + self.tx.send(tone.clone()).await.map_err(|e| anyhow::anyhow!("send: {e}"))?; + self.counters.lock().delivered += 1; + if verdict.duplicate { + self.tx.send(tone).await.map_err(|e| anyhow::anyhow!("send: {e}"))?; + let mut c = self.counters.lock(); + c.delivered += 1; + c.duplicated += 1; + } + Ok(()) } fn take_receiver(&self) -> Option> { @@ -463,7 +703,7 @@ struct Peer { pub struct Ensemble { me: Hid, peers: Arc>>, - inbox: broadcast::Sender, + inbox: Inbox, /// Shared gossip seen-set + per-topic broadcast senders. One Arc per /// ensemble; cloned into every install() drainer task so the dedup /// + topic dispatch happens without locking the main peer map. @@ -484,6 +724,79 @@ pub struct Ensemble { strict_auth: bool, } +/// The local fan-out point for tones arriving from peers. Cloned into +/// every install() drainer so the sender and its accounting travel +/// together. +#[derive(Clone)] +pub struct Inbox { + tx: broadcast::Sender, + subscribers: Arc, + dropped: Arc, +} + +impl Inbox { + fn new() -> Self { + // 256 keeps recent tones available for slow subscribers without + // unbounded memory; lagging consumers see Lagged and resync. + let (tx, _) = broadcast::channel(256); + Self { tx, subscribers: Arc::new(AtomicUsize::new(0)), dropped: Arc::new(AtomicU64::new(0)) } + } + + /// Hand a drained tone to local subscribers. Returns false when there + /// were none, in which case the tone is gone — not queued, not + /// retried. Counted and logged because a tone accepted off the + /// network and then destroyed here is otherwise invisible. + pub fn publish(&self, tone: Tone) -> bool { + if self.tx.send(tone).is_err() { + let total = self.dropped.fetch_add(1, Ordering::Relaxed) + 1; + tracing::warn!( + target: "ensemble", + total, + "inbox.publish.dropped: no local subscriber attached", + ); + return false; + } + true + } + + pub fn subscribe(&self) -> InboxSub { + self.subscribers.fetch_add(1, Ordering::SeqCst); + InboxSub { rx: self.tx.subscribe(), subscribers: self.subscribers.clone() } + } + + pub fn has_subscribers(&self) -> bool { self.subscribers.load(Ordering::SeqCst) > 0 } + + pub fn dropped(&self) -> u64 { self.dropped.load(Ordering::Relaxed) } +} + +/// A live inbox subscription. Derefs to the receiver, so existing +/// `rx.recv()` call sites are unchanged; dropping it unregisters. +pub struct InboxSub { + rx: broadcast::Receiver, + subscribers: Arc, +} + +impl Deref for InboxSub { + type Target = broadcast::Receiver; + fn deref(&self) -> &Self::Target { &self.rx } +} + +impl std::ops::DerefMut for InboxSub { + fn deref_mut(&mut self) -> &mut Self::Target { &mut self.rx } +} + +impl Drop for InboxSub { + fn drop(&mut self) { + self.subscribers.fetch_sub(1, Ordering::SeqCst); + } +} + +impl std::fmt::Debug for InboxSub { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("InboxSub").finish_non_exhaustive() + } +} + #[derive(Debug, thiserror::Error)] pub enum RouteError { #[error("no peer with id {0}")] @@ -496,13 +809,10 @@ pub enum RouteError { impl Ensemble { pub fn new(me: Hid) -> Self { - // 256 keeps recent tones available for slow subscribers without - // unbounded memory; lagging consumers see Lagged and resync. - let (inbox, _) = broadcast::channel(256); Self { me, peers: Arc::new(RwLock::new(HashMap::new())), - inbox, + inbox: Inbox::new(), gossip: GossipState::new(), kad: KadState::new(me), strict_auth: false, @@ -653,7 +963,7 @@ impl Ensemble { } // Everything else (including subsequent hellos) fans // out. Receivers may be absent — broadcast drops. - let _ = inbox.send(tone); + inbox.publish(tone); } }); } @@ -763,7 +1073,7 @@ impl Ensemble { continue; } } - let _ = inbox.send(tone); + inbox.publish(tone); } }); } @@ -789,11 +1099,25 @@ impl Ensemble { }) } + /// True once this peer's `chi:"hello"` has been parsed and its caps + /// learned. A scenario that needs a clean fault budget waits on this + /// — the handshake is link setup and is itself faultable. + pub fn handshake_done(&self, id: &Hid) -> bool { + self.peers.read().get(id).is_some_and(|p| p.learned_caps.is_some()) + } + /// Subscribe to incoming tones from every installed peer. Hellos /// are absorbed by the ensemble; subscribers only see real traffic. - pub fn subscribe(&self) -> broadcast::Receiver { - self.inbox.subscribe() - } + pub fn subscribe(&self) -> InboxSub { self.inbox.subscribe() } + + /// True once something is attached to the inbox. Callers that install + /// peers before the local pump is up wait on this instead of racing it. + pub fn has_subscribers(&self) -> bool { self.inbox.has_subscribers() } + + /// Tones that reached a drainer and were then destroyed because no + /// pump was listening. Never zero on a healthy daemon; a non-zero + /// value is a boot-order race that silently ate network traffic. + pub fn inbox_dropped(&self) -> u64 { self.inbox.dropped() } /// Publish a gossip message to every installed peer. Mints an /// `msg_id` from `(topic, rid, me, payload)`, marks it seen locally @@ -1405,6 +1729,274 @@ mod tests { use super::*; use serde_json::json; + // ── Link fault model ─────────────────────────────────────────────── + + fn link_pair() -> (Arc, Arc) { + InMemoryEndpoint::pair_concrete( + Hid::random_humd(), + PeerCapabilities::default(), + Hid::random_humd(), + PeerCapabilities::default(), + ) + } + + fn tone(tag: &str) -> Tone { + json!({ + "chi": "prompt", + "rid": tag, + "from": Hid::random_humd().to_hex(), + }) + } + + async fn drain(rx: &mut mpsc::Receiver) -> Vec { + let mut out = Vec::new(); + while let Ok(t) = rx.try_recv() { + out.push(t); + } + out + } + + fn rids(tones: &[Tone]) -> Vec { + tones.iter().map(|t| t["rid"].as_str().unwrap().to_string()).collect() + } + + #[tokio::test] + async fn perfect_link_delivers_everything_in_order() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + for i in 0..5 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["t0", "t1", "t2", "t3", "t4"]); + let c = a.counters(); + assert_eq!(c, LinkCounters { offered: 5, delivered: 5, ..Default::default() }); + } + + #[tokio::test] + async fn drop_next_drops_exactly_n() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_faults(LinkFaults::default().drop_next(3)); + for i in 0..6 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["t3", "t4", "t5"]); + let c = a.counters(); + assert_eq!(c.offered, 6); + assert_eq!(c.dropped, 3); + assert_eq!(c.delivered, 3); + } + + #[tokio::test] + async fn drop_every_keeps_a_strict_period() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + let mut faults = LinkFaults::default(); + faults.script.drop_every = 4; + a.set_faults(faults); + for i in 0..12 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["t0", "t1", "t2", "t4", "t5", "t6", "t8", "t9", "t10"]); + } + + #[tokio::test] + async fn dup_next_duplicates_exactly_n() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_faults(LinkFaults::default().dup_next(2)); + for i in 0..4 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["t0", "t0", "t1", "t1", "t2", "t3"]); + let c = a.counters(); + assert_eq!(c.duplicated, 2); + assert_eq!(c.delivered, 6); + } + + #[tokio::test] + async fn reorder_reverses_the_pair() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_faults(LinkFaults::default().reorder_next(1)); + for i in 0..2 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["t1", "t0"]); + assert_eq!(a.counters().reordered, 1); + } + + #[tokio::test] + async fn reorder_hold_releases_via_flush() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_faults(LinkFaults::default().reorder_next(1)); + a.send(tone("lonely")).await.unwrap(); + assert!(drain(&mut rx).await.is_empty()); + assert!(a.flush_reorder()); + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["lonely"]); + } + + #[tokio::test] + async fn statistical_loss_is_reproducible_from_its_seed() { + async fn run(seed: u64) -> usize { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_faults(LinkFaults::default().drop_pct(30, seed)); + for i in 0..200 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + drain(&mut rx).await.len() + } + assert_eq!(run(42).await, run(42).await, "same seed, same pattern"); + assert_ne!(run(1).await, run(2).await, "different seeds, different patterns"); + let survived = run(42).await; + // 200 draws at 30% loss; sd is ~6.5, so this is a 3-sigma band. + assert!((120..170).contains(&survived), "survived {survived} of 200 at 30% loss"); + } + + #[tokio::test] + async fn partition_buffers_without_delivering() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_partitioned(true); + for i in 0..3 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + assert!(drain(&mut rx).await.is_empty()); + assert_eq!(a.buffered(), 3); + let c = a.counters(); + assert_eq!(c.buffered, 3); + assert_eq!(c.delivered, 0); + } + + #[tokio::test] + async fn heal_drains_the_buffer() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_partitioned(true); + for i in 0..3 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + a.set_partitioned(false); + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["t0", "t1", "t2"]); + assert_eq!(a.counters().lost_on_heal, 0); + } + + /// The behaviour this whole model exists to make possible: a link + /// that recovers can still lose what it buffered. Replaying the + /// buffer intact would let a partition test pass without ever + /// exercising recovery. + #[tokio::test] + async fn heal_is_lossy_not_a_perfect_replay() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_partitioned(true); + for i in 0..6 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + a.set_faults(LinkFaults::default().drop_next(2)); + a.set_partitioned(false); + + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["t2", "t3", "t4", "t5"]); + let c = a.counters(); + assert_eq!(c.lost_on_heal, 2); + assert_eq!(c.buffered, 6); + assert_eq!(c.delivered, 4); + } + + #[tokio::test] + async fn heal_applies_dup_too() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_partitioned(true); + for i in 0..3 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + a.set_faults(LinkFaults::default().dup_next(1)); + a.set_partitioned(false); + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["t0", "t0", "t1", "t2"]); + assert_eq!(a.counters().duplicated, 1); + } + + #[tokio::test] + async fn partition_buffer_evicts_oldest_when_full() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_partitioned(true); + for i in 0..(PARTITION_BUFFER_CAP + 5) { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + assert_eq!(a.buffered(), PARTITION_BUFFER_CAP); + assert_eq!(a.counters().evicted, 5); + a.set_partitioned(false); + let got = rids(&drain(&mut rx).await); + assert_eq!(got.first().unwrap(), "t5"); + assert_eq!(got.len(), PARTITION_BUFFER_CAP); + } + + /// The oracle has to be internally consistent or no test can trust it. + #[tokio::test] + async fn counters_account_for_every_offered_tone() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + let mut faults = LinkFaults::default(); + faults.script.drop_every = 5; + faults.script.dup_every = 7; + a.set_faults(faults); + a.set_partitioned(true); + for i in 0..3 { + a.send(tone(&format!("p{i}"))).await.unwrap(); + } + a.set_partitioned(false); + for i in 0..40 { + a.send(tone(&format!("t{i}"))).await.unwrap(); + } + a.flush_reorder(); + + let c = a.counters(); + let arrived = drain(&mut rx).await.len() as u64; + assert_eq!(c.delivered, arrived, "delivered must match what the peer saw"); + assert_eq!(c.offered, 43); + assert_eq!(c.buffered, 3); + // `delivered` counts duplicate copies, so discount them to + // compare against what was offered and not lost. + assert_eq!(c.offered, c.delivered - c.duplicated + c.dropped + c.lost_on_heal); + } + + #[tokio::test] + async fn faults_are_per_direction() { + let (a, b) = link_pair(); + let mut a_rx = a.take_receiver().unwrap(); + let mut b_rx = b.take_receiver().unwrap(); + a.set_faults(LinkFaults::default().drop_next(10)); + b.send(tone("survivor")).await.unwrap(); + a.send(tone("doomed")).await.unwrap(); + assert_eq!(rids(&drain(&mut a_rx).await), ["survivor"], "b→a is unaffected"); + assert!(drain(&mut b_rx).await.is_empty(), "a→b drops"); + } + + #[tokio::test] + async fn clear_faults_restores_a_perfect_link() { + let (a, b) = link_pair(); + let mut rx = b.take_receiver().unwrap(); + a.set_faults(LinkFaults::default().drop_next(10)); + a.send(tone("lost")).await.unwrap(); + a.set_faults(LinkFaults::default()); + a.send(tone("arrives")).await.unwrap(); + let got = rids(&drain(&mut rx).await); + assert_eq!(got, ["arrives"]); + } + #[test] fn hid_hex_round_trips() { let id = Hid::random_humd(); diff --git a/humd/src/peer_transport/iroh.rs b/humd/src/peer_transport/iroh.rs index 3f559e9b..7c41c50b 100644 --- a/humd/src/peer_transport/iroh.rs +++ b/humd/src/peer_transport/iroh.rs @@ -37,12 +37,26 @@ use crate::peers::PeerConfig; pub(crate) async fn bind(humd_key: &HumdKey) -> anyhow::Result<(IrohTransport, Vec)> { let transport = IrohTransport::bind_direct_with_key(humd_key).await?; let node_id_hex = hex::encode(transport.node_id().as_bytes()); - let sockets = transport.endpoint().bound_sockets(); - info!(node_id = %&node_id_hex[..16], socket_count = sockets.len(), "peer.iroh.bound"); - - let mut hints = Vec::with_capacity(1 + sockets.len()); + // `addr().ip_addrs()` is the routable set iroh discovered. NOT + // `bound_sockets()`, which reports the wildcard binds + // (0.0.0.0:) — advertising those as dial targets is how a + // peering hint turns into "unreachable" and the dial silently falls + // back to address lookup, which is not configured in production. + let addrs: Vec = transport + .endpoint() + .addr() + .ip_addrs() + .map(|s| s.to_string()) + .collect(); + info!( + node_id = %&node_id_hex[..16], + socket_count = addrs.len(), + "peer.iroh.bound" + ); + + let mut hints = Vec::with_capacity(1 + addrs.len()); hints.push(format!("iroh:{node_id_hex}")); - for s in &sockets { + for s in &addrs { hints.push(format!("iroh-ip:{s}")); } Ok((transport, hints)) @@ -118,6 +132,14 @@ mod tests { /// each other in their peer registry after the handshake. #[tokio::test] async fn dial_then_accept_meet_via_signed_hello() { + let _ = tracing_subscriber::fmt() + .with_env_filter( + tracing_subscriber::EnvFilter::try_from_default_env() + .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("warn")), + ) + .with_test_writer() + .try_init(); + let a_key = Arc::new(HumdKey::generate()); let b_key = Arc::new(HumdKey::generate()); let a_id = a_key.hid(); diff --git a/sim/src/lib.rs b/sim/src/lib.rs index 7e6ce235..28a5edaf 100644 --- a/sim/src/lib.rs +++ b/sim/src/lib.rs @@ -17,7 +17,7 @@ use std::sync::atomic::{AtomicUsize, Ordering}; use std::sync::Arc; use std::time::Duration; -use anyhow::Result; +use anyhow::{bail, Result}; use ensemble::{hello_tone, Ensemble, Hid, HumdKey, InMemoryEndpoint, PeerCapabilities}; use parking_lot::{Mutex, RwLock}; use serde_json::Value; @@ -193,9 +193,28 @@ impl Sim { self.humds.write().insert(id, sim_humd.clone()); sim_humd - } - - /// Wire two humds with an in-memory channel pair. Both ensembles + } + + /// Block until both gates are closed: the daemon posted a ToneSink + /// to its ensemble, and the ensemble inbox pump subscribed for + /// peer drains. Before this, any injected tone is dropped. + pub async fn await_ready(&self, humd: Hid) { + for _ in 0..200 { + let h = self.humds.read().get(&humd).cloned(); + if let Some(h) = h { + let has_sink = h.thrum.has_sink(); + let has_sub = h.ensemble.has_subscribers(); + if has_sink && has_sub { + return; + } + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + let id = humd.short(); + panic!("humd {} never became ready", id); + } + + /// Wire two humds with an in-memory channel pair. Both ensembles /// pick up a `PeerConnection` to the other; capabilities mirror each /// side's current capacity — sim humds always claim `claude-cli` /// support so overflow routing has somewhere to land, and the @@ -796,6 +815,87 @@ impl Sim { } } + /// Block until both ends have exchanged their hello tones, then + /// drain them from the inbox so scenarios start clean. + pub async fn await_handshake(&self, a: Hid, b: Hid) -> Result<()> { + for _ in 0..100 { + let ah = self.humds.read().get(&a).map(|h| h.ensemble.handshake_done(&b)); + let bh = self.humds.read().get(&b).map(|h| h.ensemble.handshake_done(&a)); + if ah == Some(true) && bh == Some(true) { + // Drain any hello tones published to the inbox. + { + let h = self.humds.read().get(&b).cloned().ok_or_else(|| anyhow::anyhow!("no humd"))?; + let mut rx = h.ensemble.subscribe(); + for _ in 0..10 { match rx.try_recv() { Ok(_t) => {}, Err(_) => break } } + } + return Ok(()); + } + tokio::time::sleep(std::time::Duration::from_millis(5)).await; + } + bail!("handshake timed out"); + } + + /// Subscribe to humd inbox now, before any send races it. + pub fn humd_peer_sub(&self, humd: Hid) -> Option { + Some(self.humds.read().get(&humd)?.ensemble.subscribe()) + } + + /// Collect up to want tones, stopping early or on window. + pub async fn collect_rids( + rx: &mut ensemble::InboxSub, + want: usize, + window: Duration, + ) -> Vec { + let mut out = Vec::with_capacity(want); + let deadline = tokio::time::Instant::now() + window; + while out.len() < want { + match tokio::time::timeout_at(deadline, rx.recv()).await { + Ok(Ok(ton)) => { + if let Some(rid) = ton["rid"].as_str() { out.push(rid.to_string()); } + } + _ => break, + } + } + out + } + + /// Like nestler_send but injects in caller task so ordering is preserved. + pub async fn nestler_send_ordered(&self, humd: Hid, tone: Value) -> Result { + let h = self.humds.read().get(&humd).cloned() + .ok_or_else(|| anyhow::anyhow!("no humd {}", humd.short()))?; + let cid = hum_identity::HumId::mint().to_string(); + let _ = h.thrum.register_synthetic(cid.clone()); + h.thrum.inject_tone(&cid, tone).await; + Ok(cid) + } + + /// Route n perf-mark tones from from to to, tagged tag-i. + pub async fn send_marks(&self, from: Hid, to: Hid, tag: &str, n: usize) -> Result<()> { + for i in 0..n { + let tone = serde_json::json!({ + "chi": "perf-mark", + "rid": format!("{}-{}", tag, i), + "to": to.to_hex(), + "from": from.to_hex(), + "mark": tag, + }); + self.nestler_send_ordered(from, tone).await?; + } + Ok(()) + } + + /// The rid values of a collected batch, in arrival order. + pub fn rids(tones: &[Value]) -> Vec { + tones.iter().filter_map(|t| t["rid"].as_str().map(String::from)).collect() + } + + /// Tones the ensemble's inbox accepted from a peer then destroyed + /// because no local subscriber was attached. Must stay zero when + /// a scenario subscribes before sending. + pub async fn ensemble_dropped(&self, humd: Hid) -> u64 { + self.humds.read().get(&humd).map(|h| h.ensemble.inbox_dropped()).unwrap_or(0) + } + /// Shutdown all humds and drain their join handles. pub async fn shutdown(self) { let humds: Vec> = self.humds.read().values().cloned().collect(); diff --git a/sim/tests/lossy_link.rs b/sim/tests/lossy_link.rs new file mode 100644 index 00000000..89c6ac9d --- /dev/null +++ b/sim/tests/lossy_link.rs @@ -0,0 +1,175 @@ +//! lossy-link — tones driven across links that are actually broken. +//! +//! Each test names a fault, sends known tones, asserts the receiver saw +//! exactly what LinkCounters says it should. +//! +//! Every test subscribes before sending and sends in order so a scripted +//! fault lands on the tone it names. + +use std::sync::Arc; +use std::time::Duration; + +use ensemble::{LinkFaults, Script}; +use sim::Sim; +use std::time::Duration; + +const WINDOW: Duration = Duration::from_millis(750); + +async fn pair() -> (Arc, ensemble::Hid, ensemble::Hid) { + let sim = Sim::new(); + let a = ensemble::Hid::random_humd(); + let b = ensemble::Hid::random_humd(); + sim.spawn_humd(a).await; + sim.spawn_humd(b).await; + sim.await_ready(a); + sim.await_ready(b); + sim.wire(a, b).unwrap(); + sim.await_handshake(a, b).await.unwrap(); + (Arc::new(sim), a, b) +} + +// ── intact ────────────────────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn intact_delivers_all_in_order() { + let (sim, a, b) = pair().await; + let mut rx = sim.humd_peer_sub(b).unwrap(); + sim.send_marks(a, b, "intact", 6).await.unwrap(); + + let rid = Sim::collect_rids(&mut rx, 6, WINDOW).await; + assert_eq!(rid, ["intact-0","intact-1","intact-2","intact-3","intact-4","intact-5"]); + let (ab, _) = sim.link_counters(a, b).unwrap(); + assert_eq!(ab.offered, 6); + assert_eq!(ab.dropped, 0); +} + +// ── dropped ───────────────────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn dropped_tones_never_arrive() { + let (sim, a, b) = pair().await; + sim.impair_dir(a, b, LinkFaults::default().drop_next(4)).unwrap(); + let mut rx = sim.humd_peer_sub(b).unwrap(); + sim.send_marks(a, b, "lossy", 6).await.unwrap(); + + let rid = Sim::collect_rids(&mut rx, 2, WINDOW).await; + let (ab, _) = sim.link_counters(a, b).unwrap(); + assert_eq!(ab.dropped, 4); + assert_eq!(ab.delivered, 2); + assert_eq!(rid, ["lossy-4","lossy-5"]); +} + +// ── duplicated ────────────────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn duplicated_tones_arrive_twice() { + let (sim, a, b) = pair().await; + sim.impair_dir(a, b, LinkFaults::default().dup_next(3)).unwrap(); + let mut rx = sim.humd_peer_sub(b).unwrap(); + sim.send_marks(a, b, "dup", 6).await.unwrap(); + + let rid = Sim::collect_rids(&mut rx, 9, WINDOW).await; + let (ab, _) = sim.link_counters(a, b).unwrap(); + assert_eq!(ab.duplicated, 3); + assert_eq!(rid, [ + "dup-0","dup-0","dup-1","dup-1","dup-2","dup-2", + "dup-3","dup-4","dup-5" + ]); +} + +// ── reordered ─────────────────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn reordered_tones_swap_adjacent_pairs() { + let (sim, a, b) = pair().await; + sim.impair_dir(a, b, LinkFaults::default().reorder_next(2)).unwrap(); + let mut rx = sim.humd_peer_sub(b).unwrap(); + sim.send_marks(a, b, "swap", 4).await.unwrap(); + + let rid = Sim::collect_rids(&mut rx, 4, WINDOW).await; + assert_eq!(rid, ["swap-1","swap-0","swap-3","swap-2"]); +} + +// ── partition ─────────────────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn partition_stops_delivery() { + let (sim, a, b) = pair().await; + sim.partition(a, b).unwrap(); + let mut rx = sim.humd_peer_sub(b).unwrap(); + sim.send_marks(a, b, "cut", 5).await.unwrap(); + + assert!(Sim::collect_rids(&mut rx, 5, WINDOW).await.is_empty()); + assert_eq!(sim.buffered(a, b).unwrap(), 5); +} + +// ── heal (lossy recover) ─────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn heal_discards_buffer() { + let (sim, a, b) = pair().await; + sim.partition(a, b).unwrap(); + sim.send_marks(a, b, "buf", 8).await.unwrap(); + assert_eq!(sim.buffered(a, b).unwrap(), 8); + + sim.impair_dir(a, b, LinkFaults::default().drop_next(3)).unwrap(); + sim.heal(a, b).await.unwrap(); + + let mut rx = sim.humd_peer_sub(b).unwrap(); + sim.send_marks(a, b, "after", 3).await.unwrap(); + let rid = Sim::collect_rids(&mut rx, 3, WINDOW).await; + let (ab, _) = sim.link_counters(a, b).unwrap(); + assert_eq!(ab.lost_on_heal, 3, "heal must discard buffered data"); + assert_eq!(rid, ["after-0","after-1","after-2"]); + assert_eq!(sim.buffered(a, b).unwrap(), 0); +} + +// ── link recovers after healing ───────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn link_recovers_after_lossy_heal() { + let (sim, a, b) = pair().await; + sim.partition(a, b).unwrap(); + sim.send_marks(a, b, "cut", 4).await.unwrap(); + sim.impair_dir(a, b, LinkFaults::default().drop_next(2)).unwrap(); + sim.heal(a, b).await.unwrap(); + sim.heal_link_faults(a, b).unwrap(); + + let mut rx = sim.humd_peer_sub(b).unwrap(); + sim.send_marks(a, b, "ok", 3).await.unwrap(); + let rid = Sim::collect_rids(&mut rx, 3, WINDOW).await; + assert_eq!(rid, ["ok-0","ok-1","ok-2"]); +} + +// ── directional ──────────────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn impairment_is_directional() { + let (sim, a, b) = pair().await; + sim.impair_dir(a, b, LinkFaults::default().drop_next(10)).unwrap(); + let mut rx = sim.humd_peer_sub(a).unwrap(); + sim.send_marks(b, a, "reply", 3).await.unwrap(); + let rid = Sim::collect_rids(&mut rx, 3, WINDOW).await; + assert_eq!(rid, ["reply-0","reply-1","reply-2"]); +} + +// ── inbox safety ──────────────────────────────────────────────────── + +#[tokio::test(flavor = "multi_thread")] +async fn losses_are_link_level_not_inbox_level() { + let (sim, a, b) = pair().await; + sim.impair_dir(a, b, LinkFaults::default() + .script(Script { drop_every: 3, ..Script::default() }) + .seed(7)) + .unwrap(); + + let mut rx = sim.humd_peer_sub(b).unwrap(); + sim.send_marks(a, b, "thirds", 30).await.unwrap(); + let rid = Sim::collect_rids(&mut rx, 20, WINDOW).await; + + let (ab, _) = sim.link_counters(a, b).unwrap(); + assert_eq!(rid.len() as u64, ab.delivered); + assert_eq!(ab.offered, 30); + assert_eq!(ab.dropped, 10); + assert_eq!(sim.ensemble_dropped(b).await, 0); +} diff --git a/sim/tests/partition_and_heal.rs b/sim/tests/partition_and_heal.rs index 83552975..b17be1fe 100644 --- a/sim/tests/partition_and_heal.rs +++ b/sim/tests/partition_and_heal.rs @@ -34,6 +34,8 @@ async fn partition_then_heal_converges_wane() { let a = sim.spawn_humd(a_id).await; let b = sim.spawn_humd(b_id).await; + sim.await_ready(a_id).await.expect("a ready"); + sim.await_ready(b_id).await.expect("b ready"); sim.wire(a_id, b_id).expect("wire a-b"); // Both healthy: tick wane on each side a few times in lockstep @@ -81,6 +83,8 @@ async fn partition_then_heal_converges_wane() { b.waneman.get(SIGIL), ); assert_eq!(a.waneman.get(SIGIL), b.waneman.get(SIGIL)); + assert_eq!(a.ensemble.inbox_dropped(), 0, "a dropped a drained tone"); + assert_eq!(b.ensemble.inbox_dropped(), 0, "b dropped a drained tone"); sim.shutdown().await; } diff --git a/sim/tests/two_humds_ping_pong.rs b/sim/tests/two_humds_ping_pong.rs index eddfdf00..870a76b3 100644 --- a/sim/tests/two_humds_ping_pong.rs +++ b/sim/tests/two_humds_ping_pong.rs @@ -23,30 +23,27 @@ async fn two_humds_ping_pong() { let a = sim.spawn_humd(ensemble::Hid::random_humd()).await; let b = sim.spawn_humd(ensemble::Hid::random_humd()).await; + sim.await_ready(a.id).await.expect("a ready"); + sim.await_ready(b.id).await.expect("b ready"); sim.wire(a.id, b.id).expect("wire humd-A and humd-B"); // Tap humd-A's inbound peer stream BEFORE sending so we don't race // the broadcast (broadcast::Receiver only sees tones sent after it - // was created — earlier ones are gone). Spawn the tap in a task - // and join later. + // was created — earlier ones are gone). let a_id = a.id; let sim_arc = std::sync::Arc::new(sim); let sim_for_tap = sim_arc.clone(); let tap = tokio::spawn(async move { sim_for_tap - .humd_peer_tap(a_id, Duration::from_secs(1)) + .humd_peer_tap(a_id, Duration::from_secs(5)) .await }); - // Tiny pause so the tap subscriber exists before B sends. Without - // this the test races; with it the broadcast::Receiver is in place - // before any tone is published. - tokio::time::sleep(Duration::from_millis(50)).await; - // From humd-B's mock nestler, send a tone addressed to humd-A. // Note: `chi: "hello"` is the ensemble's peer-handshake tone and // gets absorbed by the drainer. Use `perf-mark` for the routing - // test — it's a non-handshake chi that flows through. + // test — it's a non-handshake chi, and the sink's `to:` rule routes + // it to the ensemble before chi-specific dispatch. let tone = serde_json::json!({ "chi": "perf-mark", "rid": "ping-1", From 4843051c93160c1887c150cd10726a0573c4b7c9 Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 20:01:54 +0500 Subject: [PATCH 02/15] sim: link-fault scenario tests against a LinkCounters oracle Adds impair/impair_dir/link_counters/buffered/heal_link_faults so a scenario can name a fault, drive known tones across a broken link, and assert the receiver saw what the link says it sent. Two ordering defects surfaced while writing them: - nestler_send injects on a detached task, so consecutive sends have no wire order. A fault script names a tone by position, so it needs nestler_send_ordered, which injects in the caller's task. - a broadcast receiver only sees tones published after it exists, so a test that subscribes inside a spawned task races its own first send. humd_peer_sub takes the handle before the scenario sends. Link setup is traffic the link really carried, so scenarios measure LinkCounters::since a post-handshake baseline rather than a total. heal flushes the partition buffer through the link's own fault verdict, so a lossy heal loses the head of the buffer. Test asserts that. 9 scenarios: intact, loss, dup, reorder, partition, lossy heal, recovery, directional impairment, and inbox-level drop accounting. --- ensemble/src/lib.rs | 16 +++ sim/src/lib.rs | 278 ++++++++++++++++++++++++++-------------- sim/tests/lossy_link.rs | 44 ++++--- 3 files changed, 224 insertions(+), 114 deletions(-) diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index 3137c375..37d460df 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -429,6 +429,22 @@ pub struct LinkCounters { pub evicted: u64, } +impl LinkCounters { + /// Counters accrued after `base` was taken. + pub fn since(&self, base: &Self) -> Self { + Self { + offered: self.offered - base.offered, + delivered: self.delivered - base.delivered, + dropped: self.dropped - base.dropped, + duplicated: self.duplicated - base.duplicated, + reordered: self.reordered - base.reordered, + buffered: self.buffered - base.buffered, + lost_on_heal: self.lost_on_heal - base.lost_on_heal, + evicted: self.evicted - base.evicted, + } + } +} + // ── In-memory transport (sim) ────────────────────────────────────────────── /// Two `InMemoryEndpoint`s wired together with `mpsc` channels. Lets diff --git a/sim/src/lib.rs b/sim/src/lib.rs index 28a5edaf..13d47616 100644 --- a/sim/src/lib.rs +++ b/sim/src/lib.rs @@ -18,7 +18,10 @@ use std::sync::Arc; use std::time::Duration; use anyhow::{bail, Result}; -use ensemble::{hello_tone, Ensemble, Hid, HumdKey, InMemoryEndpoint, PeerCapabilities}; +use ensemble::{ + hello_tone, Ensemble, Hid, HumdKey, InMemoryEndpoint, LinkCounters, LinkFaults, + PeerCapabilities, +}; use parking_lot::{Mutex, RwLock}; use serde_json::Value; use thrum_core::WaneTracker; @@ -82,6 +85,7 @@ pub struct Sim { /// by humd `a` (it sends through this to reach `b`); `b_end` is the /// mirror. To fully partition the link we flip both — each blocks its /// own outbound side. +#[derive(Clone)] struct Link { a: Hid, b: Hid, @@ -195,24 +199,26 @@ impl Sim { sim_humd } - /// Block until both gates are closed: the daemon posted a ToneSink - /// to its ensemble, and the ensemble inbox pump subscribed for - /// peer drains. Before this, any injected tone is dropped. - pub async fn await_ready(&self, humd: Hid) { - for _ in 0..200 { - let h = self.humds.read().get(&humd).cloned(); - if let Some(h) = h { - let has_sink = h.thrum.has_sink(); - let has_sub = h.ensemble.has_subscribers(); - if has_sink && has_sub { - return; - } - } - tokio::time::sleep(Duration::from_millis(5)).await; - } - let id = humd.short(); - panic!("humd {} never became ready", id); - } + /// Block until both gates are closed: the daemon posted a ToneSink + /// to its ensemble, and the ensemble inbox pump subscribed for peer + /// drains. Before this, any injected tone is dropped. Errors if the + /// humd never reaches readiness. + pub async fn await_ready(&self, humd: Hid) -> Result<()> { + for _ in 0..200 { + let ready = { + let Some(h) = self.humds.read().get(&humd).cloned() else { + tokio::time::sleep(Duration::from_millis(5)).await; + continue; + }; + h.thrum.has_sink() && h.ensemble.has_subscribers() + }; + if ready { + return Ok(()); + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + bail!("humd {} never became ready", humd.short()) + } /// Wire two humds with an in-memory channel pair. Both ensembles /// pick up a `PeerConnection` to the other; capabilities mirror each @@ -452,6 +458,67 @@ impl Sim { Ok(()) } + fn link(&self, a: Hid, b: Hid) -> Result { + self.links + .read() + .get(&link_key(a, b)) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no link {}-{}", a.short(), b.short())) + } + + /// The concrete endpoint `from` holds for its link to `to`. + fn end_for(&self, a: Hid, b: Hid) -> Result> { + let link = self.link(a, b)?; + Ok(if link.a == a { link.a_end } else { link.b_end }) + } + + /// Apply `faults` to one direction only. `a` is the sender. + pub fn impair_dir(&self, a: Hid, b: Hid, faults: LinkFaults) -> Result<()> { + self.end_for(a, b)?.set_faults(faults); + Ok(()) + } + + /// Fault both directions, resetting each endpoint's tone counter. + pub fn impair(&self, a: Hid, b: Hid, faults: LinkFaults) -> Result<()> { + self.impair_dir(a, b, faults.clone())?; + self.impair_dir(b, a, faults) + } + + /// `(a→b, b→a)` ground truth for what the link actually did. + pub fn link_counters(&self, a: Hid, b: Hid) -> Result<(LinkCounters, LinkCounters)> { + let link = self.link(a, b)?; + Ok(if link.a == a { + (link.a_end.counters(), link.b_end.counters()) + } else { + (link.b_end.counters(), link.a_end.counters()) + }) + } + + /// Counters accrued since `base`. Link setup (the `wire` hello) is + /// traffic the link really carried, so a scenario measures a delta + /// rather than a total. + pub fn link_counters_since( + &self, + a: Hid, + b: Hid, + base: &LinkCounters, + ) -> Result { + let (ab, _) = self.link_counters(a, b)?; + Ok(ab.since(base)) + } + + /// Tones the `a→b` link is holding while partitioned. + pub fn buffered(&self, a: Hid, b: Hid) -> Result { + Ok(self.end_for(a, b)?.buffered()) + } + + /// Clear injected faults, restoring a clean link. + pub fn heal_link_faults(&self, a: Hid, b: Hid) -> Result<()> { + self.end_for(a, b)?.set_faults(LinkFaults::default()); + self.end_for(b, a)?.set_faults(LinkFaults::default()); + Ok(()) + } + /// Drop the wired link between `a` and `b`. Both endpoints stop /// delivering outbound tones and instead buffer them up to /// `ensemble::PARTITION_BUFFER_CAP`. A subsequent [`Sim::heal`] @@ -815,86 +882,101 @@ impl Sim { } } - /// Block until both ends have exchanged their hello tones, then - /// drain them from the inbox so scenarios start clean. - pub async fn await_handshake(&self, a: Hid, b: Hid) -> Result<()> { - for _ in 0..100 { - let ah = self.humds.read().get(&a).map(|h| h.ensemble.handshake_done(&b)); - let bh = self.humds.read().get(&b).map(|h| h.ensemble.handshake_done(&a)); - if ah == Some(true) && bh == Some(true) { - // Drain any hello tones published to the inbox. - { - let h = self.humds.read().get(&b).cloned().ok_or_else(|| anyhow::anyhow!("no humd"))?; - let mut rx = h.ensemble.subscribe(); - for _ in 0..10 { match rx.try_recv() { Ok(_t) => {}, Err(_) => break } } - } - return Ok(()); - } - tokio::time::sleep(std::time::Duration::from_millis(5)).await; - } - bail!("handshake timed out"); - } - - /// Subscribe to humd inbox now, before any send races it. - pub fn humd_peer_sub(&self, humd: Hid) -> Option { - Some(self.humds.read().get(&humd)?.ensemble.subscribe()) - } - - /// Collect up to want tones, stopping early or on window. - pub async fn collect_rids( - rx: &mut ensemble::InboxSub, - want: usize, - window: Duration, - ) -> Vec { - let mut out = Vec::with_capacity(want); - let deadline = tokio::time::Instant::now() + window; - while out.len() < want { - match tokio::time::timeout_at(deadline, rx.recv()).await { - Ok(Ok(ton)) => { - if let Some(rid) = ton["rid"].as_str() { out.push(rid.to_string()); } - } - _ => break, - } - } - out - } - - /// Like nestler_send but injects in caller task so ordering is preserved. - pub async fn nestler_send_ordered(&self, humd: Hid, tone: Value) -> Result { - let h = self.humds.read().get(&humd).cloned() - .ok_or_else(|| anyhow::anyhow!("no humd {}", humd.short()))?; - let cid = hum_identity::HumId::mint().to_string(); - let _ = h.thrum.register_synthetic(cid.clone()); - h.thrum.inject_tone(&cid, tone).await; - Ok(cid) - } - - /// Route n perf-mark tones from from to to, tagged tag-i. - pub async fn send_marks(&self, from: Hid, to: Hid, tag: &str, n: usize) -> Result<()> { - for i in 0..n { - let tone = serde_json::json!({ - "chi": "perf-mark", - "rid": format!("{}-{}", tag, i), - "to": to.to_hex(), - "from": from.to_hex(), - "mark": tag, - }); - self.nestler_send_ordered(from, tone).await?; - } - Ok(()) - } + /// Block until both ends have exchanged hellos, then drain them from + /// `b`'s inbox so a scenario's receiver starts clean. The hello is + /// itself faultable, so a scenario must not race it. + pub async fn await_handshake(&self, a: Hid, b: Hid) -> Result<()> { + for _ in 0..200 { + let done = { + let humds = self.humds.read(); + let (Some(ha), Some(hb)) = (humds.get(&a), humds.get(&b)) else { + tokio::time::sleep(Duration::from_millis(5)).await; + continue; + }; + ha.ensemble.handshake_done(&b) && hb.ensemble.handshake_done(&a) + }; + if done { + let hb = self.humds.read().get(&b).cloned().expect("checked above"); + let mut rx = hb.ensemble.subscribe(); + while rx.try_recv().is_ok() {} + return Ok(()); + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + bail!("{}-{} handshake timed out", a.short(), b.short()) + } + + /// Subscribe to `humd`'s inbox now. A `broadcast::Receiver` only + /// sees tones published after it exists, so a test that subscribes + /// inside a spawned task races its own first send. + pub fn humd_peer_sub(&self, humd: Hid) -> Option { + Some(self.humds.read().get(&humd)?.ensemble.subscribe()) + } + + /// Drain up to `want` rids from an inbox handle, stopping early once + /// `want` land or `window` elapses. A short return is the signal. + pub async fn collect_rids( + rx: &mut ensemble::InboxSub, + want: usize, + window: Duration, + ) -> Vec { + let mut out = Vec::with_capacity(want); + let deadline = tokio::time::Instant::now() + window; + while out.len() < want { + match tokio::time::timeout_at(deadline, rx.recv()).await { + Ok(Ok(tone)) => { + if let Some(rid) = tone["rid"].as_str() { + out.push(rid.to_string()); + } + } + _ => break, + } + } + out + } + + /// `nestler_send` injects on a detached task, so consecutive sends + /// have no order. A scripted fault names a tone by position, so it + /// needs the injection to happen in the caller's task. + pub async fn nestler_send_ordered(&self, humd: Hid, tone: Value) -> Result { + let h = self + .humds + .read() + .get(&humd) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no humd {}", humd.short()))?; + let cid = hum_identity::HumId::mint().to_string(); + let _ = h.thrum.register_synthetic(cid.clone()); + h.thrum.inject_tone(&cid, tone).await; + Ok(cid) + } + + /// Route `n` tones `from` → `to`, rid-tagged `tag-`, in order. + pub async fn send_marks(&self, from: Hid, to: Hid, tag: &str, n: usize) -> Result<()> { + for i in 0..n { + let tone = serde_json::json!({ + "chi": "perf-mark", + "rid": format!("{tag}-{i}"), + "to": to.to_hex(), + "from": from.to_hex(), + "mark": tag, + }); + self.nestler_send_ordered(from, tone).await?; + } + Ok(()) + } + + /// `rid`s of a collected batch, in arrival order. + pub fn rids(tones: &[Value]) -> Vec { + tones.iter().filter_map(|t| t["rid"].as_str().map(String::from)).collect() + } - /// The rid values of a collected batch, in arrival order. - pub fn rids(tones: &[Value]) -> Vec { - tones.iter().filter_map(|t| t["rid"].as_str().map(String::from)).collect() - } - - /// Tones the ensemble's inbox accepted from a peer then destroyed - /// because no local subscriber was attached. Must stay zero when - /// a scenario subscribes before sending. - pub async fn ensemble_dropped(&self, humd: Hid) -> u64 { - self.humds.read().get(&humd).map(|h| h.ensemble.inbox_dropped()).unwrap_or(0) - } + /// Tones the inbox accepted from a peer then destroyed because no + /// local subscriber was attached. Must stay zero when a scenario + /// subscribes before sending — loss belongs to the link, not here. + pub fn ensemble_dropped(&self, humd: Hid) -> u64 { + self.humds.read().get(&humd).map(|h| h.ensemble.inbox_dropped()).unwrap_or(0) + } /// Shutdown all humds and drain their join handles. pub async fn shutdown(self) { diff --git a/sim/tests/lossy_link.rs b/sim/tests/lossy_link.rs index 89c6ac9d..8cf4b144 100644 --- a/sim/tests/lossy_link.rs +++ b/sim/tests/lossy_link.rs @@ -11,7 +11,6 @@ use std::time::Duration; use ensemble::{LinkFaults, Script}; use sim::Sim; -use std::time::Duration; const WINDOW: Duration = Duration::from_millis(750); @@ -21,24 +20,30 @@ async fn pair() -> (Arc, ensemble::Hid, ensemble::Hid) { let b = ensemble::Hid::random_humd(); sim.spawn_humd(a).await; sim.spawn_humd(b).await; - sim.await_ready(a); - sim.await_ready(b); + sim.await_ready(a).await.expect("a ready"); + sim.await_ready(b).await.expect("b ready"); sim.wire(a, b).unwrap(); sim.await_handshake(a, b).await.unwrap(); (Arc::new(sim), a, b) } +// Counters taken after setup so a scenario measures its own traffic. +fn base(sim: &Sim, a: ensemble::Hid, b: ensemble::Hid) -> ensemble::LinkCounters { + sim.link_counters(a, b).unwrap().0 +} + // ── intact ────────────────────────────────────────────────────────── #[tokio::test(flavor = "multi_thread")] async fn intact_delivers_all_in_order() { let (sim, a, b) = pair().await; + let b0 = base(&sim, a, b); let mut rx = sim.humd_peer_sub(b).unwrap(); sim.send_marks(a, b, "intact", 6).await.unwrap(); let rid = Sim::collect_rids(&mut rx, 6, WINDOW).await; assert_eq!(rid, ["intact-0","intact-1","intact-2","intact-3","intact-4","intact-5"]); - let (ab, _) = sim.link_counters(a, b).unwrap(); + let ab = sim.link_counters_since(a, b, &b0).unwrap(); assert_eq!(ab.offered, 6); assert_eq!(ab.dropped, 0); } @@ -48,12 +53,13 @@ async fn intact_delivers_all_in_order() { #[tokio::test(flavor = "multi_thread")] async fn dropped_tones_never_arrive() { let (sim, a, b) = pair().await; + let b0 = base(&sim, a, b); sim.impair_dir(a, b, LinkFaults::default().drop_next(4)).unwrap(); let mut rx = sim.humd_peer_sub(b).unwrap(); sim.send_marks(a, b, "lossy", 6).await.unwrap(); let rid = Sim::collect_rids(&mut rx, 2, WINDOW).await; - let (ab, _) = sim.link_counters(a, b).unwrap(); + let ab = sim.link_counters_since(a, b, &b0).unwrap(); assert_eq!(ab.dropped, 4); assert_eq!(ab.delivered, 2); assert_eq!(rid, ["lossy-4","lossy-5"]); @@ -64,12 +70,13 @@ async fn dropped_tones_never_arrive() { #[tokio::test(flavor = "multi_thread")] async fn duplicated_tones_arrive_twice() { let (sim, a, b) = pair().await; + let b0 = base(&sim, a, b); sim.impair_dir(a, b, LinkFaults::default().dup_next(3)).unwrap(); let mut rx = sim.humd_peer_sub(b).unwrap(); sim.send_marks(a, b, "dup", 6).await.unwrap(); let rid = Sim::collect_rids(&mut rx, 9, WINDOW).await; - let (ab, _) = sim.link_counters(a, b).unwrap(); + let ab = sim.link_counters_since(a, b, &b0).unwrap(); assert_eq!(ab.duplicated, 3); assert_eq!(rid, [ "dup-0","dup-0","dup-1","dup-1","dup-2","dup-2", @@ -95,6 +102,7 @@ async fn reordered_tones_swap_adjacent_pairs() { #[tokio::test(flavor = "multi_thread")] async fn partition_stops_delivery() { let (sim, a, b) = pair().await; + let b0 = base(&sim, a, b); sim.partition(a, b).unwrap(); let mut rx = sim.humd_peer_sub(b).unwrap(); sim.send_marks(a, b, "cut", 5).await.unwrap(); @@ -105,23 +113,25 @@ async fn partition_stops_delivery() { // ── heal (lossy recover) ─────────────────────────────────────────── +/// A lossy heal is still lossy: the buffer flushes, but the link's +/// pending verdict eats the first tones off it. #[tokio::test(flavor = "multi_thread")] -async fn heal_discards_buffer() { +async fn heal_flushes_buffer_through_link_faults() { let (sim, a, b) = pair().await; + let b0 = base(&sim, a, b); sim.partition(a, b).unwrap(); sim.send_marks(a, b, "buf", 8).await.unwrap(); assert_eq!(sim.buffered(a, b).unwrap(), 8); sim.impair_dir(a, b, LinkFaults::default().drop_next(3)).unwrap(); + let mut rx = sim.humd_peer_sub(b).unwrap(); sim.heal(a, b).await.unwrap(); - let mut rx = sim.humd_peer_sub(b).unwrap(); - sim.send_marks(a, b, "after", 3).await.unwrap(); - let rid = Sim::collect_rids(&mut rx, 3, WINDOW).await; - let (ab, _) = sim.link_counters(a, b).unwrap(); - assert_eq!(ab.lost_on_heal, 3, "heal must discard buffered data"); - assert_eq!(rid, ["after-0","after-1","after-2"]); - assert_eq!(sim.buffered(a, b).unwrap(), 0); + let got = Sim::collect_rids(&mut rx, 5, WINDOW).await; + let ab = sim.link_counters_since(a, b, &b0).unwrap(); + assert_eq!(ab.lost_on_heal, 3, "the first 3 buffered tones are lost"); + assert_eq!(got, ["buf-3", "buf-4", "buf-5", "buf-6", "buf-7"]); + assert_eq!(sim.buffered(a, b).unwrap(), 0, "buffer fully drained"); } // ── link recovers after healing ───────────────────────────────────── @@ -146,6 +156,7 @@ async fn link_recovers_after_lossy_heal() { #[tokio::test(flavor = "multi_thread")] async fn impairment_is_directional() { let (sim, a, b) = pair().await; + let b0 = base(&sim, a, b); sim.impair_dir(a, b, LinkFaults::default().drop_next(10)).unwrap(); let mut rx = sim.humd_peer_sub(a).unwrap(); sim.send_marks(b, a, "reply", 3).await.unwrap(); @@ -158,6 +169,7 @@ async fn impairment_is_directional() { #[tokio::test(flavor = "multi_thread")] async fn losses_are_link_level_not_inbox_level() { let (sim, a, b) = pair().await; + let b0 = base(&sim, a, b); sim.impair_dir(a, b, LinkFaults::default() .script(Script { drop_every: 3, ..Script::default() }) .seed(7)) @@ -167,9 +179,9 @@ async fn losses_are_link_level_not_inbox_level() { sim.send_marks(a, b, "thirds", 30).await.unwrap(); let rid = Sim::collect_rids(&mut rx, 20, WINDOW).await; - let (ab, _) = sim.link_counters(a, b).unwrap(); + let ab = sim.link_counters_since(a, b, &b0).unwrap(); assert_eq!(rid.len() as u64, ab.delivered); assert_eq!(ab.offered, 30); assert_eq!(ab.dropped, 10); - assert_eq!(sim.ensemble_dropped(b).await, 0); + assert_eq!(sim.ensemble_dropped(b), 0); } From 8784e075a0d9cd1ea5394365a88a59e67edd3b46 Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 20:29:54 +0500 Subject: [PATCH 03/15] ensemble: peer liveness lease, probe, and eviction A peer is live while traffic has arrived within ttl. The drainer stamps a per-peer lease on every inbound tone, so a busy peer needs no probing; a wedged link stops stamping and goes stale. The drainer previously exited on a closed transport leaving the registry entry orphaned, and nothing anywhere removed a peer except the peer-remove CLI. The exit now records TransportClosed, which a sweep can reap. chi:peer-ping / chi:peer-pong as a literal pair, matching how gossip-publish and kad-find-node avoid pulling the chi enum into this crate. Probes are answered in the drainer and never reach subscribers. Dead outranks stale: the transport's word about its own link is a fact, whereas staleness is inferred from a quiet clock. --- ensemble/src/lib.rs | 101 ++++++++++++++++++- ensemble/src/liveness.rs | 208 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 307 insertions(+), 2 deletions(-) create mode 100644 ensemble/src/liveness.rs diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index 37d460df..4c44047b 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -57,6 +57,11 @@ pub use iroh::{IrohEndpoint, IrohTransport, IROH_ALPN}; pub mod gossip; pub use gossip::{gossip_tone, mint_msg_id, GossipState, GOSSIP_CHI, GOSSIP_SEEN_CAP}; +pub mod liveness; +pub use liveness::{ + ping_tone, pong_tone, probe_seq, Lease, Liveness, LivenessSignal, PING_CHI, PONG_CHI, +}; + pub mod kad; pub use kad::{ find_node_resp_tone, find_node_tone, mint_query_id, parse_find_node, parse_find_node_resp, @@ -707,6 +712,8 @@ impl PeerConnection for InMemoryEndpoint { struct Peer { conn: Arc, learned_caps: Option, + /// Liveness lease, stamped by the drainer on every inbound tone. + lease: Lease, } /// One humd's view of the ensemble: peers it knows about, their @@ -875,7 +882,7 @@ impl Ensemble { let rx = conn.take_receiver(); self.peers.write().insert( id, - Peer { conn: conn.clone(), learned_caps: None }, + Peer { conn: conn.clone(), learned_caps: None, lease: Lease::new() }, ); // Bootstrap the kad routing table with the peer we just wired. // The HumdAddr from the transport carries whatever dial hints @@ -959,6 +966,9 @@ impl Ensemble { // subscribers, and re-fanned to every OTHER peer. // Falls through to the inbox fan-out if the tone is // malformed (treats it as opaque application data). + if handle_liveness(&peers, &my_id, &id, &conn_for_drain, &tone).await { + continue; + } if tone.get("chi").and_then(|v| v.as_str()) == Some(GOSSIP_CHI) { if handle_gossip(&gossip, &peers, &id, &tone).await { continue; @@ -981,6 +991,13 @@ impl Ensemble { // out. Receivers may be absent — broadcast drops. inbox.publish(tone); } + // The transport's receiver closed. Mark the lease dead + // so a sweep can reap it; the registry entry itself + // outlives the drainer by design, because the daemon + // owns eviction and redial. + if let Some(p) = peers.write().get_mut(&id) { + p.lease.observe(LivenessSignal::TransportClosed); + } }); } } @@ -1020,7 +1037,7 @@ impl Ensemble { let rx = conn.take_receiver(); self.peers.write().insert( id, - Peer { conn: conn.clone(), learned_caps: None }, + Peer { conn: conn.clone(), learned_caps: None, lease: Lease::new() }, ); // Bootstrap the kad routing table — same as `install`. self.kad.note_peer(conn.peer().clone()); @@ -1076,6 +1093,9 @@ impl Ensemble { } continue; } + if handle_liveness(&peers, &my_id, &id, &conn_for_drain, &tone).await { + continue; + } if tone.get("chi").and_then(|v| v.as_str()) == Some(GOSSIP_CHI) { if handle_gossip(&gossip, &peers, &id, &tone).await { continue; @@ -1091,10 +1111,57 @@ impl Ensemble { } inbox.publish(tone); } + // The transport's receiver closed. Mark the lease dead + // so a sweep can reap it; the registry entry itself + // outlives the drainer by design, because the daemon + // owns eviction and redial. + if let Some(p) = peers.write().get_mut(&id) { + p.lease.observe(LivenessSignal::TransportClosed); + } }); } } + /// Send a `chi:"peer-ping"` to every installed peer. A peer whose + /// link is wedged will not answer, and the answer is what renews + /// its lease — so a sweep after `ttl` reaps it. + pub async fn probe_all(&self, seq: u64) { + let peers: Vec<(Hid, Arc)> = self + .peers + .read() + .iter() + .map(|(id, p)| (*id, p.conn.clone())) + .collect(); + for (id, conn) in peers { + let _ = conn.send(ping_tone(&self.me, &id, seq)).await; + } + } + + /// Liveness of one peer. `None` if it isn't installed. + pub fn peer_liveness(&self, id: &Hid, ttl: std::time::Duration) -> Option { + self.peers.read().get(id).map(|p| p.lease.state(ttl)) + } + + /// Every peer that has stopped answering, and is due for eviction. + pub fn expired_peers(&self, ttl: std::time::Duration) -> Vec { + self.peers + .read() + .iter() + .filter(|(_, p)| p.lease.expired(ttl)) + .map(|(id, _)| *id) + .collect() + } + + /// Evict every peer whose lease has run out, closing each link. + /// Returns the evicted ids so the caller can redial them. + pub fn evict_expired(&self, ttl: std::time::Duration) -> Vec { + let expired = self.expired_peers(ttl); + for id in &expired { + self.remove_peer(id); + } + expired + } + pub fn remove_peer(&self, id: &Hid) { if let Some(p) = self.peers.write().remove(id) { p.conn.close(); @@ -1696,6 +1763,36 @@ async fn handle_kad( } } +/// Renew a peer's lease and answer any probe. Returns true when the +/// tone was a liveness control message and should not reach +/// subscribers — a `peer-ping` is answered here and swallowed, a +/// `peer-pong` is absorbed silently, and everything else falls through. +async fn handle_liveness( + peers: &Arc>>, + me: &Hid, + arrived_from: &Hid, + conn: &Arc, + tone: &Tone, +) -> bool { + // The arrival itself is the lease renewal, for every inbound tone. + if let Some(p) = peers.write().get_mut(arrived_from) { + p.lease.observe(LivenessSignal::Traffic); + } + let chi = tone.get("chi").and_then(|v| v.as_str()); + if chi == Some(PONG_CHI) { + return true; + } + if chi != Some(PING_CHI) { + return false; + } + // Answer the probe, then swallow it: a probe is link maintenance, + // not application traffic. + if let Some(seq) = probe_seq(tone) { + let _ = conn.send(pong_tone(me, arrived_from, seq)).await; + } + true +} + async fn handle_gossip( gossip: &Arc, peers: &Arc>>, diff --git a/ensemble/src/liveness.rs b/ensemble/src/liveness.rs new file mode 100644 index 00000000..7c46c20e --- /dev/null +++ b/ensemble/src/liveness.rs @@ -0,0 +1,208 @@ +//! Peer liveness: probe, expire, evict. +//! +//! A peer is *live* while traffic has arrived from it within `ttl`. A +//! peer is *dead* once nothing has, and a dead peer is evicted from the +//! registry so the caller can redial it. `chi:"peer-ping"` is the +//! probe; any inbound tone at all renews the lease, so a busy peer +//! never needs pinging. +//! +//! The peer registry has no clock of its own — the daemon owns the +//! sweep, because the daemon owns the dialer that has something to do +//! with an eviction. + +use std::time::Duration; + +use crate::{Hid, Tone}; + +/// Wire-level chi for a liveness probe. Answered with `peer-pong`. +pub const PING_CHI: &str = "peer-ping"; + +/// Wire-level chi for a probe reply. +pub const PONG_CHI: &str = "peer-pong"; + +/// What the registry knows about a peer's reachability. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Liveness { + /// Traffic seen within `ttl`. + Live, + /// Nothing seen within `ttl`. Still installed, still routable, and + /// expected to be reaped on the next sweep. + Stale, + /// The transport reported the link closed. Distinguished from + /// `Stale` because it is a fact rather than an inference, so a + /// redial is worth attempting immediately instead of after a sweep. + Dead, +} + +/// How a peer lease is renewed. The drainer stamps on every inbound +/// tone; a probe reply is just a tone that says "I'm still here" with +/// no payload behind it. +#[derive(Debug, Clone, PartialEq)] +pub enum LivenessSignal { + /// Any inbound traffic — a probe reply is the cheapest kind. + Traffic, + /// The transport's receiver closed. Terminal for this connection. + TransportClosed, +} + +impl LivenessSignal { + /// Whether this signal keeps a peer on the `Live` side of the + /// lease. `TransportClosed` cannot: the link is gone regardless of + /// how recently we heard from it. + pub fn renews(self) -> bool { + matches!(self, LivenessSignal::Traffic) + } +} + +/// One peer's lease. +#[derive(Debug, Clone)] +pub struct Lease { + /// Traffic last seen from this peer, or `None` before the first + /// inbound tone. A peer is `Stale` until it proves itself. + pub last_seen: Option, + /// Set when the transport reported the link closed. + pub closed: bool, +} + +impl Default for Lease { + fn default() -> Self { + Self { last_seen: None, closed: false } + } +} + +impl Lease { + pub fn new() -> Self { + Self::default() + } + + /// Apply an inbound signal. + pub fn observe(&mut self, signal: LivenessSignal) { + match signal { + LivenessSignal::Traffic => self.last_seen = Some(std::time::Instant::now()), + LivenessSignal::TransportClosed => self.closed = true, + } + } + + /// Classify against `ttl`. A closed link is `Dead` regardless of + /// the clock — the transport told us, and no amount of recent + /// traffic overturns that. + pub fn state(&self, ttl: Duration) -> Liveness { + if self.closed { + return Liveness::Dead; + } + match self.last_seen { + // Never heard from: a link that hasn't finished its + // handshake yet is not stale, it's unproven. + None => Liveness::Live, + Some(t) if t.elapsed() > ttl => Liveness::Stale, + Some(_) => Liveness::Live, + } + } + + /// Whether the next sweep should reap this peer. + pub fn expired(&self, ttl: Duration) -> bool { + self.state(ttl) != Liveness::Live + } +} + +/// A `peer-ping` carrying our own identity so the far end can key its +/// reply at us without a registry lookup. +pub fn ping_tone(from: &Hid, to: &Hid, seq: u64) -> Tone { + serde_json::json!({ + "chi": PING_CHI, + "rid": hum_identity::HumId::mint().to_string(), + "from": from.to_hex(), + "to": to.to_hex(), + "seq": seq, + }) +} + +/// A `peer-pong`, echoing the probe's seq so a caller can pair +/// request with reply. +pub fn pong_tone(from: &Hid, to: &Hid, seq: u64) -> Tone { + serde_json::json!({ + "chi": PONG_CHI, + "rid": hum_identity::HumId::mint().to_string(), + "from": from.to_hex(), + "to": to.to_hex(), + "seq": seq, + }) +} + +/// The seq a probe carries, or `None` if this isn't a probe. +pub fn probe_seq(tone: &Tone) -> Option { + if tone.get("chi").and_then(|v| v.as_str()) != Some(PING_CHI) { + return None; + } + tone.get("seq").and_then(|v| v.as_u64()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn fresh_lease_is_live() { + assert_eq!(Lease::new().state(Duration::from_secs(30)), Liveness::Live); + } + + #[test] + fn traffic_keeps_a_peer_live() { + let mut l = Lease::new(); + l.observe(LivenessSignal::Traffic); + assert_eq!(l.state(Duration::from_secs(30)), Liveness::Live); + assert!(!l.expired(Duration::from_secs(30))); + } + + #[test] + fn quiet_past_ttl_goes_stale() { + let mut l = Lease::new(); + l.observe(LivenessSignal::Traffic); + // Backdate rather than sleep. + l.last_seen = Some(std::time::Instant::now() - Duration::from_secs(31)); + assert_eq!(l.state(Duration::from_secs(30)), Liveness::Stale); + assert!(l.expired(Duration::from_secs(30))); + } + + #[test] + fn closed_is_dead_even_when_fresh() { + let mut l = Lease::new(); + l.observe(LivenessSignal::Traffic); + l.observe(LivenessSignal::TransportClosed); + assert_eq!(l.state(Duration::from_secs(30)), Liveness::Dead); + } + + #[test] + fn closed_survives_later_traffic() { + // The transport's word outranks the clock: a link it has + // dropped is dead even if a tone had arrived a moment ago. + let mut l = Lease::new(); + l.observe(LivenessSignal::TransportClosed); + l.observe(LivenessSignal::Traffic); + assert_eq!(l.state(Duration::from_secs(30)), Liveness::Dead); + assert!(l.expired(Duration::from_secs(30))); + } + + #[test] + fn probe_seq_reads_only_pings() { + let p = ping_tone(&Hid::random_humd(), &Hid::random_humd(), 7); + assert_eq!(probe_seq(&p), Some(7)); + let c = pong_tone(&Hid::random_humd(), &Hid::random_humd(), 7); + assert_eq!(probe_seq(&c), None); + } + + #[test] + fn ping_and_pong_carry_chi_and_seq() { + let a = Hid::random_humd(); + let b = Hid::random_humd(); + assert_eq!(ping_tone(&a, &b, 3)["chi"], PING_CHI); + assert_eq!(pong_tone(&a, &b, 3)["chi"], PONG_CHI); + assert_eq!(ping_tone(&a, &b, 3)["seq"], 3); + } + + #[test] + fn transport_closed_does_not_renew() { + assert!(!LivenessSignal::TransportClosed.renews()); + assert!(LivenessSignal::Traffic.renews()); + } +} From 655d9c03933006d4e74837cd124091721ab56797 Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 20:34:06 +0500 Subject: [PATCH 04/15] humd: redial supervisor with exponential backoff Boot dialed every peer once and never looked back, so a peer that died stayed in the registry forever: routing kept addressing a dead link, and a peer that restarted was unreachable because nobody dialed it again. Recovery required restarting the daemon. The supervisor probes, then sweeps. A peer that stops answering goes stale; after ttl it is evicted and redialled. Backoff is per-peer with full jitter, so one dead peer doesn't slow the others and a mesh restarting together doesn't synchronise onto the same retry instant. tick() is public and returns a SweepReport so a scenario can drive the supervisor deterministically instead of waiting on a timer. Each transport gains dial_one, so a redial failure is attributable to one peer rather than a whole sweep. A transport that isn't configured is skipped rather than failed. --- humd/src/lib.rs | 27 ++++- humd/src/peer_transport/iroh.rs | 52 ++++++--- humd/src/peer_transport/tcp.rs | 48 +++++--- humd/src/redial.rs | 146 +++++++++++++++++++++++ humd/src/supervisor.rs | 198 ++++++++++++++++++++++++++++++++ 5 files changed, 434 insertions(+), 37 deletions(-) create mode 100644 humd/src/redial.rs create mode 100644 humd/src/supervisor.rs diff --git a/humd/src/lib.rs b/humd/src/lib.rs index c7fcdcc5..9eef5f8e 100644 --- a/humd/src/lib.rs +++ b/humd/src/lib.rs @@ -27,6 +27,8 @@ mod identity; mod peer_transport; mod peers; mod penny; +pub mod redial; +mod supervisor; pub mod thrumd; pub use identity::{key_path, load_or_mint_key, read_key}; pub use peers::{peers_path, PeerConfig}; @@ -206,19 +208,22 @@ where // so the peer registry stays transport-agnostic. Failures are // logged and non-fatal. let mut peer_reach: Vec = Vec::new(); + let mut supervisor: Option = None; if let (Some(ens), Some(key)) = (&ensemble_opt, &cfg.humd_key) { let my_caps = my_capabilities(&cfg); + let mut iroh_transport: Option> = None; match peer_transport::iroh::bind(key).await { Ok((transport, hints)) => { let transport = Arc::new(transport); peer_transport::iroh::dial_all(&transport, ens, key, &cfg.bootstrap_peers, &my_caps).await; peer_transport::iroh::spawn_listener( - transport, + transport.clone(), ens.clone(), key.clone(), my_caps.clone(), ); + iroh_transport = Some(transport); peer_reach.extend(hints); } Err(e) => warn!(err = %e, "peer.iroh.bind_failed"), @@ -232,6 +237,21 @@ where } peer_transport::tcp::dial_all(ens, key, &cfg.bootstrap_peers, &my_caps).await; + + // Boot dialed every peer once. The supervisor is what makes the + // peer set converge afterwards: a peer that dies gets evicted + // and redialled, so a restarted peer becomes reachable again + // without restarting this humd. + if !cfg.bootstrap_peers.is_empty() { + supervisor = Some(supervisor::Supervisor::new( + ens.clone(), + key.clone(), + Arc::new(cfg.bootstrap_peers.clone()), + my_caps, + iroh_transport, + supervisor::LivenessConfig::default(), + )); + } } else if !cfg.bootstrap_peers.is_empty() { warn!( count = cfg.bootstrap_peers.len(), @@ -243,6 +263,11 @@ where // ensemble instead of just cfg.ensemble. let ensemble_for_sink = ensemble_opt.clone(); + // Detached so a failing redial can never take the daemon down. + if let Some(sup) = supervisor { + tokio::spawn(sup.run()); + } + // humd no longer hosts an in-process nest pool — worker bees // register over thrum as separate processes and own their own // MCP servers (when their compute speaks MCP at all). humd diff --git a/humd/src/peer_transport/iroh.rs b/humd/src/peer_transport/iroh.rs index 7c41c50b..241216ab 100644 --- a/humd/src/peer_transport/iroh.rs +++ b/humd/src/peer_transport/iroh.rs @@ -62,6 +62,38 @@ pub(crate) async fn bind(humd_key: &HumdKey) -> anyhow::Result<(IrohTransport, V Ok((transport, hints)) } +/// Dial one bootstrap peer and install it. Returns true when the +/// connection is open. The redial supervisor calls this per peer so a +/// failure is attributable to one peer rather than a whole sweep. +pub(crate) async fn dial_one( + transport: &IrohTransport, + ens: &Arc, + key: &HumdKey, + peer: &PeerConfig, + my_caps: &PeerCapabilities, +) -> bool { + use ensemble::Transport as _; + + if !peer.hints.iter().any(|h| h.starts_with(ensemble::iroh::IROH_HINT)) { + return false; + } + let mut peer_addr = HumdAddr::new(peer.humd_id); + for h in &peer.hints { + peer_addr.hints.push(h.clone()); + } + match transport.connect(&peer_addr).await { + Ok(conn) => { + info!(peer = %peer.humd_id.short(), "peer.iroh.dial.ok"); + ens.install(conn, my_caps.clone(), key); + true + } + Err(e) => { + warn!(peer = %peer.humd_id.short(), err = %e, "peer.iroh.dial.failed"); + false + } + } +} + /// Open one iroh connection per bootstrap peer entry, install signed. /// Entries without an `iroh:` hint are skipped — those are for other /// transports. @@ -72,26 +104,8 @@ pub(crate) async fn dial_all( peers: &[PeerConfig], my_caps: &PeerCapabilities, ) { - use ensemble::Transport as _; - for peer in peers { - if !peer.hints.iter().any(|h| h.starts_with(ensemble::iroh::IROH_HINT)) { - trace!(peer = %peer.humd_id.short(), "peer.iroh.skip.no_hint"); - continue; - } - let mut peer_addr = HumdAddr::new(peer.humd_id); - for h in &peer.hints { - peer_addr.hints.push(h.clone()); - } - match transport.connect(&peer_addr).await { - Ok(conn) => { - info!(peer = %peer.humd_id.short(), "peer.iroh.dial.ok"); - ens.install(conn, my_caps.clone(), key); - } - Err(e) => { - warn!(peer = %peer.humd_id.short(), err = %e, "peer.iroh.dial.failed"); - } - } + dial_one(transport, ens, key, peer, my_caps).await; } } diff --git a/humd/src/peer_transport/tcp.rs b/humd/src/peer_transport/tcp.rs index ccf25978..6cffab60 100644 --- a/humd/src/peer_transport/tcp.rs +++ b/humd/src/peer_transport/tcp.rs @@ -32,6 +32,36 @@ use crate::peers::PeerConfig; /// Open one TCP connection per bootstrap peer entry, install signed. /// Entries without a `tcp:` hint are skipped — those are for other /// transports. +/// Dial one bootstrap peer and install it. Returns true when the +/// connection is open, so the redial supervisor can arm or clear that +/// peer's backoff. +pub(crate) async fn dial_one( + ens: &Arc, + key: &HumdKey, + peer: &PeerConfig, + my_caps: &PeerCapabilities, +) -> bool { + let Some(addr) = peer.hints.iter().find_map(|h| h.strip_prefix("tcp:")) else { + trace!(peer = %peer.humd_id.short(), "peer.tcp.skip.no_hint"); + return false; + }; + let mut peer_addr = HumdAddr::new(peer.humd_id); + for h in &peer.hints { + peer_addr.hints.push(h.clone()); + } + match TcpEndpoint::connect(addr, peer_addr, PeerCapabilities::default()).await { + Ok(conn) => { + info!(peer = %peer.humd_id.short(), addr, "peer.tcp.dial.ok"); + ens.install(conn as Arc, my_caps.clone(), key); + true + } + Err(e) => { + warn!(peer = %peer.humd_id.short(), addr, err = %e, "peer.tcp.dial.failed"); + false + } + } +} + pub(crate) async fn dial_all( ens: &Arc, key: &HumdKey, @@ -39,23 +69,7 @@ pub(crate) async fn dial_all( my_caps: &PeerCapabilities, ) { for peer in peers { - let Some(addr) = peer.hints.iter().find_map(|h| h.strip_prefix("tcp:")) else { - trace!(peer = %peer.humd_id.short(), "peer.tcp.skip.no_hint"); - continue; - }; - let mut peer_addr = HumdAddr::new(peer.humd_id); - for h in &peer.hints { - peer_addr.hints.push(h.clone()); - } - match TcpEndpoint::connect(addr, peer_addr, PeerCapabilities::default()).await { - Ok(conn) => { - info!(peer = %peer.humd_id.short(), addr, "peer.tcp.dial.ok"); - ens.install(conn as Arc, my_caps.clone(), key); - } - Err(e) => { - warn!(peer = %peer.humd_id.short(), addr, err = %e, "peer.tcp.dial.failed"); - } - } + dial_one(ens, key, peer, my_caps).await; } } diff --git a/humd/src/redial.rs b/humd/src/redial.rs new file mode 100644 index 00000000..56b23391 --- /dev/null +++ b/humd/src/redial.rs @@ -0,0 +1,146 @@ +//! Exponential backoff with full jitter, for peer redial. +//! +//! A peer that just died will often still be down on the next attempt. +//! Retrying immediately burns the daemon's budget and, on a mesh +//! restarting together, synchronises every node onto the same retry +//! instant. Jitter is what breaks that tie. + +use std::time::Duration; + +/// Retry schedule for one peer. Per-peer, not global: one dead peer +/// must not slow the recovery of the others. +#[derive(Debug, Clone)] +pub struct Backoff { + base: Duration, + max: Duration, + /// Consecutive failures so far. Reset by a successful dial. + attempt: u32, + /// When the next attempt is allowed. `None` means "never failed", + /// so the first dial is immediate. + ready_at: Option, +} + +impl Backoff { + pub fn new(base: Duration, max: Duration) -> Self { + Self { base, max, attempt: 0, ready_at: None } + } + + /// Consecutive failed attempts. + pub fn attempts(&self) -> u32 { + self.attempt + } + + /// Whether a dial may be attempted right now. + pub fn ready(&self) -> bool { + match self.ready_at { + None => true, + Some(t) => std::time::Instant::now() >= t, + } + } + + /// The delay before the next attempt would be allowed. + pub fn wait(&self) -> Option { + self.ready_at.map(|t| t.saturating_duration_since(std::time::Instant::now())) + } + + /// Record a failed attempt and arm the next one. + pub fn fail(&mut self) -> Duration { + // 2^attempt, clamped before the shift: `1u32 << 32` is itself an + // overflow, and saturating the multiplier is what stops a + // long-dead peer from wrapping to a *short* delay. + let exp = self.attempt.min(31); + let scaled = self.base.saturating_mul(1u32 << exp); + let ceiling = scaled.min(self.max); + let delay = jitter(ceiling); + self.attempt = self.attempt.saturating_add(1); + self.ready_at = Some(std::time::Instant::now() + delay); + delay + } + + /// Record a successful dial. The peer is reachable, so the next + /// failure starts from the bottom of the schedule again. + pub fn succeed(&mut self) { + self.attempt = 0; + self.ready_at = None; + } +} + +/// Full jitter: uniform over `[ceiling/2, ceiling]`. Keeps a floor so a +/// fast peer isn't hammered, while spreading the herd across the top +/// half of the window. +fn jitter(ceiling: Duration) -> Duration { + let half = ceiling / 2; + let span = ceiling - half; + if span.is_zero() { + return half; + } + let extra = rand::random::() % (span.as_millis() as u64).max(1); + half + Duration::from_millis(extra) +} + +impl Default for Backoff { + fn default() -> Self { + Self::new(Duration::from_secs(1), Duration::from_secs(300)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn first_attempt_is_immediate() { + assert!(Backoff::default().ready()); + } + + #[test] + fn failing_arms_the_next_attempt() { + let mut b = Backoff::new(Duration::from_millis(10), Duration::from_secs(60)); + b.fail(); + assert!(!b.ready(), "just failed, must not retry instantly"); + assert!(b.wait().is_some()); + } + + #[test] + fn success_rearms_immediately() { + let mut b = Backoff::default(); + b.fail(); + b.succeed(); + assert!(b.ready()); + assert_eq!(b.attempts(), 0); + } + + #[test] + fn delay_grows_and_is_capped() { + let mut b = Backoff::new(Duration::from_millis(100), Duration::from_secs(10)); + let mut prev = Duration::ZERO; + for i in 0..12 { + let d = b.fail(); + // Full jitter floors each window at half its ceiling, so + // growth is observable even though exact values vary. + assert!(d <= Duration::from_secs(10), "attempt {i} exceeded max"); + if i > 0 { + assert!(d >= prev / 2, "attempt {i} shrank past its jitter floor"); + } + prev = d; + } + } + + #[test] + fn many_failures_do_not_overflow_to_short_delays() { + let mut b = Backoff::new(Duration::from_millis(100), Duration::from_secs(10)); + for _ in 0..200 { + assert!(b.fail() <= Duration::from_secs(10)); + } + assert!(!b.ready()); + } + + #[test] + fn jitter_stays_within_half_the_ceiling() { + let ceiling = Duration::from_millis(800); + for _ in 0..200 { + let d = jitter(ceiling); + assert!(d >= ceiling / 2 && d <= ceiling, "jitter {d:?} escaped window"); + } + } +} diff --git a/humd/src/supervisor.rs b/humd/src/supervisor.rs new file mode 100644 index 00000000..285f5fcc --- /dev/null +++ b/humd/src/supervisor.rs @@ -0,0 +1,198 @@ +//! The redial supervisor: one loop that keeps the peer set matching +//! the bootstrap set. +//! +//! Boot dials every peer once and never looks back, so a peer that +//! dies stays in the registry forever: routing keeps addressing a dead +//! link, and a peer that restarts is unreachable because nobody dials +//! it again. This loop closes that gap. +//! +//! Each tick probes, then sweeps. A peer that stops answering goes +//! `Stale`; after `ttl` it is evicted, and the redial attempt that +//! follows carries its own exponential backoff so a peer that stays +//! down doesn't turn into a dial spin. + +use std::collections::HashMap; +use std::sync::Arc; +use std::time::Duration; + +use ensemble::{Ensemble, Hid, HumdKey, Liveness, PeerCapabilities}; +use tracing::{debug, info, warn}; + +use crate::peer_transport::{iroh, tcp}; +use crate::peers::PeerConfig; +use crate::redial::Backoff; + +/// Tuning for the supervisor. Timeouts are short by default: a +/// distributed peer is expected to be a LAN hop or a relay, and a slow +/// ping should be a warning, not a half-minute stall. +#[derive(Debug, Clone)] +pub struct LivenessConfig { + /// Silence after which an un-probed peer is reaped. + pub ttl: Duration, + /// How often to probe and sweep. + pub interval: Duration, + /// Backoff bounds for a peer that fails to redial. + pub backoff_base: Duration, + pub backoff_max: Duration, +} + +impl Default for LivenessConfig { + fn default() -> Self { + Self { + ttl: Duration::from_secs(30), + interval: Duration::from_secs(10), + backoff_base: Duration::from_secs(1), + backoff_max: Duration::from_secs(300), + } + } +} + +/// What one tick decided, so a caller (or a test) can observe the +/// supervisor without reading logs. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct SweepReport { + /// Peers evicted this tick, in registry order. + pub evicted: Vec, + /// Evicted peers that were dialled again this tick. + pub redialed: Vec, + /// Evicted peers whose redial was still inside its backoff window. + pub backed_off: Vec, + /// Peers that were evicted and redialled unsuccessfully. + pub dial_failed: Vec, +} + +/// The transports a peer may be reachable over. A peer with both an +/// `iroh:` and a `tcp:` hint gets tried over both; the first that +/// connects wins and the other is not attempted. +pub struct Supervisor { + ens: Arc, + key: Arc, + peers: Arc>, + my_caps: PeerCapabilities, + iroh: Option>, + cfg: LivenessConfig, + backoff: HashMap, + seq: u64, +} + +impl Supervisor { + pub fn new( + ens: Arc, + key: Arc, + peers: Arc>, + my_caps: PeerCapabilities, + iroh: Option>, + cfg: LivenessConfig, + ) -> Self { + Self { ens, key, peers, my_caps, iroh, cfg, backoff: HashMap::new(), seq: 0 } + } + + fn backoff_for(&mut self, id: &Hid) -> &mut Backoff { + self.backoff + .entry(*id) + .or_insert_with(|| Backoff::new(self.cfg.backoff_base, self.cfg.backoff_max)) + } + + /// Probe, sweep, redial. One pass. Exposed so a test can drive the + /// supervisor deterministically instead of waiting on a timer. + pub async fn tick(&mut self) -> SweepReport { + self.seq += 1; + self.ens.probe_all(self.seq).await; + let evicted = self.ens.evict_expired(self.cfg.ttl); + + let mut report = SweepReport { evicted: evicted.clone(), ..Default::default() }; + for id in evicted { + if !self.peers.iter().any(|p| p.humd_id == id) { + // Evicted a peer we didn't dial (inbound-only). Nothing + // to redial, but it's no longer in the registry. + debug!(peer = %id.short(), "liveness.evicted.inbound_only"); + continue; + } + if !self.backoff_for(&id).ready() { + report.backed_off.push(id); + continue; + } + let peer = self.peers.iter().find(|p| p.humd_id == id).expect("checked above"); + if self.dial(peer).await { + self.backoff_for(&id).succeed(); + report.redialed.push(id); + info!(peer = %id.short(), "liveness.redial.ok"); + } else { + let delay = self.backoff_for(&id).fail(); + report.dial_failed.push(id); + warn!(peer = %id.short(), ?delay, "liveness.redial.failed"); + } + } + report + } + + /// Try each transport this peer advertises, in preference order. + /// A transport that isn't configured is skipped, not failed — an + /// iroh-only daemon shouldn't count TCP's absence against the + /// peer. + async fn dial(&self, peer: &PeerConfig) -> bool { + let mut tried = false; + if let Some(transport) = &self.iroh { + if peer.hints.iter().any(|h| h.starts_with(ensemble::iroh::IROH_HINT)) { + tried = true; + if iroh::dial_one(transport, &self.ens, &self.key, peer, &self.my_caps).await { + return true; + } + } + } + if peer.hints.iter().any(|h| h.starts_with("tcp:")) { + tried = true; + if tcp::dial_one(&self.ens, &self.key, peer, &self.my_caps).await { + return true; + } + } + !tried + } + + /// Liveness of a single peer, for reporting. + pub fn peer_liveness(&self, id: &Hid) -> Option { + self.ens.peer_liveness(id, self.cfg.ttl) + } + + /// Consecutive failed redials for a peer. + pub fn attempts(&self, id: &Hid) -> u32 { + self.backoff.get(id).map(|b| b.attempts()).unwrap_or(0) + } + + /// Run until cancelled. Probes, sweeps, and redials on every tick. + pub async fn run(mut self) { + let mut ticker = tokio::time::interval(self.cfg.interval); + ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + // The first tick fires immediately; the boot dial has already + // happened, so that pass would only duplicate work. + ticker.tick().await; + loop { + ticker.tick().await; + self.tick().await; + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn default_config_probes_far_inside_ttl() { + let c = LivenessConfig::default(); + assert!( + c.interval * 3 <= c.ttl, + "a peer must be probed repeatedly before it can expire" + ); + } + + #[test] + fn report_defaults_to_empty() { + assert_eq!(SweepReport::default(), SweepReport { + evicted: vec![], + redialed: vec![], + backed_off: vec![], + dial_failed: vec![], + }); + } +} From 4e630e68ded3c167afd75fe72332fe302c477522 Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 20:39:16 +0500 Subject: [PATCH 05/15] sim + ensemble: kill a peer, and the liveness it produces MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit InMemoryEndpoint::kill drops the outbound *sender*, which is what closes the peer's receiver and ends its drainer. close() dropped only our own receiver, so it stopped us reading while leaving the peer with a link that looked fine — a kill built on it detected nothing. kill is distinct from partition on purpose: a partitioned link stays nominally up and its peer keeps renewing its lease, so a partition heals itself and a dead peer needs a redial. A liveness sweep that reaped partitions would churn the peer set on every network blip. The sender moved behind a lock so kill can drop it. Sends go through emit/try_emit, which take the sender out for the duration rather than holding a parking_lot guard across the await — holding it would block kill, and kill would deadlock against an in-flight send. --- ensemble/src/lib.rs | 94 +++++++++++++++++++---- sim/src/lib.rs | 76 +++++++++++++++++++ sim/tests/liveness.rs | 173 ++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 329 insertions(+), 14 deletions(-) create mode 100644 sim/tests/liveness.rs diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index 4c44047b..3e5b7c8d 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -28,7 +28,7 @@ use std::collections::{HashMap, VecDeque}; use std::fmt; use std::ops::Deref; -use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering}; use std::sync::Arc; use std::time::Duration; @@ -470,7 +470,11 @@ pub const PARTITION_BUFFER_CAP: usize = 64; pub struct InMemoryEndpoint { peer: HumdAddr, caps: PeerCapabilities, - tx: mpsc::Sender, + /// Outbound sender. An Option so `kill` can drop it: dropping the + /// last sender is what closes the *peer's* receiver and ends its + /// drainer. `close` only drops our own receiver, which stops us + /// reading without telling the peer anything. + tx: Mutex>>, rx: Mutex>>, /// Sim-controlled link state. When `partitioned == true`, `send()` /// accepts the tone and buffers it (bounded VecDeque) instead of @@ -491,6 +495,11 @@ pub struct InMemoryEndpoint { /// Held tone during a reorder pair, awaiting its partner so the pair /// can be delivered in reverse. reorder_hold: Mutex>, + /// Set by [`InMemoryEndpoint::kill`]. A killed link is gone, not + /// slow: sends fail and the peer drainer sees its receiver close, + /// which is what marks the lease `TransportClosed`. Distinct from + /// `partitioned`, which keeps the link nominally up. + killed: AtomicBool, } struct PartitionState { @@ -531,7 +540,7 @@ impl InMemoryEndpoint { let a = Arc::new(InMemoryEndpoint { peer: HumdAddr::new(b_id), caps: b_caps.clone(), - tx: tx_ab, + tx: Mutex::new(Some(tx_ab)), rx: Mutex::new(Some(rx_ba)), partition: Mutex::new(PartitionState { partitioned: false, @@ -541,11 +550,12 @@ impl InMemoryEndpoint { rng: Mutex::new(StdRng::seed_from_u64(seed_a)), counters: Mutex::new(LinkCounters::default()), reorder_hold: Mutex::new(None), + killed: AtomicBool::new(false), }); let b = Arc::new(InMemoryEndpoint { peer: HumdAddr::new(a_id), caps: a_caps, - tx: tx_ba, + tx: Mutex::new(Some(tx_ba)), rx: Mutex::new(Some(rx_ab)), partition: Mutex::new(PartitionState { partitioned: false, @@ -555,6 +565,7 @@ impl InMemoryEndpoint { rng: Mutex::new(StdRng::seed_from_u64(seed_b)), counters: Mutex::new(LinkCounters::default()), reorder_hold: Mutex::new(None), + killed: AtomicBool::new(false), }); (a, b) } @@ -586,13 +597,43 @@ impl InMemoryEndpoint { } } + /// `try_send` variant for the synchronous heal-flush path, where + /// holding a lock across an await isn't an option. + fn try_emit(&self, tone: Tone) -> Result<()> { + let guard = self.tx.lock(); + let Some(tx) = guard.as_ref() else { + return Err(anyhow::anyhow!("link to {} is dead", self.peer.id.short())); + }; + tx.try_send(tone).map_err(|e| anyhow::anyhow!("push: {e}")) + } + + /// Send without holding the sender lock across the await. Taking + /// the sender out and putting it back is what keeps `kill` from + /// deadlocking against a send that has already committed to it. + async fn emit(&self, tone: Tone) -> Result<()> { + let tx = { + let mut guard = self.tx.lock(); + match guard.take() { + Some(tx) => tx, + None => return Err(anyhow::anyhow!("link to {} is dead", self.peer.id.short())), + } + }; + let sent = tx.send(tone).await; + // Reclaim the sender unless `kill` won the race and dropped it. + let mut guard = self.tx.lock(); + if guard.is_none() { + *guard = Some(tx); + } + sent.map_err(|e| anyhow::anyhow!("send: {e}")) + } + /// One tone onto the wire, optionally twice. `try_send` — a full or /// dead receiver during a heal flush is a lost tone, not an error the /// caller can act on. fn push(&self, tone: Tone, duplicate: bool) -> Result<()> { - self.tx.try_send(tone.clone()).map_err(|e| anyhow::anyhow!("push: {e}"))?; + self.try_emit(tone.clone())?; self.counters.lock().delivered += 1; - if duplicate && self.tx.try_send(tone).is_ok() { + if duplicate && self.try_emit(tone).is_ok() { let mut c = self.counters.lock(); c.delivered += 1; c.duplicated += 1; @@ -633,10 +674,32 @@ impl InMemoryEndpoint { /// Frees a tone held for reorder when no partner arrives. pub fn flush_reorder(&self) -> bool { match self.reorder_hold.lock().take() { - Some(tone) => self.tx.try_send(tone).is_ok(), + Some(tone) => self.try_emit(tone).is_ok(), None => false, } } + + /// Drop the link as if the peer's process had vanished. Sends start + /// failing and the drainer at the far end sees its receiver close, + /// which marks that peer's lease `TransportClosed`. + /// + /// A partition is not this: a partitioned link stays nominally up + /// and its peer may still be alive, so its lease keeps renewing and + /// it is never reaped. That difference is the point — a partition + /// heals itself, a dead peer needs a redial. + pub fn kill(&self) { + self.killed.store(true, Ordering::SeqCst); + // Dropping our sender is what closes the peer's receiver, which + // ends its drainer and marks its lease TransportClosed. Dropping + // our own receiver would only stop *us* reading, leaving the + // peer with a link that looks fine. + self.tx.lock().take(); + } + + /// Whether this link has been killed. + pub fn is_killed(&self) -> bool { + self.killed.load(Ordering::SeqCst) + } } #[async_trait] @@ -645,6 +708,9 @@ impl PeerConnection for InMemoryEndpoint { fn capabilities(&self) -> &PeerCapabilities { &self.caps } async fn send(&self, tone: Tone) -> Result<()> { + if self.killed.load(Ordering::SeqCst) { + return Err(anyhow::anyhow!("link to {} is dead", self.peer.id.short())); + } self.counters.lock().offered += 1; { @@ -670,9 +736,9 @@ impl PeerConnection for InMemoryEndpoint { let held = self.reorder_hold.lock().take(); if let Some(held) = held { self.counters.lock().reordered += 1; - self.tx.send(tone).await.map_err(|e| anyhow::anyhow!("send: {e}"))?; + self.emit(tone).await?; self.counters.lock().delivered += 1; - self.tx.send(held).await.map_err(|e| anyhow::anyhow!("send: {e}"))?; + self.emit(held).await?; self.counters.lock().delivered += 1; return Ok(()); } @@ -680,10 +746,10 @@ impl PeerConnection for InMemoryEndpoint { return Ok(()); } - self.tx.send(tone.clone()).await.map_err(|e| anyhow::anyhow!("send: {e}"))?; + self.emit(tone.clone()).await?; self.counters.lock().delivered += 1; if verdict.duplicate { - self.tx.send(tone).await.map_err(|e| anyhow::anyhow!("send: {e}"))?; + self.emit(tone).await?; let mut c = self.counters.lock(); c.delivered += 1; c.duplicated += 1; @@ -696,9 +762,9 @@ impl PeerConnection for InMemoryEndpoint { } fn close(&self) { - // Dropping the only sender drops the channel — receiver gets None. - // We can't drop tx through &self without interior mutability; mark - // closed by replacing rx with None so subsequent takes report empty. + // Drop both halves: the sender so the peer stops reading, the + // receiver so we stop expecting. Idempotent. + self.tx.lock().take(); let _ = self.rx.lock().take(); } } diff --git a/sim/src/lib.rs b/sim/src/lib.rs index 13d47616..fa879703 100644 --- a/sim/src/lib.rs +++ b/sim/src/lib.rs @@ -226,6 +226,13 @@ impl Sim { /// support so overflow routing has somewhere to land, and the /// advertised `free_slots` reflects the atomic set via /// [`Sim::set_capacity`] (default = unlimited). + /// Rebuild the link between `a` and `b`, replacing whatever was + /// there. This is what a redial does: a fresh endpoint pair, so the + /// previous liveness state goes with it. + pub fn rewire(&self, a: Hid, b: Hid) -> Result<()> { + self.wire(a, b) + } + pub fn wire(&self, a: Hid, b: Hid) -> Result<()> { let humds = self.humds.read(); let ha = humds @@ -519,6 +526,75 @@ impl Sim { Ok(()) } + /// Drop the `a→b` direction as if `a`'s process had died: sends + /// fail and `b`'s drainer sees its receiver close, marking `a` + /// `TransportClosed` on `b`'s side. + pub fn kill_peer(&self, a: Hid, b: Hid) -> Result<()> { + self.end_for(a, b)?.kill(); + Ok(()) + } + + /// Both directions. For "this peer is gone", which is the case a + /// liveness test actually wants. + pub fn kill_link(&self, a: Hid, b: Hid) -> Result<()> { + self.kill_peer(a, b)?; + self.kill_peer(b, a) + } + + pub fn link_killed(&self, a: Hid, b: Hid) -> Result { + Ok(self.end_for(a, b)?.is_killed()) + } + + /// Liveness of `a` as `b` sees it. The sim needs its own probe and + /// sweep because it drives the ensembles directly rather than + /// through the daemon's supervisor. + pub async fn probe(&self, a: Hid, b: Hid) -> Result<()> { + let ens = self + .humds + .read() + .get(&b) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no humd {}", b.short()))?; + ens.ensemble.probe_all(0).await; + Ok(()) + } + + /// Classify every installed peer of `observer` against `ttl`. + pub fn peer_liveness( + &self, + observer: Hid, + ttl: std::time::Duration, + ) -> Result> { + let ens = self + .humds + .read() + .get(&observer) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no humd {}", observer.short()))?; + Ok(ens + .ensemble + .peers() + .into_iter() + .filter_map(|p| ens.ensemble.peer_liveness(&p, ttl).map(|l| (p, l))) + .collect()) + } + + /// Evict `observer`'s expired peers, returning who was reaped. + pub fn evict_expired(&self, observer: Hid, ttl: std::time::Duration) -> Result> { + let ens = self + .humds + .read() + .get(&observer) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no humd {}", observer.short()))?; + Ok(ens.ensemble.evict_expired(ttl)) + } + + /// Peers `observer` still has installed. + pub fn peer_count(&self, observer: Hid) -> usize { + self.humds.read().get(&observer).map(|h| h.ensemble.peers().len()).unwrap_or(0) + } + /// Drop the wired link between `a` and `b`. Both endpoints stop /// delivering outbound tones and instead buffer them up to /// `ensemble::PARTITION_BUFFER_CAP`. A subsequent [`Sim::heal`] diff --git a/sim/tests/liveness.rs b/sim/tests/liveness.rs new file mode 100644 index 00000000..7ad8e8ea --- /dev/null +++ b/sim/tests/liveness.rs @@ -0,0 +1,173 @@ +//! liveness — a peer that dies is detected, reaped, and comes back. +//! +//! The lossy-link suite proves a broken link loses data. This one proves +//! the system *notices*, which is the property that makes recovery +//! possible at all: routing must stop addressing a dead peer, and a +//! peer that restarts must become reachable without a daemon restart. +//! +//! TTLs here are milliseconds. A liveness test that sleeps for a real +//! TTL is slow, and a slow test is one that gets skipped. + +use std::time::Duration; + +use ensemble::Liveness; +use sim::Sim; + +const TTL: Duration = Duration::from_millis(120); + +async fn pair() -> (std::sync::Arc, ensemble::Hid, ensemble::Hid) { + let sim = Sim::new(); + let a = ensemble::Hid::random_humd(); + let b = ensemble::Hid::random_humd(); + sim.spawn_humd(a).await; + sim.spawn_humd(b).await; + sim.await_ready(a).await.expect("a ready"); + sim.await_ready(b).await.expect("b ready"); + sim.wire(a, b).expect("wire"); + sim.await_handshake(a, b).await.expect("handshake"); + (std::sync::Arc::new(sim), a, b) +} + +fn liveness(sim: &Sim, observer: ensemble::Hid, peer: ensemble::Hid) -> Option { + sim.peer_liveness(observer, TTL) + .expect("observer exists") + .into_iter() + .find(|(p, _)| *p == peer) + .map(|(_, l)| l) +} + +/// A peer that is answering is Live, and a sweep leaves it alone. +#[tokio::test(flavor = "multi_thread")] +async fn a_talking_peer_is_live_and_survives_a_sweep() { + let (sim, a, b) = pair().await; + sim.probe(a, b).await.expect("probe"); + assert_eq!(liveness(&sim, a, b), Some(Liveness::Live)); + assert_eq!(sim.evict_expired(a, TTL).expect("sweep"), vec![]); + assert_eq!(sim.peer_count(a), 1, "sweep must not reap a live peer"); +} + +/// Traffic is what renews the lease. This is the whole point: a busy +/// peer never needs to be probed. +#[tokio::test(flavor = "multi_thread")] +async fn traffic_renews_the_lease_without_probing() { + let (sim, a, b) = pair().await; + // Drain off a peer stamps the lease in the drainer, so a real tone + // is enough. + for i in 0..3 { + sim.send_marks(b, a, &format!("keep{i}"), 1).await.expect("send"); + } + // Give the drainer a moment to publish. + tokio::time::sleep(Duration::from_millis(30)).await; + assert_eq!(liveness(&sim, a, b), Some(Liveness::Live)); + assert_eq!(sim.evict_expired(a, TTL).expect("sweep"), vec![]); +} + +/// A killed peer's link stops answering, and the drainer notices the +/// transport is gone rather than merely quiet. +#[tokio::test(flavor = "multi_thread")] +async fn a_killed_peer_goes_dead() { + let (sim, a, b) = pair().await; + sim.probe(a, b).await.expect("probe"); + assert_eq!(liveness(&sim, a, b), Some(Liveness::Live)); + + sim.kill_peer(b, a).expect("kill b→a"); + // The drainer wakes on its closed receiver. + tokio::time::sleep(Duration::from_millis(40)).await; + assert_eq!(liveness(&sim, a, b), Some(Liveness::Dead)); +} + +/// A dead peer is still in the registry until a sweep runs. Eviction is +/// the supervisor's call, not the transport's. +#[tokio::test(flavor = "multi_thread")] +async fn a_dead_peer_survives_until_swept() { + let (sim, a, b) = pair().await; + sim.kill_link(a, b).expect("kill"); + tokio::time::sleep(Duration::from_millis(40)).await; + assert_eq!(sim.peer_count(a), 1, "transport death does not self-evict"); + + let evicted = sim.evict_expired(a, TTL).expect("sweep"); + assert_eq!(evicted, vec![b]); + assert_eq!(sim.peer_count(a), 0, "sweep reaps the dead peer"); +} + +/// Silence past the TTL is staleness — the peer never reported a closed +/// link, it just stopped answering. +#[tokio::test(flavor = "multi_thread")] +async fn silence_past_ttl_goes_stale_and_is_reaped() { + let (sim, a, b) = pair().await; + sim.probe(a, b).await.expect("probe"); + assert_eq!(liveness(&sim, a, b), Some(Liveness::Live)); + + // Drop every answer without closing the link, so the transport + // still looks up. This is the wedged-peer case: a connection that + // accepts writes but delivers nothing. + sim.impair_dir(b, a, ensemble::LinkFaults::default().drop_pct(100, 3)) + .expect("silence b→a"); + sim.probe(a, b).await.expect("probe"); + // A:→B still delivers, so the link is not killed — but B never + // answers A's probe. + tokio::time::sleep(TTL + Duration::from_millis(60)).await; + assert_eq!(liveness(&sim, a, b), Some(Liveness::Stale)); + assert_eq!(sim.evict_expired(a, TTL).expect("sweep"), vec![b]); +} + +/// The asymmetry that makes partitions and death different things: a +/// partitioned peer is not reaped, because it is still there. If +/// partitions evicted, every blip would churn the peer set. +#[tokio::test(flavor = "multi_thread")] +async fn a_partitioned_peer_is_not_dead() { + let (sim, a, b) = pair().await; + sim.partition(a, b).expect("partition"); + tokio::time::sleep(TTL + Duration::from_millis(60)).await; + assert_eq!( + liveness(&sim, a, b), + Some(Liveness::Live), + "a partition is not a death: the link is still up" + ); + assert_eq!(sim.evict_expired(a, TTL).expect("sweep"), vec![]); + assert_eq!(sim.peer_count(a), 1); +} + +/// After reaping, a rewired peer is reachable again — the recovery the +/// whole subsystem exists for. The liveness state must not be sticky. +#[tokio::test(flavor = "multi_thread")] +async fn a_rewired_peer_is_live_again() { + let (sim, a, b) = pair().await; + sim.kill_link(a, b).expect("kill"); + tokio::time::sleep(Duration::from_millis(40)).await; + assert_eq!(sim.evict_expired(a, TTL).expect("sweep"), vec![b]); + assert_eq!(sim.peer_count(a), 0); + + // Rewire, as the supervisor's redial would. + sim.rewire(a, b).expect("rewire"); + sim.await_handshake(a, b).await.expect("handshake again"); + assert_eq!(liveness(&sim, a, b), Some(Liveness::Live)); + assert_eq!(sim.peer_count(a), 1); +} + +/// Eviction is per-observer. B dying must not disturb A's other peers. +#[tokio::test(flavor = "multi_thread")] +async fn eviction_only_touches_the_dead_peer() { + let sim = Sim::new(); + let a = ensemble::Hid::random_humd(); + let b = ensemble::Hid::random_humd(); + let c = ensemble::Hid::random_humd(); + for id in [a, b, c] { + sim.spawn_humd(id).await; + sim.await_ready(id).await.expect("ready"); + } + sim.wire(a, b).expect("wire ab"); + sim.wire(a, c).expect("wire ac"); + sim.await_handshake(a, b).await.expect("ab"); + sim.await_handshake(a, c).await.expect("ac"); + let sim = std::sync::Arc::new(sim); + + sim.probe(a, b).await.expect("probe"); + sim.probe(a, c).await.expect("probe"); + sim.kill_link(b, a).expect("kill b"); + tokio::time::sleep(Duration::from_millis(40)).await; + + assert_eq!(sim.evict_expired(a, TTL).expect("sweep"), vec![b]); + assert_eq!(liveness(&sim, a, c), Some(Liveness::Live), "c is untouched"); + assert_eq!(sim.peer_count(a), 1, "only b was reaped"); +} From 7ec3b37371cb90f4c3ccaf703a82a6ec0fc4c0d1 Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 20:57:59 +0500 Subject: [PATCH 06/15] humd: give each drift test its own temp directory tmp() keyed the directory on pid + now_ms(). Tests run in parallel, so two starting in the same millisecond shared a directory; whichever finished first called remove_dir_all and the other found it empty. Seen as flag_persists_through_wilt failing in a full workspace run and passing alone. Add a per-call counter to the name. --- humd/src/drift.rs | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/humd/src/drift.rs b/humd/src/drift.rs index 27ba4bd9..cb5c5cac 100644 --- a/humd/src/drift.rs +++ b/humd/src/drift.rs @@ -319,7 +319,9 @@ fn prune_dir(dir: &Path, days: u32) -> Result { continue; } let Ok(meta) = entry.metadata() else { continue }; - let Ok(modified) = meta.modified() else { continue }; + let Ok(modified) = meta.modified() else { + continue; + }; if modified < cutoff { if let Err(e) = fs::remove_file(&path) { tracing::warn!(error = %e, path = %path.display(), "drift.prune.unlink.failed"); @@ -361,11 +363,18 @@ mod tests { use std::env; fn tmp() -> PathBuf { + // Tests run in parallel and each one removes its directory on + // the way out, so the name has to be unique per call. A + // millisecond timestamp alone collides: two tests starting in + // the same millisecond shared a directory, and whichever + // finished first deleted the other's files. + static SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); let mut p = env::temp_dir(); p.push(format!( - "drift-test-{}-{}", + "drift-test-{}-{}-{}", std::process::id(), - now_ms() + now_ms(), + SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed) )); p } From 9fa8fa84f0e94b260e71c98dd95a897743c9f2d8 Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 20:58:08 +0500 Subject: [PATCH 07/15] liveness: an unproven peer was immortal, and a probe ran the wrong way MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three defects, all found by driving the real supervisor and sim rather than the units underneath them. An unproven peer never expired. Lease::state returned Live whenever last_seen was None, so a link that swallowed every probe stayed Live forever — "unproven" had quietly become "immortal", and the black-hole case was the one case that could not be detected at all. A lease is a deadline: count silence from install, not from the first reply. Sim::probe(a, b) looked up b and probed b's peers, so it ignored a entirely and reported on the reverse direction. It happened to agree with the fault scripts written alongside it, which is how the bug survived. Add Ensemble::probe_one so a probe says something about the direction it travels; clone the conn out from under the registry lock while we are here, since holding that read lock across a send blocks every add_peer/remove_peer for the length of the send. The redial set was "just evicted", which misses every peer that failed its *first* dial: those were never installed, so no sweep can name them, and a peer down at boot stayed down permanently. It is now "configured but absent". Related, dial() returned !tried, so a peer with no usable hint counted as a successful redial, cleared its own backoff, and would be re-dialled forever. Correction to 4e630e6: I claimed a sweep that reaped partitions would churn the peer set. It does reap them, and that is right. A partition is not distinguishable from a death by observation — both are silence — so sweeping on silence is honest, and backoff plus redial is what prevents churn. What a partition is not is Dead, which stays reserved for a transport that reported closure: a fact rather than an inference, and the only case that earns an immediate redial. --- ensemble/src/lib.rs | 22 +++++ ensemble/src/liveness.rs | 26 ++++-- humd/src/lib.rs | 4 +- humd/src/supervisor.rs | 77 +++++++++++---- humd/tests/supervisor.rs | 196 +++++++++++++++++++++++++++++++++++++++ sim/src/lib.rs | 44 +++++---- sim/tests/liveness.rs | 38 ++++++-- 7 files changed, 352 insertions(+), 55 deletions(-) create mode 100644 humd/tests/supervisor.rs diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index 3e5b7c8d..0c8ed6f6 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -1203,6 +1203,19 @@ impl Ensemble { } } + /// Ping one peer. Direction matters: a probe answers a question + /// about the link in the direction it travels, so probing a->b + /// says nothing about whether b->a still works. + pub async fn probe_one(&self, id: &Hid, seq: u64) { + // Clone out from under the lock: holding the registry read + // lock across the send would block add_peer/remove_peer for + // as long as the link takes. + let conn = self.peers.read().get(id).map(|p| p.conn.clone()); + if let Some(conn) = conn { + let _ = conn.send(ping_tone(&self.me, id, seq)).await; + } + } + /// Liveness of one peer. `None` if it isn't installed. pub fn peer_liveness(&self, id: &Hid, ttl: std::time::Duration) -> Option { self.peers.read().get(id).map(|p| p.lease.state(ttl)) @@ -1218,6 +1231,15 @@ impl Ensemble { .collect() } + /// Mark a peer dead as if its transport had closed. Lets a scenario + /// produce a death without waiting out a TTL, which is the only way + /// to test eviction deterministically. + pub fn expire_peer(&self, id: &Hid) { + if let Some(p) = self.peers.write().get_mut(id) { + p.lease.observe(LivenessSignal::TransportClosed); + } + } + /// Evict every peer whose lease has run out, closing each link. /// Returns the evicted ids so the caller can redial them. pub fn evict_expired(&self, ttl: std::time::Duration) -> Vec { diff --git a/ensemble/src/liveness.rs b/ensemble/src/liveness.rs index 7c46c20e..80e40356 100644 --- a/ensemble/src/liveness.rs +++ b/ensemble/src/liveness.rs @@ -57,8 +57,11 @@ impl LivenessSignal { /// One peer's lease. #[derive(Debug, Clone)] pub struct Lease { - /// Traffic last seen from this peer, or `None` before the first - /// inbound tone. A peer is `Stale` until it proves itself. + /// When this peer was installed. Silence counts from here, so a + /// peer that never proves itself still expires: a black-hole link + /// that swallows every probe would otherwise read as `Live` forever. + pub since: std::time::Instant, + /// Traffic last seen from this peer, `None` until it first answers. pub last_seen: Option, /// Set when the transport reported the link closed. pub closed: bool, @@ -66,7 +69,11 @@ pub struct Lease { impl Default for Lease { fn default() -> Self { - Self { last_seen: None, closed: false } + Self { + since: std::time::Instant::now(), + last_seen: None, + closed: false, + } } } @@ -90,12 +97,13 @@ impl Lease { if self.closed { return Liveness::Dead; } - match self.last_seen { - // Never heard from: a link that hasn't finished its - // handshake yet is not stale, it's unproven. - None => Liveness::Live, - Some(t) if t.elapsed() > ttl => Liveness::Stale, - Some(_) => Liveness::Live, + // An unproven peer is still on probation, not immortal: quiet + // since install ages it out just like a peer that went quiet. + let quiet = self.last_seen.unwrap_or(self.since); + if quiet.elapsed() > ttl { + Liveness::Stale + } else { + Liveness::Live } } diff --git a/humd/src/lib.rs b/humd/src/lib.rs index 9eef5f8e..95d16d60 100644 --- a/humd/src/lib.rs +++ b/humd/src/lib.rs @@ -25,10 +25,10 @@ mod drone; mod drift; mod identity; mod peer_transport; -mod peers; +pub mod peers; mod penny; pub mod redial; -mod supervisor; +pub mod supervisor; pub mod thrumd; pub use identity::{key_path, load_or_mint_key, read_key}; pub use peers::{peers_path, PeerConfig}; diff --git a/humd/src/supervisor.rs b/humd/src/supervisor.rs index 285f5fcc..9ccb0335 100644 --- a/humd/src/supervisor.rs +++ b/humd/src/supervisor.rs @@ -84,7 +84,16 @@ impl Supervisor { iroh: Option>, cfg: LivenessConfig, ) -> Self { - Self { ens, key, peers, my_caps, iroh, cfg, backoff: HashMap::new(), seq: 0 } + Self { + ens, + key, + peers, + my_caps, + iroh, + cfg, + backoff: HashMap::new(), + seq: 0, + } } fn backoff_for(&mut self, id: &Hid) -> &mut Backoff { @@ -100,20 +109,36 @@ impl Supervisor { self.ens.probe_all(self.seq).await; let evicted = self.ens.evict_expired(self.cfg.ttl); - let mut report = SweepReport { evicted: evicted.clone(), ..Default::default() }; + let mut report = SweepReport { + evicted: evicted.clone(), + ..Default::default() + }; for id in evicted { if !self.peers.iter().any(|p| p.humd_id == id) { // Evicted a peer we didn't dial (inbound-only). Nothing // to redial, but it's no longer in the registry. debug!(peer = %id.short(), "liveness.evicted.inbound_only"); - continue; } + } + + // The redial set is "configured but absent", not "just evicted". + // A peer whose *first* dial failed was never installed, so no + // sweep will ever name it — and a peer that is down at boot is + // exactly the one that must come back. + let wanted: Vec = self + .peers + .iter() + .filter(|p| !self.ens.peers().contains(&p.humd_id)) + .cloned() + .collect(); + + for peer in wanted { + let id = peer.humd_id; if !self.backoff_for(&id).ready() { report.backed_off.push(id); continue; } - let peer = self.peers.iter().find(|p| p.humd_id == id).expect("checked above"); - if self.dial(peer).await { + if self.dial(&peer).await { self.backoff_for(&id).succeed(); report.redialed.push(id); info!(peer = %id.short(), "liveness.redial.ok"); @@ -127,26 +152,37 @@ impl Supervisor { } /// Try each transport this peer advertises, in preference order. - /// A transport that isn't configured is skipped, not failed — an - /// iroh-only daemon shouldn't count TCP's absence against the - /// peer. + /// A transport this daemon isn't running is skipped rather than + /// failed — an iroh-only daemon shouldn't count TCP's absence + /// against the peer. + /// + /// A peer that advertises nothing we can dial fails: reporting + /// success would clear its backoff, and the peer would be + /// re-evicted and re-"dialled" on every tick forever. async fn dial(&self, peer: &PeerConfig) -> bool { - let mut tried = false; + let mut dialable = false; if let Some(transport) = &self.iroh { - if peer.hints.iter().any(|h| h.starts_with(ensemble::iroh::IROH_HINT)) { - tried = true; + if peer + .hints + .iter() + .any(|h| h.starts_with(ensemble::iroh::IROH_HINT)) + { + dialable = true; if iroh::dial_one(transport, &self.ens, &self.key, peer, &self.my_caps).await { return true; } } } if peer.hints.iter().any(|h| h.starts_with("tcp:")) { - tried = true; + dialable = true; if tcp::dial_one(&self.ens, &self.key, peer, &self.my_caps).await { return true; } } - !tried + if !dialable { + warn!(peer = %peer.humd_id.short(), "liveness.redial.undialable"); + } + false } /// Liveness of a single peer, for reporting. @@ -188,11 +224,14 @@ mod tests { #[test] fn report_defaults_to_empty() { - assert_eq!(SweepReport::default(), SweepReport { - evicted: vec![], - redialed: vec![], - backed_off: vec![], - dial_failed: vec![], - }); + assert_eq!( + SweepReport::default(), + SweepReport { + evicted: vec![], + redialed: vec![], + backed_off: vec![], + dial_failed: vec![], + } + ); } } diff --git a/humd/tests/supervisor.rs b/humd/tests/supervisor.rs new file mode 100644 index 00000000..b0036a72 --- /dev/null +++ b/humd/tests/supervisor.rs @@ -0,0 +1,196 @@ +//! supervisor — eviction and redial, as one decision. +//! +//! `ensemble::liveness` proves the ensemble detects a dead peer. This +//! proves the daemon *acts* on it: evicts, then dials again, and +//! remembers a peer's failure so a peer that stays down doesn't turn +//! into a dial spin. +//! +//! Driving `Supervisor::tick` directly is deliberate. The production +//! loop runs on a 10s interval, and a test that waits on a real +//! interval is a test that either takes ten seconds or gets skipped. + +use std::sync::Arc; +use std::time::Duration; + +use ensemble::{Ensemble, Hid, HumdKey, PeerCapabilities}; +use humd::peers::PeerConfig; +use humd::redial::Backoff; +use humd::supervisor::{LivenessConfig, Supervisor}; + +/// No transports configured, so a dial cannot succeed — which is the +/// case that matters here: a peer that stays down must back off rather +/// than spin. +fn supervisor_for(ens: Arc, peer: Hid) -> Supervisor { + Supervisor::new( + ens, + Arc::new(HumdKey::generate()), + Arc::new(vec![PeerConfig { + humd_id: peer, + hints: vec![], + alias: None, + }]), + PeerCapabilities::default(), + None, + LivenessConfig::default(), + ) +} + +/// An endpoint claiming `peer` as its peer. The far side is returned +/// too and must be held: dropping it closes the channel, which is +/// exactly the death these tests are about. +fn endpoint( + peer: Hid, +) -> ( + Arc, + Arc, +) { + ensemble::InMemoryEndpoint::pair( + Hid::random_humd(), + PeerCapabilities::default(), + peer, + PeerCapabilities::default(), + ) +} + +/// An unproven peer is Live: no traffic and no reported close is not +/// evidence of death, and reaping it would evict every peer at boot. +#[tokio::test(flavor = "multi_thread")] +async fn an_unproven_peer_is_not_reaped() { + let me = Hid::random_humd(); + let peer = Hid::random_humd(); + let ens = Arc::new(Ensemble::new(me)); + let (mine, _theirs) = endpoint(peer); + ens.add_peer(mine); + let mut sup = supervisor_for(ens.clone(), peer); + + let report = sup.tick().await; + assert!( + report.evicted.is_empty(), + "unproven peer must survive a sweep" + ); + assert!(ens.peers().contains(&peer)); +} + +/// The core decision: a dead peer is evicted and its redial is +/// attempted in the same pass. +#[tokio::test(flavor = "multi_thread")] +async fn a_dead_peer_is_evicted_and_redial_attempted() { + let me = Hid::random_humd(); + let peer = Hid::random_humd(); + let ens = Arc::new(Ensemble::new(me)); + let (mine, _theirs) = endpoint(peer); + ens.add_peer(mine); + ens.expire_peer(&peer); + let mut sup = supervisor_for(ens.clone(), peer); + + let report = sup.tick().await; + assert_eq!(report.evicted, vec![peer], "the dead peer is reaped"); + assert_eq!(report.dial_failed, vec![peer], "and redialled this pass"); + assert!(!ens.peers().contains(&peer), "and gone from the registry"); +} + +/// Backoff is the difference between a peer being down and a peer being +/// a busy loop. +#[tokio::test(flavor = "multi_thread")] +async fn a_peer_that_stays_down_backs_off() { + let me = Hid::random_humd(); + let peer = Hid::random_humd(); + let ens = Arc::new(Ensemble::new(me)); + let (mine, _theirs) = endpoint(peer); + ens.add_peer(mine); + ens.expire_peer(&peer); + let mut sup = supervisor_for(ens.clone(), peer); + + let first = sup.tick().await; + assert_eq!(first.dial_failed, vec![peer], "first attempt is made"); + assert_eq!(sup.attempts(&peer), 1); + + // The peer is already evicted, so the next pass has nothing to reap — + // and must not dial again inside the backoff window. + let second = sup.tick().await; + assert!(second.evicted.is_empty()); + assert!( + second.dial_failed.is_empty() && second.redialed.is_empty(), + "must not redial while inside the backoff window" + ); +} + +/// The probe must reach peers that are still installed, and must not +/// resurrect an evicted one. +#[tokio::test(flavor = "multi_thread")] +async fn tick_probes_before_sweeping() { + let me = Hid::random_humd(); + let peer = Hid::random_humd(); + let ens = Arc::new(Ensemble::new(me)); + let (mine, _theirs) = endpoint(peer); + ens.add_peer(mine); + let mut sup = supervisor_for(ens.clone(), peer); + + sup.tick().await; + // The probe landed, so the peer's lease now has traffic on it. + assert_eq!( + ens.peer_liveness(&peer, LivenessConfig::default().ttl), + Some(ensemble::Liveness::Live) + ); +} + +/// Backoff resets on success, so a peer that flaps doesn't accumulate +/// delay forever. +#[test] +fn a_recovered_peer_resets_its_backoff() { + let mut b = Backoff::new(Duration::from_millis(10), Duration::from_secs(60)); + for _ in 0..5 { + b.fail(); + } + assert_eq!(b.attempts(), 5); + b.succeed(); + assert_eq!(b.attempts(), 0, "a reachable peer starts from the bottom"); + assert!(b.ready(), "and may be redialled immediately"); +} + +/// A healthy peer must survive several sweeps. This is the guard +/// against a config that probes slower than its TTL, which would evict +/// live peers on a timer. +#[tokio::test(flavor = "multi_thread")] +async fn a_healthy_peer_survives_repeated_sweeps() { + let me = Hid::random_humd(); + let peer = Hid::random_humd(); + let ens = Arc::new(Ensemble::new(me)); + let (mine, _theirs) = endpoint(peer); + ens.add_peer(mine); + let mut sup = supervisor_for(ens.clone(), peer); + + for _ in 0..5 { + sup.tick().await; + } + assert!( + ens.peers().contains(&peer), + "a live peer must never be reaped" + ); +} + +#[test] +fn config_probes_far_inside_ttl() { + let c = LivenessConfig::default(); + assert!( + c.interval * 3 <= c.ttl, + "a peer must be probed repeatedly before a sweep can reap it" + ); +} + +/// A peer that was down at boot never entered the registry, so no sweep +/// will ever name it. If the redial set were "just evicted", it would +/// never be retried — the boot-time outage would be permanent. +#[tokio::test(flavor = "multi_thread")] +async fn a_peer_that_never_installed_is_still_retried() { + let me = Hid::random_humd(); + let peer = Hid::random_humd(); + let ens = Arc::new(Ensemble::new(me)); + // Note: no add_peer. The peer is configured and simply absent. + let mut sup = supervisor_for(ens.clone(), peer); + + let report = sup.tick().await; + assert!(report.evicted.is_empty(), "nothing was installed to reap"); + assert_eq!(report.dial_failed, vec![peer], "but it is still dialled"); + assert_eq!(sup.attempts(&peer), 1); +} diff --git a/sim/src/lib.rs b/sim/src/lib.rs index fa879703..f5f6af80 100644 --- a/sim/src/lib.rs +++ b/sim/src/lib.rs @@ -548,14 +548,16 @@ impl Sim { /// Liveness of `a` as `b` sees it. The sim needs its own probe and /// sweep because it drives the ensembles directly rather than /// through the daemon's supervisor. + /// Send a liveness probe from `a` to `b`. Direction is the point: + /// this says whether a can still reach b, not the reverse. pub async fn probe(&self, a: Hid, b: Hid) -> Result<()> { let ens = self .humds .read() - .get(&b) + .get(&a) .cloned() - .ok_or_else(|| anyhow::anyhow!("no humd {}", b.short()))?; - ens.ensemble.probe_all(0).await; + .ok_or_else(|| anyhow::anyhow!("no humd {}", a.short()))?; + ens.ensemble.probe_one(&b, 0).await; Ok(()) } @@ -631,16 +633,12 @@ impl Sim { link.b_end.set_partitioned(false); } - let humds = self.humds.read(); - let ha = humds - .get(&link_a) - .cloned() - .ok_or_else(|| anyhow::anyhow!("no humd {}", link_a.short()))?; - let hb = humds - .get(&link_b) - .cloned() - .ok_or_else(|| anyhow::anyhow!("no humd {}", link_b.short()))?; - drop(humds); + let (ha, hb) = { + let humds = self.humds.read(); + (humds.get(&link_a).cloned(), humds.get(&link_b).cloned()) + }; + let ha = ha.ok_or_else(|| anyhow::anyhow!("no humd {}", link_a.short()))?; + let hb = hb.ok_or_else(|| anyhow::anyhow!("no humd {}", link_b.short()))?; for (from, to) in [(&ha, &hb), (&hb, &ha)] { let snapshot = from.waneman.snapshot(); @@ -812,6 +810,7 @@ impl Sim { // each receiver out, drain non-blockingly, then put it back. // This keeps the per-humd state simple — no long-lived fanout // task per synthetic client. + { let mut queues = h.out_queues.lock(); for (_cid, rx) in queues.iter_mut() { while let Ok(tone) = rx.try_recv() { @@ -837,7 +836,7 @@ impl Sim { } } } - drop(queues); + } // Now await the named sid's mailbox. let mut rx_opt = h.sid_mailboxes.lock().remove(sid)?; @@ -964,12 +963,19 @@ impl Sim { pub async fn await_handshake(&self, a: Hid, b: Hid) -> Result<()> { for _ in 0..200 { let done = { - let humds = self.humds.read(); - let (Some(ha), Some(hb)) = (humds.get(&a), humds.get(&b)) else { - tokio::time::sleep(Duration::from_millis(5)).await; - continue; + let (ha, hb) = { + let humds = self.humds.read(); + (humds.get(&a).cloned(), humds.get(&b).cloned()) }; - ha.ensemble.handshake_done(&b) && hb.ensemble.handshake_done(&a) + match (ha, hb) { + (Some(ha), Some(hb)) => { + ha.ensemble.handshake_done(&b) && hb.ensemble.handshake_done(&a) + } + _ => { + tokio::time::sleep(Duration::from_millis(5)).await; + continue; + } + } }; if done { let hb = self.humds.read().get(&b).cloned().expect("checked above"); diff --git a/sim/tests/liveness.rs b/sim/tests/liveness.rs index 7ad8e8ea..10239c9a 100644 --- a/sim/tests/liveness.rs +++ b/sim/tests/liveness.rs @@ -54,7 +54,9 @@ async fn traffic_renews_the_lease_without_probing() { // Drain off a peer stamps the lease in the drainer, so a real tone // is enough. for i in 0..3 { - sim.send_marks(b, a, &format!("keep{i}"), 1).await.expect("send"); + sim.send_marks(b, a, &format!("keep{i}"), 1) + .await + .expect("send"); } // Give the drainer a moment to publish. tokio::time::sleep(Duration::from_millis(30)).await; @@ -119,13 +121,23 @@ async fn a_partitioned_peer_is_not_dead() { let (sim, a, b) = pair().await; sim.partition(a, b).expect("partition"); tokio::time::sleep(TTL + Duration::from_millis(60)).await; - assert_eq!( + + // Silence is silence: from here a partition and a death look the + // same, and the sweep reaps both. What a partition is *not* is + // `Dead`, which is reserved for a transport that told us it closed + // — a fact rather than an inference, and the only case that earns + // an immediate redial. + assert_ne!( liveness(&sim, a, b), - Some(Liveness::Live), - "a partition is not a death: the link is still up" + Some(Liveness::Dead), + "no transport reported closure, so this is not a proven death" + ); + assert_eq!(liveness(&sim, a, b), Some(Liveness::Stale)); + assert_eq!( + sim.peer_count(a), + 1, + "a partition does not evict on its own" ); - assert_eq!(sim.evict_expired(a, TTL).expect("sweep"), vec![]); - assert_eq!(sim.peer_count(a), 1); } /// After reaping, a rewired peer is reachable again — the recovery the @@ -171,3 +183,17 @@ async fn eviction_only_touches_the_dead_peer() { assert_eq!(liveness(&sim, a, c), Some(Liveness::Live), "c is untouched"); assert_eq!(sim.peer_count(a), 1, "only b was reaped"); } +/// A link that swallows every probe must still be reaped. Without an +/// install timestamp such a peer is unproven forever, and "unproven" +/// quietly becomes "immortal". +#[tokio::test(flavor = "multi_thread")] +async fn a_peer_that_never_answers_goes_stale() { + let (sim, a, b) = pair().await; + // Black-hole both directions: a's probe arrives, b's reply does not. + sim.impair(a, b, ensemble::LinkFaults::default().drop_pct(100, 7)) + .expect("black hole"); + sim.probe(a, b).await.expect("probe"); + tokio::time::sleep(TTL + Duration::from_millis(60)).await; + assert_eq!(liveness(&sim, a, b), Some(Liveness::Stale)); + assert_eq!(sim.evict_expired(a, TTL).expect("sweep"), vec![b]); +} From 8a9ea8f87f2a8875650459486ed304ac505726f4 Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 21:06:12 +0500 Subject: [PATCH 08/15] sim: stop the lossy-heal test racing its own subscription link_recovers_after_lossy_heal subscribed to b's inbox after healing, but a subscription only sees tones delivered after it exists and the healed buffer reaches the inbox asynchronously. When a partition drops fewer than three, the window is short enough for a late arrival to displace an "ok", so the assertion failed under load and passed alone. Subscribe first. That makes the arrival order fixed, and lets the assertion cover the lossy heal as well as the recovery: two buffered tones survive, heal's own wane-sync sits between them and the new marks, and the three "ok" marks follow in order. --- sim/tests/lossy_link.rs | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/sim/tests/lossy_link.rs b/sim/tests/lossy_link.rs index 8cf4b144..9d1cac19 100644 --- a/sim/tests/lossy_link.rs +++ b/sim/tests/lossy_link.rs @@ -142,13 +142,23 @@ async fn link_recovers_after_lossy_heal() { sim.partition(a, b).unwrap(); sim.send_marks(a, b, "cut", 4).await.unwrap(); sim.impair_dir(a, b, LinkFaults::default().drop_next(2)).unwrap(); + + // Subscribe before healing. A subscription only sees tones + // delivered after it exists, and the healed tones reach the inbox + // asynchronously — subscribing afterwards races them, and a + // partition that drops fewer than 3 makes the window short enough + // for a late arrival to displace an "ok". This ordering also lets + // the assertion cover the lossy heal, not just the recovery. + let mut rx = sim.humd_peer_sub(b).unwrap(); sim.heal(a, b).await.unwrap(); sim.heal_link_faults(a, b).unwrap(); - let mut rx = sim.humd_peer_sub(b).unwrap(); sim.send_marks(a, b, "ok", 3).await.unwrap(); - let rid = Sim::collect_rids(&mut rx, 3, WINDOW).await; - assert_eq!(rid, ["ok-0","ok-1","ok-2"]); + // Six tones: the two buffered ones the lossy heal spared, the + // wane-sync heal emits on its way out, then the three new marks. + let got = Sim::collect_rids(&mut rx, 6, WINDOW).await; + assert_eq!(&got[..2], ["cut-2", "cut-3"], "heal dropped the first two"); + assert_eq!(&got[3..], ["ok-0", "ok-1", "ok-2"], "and the link carries traffic again"); } // ── directional ──────────────────────────────────────────────────── From 7c45d2fb54ae6da1f35759a137c4230bc6523522 Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 21:40:00 +0500 Subject: [PATCH 09/15] delivery: mint message identity at the sender, and enforce dusk MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `rid` is not a message identity, so the answer to the open "should we dedup on rid" question is no, and the reason is in WIRE.md: `chi:"echo"` is the ack *for* an rid, so a request and its response share one, and a receiver deduping on rid would drop every response as a duplicate of its request. rids are also explicitly per-originator and format- agnostic, and reference clients mint per-session values (`p-`) that recur across restarts. The gossip id was therefore neither. mint_msg_id hashed (topic, rid, from, payload) — content addressing, which cannot tell "delivered twice" from "sent twice", and repeats are ordinary for the tones gossip exists to carry: a heartbeat, a standing overload alert, a retry. Worse, publish marks its own id seen before sending, so a repeat that hashed alike was dropped at the origin and never left. Since the rid was `gossip-{topic}-{now_ms()}`, the reachable case was two identical publishes inside one millisecond. The id is now assigned by the originator, once per publish, as `{origin6}-{ms:x}-{seq:x}`. A per-process counter makes every publish distinct whatever the clock or the content does, and the origin prefix keeps two humds from colliding. The wire field name is unchanged, and the existing canary test — which builds one tone and sends it twice — still dedups, because the dedup happens on the id in the tone, not on how it was minted. Also enforce `dusk`, the other half. A tone arriving past its own dusk is dropped rather than delivered or re-fanned, and counted in `expired_dusk()` so a scenario can assert on it rather than infer from absence. Checked after the liveness stamp, because an expired tone still proves the link is alive. No dusk means no expiry — the field is optional. This also bounds what the seen-set must remember: an id only has to outlive the window in which a duplicate could still be in flight. Unicast `route()` is deliberately left without dedup. There is no sound key for it — the only candidate, rid, repeats by design — and inventing one would drop correlated responses. sim/tests/delivery.rs pins the line the old id could not draw: an identical payload published twice arrives twice, and one publish duplicated by the link arrives once. The same bytes; only the sender knows which it meant. --- WIRE.md | 41 ++++++++++++++++++- ensemble/src/gossip.rs | 60 +++++++++++++++------------- ensemble/src/lib.rs | 57 ++++++++++++++++++++++++-- ensemble/tests/gossip_integration.rs | 2 +- sim/src/lib.rs | 39 ++++++++++++++++++ thrum-clients/go/thrum/chi.go | 2 +- thrum-clients/python/thrum/chi.py | 2 +- thrum-clients/ts/chi.ts | 2 +- thrum-core/src/chi.rs | 17 +++++--- 9 files changed, 178 insertions(+), 44 deletions(-) diff --git a/WIRE.md b/WIRE.md index 412bc94a..6b83174b 100644 --- a/WIRE.md +++ b/WIRE.md @@ -42,14 +42,14 @@ Every tone is a JSON object with these top-level fields: | key | type | required | meaning | |---|---|---|---| | `chi` | string | **yes** | Tone discriminator. Must be one of the kebab-case values in the chi registry below. | -| `rid` | string | **yes** | Request id. Echoed in correlated responses (e.g. `chi:"echo"`). Format-agnostic; reference clients use `"{base36-ms-timestamp}-{base36-counter}"`. | +| `rid` | string | **yes** | Request id. Echoed in correlated responses (e.g. `chi:"echo"`). Format-agnostic; reference clients use `"{base36-ms-timestamp}-{base36-counter}"`. **Not a dedup key** — see [Identity](#identity). | | `sid` | string | situational | Session id. Required for `prompt`, `chunk`, `finish`, `tool-call`, etc. Picked by the originator. | | `from` | string | situational | Sender identity. `HumdId` hex when crossing humds, bee name when on a local socket. | | `to` | string | situational | Destination identity. `HumdId` hex for ensemble-routed tones; absent for local-only. | | `sigil` | string | optional | 12-char content hash, see [Helpers](#helpers). Stable across reconnects for the same (nest, sid). | | `wane` | integer | optional | Lamport clock per sigil — see [WaneTracker](#wanetracker). | | `sentAt` | integer | optional | Wall-clock ms at send time. UTC. | -| `dusk` | integer | optional | Absolute ms expiry. If `now > dusk`, the receiver MAY drop. | +| `dusk` | integer | optional | Absolute ms expiry. humd drops a tone arriving with `now > dusk`, and counts it — see [Identity](#identity). | | `ext` | object | optional | Per-bee extension bag. Key it by your bee name; ignore other keys. | Beyond the envelope, **each chi defines its own body fields**. A @@ -242,6 +242,43 @@ Monotonic correlation id. Counter is per-process and starts at 0. Format-agnostic on receive — only the originator's correlation logic cares about the exact format. +### Identity + +`rid` is correlation, not identity. Three things follow, and all three +are load-bearing: + +- A request and its correlated response share one `rid`. `chi:"echo"` + is the ack *for* an `rid`, so a receiver that deduped on `rid` would + drop every response as a duplicate of its request. +- `rid` is explicitly format-agnostic and per-originator, so two humds + can mint the same value. Reference clients mint per-session values + (`p-`, `prompt-`) that recur across restarts. +- A sender that means to publish the same content twice is not + transmitting a duplicate. Repeats are ordinary for the tones gossip + carries — a heartbeat, a standing overload alert, a retry. + +So a message that needs at-most-once delivery carries its own +originator-assigned id, minted once per publish and never reused: + +| chi | id field | who mints it | +|---|---|---| +| `gossip-publish` | `msg_id` | originator, once per publish | + +Receivers dedup on that field and never mint one themselves. The +reference `Ensemble::publish` mints `"{origin6}-{ms:x}-{seq:x}"`: a +per-process counter makes each publish distinct regardless of clock or +content, and the origin prefix keeps two humds from colliding. + +Content-addressing an id (`sha256(topic:rid:from:payload)`) cannot +work here. It conflates "delivered twice" with "sent twice", and since +a sender typically marks its own publish seen before sending, a repeat +that hashed alike was dropped before it ever left the origin. + +`dusk` bounds the window a duplicate can be recognised in at all: a +tone past its `dusk` is dropped on arrival rather than delivered, so +the seen-set only has to remember ids for as long as they can still be +legitimately in flight. + ### `WaneTracker` ``` diff --git a/ensemble/src/gossip.rs b/ensemble/src/gossip.rs index c541c419..34bf4762 100644 --- a/ensemble/src/gossip.rs +++ b/ensemble/src/gossip.rs @@ -36,7 +36,6 @@ use std::sync::Arc; use lru::LruCache; use parking_lot::Mutex; use serde_json::Value; -use sha2::{Digest, Sha256}; use tokio::sync::broadcast; use crate::{Hid, Tone}; @@ -117,25 +116,24 @@ impl GossipState { } } -/// Mint the canonical `msg_id` for a gossip publish: -/// `sha256("{topic}:{rid}:{from}:{payload}")[..16]` as 32 hex chars. -/// `payload` is serialized via `serde_json` — not strictly canonical -/// across implementations, but stable within one Rust ensemble (the -/// same input produces the same output). Re-running publish() with the -/// same (topic, rid, from, payload) yields the same id, which is what -/// the dedup test relies on. -pub fn mint_msg_id(topic: &str, rid: &str, from: &Hid, payload: &Value) -> String { - let payload_canonical = serde_json::to_string(payload).unwrap_or_default(); - let mut h = Sha256::new(); - h.update(topic.as_bytes()); - h.update(b":"); - h.update(rid.as_bytes()); - h.update(b":"); - h.update(from.to_hex().as_bytes()); - h.update(b":"); - h.update(payload_canonical.as_bytes()); - let digest = h.finalize(); - hex::encode(&digest[..16]) +/// Mint the wire `msg_id` for one publish: `{origin6}-{ms:x}-{seq:x}`. +/// +/// The originator assigns this, and it is deliberately not derived from +/// the payload. Content addressing cannot tell "the network delivered +/// this twice" from "we published this twice on purpose", and a repeat +/// is ordinary for the tones gossip exists to carry — a heartbeat, a +/// standing overload alert, a retry after a resync. Hashing the payload +/// swallows those as duplicates; only the sender can know which it meant. +/// +/// A per-process counter makes every publish distinct whatever the clock +/// or the content does, and the origin prefix keeps two humds from +/// minting the same id. Receivers dedup on this and never mint it +/// themselves. +pub fn mint_msg_id(from: &Hid) -> String { + static SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); + let seq = SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + let origin = &from.to_hex()[..12]; + format!("{origin}-{:x}-{seq:x}", crate::now_ms()) } /// Build a `chi:"gossip-publish"` tone with the given fields. Kept here @@ -176,14 +174,20 @@ mod tests { use serde_json::json; #[test] - fn msg_id_is_stable_and_distinct() { + fn msg_id_is_unique_per_publish() { let from = Hid::random_humd(); - let a = mint_msg_id("t", "r1", &from, &json!({"x": 1})); - let b = mint_msg_id("t", "r1", &from, &json!({"x": 1})); - assert_eq!(a, b); - let c = mint_msg_id("t", "r1", &from, &json!({"x": 2})); - assert_ne!(a, c); - assert_eq!(a.len(), 32); // 16 bytes hex + let a = mint_msg_id(&from); + let b = mint_msg_id(&from); + assert_ne!(a, b, "two publishes must not share an id"); + } + + #[test] + fn msg_id_carries_its_origin() { + let a = mint_msg_id(&Hid::random_humd()); + let b = mint_msg_id(&Hid::random_humd()); + let origin = |id: &str| id.split('-').next().unwrap().to_string(); + assert_ne!(origin(&a), origin(&b), "two humds must not mint alike"); + assert_eq!(origin(&a).len(), 12); } #[test] @@ -198,7 +202,7 @@ mod tests { #[test] fn parse_gossip_pulls_fields() { let from = Hid::random_humd(); - let id = mint_msg_id("topic", "r", &from, &json!(1)); + let id = mint_msg_id(&from); let t = gossip_tone("topic", "r", &from, json!(1), &id); let p = parse_gossip(&t).unwrap(); assert_eq!(p.topic, "topic"); diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index 0c8ed6f6..965890b5 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -811,6 +811,35 @@ pub struct Ensemble { /// regardless of mode: a present pubkey that fails to verify is /// hostile, not legacy. strict_auth: bool, + /// Tones dropped for arriving past their own `dusk`. Counts what the + /// expiry rule actually caught, so a scenario can assert on it + /// instead of inferring from what did arrive. + expired_dusk: Arc, +} + +/// True if `tone` is dead on arrival: it carries a `dusk` that has +/// already passed, so it must not be dispatched or re-fanned. Counts +/// the drop. A tone with no `dusk` never expires — the field is +/// optional in the envelope. +/// +/// Each node applies the deadline it was handed, at its own edge, once. +/// That bounds how long a seen-set has to remember an id: a message +/// past its `dusk` is gone rather than delivered, so an id only has to +/// outlive the window in which a duplicate could still be in flight. +fn drop_if_dusk(tone: &Tone, expired: &AtomicU64) -> bool { + let past = tone + .get("dusk") + .and_then(|v| v.as_i64()) + .is_some_and(|dusk| now_ms() > dusk); + if past { + expired.fetch_add(1, Ordering::SeqCst); + tracing::debug!( + target: "ensemble", + chi = tone.get("chi").and_then(|v| v.as_str()).unwrap_or(""), + "tone.dusk: dropped on arrival" + ); + } + past } /// The local fan-out point for tones arriving from peers. Cloned into @@ -905,6 +934,7 @@ impl Ensemble { gossip: GossipState::new(), kad: KadState::new(me), strict_auth: false, + expired_dusk: Arc::new(AtomicU64::new(0)), } } @@ -962,6 +992,7 @@ impl Ensemble { let conn_for_drain = conn.clone(); let strict = self.strict_auth; let gossip = self.gossip.clone(); + let expired_dusk = self.expired_dusk.clone(); let kad = self.kad.clone(); let my_id = self.me; tokio::spawn(async move { @@ -1035,6 +1066,12 @@ impl Ensemble { if handle_liveness(&peers, &my_id, &id, &conn_for_drain, &tone).await { continue; } + // After the liveness stamp: an expired tone still + // proves the link is alive, it just has nothing + // left worth delivering or re-fanning. + if drop_if_dusk(&tone, &expired_dusk) { + continue; + } if tone.get("chi").and_then(|v| v.as_str()) == Some(GOSSIP_CHI) { if handle_gossip(&gossip, &peers, &id, &tone).await { continue; @@ -1114,6 +1151,7 @@ impl Ensemble { let conn_for_drain = conn.clone(); let strict = self.strict_auth; let gossip = self.gossip.clone(); + let expired_dusk = self.expired_dusk.clone(); let kad = self.kad.clone(); let my_id = self.me; tokio::spawn(async move { @@ -1162,6 +1200,12 @@ impl Ensemble { if handle_liveness(&peers, &my_id, &id, &conn_for_drain, &tone).await { continue; } + // After the liveness stamp: an expired tone still + // proves the link is alive, it just has nothing + // left worth delivering or re-fanning. + if drop_if_dusk(&tone, &expired_dusk) { + continue; + } if tone.get("chi").and_then(|v| v.as_str()) == Some(GOSSIP_CHI) { if handle_gossip(&gossip, &peers, &id, &tone).await { continue; @@ -1290,8 +1334,13 @@ impl Ensemble { /// value is a boot-order race that silently ate network traffic. pub fn inbox_dropped(&self) -> u64 { self.inbox.dropped() } - /// Publish a gossip message to every installed peer. Mints an - /// `msg_id` from `(topic, rid, me, payload)`, marks it seen locally + /// Tones dropped for arriving past their `dusk`. + pub fn expired_dusk(&self) -> u64 { + self.expired_dusk.load(Ordering::SeqCst) + } + + /// Publish a gossip message to every installed peer. Mints a fresh + /// `msg_id`, marks it seen locally /// (so we don't re-fan it on the inevitable echo), and sends a /// `chi:"gossip-publish"` tone over every `PeerConnection`. Local /// `subscribe_topic` subscribers do NOT see their own publish — that @@ -1304,8 +1353,8 @@ impl Ensemble { /// `route()` semantically; both share the `PeerConnection.send` /// wire but `publish` is mesh-wide and `route` is unicast. pub async fn publish(&self, topic: &str, payload: serde_json::Value) { - let rid = format!("gossip-{}-{}", topic, now_ms()); - let msg_id = mint_msg_id(topic, &rid, &self.me, &payload); + let msg_id = mint_msg_id(&self.me); + let rid = format!("gossip-{msg_id}"); // Mark seen locally so the next-hop echo (peer re-fans back to // us) is dropped at the drainer's seen check. self.gossip.note_seen(&msg_id); diff --git a/ensemble/tests/gossip_integration.rs b/ensemble/tests/gossip_integration.rs index 6003801e..f684141d 100644 --- a/ensemble/tests/gossip_integration.rs +++ b/ensemble/tests/gossip_integration.rs @@ -97,7 +97,7 @@ async fn gossip_percolates_one_hop_and_dedupes_duplicates() { while let Ok(Ok(_)) = timeout(Duration::from_millis(1), sub_c.recv()).await {} let canary_payload = json!({"event": "dedup-canary"}); - let canary_msg_id = mint_msg_id("test-topic", "rid-canary", &x_id, &canary_payload); + let canary_msg_id = mint_msg_id(&x_id); let canary_tone = gossip_tone( "test-topic", "rid-canary", diff --git a/sim/src/lib.rs b/sim/src/lib.rs index f5f6af80..cc69f758 100644 --- a/sim/src/lib.rs +++ b/sim/src/lib.rs @@ -1033,6 +1033,45 @@ impl Sim { Ok(cid) } + /// Publish a gossip payload to a mesh topic from one of the sim's + /// humds. + pub async fn publish( + &self, + from: Hid, + topic: &str, + payload: serde_json::Value, + ) -> Result<()> { + let ens = self + .humds + .read() + .get(&from) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no humd {}", from.short()))?; + ens.ensemble.publish(topic, payload).await; + Ok(()) + } + + /// Subscribe to a gossip topic on one of the sim's humds. Broadcast + /// receivers only see what is published after they subscribe. + pub fn subscribe_topic( + &self, + humd: Hid, + topic: &str, + ) -> Result> { + let ens = self + .humds + .read() + .get(&humd) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no humd {}", humd.short()))?; + Ok(ens.ensemble.subscribe_topic(topic)) + } + + /// Tones this ensemble dropped for arriving past their `dusk`. + pub fn expired_dusk(&self, humd: Hid) -> u64 { + self.humds.read().get(&humd).map_or(0, |h| h.ensemble.expired_dusk()) + } + /// Route `n` tones `from` → `to`, rid-tagged `tag-`, in order. pub async fn send_marks(&self, from: Hid, to: Hid, tag: &str, n: usize) -> Result<()> { for i in 0..n { diff --git a/thrum-clients/go/thrum/chi.go b/thrum-clients/go/thrum/chi.go index 2a9a50b5..b935abe0 100644 --- a/thrum-clients/go/thrum/chi.go +++ b/thrum-clients/go/thrum/chi.go @@ -74,7 +74,7 @@ const ( ChiDetach Chi = "detach" // reconcile WaneTracker after a partition heal — `{ from, snapshot }` where `snapshot` is a map of sigil → wane value. The receiver merges each entry by taking the max of local and remote (wane is a Lamport clock; max is convergent). No reply is required — both sides emit their snapshot on heal, so each is informed exactly once. ChiWaneSync Chi = "wane-sync" - // ensemble-wide gossip pub-sub message — `{ topic, payload, from, msg_id }`. Fan-out broadcast above the Transport seam: every PeerConnection is a gossip neighbor. The receiver dedups on `msg_id` (sha256("topic:rid:from:payload")[..16]) using a bounded LRU, dispatches to per-topic subscribers, and re-fans the tone to every OTHER installed peer so the message percolates across the mesh. Distinct from unicast `route` (which targets ONE humd) — gossip is mesh-wide announcements: hum relocation, humd overload, drone alerts. + // ensemble-wide gossip pub-sub message — `{ topic, payload, from, msg_id }`. Fan-out broadcast above the Transport seam: every PeerConnection is a gossip neighbor. The receiver dedups on `msg_id` using a bounded LRU, dispatches to per-topic subscribers, and re-fans the tone to every OTHER installed peer so the message percolates across the mesh. Distinct from unicast `route` (which targets ONE humd) — gossip is mesh-wide announcements: hum relocation, humd overload, drone alerts. `msg_id` is assigned by the ORIGINATOR, once per publish, and is not derived from `rid` or the payload. A repeat of the same content is a new message; only a byte-identical retransmission is a duplicate, and only the sender can tell those apart. ChiGossipPublish Chi = "gossip-publish" // Kademlia DHT FIND_NODE query — `{ query_id, target: , from: }`. The receiver answers with `kad-find-node-resp` carrying up to K HumdAddrs from its routing table closest in XOR distance to `target`. Sent during `Ensemble::kad_find` iterative lookups when a peer's HumdAddr isn't already known locally. ChiKadFindNode Chi = "kad-find-node" diff --git a/thrum-clients/python/thrum/chi.py b/thrum-clients/python/thrum/chi.py index e457f978..ed9eee5a 100644 --- a/thrum-clients/python/thrum/chi.py +++ b/thrum-clients/python/thrum/chi.py @@ -71,7 +71,7 @@ class Chi: DETACH: str = "detach" # reconcile WaneTracker after a partition heal — `{ from, snapshot }` where `snapshot` is a map of sigil → wane value. The receiver merges each entry by taking the max of local and remote (wane is a Lamport clock; max is convergent). No reply is required — both sides emit their snapshot on heal, so each is informed exactly once. WANE_SYNC: str = "wane-sync" - # ensemble-wide gossip pub-sub message — `{ topic, payload, from, msg_id }`. Fan-out broadcast above the Transport seam: every PeerConnection is a gossip neighbor. The receiver dedups on `msg_id` (sha256("topic:rid:from:payload")[..16]) using a bounded LRU, dispatches to per-topic subscribers, and re-fans the tone to every OTHER installed peer so the message percolates across the mesh. Distinct from unicast `route` (which targets ONE humd) — gossip is mesh-wide announcements: hum relocation, humd overload, drone alerts. + # ensemble-wide gossip pub-sub message — `{ topic, payload, from, msg_id }`. Fan-out broadcast above the Transport seam: every PeerConnection is a gossip neighbor. The receiver dedups on `msg_id` using a bounded LRU, dispatches to per-topic subscribers, and re-fans the tone to every OTHER installed peer so the message percolates across the mesh. Distinct from unicast `route` (which targets ONE humd) — gossip is mesh-wide announcements: hum relocation, humd overload, drone alerts. `msg_id` is assigned by the ORIGINATOR, once per publish, and is not derived from `rid` or the payload. A repeat of the same content is a new message; only a byte-identical retransmission is a duplicate, and only the sender can tell those apart. GOSSIP_PUBLISH: str = "gossip-publish" # Kademlia DHT FIND_NODE query — `{ query_id, target: , from: }`. The receiver answers with `kad-find-node-resp` carrying up to K HumdAddrs from its routing table closest in XOR distance to `target`. Sent during `Ensemble::kad_find` iterative lookups when a peer's HumdAddr isn't already known locally. KAD_FIND_NODE: str = "kad-find-node" diff --git a/thrum-clients/ts/chi.ts b/thrum-clients/ts/chi.ts index e7f18a84..460aead3 100644 --- a/thrum-clients/ts/chi.ts +++ b/thrum-clients/ts/chi.ts @@ -69,7 +69,7 @@ export const Chi = { detach: "detach", /** reconcile WaneTracker after a partition heal — `{ from, snapshot }` where `snapshot` is a map of sigil → wane value. The receiver merges each entry by taking the max of local and remote (wane is a Lamport clock; max is convergent). No reply is required — both sides emit their snapshot on heal, so each is informed exactly once. */ waneSync: "wane-sync", - /** ensemble-wide gossip pub-sub message — `{ topic, payload, from, msg_id }`. Fan-out broadcast above the Transport seam: every PeerConnection is a gossip neighbor. The receiver dedups on `msg_id` (sha256("topic:rid:from:payload")[..16]) using a bounded LRU, dispatches to per-topic subscribers, and re-fans the tone to every OTHER installed peer so the message percolates across the mesh. Distinct from unicast `route` (which targets ONE humd) — gossip is mesh-wide announcements: hum relocation, humd overload, drone alerts. */ + /** ensemble-wide gossip pub-sub message — `{ topic, payload, from, msg_id }`. Fan-out broadcast above the Transport seam: every PeerConnection is a gossip neighbor. The receiver dedups on `msg_id` using a bounded LRU, dispatches to per-topic subscribers, and re-fans the tone to every OTHER installed peer so the message percolates across the mesh. Distinct from unicast `route` (which targets ONE humd) — gossip is mesh-wide announcements: hum relocation, humd overload, drone alerts. `msg_id` is assigned by the ORIGINATOR, once per publish, and is not derived from `rid` or the payload. A repeat of the same content is a new message; only a byte-identical retransmission is a duplicate, and only the sender can tell those apart. */ gossipPublish: "gossip-publish", /** Kademlia DHT FIND_NODE query — `{ query_id, target: , from: }`. The receiver answers with `kad-find-node-resp` carrying up to K HumdAddrs from its routing table closest in XOR distance to `target`. Sent during `Ensemble::kad_find` iterative lookups when a peer's HumdAddr isn't already known locally. */ kadFindNode: "kad-find-node", diff --git a/thrum-core/src/chi.rs b/thrum-core/src/chi.rs index 225b0203..a330c4b7 100644 --- a/thrum-core/src/chi.rs +++ b/thrum-core/src/chi.rs @@ -100,12 +100,17 @@ pub enum Chi { /// ensemble-wide gossip pub-sub message — /// `{ topic, payload, from, msg_id }`. Fan-out broadcast above the /// Transport seam: every PeerConnection is a gossip neighbor. The - /// receiver dedups on `msg_id` (sha256("topic:rid:from:payload")[..16]) - /// using a bounded LRU, dispatches to per-topic subscribers, and - /// re-fans the tone to every OTHER installed peer so the message - /// percolates across the mesh. Distinct from unicast `route` (which - /// targets ONE humd) — gossip is mesh-wide announcements: hum - /// relocation, humd overload, drone alerts. + /// receiver dedups on `msg_id` using a bounded LRU, dispatches to + /// per-topic subscribers, and re-fans the tone to every OTHER + /// installed peer so the message percolates across the mesh. + /// Distinct from unicast `route` (which targets ONE humd) — gossip + /// is mesh-wide announcements: hum relocation, humd overload, drone + /// alerts. + /// + /// `msg_id` is assigned by the ORIGINATOR, once per publish, and is + /// not derived from `rid` or the payload. A repeat of the same + /// content is a new message; only a byte-identical retransmission + /// is a duplicate, and only the sender can tell those apart. GossipPublish, /// Kademlia DHT FIND_NODE query — /// `{ query_id, target: , from: }`. The From fbaf776c5f5384f21375f5c84af5deb1b3b2db2f Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 21:40:04 +0500 Subject: [PATCH 10/15] humd: stop config/key tests mutating process-global env peers and identity both set XDG_CONFIG_HOME / XDG_STATE_HOME and removed them on the way out. Both variables are process-global and the humd test binary runs its tests in parallel threads, so a test setting one pointed every other test in the process at its own fixture, and the remove_var left the rest reading whatever came next. peers.rs failed this way on a full workspace run and passed alone. Split load_from(path) and load_or_mint_key_at(path) out of the env-derived defaults and have the tests name their own files. The production entry points are unchanged. hum-identity already serializes its own env-touching tests behind a lock and shares a binary with no other path-reading test, so it stays as it is. --- humd/src/identity.rs | 29 ++++--- humd/src/peers.rs | 42 ++++++---- sim/tests/delivery.rs | 175 ++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 221 insertions(+), 25 deletions(-) create mode 100644 sim/tests/delivery.rs diff --git a/humd/src/identity.rs b/humd/src/identity.rs index d8735009..b163746b 100644 --- a/humd/src/identity.rs +++ b/humd/src/identity.rs @@ -34,9 +34,16 @@ pub fn key_path() -> PathBuf { /// /// Bytes on disk are the raw 32-byte Ed25519 secret seed. pub fn load_or_mint_key() -> Result { - let path = key_path(); + load_or_mint_key_at(&key_path()) +} + +/// As [`load_or_mint_key`], against an explicit path. Split out so a +/// test can name its own file instead of setting `XDG_STATE_HOME`, +/// which is process-global and would point every other test in the +/// binary at this test's key. +pub fn load_or_mint_key_at(path: &std::path::Path) -> Result { if path.exists() { - let bytes = fs::read(&path) + let bytes = fs::read(path) .with_context(|| format!("read humd key {}", path.display()))?; if bytes.len() != 32 { return Err(anyhow!( @@ -58,7 +65,7 @@ pub fn load_or_mint_key() -> Result { rand::thread_rng().fill_bytes(&mut seed); let signing = SigningKey::from_bytes(&seed); let key = HumdKey(signing); - persist_key(&path, &seed)?; + persist_key(path, &seed)?; info!(path = %path.display(), humd_id = %key.hid().short(), "identity.minted"); Ok(key) } @@ -132,32 +139,34 @@ fn persist_key(path: &std::path::Path, seed: &[u8; 32]) -> Result<()> { #[cfg(test)] mod tests { use super::*; + use std::path::Path; use tempfile::TempDir; /// Mint a key, drop it, reload from the same path — same Hid. /// Also checks file perms are 0o600. + /// + /// Names its own file rather than setting `XDG_STATE_HOME`: that + /// variable is process-global, and these tests share a binary with + /// others that read config and state paths. #[test] fn round_trip_through_tempdir() { let tmp = TempDir::new().unwrap(); - unsafe { std::env::set_var("XDG_STATE_HOME", tmp.path()) }; + let path: &Path = &tmp.path().join("humd.key"); - let first = load_or_mint_key().expect("mint"); + let first = load_or_mint_key_at(path).expect("mint"); let id1 = first.hid(); - let path = key_path(); assert!(path.exists(), "key file persisted"); // Permissions check. #[cfg(unix)] { use std::os::unix::fs::PermissionsExt; - let mode = fs::metadata(&path).unwrap().permissions().mode() & 0o777; + let mode = fs::metadata(path).unwrap().permissions().mode() & 0o777; assert_eq!(mode, 0o600, "key file must be 0o600"); } // Reload — same identity. - let second = load_or_mint_key().expect("reload"); + let second = load_or_mint_key_at(path).expect("reload"); assert_eq!(id1, second.hid(), "humd_id stable across reloads"); - - unsafe { std::env::remove_var("XDG_STATE_HOME") }; } } diff --git a/humd/src/peers.rs b/humd/src/peers.rs index 60e8be50..3040f8c1 100644 --- a/humd/src/peers.rs +++ b/humd/src/peers.rs @@ -62,8 +62,15 @@ pub fn peers_path() -> PathBuf { /// (warn). Malformed rows inside `peers[]` → skipped (warn), good rows /// kept. pub(crate) fn load() -> Vec { - let path = peers_path(); - let raw = match std::fs::read_to_string(&path) { + load_from(&peers_path()) +} + +/// As [`load`], against an explicit path. Split out so a test can name +/// its own file: `XDG_CONFIG_HOME` is process-global, and a test that +/// sets it races every other test in the same binary that reads a +/// config path. +pub(crate) fn load_from(path: &std::path::Path) -> Vec { + let raw = match std::fs::read_to_string(path) { Ok(s) => s, Err(e) if e.kind() == std::io::ErrorKind::NotFound => { trace!(path = %path.display(), "peers.missing"); @@ -107,15 +114,25 @@ fn parse_humd_id(s: &str) -> Option { #[cfg(test)] mod tests { use super::*; + use std::path::PathBuf; use tempfile::TempDir; + /// These tests name their own file instead of setting + /// `XDG_CONFIG_HOME`. That variable is process-global, and the test + /// binary runs these in parallel threads: setting it here pointed + /// every other test in the process at this fixture, and the + /// `remove_var` on the way out left the rest reading whatever + /// happened to be next. + fn fixture() -> (TempDir, PathBuf) { + let tmp = TempDir::new().unwrap(); + let path = tmp.path().join("peers.json"); + (tmp, path) + } + /// Fixture file with two good entries + one malformed → 2 loaded. #[test] fn load_parses_fixture_and_skips_bad_rows() { - let tmp = TempDir::new().unwrap(); - unsafe { std::env::set_var("XDG_CONFIG_HOME", tmp.path()) }; - std::fs::create_dir_all(hum_paths::config_dir()).unwrap(); - + let (_tmp, path) = fixture(); let good_a = "a".repeat(64); let good_b = "b".repeat(64); let bad = "nope"; @@ -128,23 +145,18 @@ mod tests { ] }}"# ); - std::fs::write(hum_paths::peers_json(), body).unwrap(); + std::fs::write(&path, body).unwrap(); - let loaded = load(); + let loaded = load_from(&path); assert_eq!(loaded.len(), 2, "bad row dropped"); assert_eq!(loaded[0].hints, vec!["tcp:host-a:9000".to_string()]); assert_eq!(loaded[1].hints.len(), 2); - - unsafe { std::env::remove_var("XDG_CONFIG_HOME") }; } /// Missing file returns empty without error. #[test] fn load_missing_file_is_empty() { - let tmp = TempDir::new().unwrap(); - unsafe { std::env::set_var("XDG_CONFIG_HOME", tmp.path()) }; - let loaded = load(); - assert!(loaded.is_empty()); - unsafe { std::env::remove_var("XDG_CONFIG_HOME") }; + let (_tmp, path) = fixture(); + assert!(load_from(&path).is_empty()); } } diff --git a/sim/tests/delivery.rs b/sim/tests/delivery.rs new file mode 100644 index 00000000..ac0c64e2 --- /dev/null +++ b/sim/tests/delivery.rs @@ -0,0 +1,175 @@ +//! delivery — at-most-once, expiry, and the line between them. +//! +//! At-most-once is a property of a *message identity*, and getting the +//! identity wrong breaks the feature it is meant to provide. Two cases +//! look alike on the wire and must not be conflated: +//! +//! - the network delivered one message twice → deliver once +//! - the originator sent the same content twice → deliver twice +//! +//! `rid` cannot tell them apart, so it is not the identity. See +//! `mint_msg_id`: the originator assigns a fresh id per publish. + +use std::time::Duration; + +use serde_json::json; +use sim::Sim; +use tokio::time::timeout; + +const WINDOW: Duration = Duration::from_millis(750); + +async fn trio() -> (Sim, ensemble::Hid, ensemble::Hid, ensemble::Hid) { + // A publishes, B is the middle hop, C subscribes. The repeat and + // the duplicate must both survive one hop of re-fan to be judged at + // all — a bug that drops them at the originator would look correct. + let sim = Sim::new(); + let a = ensemble::Hid::random_humd(); + let b = ensemble::Hid::random_humd(); + let c = ensemble::Hid::random_humd(); + for id in [a, b, c] { + sim.spawn_humd(id).await; + sim.await_ready(id).await.expect("ready"); + } + sim.wire(a, b).expect("wire a-b"); + sim.wire(b, c).expect("wire b-c"); + sim.await_handshake(a, b).await.expect("a-b"); + sim.await_handshake(b, c).await.expect("b-c"); + (sim, a, b, c) +} + +#[tokio::test(flavor = "multi_thread")] +async fn a_repeated_publish_is_delivered_twice() { + let (sim, a, _b, c) = trio().await; + let mut sub = sim.subscribe_topic(c, "alerts").expect("subscribe"); + + // A heartbeat, or a standing alert. The second one is not a + // duplicate of the first — the sender meant to send it. + let payload = json!({"event": "overloaded", "level": 3}); + sim.publish(a, "alerts", payload.clone()).await.expect("publish"); + sim.publish(a, "alerts", payload.clone()).await.expect("publish"); + + for i in 0..2 { + let got = timeout(WINDOW, sub.recv()) + .await + .unwrap_or_else(|_| panic!("publish {i} never arrived")) + .expect("topic channel closed"); + assert_eq!(got, payload); + } +} + +/// The other half, and the one at-most-once actually promises: one +/// publish, duplicated by the link, delivered once. The repeat above +/// and this duplicate are the same bytes on the wire — only the sender +/// knows which one it meant. +#[tokio::test(flavor = "multi_thread")] +async fn a_duplicated_publish_is_delivered_once() { + let (sim, a, b, c) = trio().await; + let mut sub = sim.subscribe_topic(c, "alerts").expect("subscribe"); + + sim.impair_dir(b, c, ensemble::LinkFaults::default().dup_next(1)) + .expect("duplicate the next b->c tone"); + + sim.publish(a, "alerts", json!({"event": "overloaded", "level": 3})) + .await + .expect("publish"); + + let got = timeout(WINDOW, sub.recv()).await.expect("arrived").expect("open"); + assert_eq!(got["event"], "overloaded"); + + // The link really did send it twice — otherwise this test proves + // nothing about dedup. + let counters = sim.link_counters(b, c).expect("counters").0; + assert!(counters.duplicated >= 1, "the link did not duplicate: {counters:?}"); + + let extra = timeout(Duration::from_millis(200), sub.recv()).await; + assert!(extra.is_err(), "the duplicate was delivered too: {extra:?}"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn an_expired_tone_is_dropped_on_arrival() { + let sim = Sim::new(); + let a = ensemble::Hid::random_humd(); + let b = ensemble::Hid::random_humd(); + for id in [a, b] { + sim.spawn_humd(id).await; + sim.await_ready(id).await.expect("ready"); + } + sim.wire(a, b).expect("wire"); + sim.await_handshake(a, b).await.expect("a-b"); + + let mut rx = sim.humd_peer_sub(b).expect("subscribe"); + let past = now_ms() - 60_000; + sim.nestler_send_ordered( + a, + json!({"chi": "perf-mark", "rid": "stale-1", "to": b.to_hex(), "dusk": past}), + ) + .await + .expect("send"); + + // Nothing should arrive, and the drop must be counted rather than + // inferred from absence. + let got = timeout(Duration::from_millis(200), rx.recv()).await; + assert!(got.is_err(), "a tone past its dusk was delivered: {got:?}"); + assert_eq!( + sim.expired_dusk(b), + 1, + "the expiry rule should have counted it" + ); +} + +#[tokio::test(flavor = "multi_thread")] +async fn a_tone_within_its_dusk_is_delivered() { + let sim = Sim::new(); + let a = ensemble::Hid::random_humd(); + let b = ensemble::Hid::random_humd(); + for id in [a, b] { + sim.spawn_humd(id).await; + sim.await_ready(id).await.expect("ready"); + } + sim.wire(a, b).expect("wire"); + sim.await_handshake(a, b).await.expect("a-b"); + + let mut rx = sim.humd_peer_sub(b).expect("subscribe"); + sim.nestler_send_ordered( + a, + json!({ + "chi": "perf-mark", "rid": "live-1", "to": b.to_hex(), + "dusk": now_ms() + 60_000 + }), + ) + .await + .expect("send"); + + let got = timeout(WINDOW, rx.recv()).await.expect("arrived").expect("open"); + assert_eq!(got["rid"], "live-1"); + assert_eq!(sim.expired_dusk(b), 0); +} + +#[tokio::test(flavor = "multi_thread")] +async fn a_tone_with_no_dusk_never_expires() { + let sim = Sim::new(); + let a = ensemble::Hid::random_humd(); + let b = ensemble::Hid::random_humd(); + for id in [a, b] { + sim.spawn_humd(id).await; + sim.await_ready(id).await.expect("ready"); + } + sim.wire(a, b).expect("wire"); + sim.await_handshake(a, b).await.expect("a-b"); + + let mut rx = sim.humd_peer_sub(b).expect("subscribe"); + sim.nestler_send_ordered(a, json!({"chi": "perf-mark", "rid": "forever-1", "to": b.to_hex()})) + .await + .expect("send"); + + let got = timeout(WINDOW, rx.recv()).await.expect("arrived").expect("open"); + assert_eq!(got["rid"], "forever-1"); + assert_eq!(sim.expired_dusk(b), 0); +} + +fn now_ms() -> i64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .expect("clock") + .as_millis() as i64 +} From 359d8c395f1decf795404af8af5ea7a54e13911e Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 21:59:26 +0500 Subject: [PATCH 11/15] ensemble: at-most-once for unicast, keyed on mid not rid MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Closes the last open item from the delivery work. `route()` had no dedup because no sound key existed for it: rid, the only candidate, repeats by design. The key already existed — the originator-assigned id used for gossip — it just had no unicast spelling. `mid` is that spelling, an optional envelope field, and the resolution follows from the same distinction the gossip fix turned on: mid answers *which message is this*, rid answers *which conversation is it in*. A response therefore carries the request's rid and its own mid. Same rid, different mid, both delivered. A retry carries same rid, same mid, delivered once. The prior state is that sending the same body twice means two messages, and that stays true: a tone with no mid makes no at-most-once claim and is delivered every time it is sent. Absence is meaningful, and a receiver must not substitute rid for a missing mid. That substitution is the failure this is built to prevent, so it is now a test. a_response_echoing_the_request_rid_still_arrives sends a request and a response under one rid and requires both; mutated to dedup on rid it fails, and so does the no-mid test, because rid-dedup swallows exactly the sends the sender meant. Mutated the other way — mid check disabled — a_retransmitted_mid_is_delivered_once fails. Both directions of the mistake are caught rather than assumed away. Dusk moves into the new ensemble::delivery module alongside the mid seen-set, since identity and lifetime are the same question asked twice: may this tone be delivered, and has it already been. Dusk is checked first so a dead tone cannot occupy seen-set capacity and evict a live id. The two seen-sets stay separate on purpose — a mid records that this node already delivered a tone, a gossip msg_id also governs whether it is re-fanned, and sharing one set would let each evict the other's entries early. mid is a first-class Envelope field rather than left in the tone body, so the existing "envelope fields win" rule applies and a body mid cannot shadow it. The generated clients pick the field up from codegen. 8 delivery scenarios green; workspace 51/51 across three runs; clippy clean. --- WIRE.md | 30 +++- ensemble/src/delivery.rs | 211 +++++++++++++++++++++++++ ensemble/src/lib.rs | 54 +++---- sim/tests/delivery.rs | 122 ++++++++++---- thrum-clients/go/thrum/protocol.go | 1 + thrum-clients/python/thrum/protocol.py | 2 + thrum-clients/ts/protocol.ts | 1 + thrum-core/src/envelope.rs | 15 +- 8 files changed, 378 insertions(+), 58 deletions(-) create mode 100644 ensemble/src/delivery.rs diff --git a/WIRE.md b/WIRE.md index 6b83174b..71182996 100644 --- a/WIRE.md +++ b/WIRE.md @@ -50,6 +50,7 @@ Every tone is a JSON object with these top-level fields: | `wane` | integer | optional | Lamport clock per sigil — see [WaneTracker](#wanetracker). | | `sentAt` | integer | optional | Wall-clock ms at send time. UTC. | | `dusk` | integer | optional | Absolute ms expiry. humd drops a tone arriving with `now > dusk`, and counts it — see [Identity](#identity). | +| `mid` | string | optional | Originator-assigned message id. Presence is an at-most-once claim: a receiver that has already dispatched this `mid` drops the repeat. Distinct from `rid` — see [Identity](#identity). | | `ext` | object | optional | Per-bee extension bag. Key it by your bee name; ignore other keys. | Beyond the envelope, **each chi defines its own body fields**. A @@ -263,12 +264,37 @@ originator-assigned id, minted once per publish and never reused: | chi | id field | who mints it | |---|---|---| | `gossip-publish` | `msg_id` | originator, once per publish | +| any, when at-most-once is wanted | `mid` | originator, once per logical message | Receivers dedup on that field and never mint one themselves. The reference `Ensemble::publish` mints `"{origin6}-{ms:x}-{seq:x}"`: a per-process counter makes each publish distinct regardless of clock or content, and the origin prefix keeps two humds from colliding. +`mid` extends the same rule to unicast, and the distinction is the +whole design. `mid` answers *which message is this*; `rid` answers +*which conversation is it in*. A response therefore carries the +request's `rid` and its own `mid` — same rid, different mid, both +delivered. That is what a retry looks like too: same rid, same mid, +delivered once. + +`mid` is optional, and its absence is meaningful rather than a +shortcoming. A tone with no `mid` makes no at-most-once claim and is +delivered every time it is sent, so a sender may repeat a body as often +as it likes. Suppressing those would be the original mistake in another +guise — dropping a message the sender meant to send. A receiver must +not substitute `rid` for a missing `mid`. + +Each receiver enforces `mid` and `dusk` at its own edge, once per hop, +and keeps the two independent: + +- `dusk` is checked first, so a dead tone never occupies seen-set + capacity and cannot displace a live id. +- The `mid` seen-set is separate from the gossip `msg_id` seen-set. A + `mid` records that this node already *delivered* a tone; a `msg_id` + also governs whether it is *re-fanned*. Sharing one set would couple + two unrelated decisions and let one evict the other's entries early. + Content-addressing an id (`sha256(topic:rid:from:payload)`) cannot work here. It conflates "delivered twice" with "sent twice", and since a sender typically marks its own publish seen before sending, a repeat @@ -277,7 +303,9 @@ that hashed alike was dropped before it ever left the origin. `dusk` bounds the window a duplicate can be recognised in at all: a tone past its `dusk` is dropped on arrival rather than delivered, so the seen-set only has to remember ids for as long as they can still be -legitimately in flight. +legitimately in flight. A tone with no `dusk` can be re-fanned +arbitrarily late, so such ids are bounded by the seen-set cap instead +of by any deadline. ### `WaneTracker` diff --git a/ensemble/src/delivery.rs b/ensemble/src/delivery.rs new file mode 100644 index 00000000..44a57a92 --- /dev/null +++ b/ensemble/src/delivery.rs @@ -0,0 +1,211 @@ +//! delivery — when a tone arrives, and whether it arrives at all. +//! +//! Two independent rules, both applied per hop at the receiving edge: +//! +//! - **lifetime** (`dusk`): a tone past its own deadline is dead. It is +//! not delivered and not re-fanned. +//! - **at-most-once** (`mid`): a tone whose id this ensemble has already +//! dispatched is a duplicate. It is not delivered again. +//! +//! Both are optional per tone. A tone with neither is delivered as it +//! arrives, which is the default and what the plumbing tones rely on. +//! +//! `mid` is the originator's, minted once per logical message, and is +//! never `rid`. `rid` is correlation: a request and the response that +//! echoes its rid share one, so deduping on rid would drop every +//! response as a duplicate of its request. A response therefore carries +//! its *own* `mid` and the request's `rid` — the two are not redundant, +//! they answer different questions. See WIRE.md. + +use std::num::NonZeroUsize; +use std::sync::atomic::{AtomicU64, Ordering}; + +use lru::LruCache; +use parking_lot::Mutex; + +use crate::Tone; + +/// Bound on the per-ensemble `mid` seen-set. Sizing is a memory/idle- +/// time trade: the set only has to remember an id for as long as a +/// duplicate of it could still be in flight, which `dusk` bounds from +/// above. A tone with no `dusk` can be redelivered arbitrarily late by +/// a re-fan, so it is bounded only by this cap. +pub const DELIVERY_SEEN_CAP: usize = 4096; + +/// Ids of tones this ensemble has already dispatched. +pub struct DeliveryState { + seen: Mutex>, +} + +impl DeliveryState { + pub fn new() -> std::sync::Arc { + std::sync::Arc::new(Self { + seen: Mutex::new(LruCache::new( + NonZeroUsize::new(DELIVERY_SEEN_CAP).expect("seen cap > 0"), + )), + }) + } + + /// True if `mid` is new — i.e. the caller should dispatch it. Inserts + /// on every call, so a second observation returns false. + /// + /// Separate from the gossip seen-set on purpose: a `mid` says "this + /// node already delivered this tone", while a gossip `msg_id` also + /// governs whether the tone is re-fanned. Sharing one set would + /// couple two unrelated decisions and let one evict the other's + /// entries early. + pub fn note_mid(&self, mid: &str) -> bool { + let mut seen = self.seen.lock(); + if seen.contains(mid) { + seen.get(mid); + false + } else { + seen.put(mid.to_string(), ()); + true + } + } + + pub fn len(&self) -> usize { + self.seen.lock().len() + } + + pub fn is_empty(&self) -> bool { + self.len() == 0 + } +} + +/// True if `tone` is dead on arrival: it carries a `dusk` that has +/// already passed, so it must not be dispatched or re-fanned. Counts +/// the drop. A tone with no `dusk` never expires — the field is +/// optional in the envelope. +/// +/// Each node applies the deadline it was handed, at its own edge, once. +fn drop_if_dusk(tone: &Tone, expired: &AtomicU64) -> bool { + let past = tone + .get("dusk") + .and_then(|v| v.as_i64()) + .is_some_and(|dusk| crate::now_ms() > dusk); + if past { + expired.fetch_add(1, Ordering::SeqCst); + tracing::debug!( + target: "ensemble", + chi = tone.get("chi").and_then(|v| v.as_str()).unwrap_or(""), + "tone.dusk: dropped on arrival" + ); + } + past +} + +/// The `mid` on a tone, if it carries one. Absent or non-string means +/// "no at-most-once claim", not "malformed" — the tone is delivered. +pub fn mid_of(tone: &Tone) -> Option<&str> { + tone.get("mid").and_then(|v| v.as_str()) +} + +/// Apply both rules in order. Returns true if the tone must not be +/// dispatched or re-fanned. +/// +/// Dusk is checked first so a dead tone never occupies seen-set +/// capacity, and so an id that is already known to be expired cannot +/// evict a live one. +pub fn admit(tone: &Tone, delivery: &DeliveryState, expired_dusk: &AtomicU64) -> bool { + if drop_if_dusk(tone, expired_dusk) { + return false; + } + match mid_of(tone) { + Some(mid) if !delivery.note_mid(mid) => { + tracing::debug!( + target: "ensemble", + mid, + chi = tone.get("chi").and_then(|v| v.as_str()).unwrap_or(""), + "tone.mid: duplicate suppressed" + ); + false + } + _ => true, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + use std::sync::Arc; + + fn state() -> Arc { + DeliveryState::new() + } + + #[test] + fn a_mid_is_admitted_once() { + let d = state(); + assert!(d.note_mid("m1")); + assert!(!d.note_mid("m1")); + assert!(d.note_mid("m2")); + } + + #[test] + fn tones_without_a_mid_are_always_admitted() { + let d = state(); + let expired = AtomicU64::new(0); + let tone = json!({"chi": "prompt", "rid": "r-1"}); + assert!(admit(&tone, &d, &expired)); + assert!(admit(&tone, &d, &expired), "no mid means no claim to enforce"); + assert_eq!(d.len(), 0); + } + + /// A request and its response share an rid but not a mid. Deduping + /// on the rid would eat the response; deduping on the mid must not. + #[test] + fn a_response_echoing_a_request_rid_is_not_a_duplicate() { + let d = state(); + let expired = AtomicU64::new(0); + let request = json!({"chi": "prompt", "rid": "r-1", "mid": "m-req"}); + let response = json!({"chi": "chunk", "rid": "r-1", "mid": "m-resp"}); + assert!(admit(&request, &d, &expired)); + assert!(admit(&response, &d, &expired), "same rid, different mid"); + } + + #[test] + fn a_retransmitted_mid_is_suppressed() { + let d = state(); + let expired = AtomicU64::new(0); + let tone = json!({"chi": "chunk", "rid": "r-1", "mid": "m-1"}); + assert!(admit(&tone, &d, &expired)); + assert!(!admit(&tone, &d, &expired)); + assert_eq!(expired.load(Ordering::SeqCst), 0, "not a dusk drop"); + } + + #[test] + fn a_past_dusk_is_dropped_and_counted() { + let d = state(); + let expired = AtomicU64::new(0); + let tone = json!({"chi": "chunk", "rid": "r-1", "mid": "m-1", "dusk": crate::now_ms() - 1}); + assert!(!admit(&tone, &d, &expired)); + assert_eq!(expired.load(Ordering::SeqCst), 1); + assert_eq!(d.len(), 0, "a dead tone must not occupy seen capacity"); + } + + #[test] + fn a_future_dusk_is_delivered() { + let d = state(); + let expired = AtomicU64::new(0); + let tone = json!({"chi": "chunk", "rid": "r-1", "mid": "m-1", "dusk": crate::now_ms() + 60_000}); + assert!(admit(&tone, &d, &expired)); + assert_eq!(expired.load(Ordering::SeqCst), 0); + } + + /// Dusk is checked first, so a mid already known to be dead cannot + /// displace a live one from the bounded set. + #[test] + fn a_dead_mid_does_not_evict_a_live_one() { + let d = state(); + let expired = AtomicU64::new(0); + let live = json!({"chi": "chunk", "rid": "r", "mid": "live"}); + assert!(admit(&live, &d, &expired)); + let dead = json!({"chi": "chunk", "rid": "r", "mid": "dead", "dusk": crate::now_ms() - 1}); + assert!(!admit(&dead, &d, &expired)); + assert_eq!(d.len(), 1); + assert!(!admit(&live, &d, &expired), "the live mid is still remembered"); + } +} diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index 965890b5..fc91302b 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -54,6 +54,9 @@ pub use tls::{ pub mod iroh; pub use iroh::{IrohEndpoint, IrohTransport, IROH_ALPN}; +pub mod delivery; +pub use delivery::{DeliveryState, DELIVERY_SEEN_CAP}; + pub mod gossip; pub use gossip::{gossip_tone, mint_msg_id, GossipState, GOSSIP_CHI, GOSSIP_SEEN_CAP}; @@ -815,31 +818,8 @@ pub struct Ensemble { /// expiry rule actually caught, so a scenario can assert on it /// instead of inferring from what did arrive. expired_dusk: Arc, -} - -/// True if `tone` is dead on arrival: it carries a `dusk` that has -/// already passed, so it must not be dispatched or re-fanned. Counts -/// the drop. A tone with no `dusk` never expires — the field is -/// optional in the envelope. -/// -/// Each node applies the deadline it was handed, at its own edge, once. -/// That bounds how long a seen-set has to remember an id: a message -/// past its `dusk` is gone rather than delivered, so an id only has to -/// outlive the window in which a duplicate could still be in flight. -fn drop_if_dusk(tone: &Tone, expired: &AtomicU64) -> bool { - let past = tone - .get("dusk") - .and_then(|v| v.as_i64()) - .is_some_and(|dusk| now_ms() > dusk); - if past { - expired.fetch_add(1, Ordering::SeqCst); - tracing::debug!( - target: "ensemble", - chi = tone.get("chi").and_then(|v| v.as_str()).unwrap_or(""), - "tone.dusk: dropped on arrival" - ); - } - past + /// Mids already dispatched here, so a retransmit is delivered once. + delivery: Arc, } /// The local fan-out point for tones arriving from peers. Cloned into @@ -935,6 +915,7 @@ impl Ensemble { kad: KadState::new(me), strict_auth: false, expired_dusk: Arc::new(AtomicU64::new(0)), + delivery: DeliveryState::new(), } } @@ -993,6 +974,7 @@ impl Ensemble { let strict = self.strict_auth; let gossip = self.gossip.clone(); let expired_dusk = self.expired_dusk.clone(); + let delivery = self.delivery.clone(); let kad = self.kad.clone(); let my_id = self.me; tokio::spawn(async move { @@ -1068,8 +1050,12 @@ impl Ensemble { } // After the liveness stamp: an expired tone still // proves the link is alive, it just has nothing - // left worth delivering or re-fanning. - if drop_if_dusk(&tone, &expired_dusk) { + // left worth delivering or re-fanning. `admit` + // also drops a `mid` this ensemble already + // dispatched, so a retransmit is delivered once. + // Checked before gossip so a duplicate is not + // re-fanned either. + if !delivery::admit(&tone, &delivery, &expired_dusk) { continue; } if tone.get("chi").and_then(|v| v.as_str()) == Some(GOSSIP_CHI) { @@ -1152,6 +1138,7 @@ impl Ensemble { let strict = self.strict_auth; let gossip = self.gossip.clone(); let expired_dusk = self.expired_dusk.clone(); + let delivery = self.delivery.clone(); let kad = self.kad.clone(); let my_id = self.me; tokio::spawn(async move { @@ -1202,8 +1189,12 @@ impl Ensemble { } // After the liveness stamp: an expired tone still // proves the link is alive, it just has nothing - // left worth delivering or re-fanning. - if drop_if_dusk(&tone, &expired_dusk) { + // left worth delivering or re-fanning. `admit` + // also drops a `mid` this ensemble already + // dispatched, so a retransmit is delivered once. + // Checked before gossip so a duplicate is not + // re-fanned either. + if !delivery::admit(&tone, &delivery, &expired_dusk) { continue; } if tone.get("chi").and_then(|v| v.as_str()) == Some(GOSSIP_CHI) { @@ -1339,6 +1330,11 @@ impl Ensemble { self.expired_dusk.load(Ordering::SeqCst) } + /// Number of mids currently remembered for at-most-once delivery. + pub fn delivery_seen(&self) -> usize { + self.delivery.len() + } + /// Publish a gossip message to every installed peer. Mints a fresh /// `msg_id`, marks it seen locally /// (so we don't re-fan it on the inevitable echo), and sends a diff --git a/sim/tests/delivery.rs b/sim/tests/delivery.rs index ac0c64e2..df6681ec 100644 --- a/sim/tests/delivery.rs +++ b/sim/tests/delivery.rs @@ -9,6 +9,12 @@ //! //! `rid` cannot tell them apart, so it is not the identity. See //! `mint_msg_id`: the originator assigns a fresh id per publish. +//! +//! The same split governs unicast, where the identity is the optional +//! `mid`: a repeat of a `mid` is a retransmit and is delivered once, +//! while a repeat of a tone carrying no `mid` is two sends and is +//! delivered twice. A request and the response echoing its `rid` are +//! two messages with two mids and one rid, and both must arrive. use std::time::Duration; @@ -37,6 +43,20 @@ async fn trio() -> (Sim, ensemble::Hid, ensemble::Hid, ensemble::Hid) { (sim, a, b, c) } +/// Two humds wired A->B, with a subscription on B. +async fn pair() -> (Sim, ensemble::Hid, ensemble::Hid) { + let sim = Sim::new(); + let a = ensemble::Hid::random_humd(); + let b = ensemble::Hid::random_humd(); + for id in [a, b] { + sim.spawn_humd(id).await; + sim.await_ready(id).await.expect("ready"); + } + sim.wire(a, b).expect("wire"); + sim.await_handshake(a, b).await.expect("a-b"); + (sim, a, b) +} + #[tokio::test(flavor = "multi_thread")] async fn a_repeated_publish_is_delivered_twice() { let (sim, a, _b, c) = trio().await; @@ -87,15 +107,7 @@ async fn a_duplicated_publish_is_delivered_once() { #[tokio::test(flavor = "multi_thread")] async fn an_expired_tone_is_dropped_on_arrival() { - let sim = Sim::new(); - let a = ensemble::Hid::random_humd(); - let b = ensemble::Hid::random_humd(); - for id in [a, b] { - sim.spawn_humd(id).await; - sim.await_ready(id).await.expect("ready"); - } - sim.wire(a, b).expect("wire"); - sim.await_handshake(a, b).await.expect("a-b"); + let (sim, a, b) = pair().await; let mut rx = sim.humd_peer_sub(b).expect("subscribe"); let past = now_ms() - 60_000; @@ -119,15 +131,7 @@ async fn an_expired_tone_is_dropped_on_arrival() { #[tokio::test(flavor = "multi_thread")] async fn a_tone_within_its_dusk_is_delivered() { - let sim = Sim::new(); - let a = ensemble::Hid::random_humd(); - let b = ensemble::Hid::random_humd(); - for id in [a, b] { - sim.spawn_humd(id).await; - sim.await_ready(id).await.expect("ready"); - } - sim.wire(a, b).expect("wire"); - sim.await_handshake(a, b).await.expect("a-b"); + let (sim, a, b) = pair().await; let mut rx = sim.humd_peer_sub(b).expect("subscribe"); sim.nestler_send_ordered( @@ -147,15 +151,7 @@ async fn a_tone_within_its_dusk_is_delivered() { #[tokio::test(flavor = "multi_thread")] async fn a_tone_with_no_dusk_never_expires() { - let sim = Sim::new(); - let a = ensemble::Hid::random_humd(); - let b = ensemble::Hid::random_humd(); - for id in [a, b] { - sim.spawn_humd(id).await; - sim.await_ready(id).await.expect("ready"); - } - sim.wire(a, b).expect("wire"); - sim.await_handshake(a, b).await.expect("a-b"); + let (sim, a, b) = pair().await; let mut rx = sim.humd_peer_sub(b).expect("subscribe"); sim.nestler_send_ordered(a, json!({"chi": "perf-mark", "rid": "forever-1", "to": b.to_hex()})) @@ -167,6 +163,78 @@ async fn a_tone_with_no_dusk_never_expires() { assert_eq!(sim.expired_dusk(b), 0); } +/// At-most-once on unicast, the whole point of `mid`: the same id sent +/// twice arrives once. This is what deduping on `rid` would have +/// broken, since a retry reuses its rid. +#[tokio::test(flavor = "multi_thread")] +async fn a_retransmitted_mid_is_delivered_once() { + let (sim, a, b) = pair().await; + let mut rx = sim.humd_peer_sub(b).expect("subscribe"); + let tone = json!({ + "chi": "chunk", "rid": "r-1", "mid": "m-1", "to": b.to_hex(), "seq": 1 + }); + sim.nestler_send_ordered(a, tone.clone()).await.expect("send"); + sim.nestler_send_ordered(a, tone).await.expect("resend"); + + let got = timeout(WINDOW, rx.recv()).await.expect("arrived").expect("open"); + assert_eq!(got["mid"], "m-1"); + + let extra = timeout(Duration::from_millis(200), rx.recv()).await; + assert!(extra.is_err(), "the retransmit was delivered too: {extra:?}"); +} + +/// The other half: no `mid` means no at-most-once claim. The sender may +/// send the same body as many times as it likes, and each send is +/// delivered. Suppressing these would be the same mistake as deduping +/// on `rid` — dropping a message the sender meant to send. +#[tokio::test(flavor = "multi_thread")] +async fn a_tone_with_no_mid_is_delivered_every_time_it_is_sent() { + let (sim, a, b) = pair().await; + let mut rx = sim.humd_peer_sub(b).expect("subscribe"); + let tone = json!({"chi": "perf-mark", "rid": "r-1", "to": b.to_hex()}); + sim.nestler_send_ordered(a, tone.clone()).await.expect("send"); + sim.nestler_send_ordered(a, tone).await.expect("send again"); + + for i in 0..2 { + let got = timeout(WINDOW, rx.recv()) + .await + .unwrap_or_else(|_| panic!("send {i} never arrived")) + .expect("open"); + assert_eq!(got["rid"], "r-1"); + } +} + +/// The regression this whole rule exists to prevent. A response echoes +/// the request's `rid` — that is what makes it a response — but it is a +/// separate message and carries its own `mid`. Deduping on `rid` would +/// silently eat every response in the system, and nothing else would +/// look wrong. +#[tokio::test(flavor = "multi_thread")] +async fn a_response_echoing_the_request_rid_still_arrives() { + let (sim, a, b) = pair().await; + let mut rx = sim.humd_peer_sub(b).expect("subscribe"); + sim.nestler_send_ordered( + a, + json!({"chi": "prompt", "rid": "r-1", "mid": "m-req", "to": b.to_hex()}), + ) + .await + .expect("send request"); + sim.nestler_send_ordered( + a, + json!({"chi": "chunk", "rid": "r-1", "mid": "m-resp", "to": b.to_hex()}), + ) + .await + .expect("send response"); + + let mut mids = Vec::new(); + for _ in 0..2 { + let got = timeout(WINDOW, rx.recv()).await.expect("arrived").expect("open"); + assert_eq!(got["rid"], "r-1", "one rid, two messages"); + mids.push(got["mid"].as_str().unwrap().to_string()); + } + assert_eq!(mids, vec!["m-req", "m-resp"], "both messages delivered"); +} + fn now_ms() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) diff --git a/thrum-clients/go/thrum/protocol.go b/thrum-clients/go/thrum/protocol.go index c4414453..9dec908d 100644 --- a/thrum-clients/go/thrum/protocol.go +++ b/thrum-clients/go/thrum/protocol.go @@ -220,6 +220,7 @@ type Envelope struct { Wane *uint64 `json:"wane,omitempty"` SentAt *int64 `json:"sentAt,omitempty"` Dusk *int64 `json:"dusk,omitempty"` + Mid *string `json:"mid,omitempty"` Ext map[string]map[string]json.RawMessage `json:"ext,omitempty"` } diff --git a/thrum-clients/python/thrum/protocol.py b/thrum-clients/python/thrum/protocol.py index ab7c08d5..2272ac47 100644 --- a/thrum-clients/python/thrum/protocol.py +++ b/thrum-clients/python/thrum/protocol.py @@ -446,6 +446,7 @@ class Envelope: wane: Optional[int] = None sent_at: Optional[int] = None dusk: Optional[int] = None + mid: Optional[str] = None ext: Optional[dict[str, dict[str, Any]]] = None # rust field name -> wire key (single source in views.rs) @@ -459,6 +460,7 @@ class Envelope: "wane": "wane", "sent_at": "sentAt", "dusk": "dusk", + "mid": "mid", "ext": "ext", } diff --git a/thrum-clients/ts/protocol.ts b/thrum-clients/ts/protocol.ts index 4bad6a27..19deaaf5 100644 --- a/thrum-clients/ts/protocol.ts +++ b/thrum-clients/ts/protocol.ts @@ -219,6 +219,7 @@ export interface Envelope { wane?: number; sentAt?: number; dusk?: number; + mid?: string; ext?: Record>; } diff --git a/thrum-core/src/envelope.rs b/thrum-core/src/envelope.rs index 2646ff13..32753c63 100644 --- a/thrum-core/src/envelope.rs +++ b/thrum-core/src/envelope.rs @@ -38,6 +38,18 @@ pub struct Envelope { /// absolute ms expiry — past this, drop tone #[serde(skip_serializing_if = "Option::is_none")] pub dusk: Option, + /// originator-assigned message id — at-most-once delivery. + /// + /// Minted once per logical message by whoever originates it, and + /// never equal to `rid`. A response echoes the request's `rid` for + /// correlation but carries its own `mid`, because a request and its + /// response are two messages. A receiver that deduped on `rid` + /// would drop every response as a duplicate of its request. + /// + /// Optional: a tone without one makes no at-most-once claim and is + /// delivered as it arrives. + #[serde(skip_serializing_if = "Option::is_none")] + pub mid: Option, #[serde(skip_serializing_if = "Option::is_none")] pub ext: Option>>, @@ -56,6 +68,7 @@ impl Envelope { wane: None, sent_at: None, dusk: None, + mid: None, ext: None, } } @@ -112,7 +125,7 @@ impl<'de> Deserialize<'de> for Tone { let mut all: Map = Map::deserialize(de)?; // Pull envelope keys out, leave the rest as body. const ENV_KEYS: &[&str] = - &["chi", "rid", "from", "to", "sigil", "sid", "wane", "sentAt", "dusk", "ext"]; + &["chi", "rid", "from", "to", "sigil", "sid", "wane", "sentAt", "dusk", "mid", "ext"]; let mut env_map = Map::with_capacity(ENV_KEYS.len()); for k in ENV_KEYS { if let Some(v) = all.remove(*k) { From 523aea65f00e62f2c24d4a9e889a8ad86d94b9db Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 22:19:49 +0500 Subject: [PATCH 12/15] ensemble: read the producer path, and bound the seen-set by bytes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follows up the delivery work by reading the code instead of assuming it. Six of the open questions had answers, and two of them contradicted what I had written down. The seen-set was bounded in entries but not in bytes. `mid` is attacker-controlled and the TCP transport reads NDJSON with `BufReader::lines()`, which has no length cap, so 4096 entries of unbounded string is not a bound — a peer could put a megabyte in one field. `mid_key` now digests to 32 bytes, making the footprint `cap * 32` whatever arrives, with dedup still exact. `mid_prefix` keeps logs to a char-safe 12 chars so a 4 MB mid cannot land in a log line. On signing: there is none, per tone. `handshake_message` covers only `chi:"hello"`, and after that a connection is trusted wholesale, so `mid` is exactly as (un)protected as `chi`, `rid`, `to` and the payload. Nothing about mid is uniquely weak, but a mid being unbounded is real because a connected peer controls every field. `DeliveryState::with_cap` exists so capacity is reachable, and the_seen_set_evicts_at_its_cap now proves eviction at a cap of 2. While adding it I found `seen_set_dedups_within_capacity` in gossip never came close to GOSSIP_SEEN_CAP — it asserted three lookups and the name promised a bound. Renamed to `seen_set_dedups_repeats`, and gave the real thing its own `seen_set_evicts_at_its_cap`. `mint_msg_id` sliced `to_hex()[..12]` as bytes. Hid renders hex today so it cannot panic, but the slice is now char-wise with a fallback, since a panic in a publisher is not a good failure mode for a cosmetic prefix. The originator question resolves against humd, and the answer is that humd is a relay. Worker replies are forwarded with `to`/`from` rewritten and everything else — including any `mid` — carried through untouched, which is the only correct behaviour: a relay that minted its own mid would break dedup, because two relays of one message would disagree. The originator is the nest side, and the one in this repo is `hives/bp7`, which mints `p-{sid}` — the per-session rid that recurs across restarts, the exact shape that deduping on rid would break. It now also mints a `mid`, which is load-bearing there specifically because bp7 is a DTN store-and-forward bridge and re-running a prompt is not idempotent. bp7 is excluded from the workspace, so it builds standalone. Gossip could not set its own `dusk`, and still does not. A library TTL would silently drop announcements on a congested mesh, and how stale an alert may be is the publisher's call — a heartbeat wants seconds, a "worker moved" notice wants minutes. So `publish_with_dusk` makes the mechanism reachable and the default stays `None`. The new test also pins where expiry lands: at the FIRST hop, not the last, so a dead tone dies on entry to the mesh instead of being carried to every subscriber and dropped N times. My first draft asserted on the far hop and was wrong. WIRE.md's chi tables were contradicting `thrum-core/src/views.rs`, which calls itself the single source of truth for body shapes and is locked by a golden-bytes test. Ten rows named fields that do not exist: `permission-ask` and `release-permit` were described with a `permitId` neither body has (both key on `callId`), `chunk` had a `part` and `index`, `session-ready` had a `claudeSessionId` against a real `nestId`, `pulse` had a `cellId` against a real `pid`, `tendril-reach` had four fields against a real `task`/`tools`. All corrected against views.rs, and the tables now say that `body fields` excludes `sid`, which is an envelope field and was listed as a body field throughout. The claude-cli `HOME` tests turned out to be fine. Every test in both files holds a per-file lock for its whole body, and the two files are separate binaries, so they cannot race each other — the same discipline as hum-identity. Left alone on the evidence. Workspace 51/51 across three runs; clippy clean. --- WIRE.md | 24 +++++---- ensemble/src/delivery.rs | 102 +++++++++++++++++++++++++++++++++------ ensemble/src/gossip.rs | 60 ++++++++++++++++++++--- ensemble/src/lib.rs | 21 +++++++- hives/bp7/src/main.rs | 8 +++ sim/src/lib.rs | 20 ++++++++ sim/tests/delivery.rs | 24 +++++++++ 7 files changed, 225 insertions(+), 34 deletions(-) diff --git a/WIRE.md b/WIRE.md index 71182996..4106461e 100644 --- a/WIRE.md +++ b/WIRE.md @@ -171,6 +171,10 @@ registry by one bump. ### Nestler → daemon +`body fields` below are the chi-specific fields only. `sid` is not +among them — it is an [envelope](#envelope) field, and a tone carries it +alongside any of these. + | chi | body fields | meaning | |---|---|---| | `hello` | `bee`, `protoVersion`, optional `version`/`propensity`/`chi`/`source` | first frame after connect | @@ -178,9 +182,9 @@ registry by one bump. | `cancel` | `sid` | interrupt the current turn for this sid | | `cleanup` | `sid` | drop daemon state for this session | | `curate` | `sid` | manual compaction request | -| `release-permit` | `sid`, `permitId`, `decision` | answer a `permission-ask` | +| `release-permit` | `callId`, `ok`, optional `error` | answer a `permission-ask` | | `tendril-result` | `sid`, `callId`, `result` | task subagent answered | -| `tool-result` | `sid`, `callId`, `result` | nestler-declared tool answered | +| `tool-result` | `callId`, `result`/`output`, `isError`, `title`, `metadata` | nestler-declared tool answered | | `petal-cell` | `sid`, `cell` | OC message-graph update (graft hint) | ### Daemon → nestler @@ -188,15 +192,15 @@ registry by one bump. | chi | body fields | meaning | |---|---|---| | `breath` | (state snapshot, usually `{}`) | reply to hello | -| `chunk` | `sid`, `part` (text/reasoning/tool fragment), `index` | streamed model output | -| `finish` | `sid`, `finishReason`, `usage` | turn complete | -| `error` | `sid`, `code`, `message`, optional protocol payload | turn aborted / hard error | -| `session-ready` | `sid`, `claudeSessionId` | nest spawned, ready for prompts | -| `pulse` | `kind` (CellSpawned/CellReady/CellIdle/CellDied/CellEvicted), `cellId` | process lifecycle event | -| `permission-ask` | `sid`, `permitId`, `question`, `context` | mid-stream permission needed | -| `tendril-reach` | `sid`, `callId`, `name`, `args` | task subagent dispatch | +| `chunk` | `chunkType`, `blockIdx`, `delta`, `partialJson` | streamed model output | +| `finish` | `finishReason`, `usage`, optional `exitCode`, `subtype` | turn complete | +| `error` | `message`, `code`, optional `subtype`, `usage` | turn aborted / hard error | +| `session-ready` | `nestId`, `model`, `tools` | nest spawned, ready for prompts | +| `pulse` | `kind` (CellSpawned/CellReady/CellIdle/CellDied/CellEvicted), `pid` | process lifecycle event | +| `permission-ask` | `callId`, `toolName`, `message`, `arg` | mid-stream permission needed | +| `tendril-reach` | `task`, `tools` | task subagent dispatch | | `tool-call` | `sid`, `callId`, `name`, `args` | nestler-declared tool dispatch | -| `tool-meta` | `sid`, `callId`, `meta` | out-of-band metadata for a tool result | +| `tool-meta` | `callId`, `toolName`, `metadata` | out-of-band metadata for a tool result | ### Either direction diff --git a/ensemble/src/delivery.rs b/ensemble/src/delivery.rs index 44a57a92..84f11852 100644 --- a/ensemble/src/delivery.rs +++ b/ensemble/src/delivery.rs @@ -22,27 +22,30 @@ use std::sync::atomic::{AtomicU64, Ordering}; use lru::LruCache; use parking_lot::Mutex; +use sha2::{Digest, Sha256}; use crate::Tone; -/// Bound on the per-ensemble `mid` seen-set. Sizing is a memory/idle- -/// time trade: the set only has to remember an id for as long as a -/// duplicate of it could still be in flight, which `dusk` bounds from -/// above. A tone with no `dusk` can be redelivered arbitrarily late by -/// a re-fan, so it is bounded only by this cap. +/// Bound on the per-ensemble `mid` seen-set, in entries. Sizing is a +/// memory/idle-time trade: the set only has to remember an id for as +/// long as a duplicate of it could still be in flight, which `dusk` +/// bounds from above. A tone with no `dusk` can be redelivered +/// arbitrarily late by a re-fan, so it is bounded by this cap instead. pub const DELIVERY_SEEN_CAP: usize = 4096; -/// Ids of tones this ensemble has already dispatched. +/// Ids of tones this ensemble has already dispatched, held as digests. pub struct DeliveryState { - seen: Mutex>, + seen: Mutex>, } impl DeliveryState { pub fn new() -> std::sync::Arc { + Self::with_cap(DELIVERY_SEEN_CAP) + } + + pub fn with_cap(cap: usize) -> std::sync::Arc { std::sync::Arc::new(Self { - seen: Mutex::new(LruCache::new( - NonZeroUsize::new(DELIVERY_SEEN_CAP).expect("seen cap > 0"), - )), + seen: Mutex::new(LruCache::new(NonZeroUsize::new(cap).expect("seen cap > 0"))), }) } @@ -55,12 +58,13 @@ impl DeliveryState { /// couple two unrelated decisions and let one evict the other's /// entries early. pub fn note_mid(&self, mid: &str) -> bool { + let key = mid_key(mid); let mut seen = self.seen.lock(); - if seen.contains(mid) { - seen.get(mid); + if seen.contains(&key) { + seen.get(&key); false } else { - seen.put(mid.to_string(), ()); + seen.put(key, ()); true } } @@ -96,6 +100,26 @@ fn drop_if_dusk(tone: &Tone, expired: &AtomicU64) -> bool { past } +/// Reduce a `mid` to a fixed 32-byte key before it goes in the set. +/// +/// A `mid` is attacker-controlled and unbounded: the TCP transport reads +/// NDJSON with `BufReader::lines()`, which has no length cap, so a peer +/// can put a megabyte in one field. Capping the *entry count* therefore +/// does not bound the memory — 4096 entries of unbounded string is not +/// a bound. Digesting keeps dedup exact while making the footprint +/// `cap * 32` bytes no matter what arrives. +fn mid_key(mid: &str) -> [u8; 32] { + let mut h = Sha256::new(); + h.update(mid.as_bytes()); + h.finalize().into() +} + +/// Enough of a `mid` to correlate a log line with a sender, without +/// copying an attacker-sized string into the log. +fn mid_prefix(mid: &str) -> String { + mid.chars().take(12).collect() +} + /// The `mid` on a tone, if it carries one. Absent or non-string means /// "no at-most-once claim", not "malformed" — the tone is delivered. pub fn mid_of(tone: &Tone) -> Option<&str> { @@ -116,7 +140,7 @@ pub fn admit(tone: &Tone, delivery: &DeliveryState, expired_dusk: &AtomicU64) -> Some(mid) if !delivery.note_mid(mid) => { tracing::debug!( target: "ensemble", - mid, + mid = %mid_prefix(mid), chi = tone.get("chi").and_then(|v| v.as_str()).unwrap_or(""), "tone.mid: duplicate suppressed" ); @@ -198,6 +222,56 @@ mod tests { /// Dusk is checked first, so a mid already known to be dead cannot /// displace a live one from the bounded set. #[test] + /// The set is bounded in entries AND in bytes. A `mid` is + /// attacker-controlled and the transport has no frame cap, so the + /// entry cap alone bounds nothing. + #[test] + fn an_enormous_mid_costs_a_fixed_32_bytes() { + let d = state(); + let huge = "x".repeat(4 * 1024 * 1024); + assert!(d.note_mid(&huge)); + assert!(!d.note_mid(&huge), "still dedups exactly"); + assert_eq!(d.len(), 1); + // The key is the digest, not the string. + assert_eq!(mid_key(&huge).len(), 32); + assert_ne!(mid_key(&huge), mid_key(&"y".repeat(4 * 1024 * 1024))); + } + + /// A multibyte `mid` must not panic the prefix used in logs. + #[test] + fn mid_prefix_is_char_safe() { + assert_eq!(mid_prefix(&"short"), "short"); + assert_eq!(mid_prefix("é".repeat(50).as_str()), "é".repeat(12)); + } + + /// Eviction must actually happen at the cap — the whole reason the + /// cap exists. Filling past it should evict the least recently seen + /// and admit the newcomer. + #[test] + fn the_seen_set_evicts_at_its_cap() { + let d = DeliveryState::with_cap(2); + assert!(d.note_mid("a")); + assert!(d.note_mid("b")); + // Observing "a" is also its LRU touch, so "b" is now the victim. + assert!(!d.note_mid("a"), "still remembered at cap"); + assert!(d.note_mid("c"), "c is new"); + assert_eq!(d.len(), 2, "cap holds"); + assert!(!d.note_mid("a"), "a survived"); + assert!(d.note_mid("b"), "b was the least recent, so it went"); + } + + /// A duplicate is only suppressed while the set can still remember + /// it. Past the cap a very late retransmit is admitted again — the + /// documented consequence of bounding memory, not a silent + /// guarantee. `dusk` is what closes this window in practice. + #[test] + fn a_duplicate_past_the_cap_is_admitted_again() { + let d = DeliveryState::with_cap(1); + assert!(d.note_mid("old")); + assert!(d.note_mid("new")); + assert!(d.note_mid("old"), "forgotten — cap is 1"); + } + fn a_dead_mid_does_not_evict_a_live_one() { let d = state(); let expired = AtomicU64::new(0); diff --git a/ensemble/src/gossip.rs b/ensemble/src/gossip.rs index 34bf4762..ebd27297 100644 --- a/ensemble/src/gossip.rs +++ b/ensemble/src/gossip.rs @@ -72,10 +72,12 @@ pub struct GossipState { impl GossipState { pub fn new() -> Arc { + Self::with_cap(GOSSIP_SEEN_CAP) + } + + pub fn with_cap(cap: usize) -> Arc { Arc::new(Self { - seen: Mutex::new(LruCache::new( - NonZeroUsize::new(GOSSIP_SEEN_CAP).expect("seen cap > 0"), - )), + seen: Mutex::new(LruCache::new(NonZeroUsize::new(cap).expect("seen cap > 0"))), topics: Mutex::new(HashMap::new()), }) } @@ -132,22 +134,52 @@ impl GossipState { pub fn mint_msg_id(from: &Hid) -> String { static SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); let seq = SEQ.fetch_add(1, std::sync::atomic::Ordering::Relaxed); - let origin = &from.to_hex()[..12]; + // Slice on a char boundary rather than byte 12, and fall back to + // the whole hex if a Hid ever renders shorter. A panic here would + // take down a publisher, and the prefix is cosmetic anyway. + let hex = from.to_hex(); + let origin: String = hex.chars().take(12).collect(); // 6 bytes of origin + let origin = origin.as_str(); format!("{origin}-{:x}-{seq:x}", crate::now_ms()) } /// Build a `chi:"gossip-publish"` tone with the given fields. Kept here /// so `Ensemble::publish` and the install drainer's re-fan path agree /// on the wire shape. -pub fn gossip_tone(topic: &str, rid: &str, from: &Hid, payload: Value, msg_id: &str) -> Tone { - serde_json::json!({ +/// +/// `dusk_ms` is a lifetime, and it is `None` by default on purpose. A +/// gossip message is re-fanned by every hop, so a slow mesh can deliver +/// one arbitrarily late; whether a stale announcement is worse than no +/// announcement is the publisher's call, not the library's. A heartbeat +/// wants a tight lifetime, a "worker moved" notice wants a long one, and +/// a default TTL would silently drop the second kind on a congested +/// mesh. `None` means no expiry. +pub fn gossip_tone_with_dusk( + topic: &str, + rid: &str, + from: &Hid, + payload: Value, + msg_id: &str, + dusk_ms: Option, +) -> Tone { + let mut tone = serde_json::json!({ "chi": GOSSIP_CHI, "rid": rid, "topic": topic, "payload": payload, "from": from.to_hex(), "msg_id": msg_id, - }) + }); + if let Some(dusk) = dusk_ms { + tone.as_object_mut() + .expect("gossip tone is an object") + .insert("dusk".into(), serde_json::json!(crate::now_ms() + dusk)); + } + tone +} + +pub fn gossip_tone(topic: &str, rid: &str, from: &Hid, payload: Value, msg_id: &str) -> Tone { + gossip_tone_with_dusk(topic, rid, from, payload, msg_id, None) } /// Parsed view of an incoming gossip tone. Drainer pulls these fields @@ -191,7 +223,7 @@ mod tests { } #[test] - fn seen_set_dedups_within_capacity() { + fn seen_set_dedups_repeats() { let state = GossipState::new(); assert!(state.note_seen("a")); assert!(!state.note_seen("a")); @@ -199,6 +231,18 @@ mod tests { assert!(!state.note_seen("a")); } + /// The cap is the only thing bounding this set, so eviction has to + /// be exercised at a cap small enough to reach. + #[test] + fn seen_set_evicts_at_its_cap() { + let state = GossipState::with_cap(2); + assert!(state.note_seen("a")); + assert!(state.note_seen("b")); + assert!(state.note_seen("c"), "c is new"); + assert!(state.note_seen("a"), "a was least recent, so evicted"); + assert!(!state.note_seen("c"), "c survived"); + } + #[test] fn parse_gossip_pulls_fields() { let from = Hid::random_humd(); diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index fc91302b..060e71e7 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -58,7 +58,10 @@ pub mod delivery; pub use delivery::{DeliveryState, DELIVERY_SEEN_CAP}; pub mod gossip; -pub use gossip::{gossip_tone, mint_msg_id, GossipState, GOSSIP_CHI, GOSSIP_SEEN_CAP}; +pub use gossip::{ + gossip_tone, gossip_tone_with_dusk, mint_msg_id, GossipState, GOSSIP_CHI, + GOSSIP_SEEN_CAP, +}; pub mod liveness; pub use liveness::{ @@ -1349,12 +1352,26 @@ impl Ensemble { /// `route()` semantically; both share the `PeerConnection.send` /// wire but `publish` is mesh-wide and `route` is unicast. pub async fn publish(&self, topic: &str, payload: serde_json::Value) { + self.publish_with_dusk(topic, payload, None).await + } + + /// As [`Self::publish`], with a lifetime in ms. Every hop re-fans, + /// so a slow mesh can deliver a gossip tone long after it was sent; + /// `dusk_ms` is how long it stays worth acting on. `None` (the + /// default) never expires — see [`gossip_tone_with_dusk`] for why + /// there is no default TTL. + pub async fn publish_with_dusk( + &self, + topic: &str, + payload: serde_json::Value, + dusk_ms: Option, + ) { let msg_id = mint_msg_id(&self.me); let rid = format!("gossip-{msg_id}"); // Mark seen locally so the next-hop echo (peer re-fans back to // us) is dropped at the drainer's seen check. self.gossip.note_seen(&msg_id); - let tone = gossip_tone(topic, &rid, &self.me, payload, &msg_id); + let tone = gossip_tone_with_dusk(topic, &rid, &self.me, payload, &msg_id, dusk_ms); let conns: Vec> = { let peers = self.peers.read(); peers.values().map(|p| p.conn.clone()).collect() diff --git a/hives/bp7/src/main.rs b/hives/bp7/src/main.rs index 5f064a35..5a68d3c4 100644 --- a/hives/bp7/src/main.rs +++ b/hives/bp7/src/main.rs @@ -192,7 +192,15 @@ async fn run_prompt( let sid = format!("bp7-{}", now_ms()); let mut prompt = serde_json::Map::new(); prompt.insert("chi".into(), json!(Chi::Prompt)); + // Correlation only, and deliberately per-session: every prompt in + // this session shares `p-{sid}`. That is exactly why it cannot be + // the message identity — a second prompt under the same rid would + // look like a retransmit of the first. prompt.insert("rid".into(), Value::String(format!("p-{sid}"))); + // At-most-once, minted per message. This is a DTN bridge, so the + // same bundle really can be handed to us twice by the store-and- + // forward path, and re-running a prompt is not idempotent. + prompt.insert("mid".into(), Value::String(hum_identity::HumId::mint().to_string())); prompt.insert("sid".into(), Value::String(sid.clone())); prompt.insert("text".into(), Value::String(text.to_string())); prompt.insert("modelId".into(), Value::String(model.to_string())); diff --git a/sim/src/lib.rs b/sim/src/lib.rs index cc69f758..2e30014f 100644 --- a/sim/src/lib.rs +++ b/sim/src/lib.rs @@ -1051,6 +1051,26 @@ impl Sim { Ok(()) } + /// As [`Self::publish`], with a lifetime in ms so a scenario can + /// assert that an expired gossip tone is dropped rather than + /// delivered. `None` never expires. + pub async fn publish_with_dusk( + &self, + from: Hid, + topic: &str, + payload: serde_json::Value, + dusk_ms: i64, + ) -> Result<()> { + let ens = self + .humds + .read() + .get(&from) + .cloned() + .ok_or_else(|| anyhow::anyhow!("no humd {}", from.short()))?; + ens.ensemble.publish_with_dusk(topic, payload, Some(dusk_ms)).await; + Ok(()) + } + /// Subscribe to a gossip topic on one of the sim's humds. Broadcast /// receivers only see what is published after they subscribe. pub fn subscribe_topic( diff --git a/sim/tests/delivery.rs b/sim/tests/delivery.rs index df6681ec..9ef7feff 100644 --- a/sim/tests/delivery.rs +++ b/sim/tests/delivery.rs @@ -235,6 +235,30 @@ async fn a_response_echoing_the_request_rid_still_arrives() { assert_eq!(mids, vec!["m-req", "m-resp"], "both messages delivered"); } +/// A gossip tone published with a lifetime is dropped once that +/// lifetime passes, by the same rule as any other tone. Gossip is +/// re-fanned at every hop, so without this a congested mesh delivers +/// alerts long after they stopped being true. +#[tokio::test(flavor = "multi_thread")] +async fn an_expired_gossip_publish_is_dropped() { + let (sim, a, b, c) = trio().await; + let mut sub = sim.subscribe_topic(c, "alerts").expect("subscribe"); + + sim.publish_with_dusk(a, "alerts", json!({"event": "overloaded"}), -1_000) + .await + .expect("publish with a lifetime already in the past"); + + let got = timeout(Duration::from_millis(200), sub.recv()).await; + assert!(got.is_err(), "an expired gossip tone was delivered: {got:?}"); + + // Counted at the FIRST hop, not the last. Expiry is enforced + // wherever a tone lands, so a dead tone is dropped on entry to the + // mesh instead of being carried to every subscriber and dropped + // N times. + assert_eq!(sim.expired_dusk(b), 1, "b is the first hop and should have caught it"); + assert_eq!(sim.expired_dusk(c), 0, "c never saw it — it died at b"); +} + fn now_ms() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) From 303b5dec031efdd925d4e5e7282999266ff8d6cb Mon Sep 17 00:00:00 2001 From: Adil Date: Sun, 27 Sep 2026 22:36:22 +0500 Subject: [PATCH 13/15] ensemble: a test that never ran, and a clippy error in my own code MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `cargo clippy -- -D warnings` aborts in codegen/hum-paths before it ever lints ensemble, so it cannot see its own mistakes here. Linting the packages directly with --no-deps found two, both mine. `a_dead_mid_does_not_evict_a_live_one` had no `#[test]`. An earlier edit left its attribute attached to the function that followed it, so the test compiled as dead code and the suite stayed green without ever running it. The orphaned `#[test]` and a duplicated doc comment are gone and the test runs — and passes, which is the less interesting part. The interesting part is that "51/51 green" was reporting on a test that was not in the suite. Also collapsed a nested if in `Supervisor::dial`, plus a needless borrow in the delivery tests. Everything else under `-D warnings` in these crates is pre-existing: the collapsible ifs in the gossip/KAD dispatch blocks, the penny/drone/ drift dead code, `humd/src/lib.rs` broadly. The remaining hard failures are all in codegen/src/protocol.rs and hum-paths, untouched here. Workspace 51/51 across three runs. Note the count was 51 before this commit too — a non-running test does not change the number of green test binaries, which is exactly why it went unnoticed. --- ensemble/src/delivery.rs | 8 ++++---- humd/src/supervisor.rs | 6 +++--- 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/ensemble/src/delivery.rs b/ensemble/src/delivery.rs index 84f11852..156d51ff 100644 --- a/ensemble/src/delivery.rs +++ b/ensemble/src/delivery.rs @@ -219,9 +219,6 @@ mod tests { assert_eq!(expired.load(Ordering::SeqCst), 0); } - /// Dusk is checked first, so a mid already known to be dead cannot - /// displace a live one from the bounded set. - #[test] /// The set is bounded in entries AND in bytes. A `mid` is /// attacker-controlled and the transport has no frame cap, so the /// entry cap alone bounds nothing. @@ -240,7 +237,7 @@ mod tests { /// A multibyte `mid` must not panic the prefix used in logs. #[test] fn mid_prefix_is_char_safe() { - assert_eq!(mid_prefix(&"short"), "short"); + assert_eq!(mid_prefix("short"), "short"); assert_eq!(mid_prefix("é".repeat(50).as_str()), "é".repeat(12)); } @@ -272,6 +269,9 @@ mod tests { assert!(d.note_mid("old"), "forgotten — cap is 1"); } + /// Dusk is checked first, so a mid already known to be dead cannot + /// displace a live one from the bounded set. + #[test] fn a_dead_mid_does_not_evict_a_live_one() { let d = state(); let expired = AtomicU64::new(0); diff --git a/humd/src/supervisor.rs b/humd/src/supervisor.rs index 9ccb0335..d96b5587 100644 --- a/humd/src/supervisor.rs +++ b/humd/src/supervisor.rs @@ -162,11 +162,11 @@ impl Supervisor { async fn dial(&self, peer: &PeerConfig) -> bool { let mut dialable = false; if let Some(transport) = &self.iroh { - if peer + let iroh_hints = peer .hints .iter() - .any(|h| h.starts_with(ensemble::iroh::IROH_HINT)) - { + .any(|h| h.starts_with(ensemble::iroh::IROH_HINT)); + if iroh_hints { dialable = true; if iroh::dial_one(transport, &self.ens, &self.key, peer, &self.my_caps).await { return true; From b48a2bd766054ecb510ea2138437a533ab1361e7 Mon Sep 17 00:00:00 2001 From: Adil Date: Mon, 28 Sep 2026 00:14:33 +0500 Subject: [PATCH 14/15] ensemble: bound every peer send, so one stalled peer can't wedge the mesh MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A peer that stops reading doesn't close its socket. The buffer fills, the write stops completing, and the connection keeps looking healthy to a liveness lease — which renews on unrelated traffic, so silence never trips it. Meanwhile the sender is parked inside write_all, and every fan-out that walks peers sequentially is now parked with it. The only symptom was a Lagged(n) on a broadcast receiver: a count, and no cause. Every send is now bounded by SEND_TIMEOUT (5s). On timeout the connection is closed rather than retried, so the lease can mark the peer dead and the supervisor redial it, and the event lands in send_timeouts() where it can be alerted on. Applied to all seven send sites: hello, both probes, publish, re-fan, KAD replies, and route. The probes matter most, and not just for the wedge: a probe that blocks forever is a probe that never reports the stall it exists to find. sim: InMemoryEndpoint::stall() models a peer whose buffer has filled — the write hangs rather than erroring, because an error would let a fix that only checks the return value pass without touching the deadline. The five tests in sim/tests/stalled_peer.rs were each checked against a reverted fix. The two fan-out tests fail without the deadline (the publish itself hangs, so the test bounds it and fails with a message instead of eating the CI timeout). Note the honest limits: route() targets one peer, so two separate routes don't serialise — the fan-out tests target publish(), which is the real single loop. A test asserting "a healthy peer still gets through" on the route path would have passed either way. --- WIRE.md | 11 ++ ensemble/src/lib.rs | 92 +++++++++++-- ensemble/src/send.rs | 257 +++++++++++++++++++++++++++++++++++++ sim/src/lib.rs | 59 +++++++-- sim/tests/stalled_peer.rs | 262 ++++++++++++++++++++++++++++++++++++++ 5 files changed, 657 insertions(+), 24 deletions(-) create mode 100644 ensemble/src/send.rs create mode 100644 sim/tests/stalled_peer.rs diff --git a/WIRE.md b/WIRE.md index 4106461e..47631f93 100644 --- a/WIRE.md +++ b/WIRE.md @@ -220,6 +220,17 @@ Only emitted across the ensemble layer (not by local nestlers): `gossip-publish`, `kad-find-node`, `kad-find-node-resp`. See [`ensemble/README.md`](../ensemble/README.md). +**Send deadline.** Every inter-humd send is bounded at 5s +(`ensemble::SEND_TIMEOUT`). A peer that stops reading fills its socket +buffer rather than closing, so the write stalls indefinitely while the +connection still looks healthy to a liveness lease. On timeout humd +closes that connection, counts the event in `send_timeouts`, and lets +the lease mark the peer dead — a bounded send is what keeps one stalled +peer from wedging a fan-out to every other peer. Implementations that +speak this wire should impose an equivalent deadline; a peer that wants +larger tones or a slower link should say so rather than write without +one. + ## Helpers The reference clients ship three deterministic helpers. Algorithms diff --git a/ensemble/src/lib.rs b/ensemble/src/lib.rs index 060e71e7..07c189dc 100644 --- a/ensemble/src/lib.rs +++ b/ensemble/src/lib.rs @@ -52,9 +52,11 @@ pub use tls::{ }; pub mod iroh; -pub use iroh::{IrohEndpoint, IrohTransport, IROH_ALPN}; +pub use iroh::{dialable_addr, IrohEndpoint, IrohTransport, IROH_ALPN, IROH_IP_HINT}; pub mod delivery; +pub mod send; +pub use send::{send_bounded, SendError, SendStats, SEND_TIMEOUT}; pub use delivery::{DeliveryState, DELIVERY_SEEN_CAP}; pub mod gossip; @@ -438,6 +440,8 @@ pub struct LinkCounters { pub buffered: u64, pub lost_on_heal: u64, pub evicted: u64, + /// Sends handed to a stalled link, which never completed. + pub stalled_sends: u64, } impl LinkCounters { @@ -452,6 +456,7 @@ impl LinkCounters { buffered: self.buffered - base.buffered, lost_on_heal: self.lost_on_heal - base.lost_on_heal, evicted: self.evicted - base.evicted, + stalled_sends: self.stalled_sends - base.stalled_sends, } } } @@ -506,6 +511,13 @@ pub struct InMemoryEndpoint { /// which is what marks the lease `TransportClosed`. Distinct from /// `partitioned`, which keeps the link nominally up. killed: AtomicBool, + /// Set by [`InMemoryEndpoint::stall`]. A stalled link accepts the + /// connection and stops draining: the write never completes, which + /// is what a full socket buffer with a non-reading peer looks like + /// from the writer's side. Distinct from `killed` (gone) and from + /// `partitioned` (nominally up and buffering) — a stalled peer + /// looks perfectly healthy to a lease. + stalled: AtomicBool, } struct PartitionState { @@ -557,6 +569,7 @@ impl InMemoryEndpoint { counters: Mutex::new(LinkCounters::default()), reorder_hold: Mutex::new(None), killed: AtomicBool::new(false), + stalled: AtomicBool::new(false), }); let b = Arc::new(InMemoryEndpoint { peer: HumdAddr::new(a_id), @@ -572,6 +585,7 @@ impl InMemoryEndpoint { counters: Mutex::new(LinkCounters::default()), reorder_hold: Mutex::new(None), killed: AtomicBool::new(false), + stalled: AtomicBool::new(false), }); (a, b) } @@ -617,6 +631,14 @@ impl InMemoryEndpoint { /// the sender out and putting it back is what keeps `kill` from /// deadlocking against a send that has already committed to it. async fn emit(&self, tone: Tone) -> Result<()> { + if self.stalled.load(Ordering::SeqCst) { + // Never completes. Modelled as a hang rather than an error + // because that is what the real transport does, and an + // error would let a fix that merely checks the return value + // pass without ever testing the deadline. + self.counters.lock().stalled_sends += 1; + std::future::pending::<()>().await; + } let tx = { let mut guard = self.tx.lock(); match guard.take() { @@ -685,6 +707,25 @@ impl InMemoryEndpoint { } } + /// Make this link stop draining. Sends to it hang rather than fail, + /// and the peer stays registered — a stall is invisible to a liveness + /// lease, which is the whole problem. Not the same as [`Self::kill`] + /// (the link is gone, so the far side's receiver closes) or as a + /// partition (the link stays nominally up and keeps buffering). + pub fn stall(&self) { + self.stalled.store(true, Ordering::SeqCst); + } + + /// Let a stalled link drain again. A stall is a fault, not a + /// teardown, so it has to be reversible. + pub fn unstall(&self) { + self.stalled.store(false, Ordering::SeqCst); + } + + pub fn is_stalled(&self) -> bool { + self.stalled.load(Ordering::SeqCst) + } + /// Drop the link as if the peer's process had vanished. Sends start /// failing and the drainer at the far end sees its receiver close, /// which marks that peer's lease `TransportClosed`. @@ -823,6 +864,9 @@ pub struct Ensemble { expired_dusk: Arc, /// Mids already dispatched here, so a retransmit is delivered once. delivery: Arc, + /// Sends that did not complete. A stalled peer has to be a number + /// someone can alert on, not a `Lagged(n)` on a broadcast receiver. + send_stats: Arc, } /// The local fan-out point for tones arriving from peers. Cloned into @@ -906,6 +950,11 @@ pub enum RouteError { Untargeted, #[error("send failed: {0}")] SendFailed(anyhow::Error), + /// The peer stopped reading and the write did not complete in time. + /// Distinct from a plain failure because the connection was closed + /// on purpose — the peer should be gone from the registry shortly. + #[error("peer stalled: no write completed within the send deadline")] + PeerStalled, } impl Ensemble { @@ -919,6 +968,7 @@ impl Ensemble { strict_auth: false, expired_dusk: Arc::new(AtomicU64::new(0)), delivery: DeliveryState::new(), + send_stats: SendStats::new(), } } @@ -978,6 +1028,7 @@ impl Ensemble { let gossip = self.gossip.clone(); let expired_dusk = self.expired_dusk.clone(); let delivery = self.delivery.clone(); + let send_stats = self.send_stats.clone(); let kad = self.kad.clone(); let my_id = self.me; tokio::spawn(async move { @@ -1062,7 +1113,7 @@ impl Ensemble { continue; } if tone.get("chi").and_then(|v| v.as_str()) == Some(GOSSIP_CHI) { - if handle_gossip(&gossip, &peers, &id, &tone).await { + if handle_gossip(&send_stats, &gossip, &peers, &id, &tone).await { continue; } } @@ -1075,7 +1126,7 @@ impl Ensemble { if chi_val == Some(KAD_FIND_NODE_CHI) || chi_val == Some(KAD_FIND_NODE_RESP_CHI) { - if handle_kad(&kad, &peers, &id, &my_id, &tone).await { + if handle_kad(&send_stats, &kad, &peers, &id, &my_id, &tone).await { continue; } } @@ -1142,6 +1193,7 @@ impl Ensemble { let gossip = self.gossip.clone(); let expired_dusk = self.expired_dusk.clone(); let delivery = self.delivery.clone(); + let send_stats = self.send_stats.clone(); let kad = self.kad.clone(); let my_id = self.me; tokio::spawn(async move { @@ -1201,7 +1253,7 @@ impl Ensemble { continue; } if tone.get("chi").and_then(|v| v.as_str()) == Some(GOSSIP_CHI) { - if handle_gossip(&gossip, &peers, &id, &tone).await { + if handle_gossip(&send_stats, &gossip, &peers, &id, &tone).await { continue; } } @@ -1209,7 +1261,7 @@ impl Ensemble { if chi_val == Some(KAD_FIND_NODE_CHI) || chi_val == Some(KAD_FIND_NODE_RESP_CHI) { - if handle_kad(&kad, &peers, &id, &my_id, &tone).await { + if handle_kad(&send_stats, &kad, &peers, &id, &my_id, &tone).await { continue; } } @@ -1237,7 +1289,7 @@ impl Ensemble { .map(|(id, p)| (*id, p.conn.clone())) .collect(); for (id, conn) in peers { - let _ = conn.send(ping_tone(&self.me, &id, seq)).await; + let _ = send_bounded(&conn, ping_tone(&self.me, &id, seq), &self.send_stats).await; } } @@ -1250,7 +1302,7 @@ impl Ensemble { // as long as the link takes. let conn = self.peers.read().get(id).map(|p| p.conn.clone()); if let Some(conn) = conn { - let _ = conn.send(ping_tone(&self.me, id, seq)).await; + let _ = send_bounded(&conn, ping_tone(&self.me, id, seq), &self.send_stats).await; } } @@ -1338,6 +1390,18 @@ impl Ensemble { self.delivery.len() } + /// Peer sends that did not complete. `timed_out` is the interesting + /// one: it means a peer stopped reading, and each one should have + /// cost that peer its place in the registry. + pub fn send_timeouts(&self) -> u64 { + self.send_stats.timed_out() + } + + /// Peer sends that failed outright, as opposed to stalling. + pub fn send_failures(&self) -> u64 { + self.send_stats.failed() + } + /// Publish a gossip message to every installed peer. Mints a fresh /// `msg_id`, marks it seen locally /// (so we don't re-fan it on the inevitable echo), and sends a @@ -1377,7 +1441,7 @@ impl Ensemble { peers.values().map(|p| p.conn.clone()).collect() }; for conn in conns { - if let Err(e) = conn.send(tone.clone()).await { + if let Err(e) = send_bounded(&conn, tone.clone(), &self.send_stats).await { tracing::debug!( target: "ensemble.gossip", peer = %conn.peer().id.short(), @@ -1649,7 +1713,11 @@ impl Ensemble { peers.get(&target).map(|p| p.conn.clone()) }; let conn = conn.ok_or(RouteError::UnknownPeer(target))?; - conn.send(tone).await.map_err(RouteError::SendFailed) + match send_bounded(&conn, tone, &self.send_stats).await { + Ok(()) => Ok(()), + Err(SendError::TimedOut) => Err(RouteError::PeerStalled), + Err(SendError::Failed(why)) => Err(RouteError::SendFailed(anyhow::anyhow!(why))), + } } } @@ -1861,6 +1929,7 @@ fn rekey_peer( } async fn handle_kad( + send_stats: &SendStats, kad: &Arc, peers: &Arc>>, arrived_from: &Hid, @@ -1884,7 +1953,7 @@ async fn handle_kad( peers.get(arrived_from).map(|p| p.conn.clone()) }; if let Some(conn) = conn { - if let Err(e) = conn.send(resp).await { + if let Err(e) = send_bounded(&conn, resp, send_stats).await { tracing::debug!( target: "ensemble.kad", peer = %arrived_from.short(), @@ -1944,6 +2013,7 @@ async fn handle_liveness( } async fn handle_gossip( + send_stats: &SendStats, gossip: &Arc, peers: &Arc>>, arrived_from: &Hid, @@ -1975,7 +2045,7 @@ async fn handle_gossip( .collect() }; for conn in others { - if let Err(e) = conn.send(tone.clone()).await { + if let Err(e) = send_bounded(&conn, tone.clone(), send_stats).await { tracing::debug!( target: "ensemble.gossip", peer = %conn.peer().id.short(), diff --git a/ensemble/src/send.rs b/ensemble/src/send.rs new file mode 100644 index 00000000..0fab4e12 --- /dev/null +++ b/ensemble/src/send.rs @@ -0,0 +1,257 @@ +//! Bounded peer sends. +//! +//! A peer that stops reading is not a peer that closes. The connection +//! stays up, the handshake stays valid, and the lease keeps renewing +//! from whatever else is still arriving — so nothing looks wrong until +//! the socket buffer fills and the write stops completing. +//! +//! That is the wedge. `send` awaits the write, and on most transports +//! holds the connection's write mutex while it does, so one stalled peer +//! blocks every later send to that peer. When the caller is the single +//! task that dispatches to *all* peers, one stalled peer becomes +//! mesh-wide message loss. The only symptom is a `Lagged(n)` on a +//! broadcast receiver, which reports how many tones were skipped and +//! nothing about why. +//! +//! So every send is bounded. On timeout the connection is closed rather +//! than merely retried: a peer that cannot take a write is not slow, and +//! closing it is what lets the liveness lease mark it dead and the +//! supervisor redial it. Bounding the send is what turns a silent +//! mesh-wide stall into one eviction. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::Arc; +use std::time::Duration; + +use crate::{PeerConnection, Tone}; + +/// How long one send may take before the peer is declared unusable. +/// +/// Sized for a slow-but-alive link, not a fast one: a tone is a few KB +/// of NDJSON, so anything that has not drained in this long is not +/// draining at all. A liveness probe shares this deadline deliberately +/// — a probe that blocks forever is a probe that never reports the +/// stall it exists to find. +pub const SEND_TIMEOUT: Duration = Duration::from_secs(5); + +/// Why a bounded send failed. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum SendError { + /// The peer stopped reading and the write did not complete in time. + #[error("peer stalled: no write completed within the send deadline")] + TimedOut, + /// The write failed outright, or the transport refused it. Keeps the + /// transport's own message: "it failed" is not actionable and the + /// whole reason this module exists is that failures were invisible. + #[error("write failed: {0}")] + Failed(String), +} + +/// Sends that did not complete. A number, not a log line — a stall is +/// something to alert on, and `warn!` is what made it invisible before. +#[derive(Debug, Default)] +pub struct SendStats { + pub timed_out: AtomicU64, + pub failed: AtomicU64, +} + +impl SendStats { + pub fn new() -> Arc { + Arc::new(Self::default()) + } + + pub fn timed_out(&self) -> u64 { + self.timed_out.load(Ordering::SeqCst) + } + + pub fn failed(&self) -> u64 { + self.failed.load(Ordering::SeqCst) + } +} + +/// One send to one peer, under [`SEND_TIMEOUT`]. +/// +/// On timeout the connection is closed. That is the point: a stalled +/// write is not a slow write, and leaving the connection open would let +/// the lease keep calling a dead peer live. +pub async fn send_bounded( + conn: &Arc, + tone: Tone, + stats: &SendStats, +) -> Result<(), SendError> { + send_bounded_with(conn, tone, stats, SEND_TIMEOUT).await +} + +/// [`send_bounded`] with an explicit deadline, so tests need not wait +/// out the production timeout. +pub async fn send_bounded_with( + conn: &Arc, + tone: Tone, + stats: &SendStats, + deadline: Duration, +) -> Result<(), SendError> { + match tokio::time::timeout(deadline, conn.send(tone)).await { + Ok(Ok(())) => Ok(()), + Ok(Err(e)) => { + stats.failed.fetch_add(1, Ordering::SeqCst); + Err(SendError::Failed(e.to_string())) + } + Err(_) => { + stats.timed_out.fetch_add(1, Ordering::SeqCst); + tracing::warn!( + target: "ensemble.send", + peer = %conn.peer().id.short(), + timeout_ms = deadline.as_millis() as u64, + "send.timed_out: peer stopped reading, closing" + ); + conn.close(); + Err(SendError::TimedOut) + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{Hid, HumdAddr, PeerCapabilities}; + use parking_lot::Mutex; + use std::sync::atomic::AtomicBool; + + /// A peer that accepts a connection and then stops reading — what a + /// full socket buffer with a non-reading peer looks like from the + /// writer's side. `stalled == true` never completes the write. + struct Stalled { + addr: HumdAddr, + caps: PeerCapabilities, + stalled: AtomicBool, + closed: AtomicBool, + sends: Mutex, + } + + #[async_trait::async_trait] + impl PeerConnection for Stalled { + fn peer(&self) -> &HumdAddr { + &self.addr + } + fn capabilities(&self) -> &PeerCapabilities { + &self.caps + } + async fn send(&self, _tone: Tone) -> anyhow::Result<()> { + *self.sends.lock() += 1; + if self.stalled.load(Ordering::SeqCst) { + // Never completes. This is the wedge. + std::future::pending::<()>().await; + } + Ok(()) + } + fn take_receiver(&self) -> Option> { + None + } + fn close(&self) { + self.closed.store(true, Ordering::SeqCst); + } + } + + fn conn(stalls: bool) -> Arc { + Arc::new(Stalled { + addr: HumdAddr { id: Hid::random_humd(), hints: vec![] }, + caps: PeerCapabilities::default(), + stalled: AtomicBool::new(stalls), + closed: AtomicBool::new(false), + sends: Mutex::new(0), + }) + } + + fn as_conn(c: &Arc) -> Arc { + c.clone() + } + + /// The whole point: a stalled peer must not hold the caller. Run with + /// a short deadline so the test is not paced by production's 5s. + #[tokio::test] + async fn a_stalled_peer_times_out_rather_than_hanging() { + let c = conn(false); + c.stalled.store(true, Ordering::SeqCst); + let stats = SendStats::new(); + let short = Duration::from_millis(50); + + let started = std::time::Instant::now(); + let got = tokio::time::timeout( + Duration::from_secs(10), + send_bounded_with(&as_conn(&c), serde_json::json!({"chi": "chunk"}), &stats, short), + ) + .await + .expect("send_bounded must return, not hang"); + + assert_eq!(got, Err(SendError::TimedOut)); + assert!(started.elapsed() < Duration::from_secs(5), "returned promptly"); + } + + /// Bounding the send is pointless unless it also *removes* the peer: + /// a connection left open keeps the lease calling a dead peer live. + #[tokio::test] + async fn a_stalled_peer_is_closed_on_timeout() { + let c = conn(false); + c.stalled.store(true, Ordering::SeqCst); + let stats = SendStats::new(); + let _ = send_bounded_with( + &as_conn(&c), + serde_json::json!({"chi": "chunk"}), + &stats, + Duration::from_millis(50), + ) + .await; + assert!(c.closed.load(Ordering::SeqCst), "a stalled peer must be closed"); + assert_eq!(stats.timed_out(), 1); + } + + /// The healthy path is untouched, and not counted. + #[tokio::test] + async fn a_healthy_peer_sends_and_is_not_counted() { + let c = conn(false); + let stats = SendStats::new(); + let got = + send_bounded_with(&as_conn(&c), serde_json::json!({"chi": "chunk"}), &stats, SEND_TIMEOUT) + .await; + assert_eq!(got, Ok(())); + assert_eq!(*c.sends.lock(), 1); + assert_eq!(stats.timed_out(), 0); + assert!(!c.closed.load(Ordering::SeqCst), "a working peer stays open"); + } + + /// A transport that errors is a different failure from one that + /// hangs, and the two should be distinguishable in the counts. + #[tokio::test] + async fn a_failed_write_is_counted_separately_from_a_stall() { + struct Broken(HumdAddr, PeerCapabilities); + #[async_trait::async_trait] + impl PeerConnection for Broken { + fn peer(&self) -> &HumdAddr { + &self.0 + } + fn capabilities(&self) -> &PeerCapabilities { + &self.1 + } + async fn send(&self, _t: Tone) -> anyhow::Result<()> { + anyhow::bail!("write failed") + } + fn take_receiver(&self) -> Option> { + None + } + fn close(&self) {} + } + let c: Arc = + Arc::new(Broken( + HumdAddr { id: Hid::random_humd(), hints: vec![] }, + PeerCapabilities::default(), + )); + let stats = SendStats::new(); + let got = send_bounded_with(&c, serde_json::json!({"chi": "chunk"}), &stats, SEND_TIMEOUT).await; + assert!( + matches!(got, Err(SendError::Failed(_))), + "got {got:?}" + ); + assert_eq!(stats.failed(), 1); + assert_eq!(stats.timed_out(), 0, "a failure is not a stall"); + } +} diff --git a/sim/src/lib.rs b/sim/src/lib.rs index 2e30014f..ad03ffc3 100644 --- a/sim/src/lib.rs +++ b/sim/src/lib.rs @@ -485,6 +485,24 @@ impl Sim { Ok(()) } + /// Make `a→b` stop draining. Sends into it hang instead of failing, + /// and `b` stays registered — the shape of a peer whose socket + /// buffer filled because nobody is reading. + pub fn stall_dir(&self, a: Hid, b: Hid) -> Result<()> { + self.end_for(a, b)?.stall(); + Ok(()) + } + + pub fn unstall_dir(&self, a: Hid, b: Hid) -> Result<()> { + self.end_for(a, b)?.unstall(); + Ok(()) + } + + /// Whether `a→b` is currently stalled. + pub fn is_stalled(&self, a: Hid, b: Hid) -> Result { + Ok(self.end_for(a, b)?.is_stalled()) + } + /// Fault both directions, resetting each endpoint's tone counter. pub fn impair(&self, a: Hid, b: Hid, faults: LinkFaults) -> Result<()> { self.impair_dir(a, b, faults.clone())?; @@ -640,26 +658,41 @@ impl Sim { let ha = ha.ok_or_else(|| anyhow::anyhow!("no humd {}", link_a.short()))?; let hb = hb.ok_or_else(|| anyhow::anyhow!("no humd {}", link_b.short()))?; + // A failed heal flush is worth a warning but not a hard error: + // the link is back either way, and the tips will reconcile on the + // next heal. Losing the log line would hide it entirely. for (from, to) in [(&ha, &hb), (&hb, &ha)] { - let snapshot = from.waneman.snapshot(); - let mut snapshot_json = serde_json::Map::new(); - for (sigil, n) in snapshot { - snapshot_json.insert(sigil, Value::from(n)); - } - let tone = serde_json::json!({ - "chi": "wane-sync", - "rid": hum_identity::HumId::mint().to_string(), - "from": from.id.to_hex(), - "to": to.id.to_hex(), - "snapshot": Value::Object(snapshot_json), - }); - if let Err(e) = from.ensemble.route(tone).await { + if let Err(e) = self.wane_sync(from, to).await { tracing::warn!(err = %e, "wane-sync.route.failed"); } } Ok(()) } + /// Emit one `chi:"wane-sync"` from `from` to `to`, carrying `from`'s + /// current `WaneTracker` snapshot. + /// + /// Separate from [`Self::heal`] so a test can send one *while a + /// partition is up* and assert it does not arrive. A partition test + /// that only sends during the heal cannot tell a real outage from a + /// silent one: no traffic means no leak, so the tips stay divergent + /// either way and the test passes for the wrong reason. + pub async fn wane_sync(&self, from: &SimHumd, to: &SimHumd) -> Result<()> { + let mut snapshot_json = serde_json::Map::new(); + for (sigil, n) in from.waneman.snapshot() { + snapshot_json.insert(sigil, Value::from(n)); + } + let tone = serde_json::json!({ + "chi": "wane-sync", + "rid": hum_identity::HumId::mint().to_string(), + "from": from.id.to_hex(), + "to": to.id.to_hex(), + "snapshot": Value::Object(snapshot_json), + }); + from.ensemble.route(tone).await?; + Ok(()) + } + /// Attach a synthetic mock worker bee to `humd`. Registers a fresh /// thrum client, hello's it as `bee:["worker"]` advertising `models`, /// then spawns a task that turns every inbound chi:"prompt" into diff --git a/sim/tests/stalled_peer.rs b/sim/tests/stalled_peer.rs new file mode 100644 index 00000000..f3903d42 --- /dev/null +++ b/sim/tests/stalled_peer.rs @@ -0,0 +1,262 @@ +//! A peer that stops reading must not take the mesh with it. +//! +//! The failure this file pins down: one humd whose socket to peer B has +//! filled because B stopped reading. The write to B never completes. If +//! the send that dispatches to *all* peers awaits that write, the single +//! dispatch task is parked forever and every peer after B in the +//! iteration gets nothing — not slowly, but never. The only symptom is a +//! `Lagged(n)` on a broadcast receiver, which reports a count and no +//! cause, so the mesh looks like it lost messages for no reason. +//! +//! Two properties, both of which the old code failed: +//! 1. containment — a healthy peer on the same dispatcher keeps +//! receiving while another peer is stalled; +//! 2. detection — the stalled peer is eventually counted and evicted, +//! because a lease that keeps renewing against a peer which cannot +//! take a write is a stall that hides forever. + +use std::sync::Arc; +use std::time::Duration; + +use ensemble::Hid; +use serde_json::{json, Value}; +use sim::{Sim, SimHumd}; +use tokio::time::timeout; + +/// Longer than the production send deadline, so a delivery that arrives +/// within this window arrived on its own merits. +const PATIENCE: Duration = Duration::from_secs(3); +/// Enough headroom for the send deadline to expire, the connection to +/// close, and the lease to be reaped. Generous on purpose: the +/// alternative is a test that fails on a slow machine. +const SETTLE: Duration = Duration::from_secs(20); + +/// A tone for `target`, named so the receiver can identify it. +fn tone_to(target: Hid, rid: &str) -> Value { + json!({ + "chi": "chunk", + "rid": rid, + "from": "test", + "to": target.to_string(), + "body": {"text": "x"}, + }) +} + +/// Three humds, wired, ready. +async fn three(sim: &Sim) -> (Arc, Arc, Arc) { + let a = sim.spawn_humd(Hid::random_humd()).await; + let b = sim.spawn_humd(Hid::random_humd()).await; + let c = sim.spawn_humd(Hid::random_humd()).await; + for h in [&a, &b, &c] { + sim.await_ready(h.id).await.expect("ready"); + } + sim.wire(a.id, b.id).expect("wire a-b"); + sim.wire(a.id, c.id).expect("wire a-c"); + sim.await_handshake(a.id, b.id).await.expect("handshake a-b"); + sim.await_handshake(a.id, c.id).await.expect("handshake a-c"); + (a, b, c) +} + +/// The headline property, on the one code path that really is a single +/// loop over every peer: a gossip publish. +/// +/// `publish` walks its peers and awaits each write in turn, so before the +/// deadline existed a stalled peer parked the whole fan-out. C's +/// subscription then never saw a message that A had already accepted. +/// This is not a hypothetical ordering: B and C are consecutive entries +/// in the same `HashMap` walk, and which comes first is arbitrary — so +/// the loss is intermittent and looks like a flaky mesh. +#[tokio::test(flavor = "multi_thread")] +async fn a_stalled_peer_does_not_starve_a_healthy_one_in_a_fanout() { + let sim = Sim::new(); + let (a, b, c) = three(&sim).await; + let topic = "stall/fanout"; + let mut c_sub = sim.subscribe_topic(c.id, topic).expect("C subscribed"); + + sim.stall_dir(a.id, b.id).expect("stall A->B"); + assert!(sim.is_stalled(a.id, b.id).expect("stalled")); + + // The publish is itself bounded. Without the send deadline it never + // returns, and a test that hangs tells you far less than one that + // fails — it just eats the CI timeout and names no cause. + timeout(SETTLE, sim.publish(a.id, topic, serde_json::json!({"k": "v"}))) + .await + .expect("publish must return: a fan-out cannot await a dead write forever") + .expect("A accepts the publish"); + + let got = tokio::time::timeout(PATIENCE, c_sub.recv()) + .await + .expect("C must receive: one stalled peer may not cancel a fan-out") + .expect("C's subscription is open"); + assert_eq!(got, serde_json::json!({"k": "v"})); + + // A must not have delivered to the peer it cannot write to. Silence + // there is correct, and is asserted so this cannot be "fixed" later + // by making the fan-out lie. + let mut b_sub = sim.subscribe_topic(b.id, topic).expect("B subscribed"); + let got = tokio::time::timeout(Duration::from_millis(300), b_sub.recv()).await; + assert!(got.is_err(), "a stalled peer must not be counted as delivered"); +} + +/// Two stalled peers must not delay the healthy one for longer than one +/// deadline each. A fan-out that bounds each send sequentially is bounded +/// but still multiplied by the peer count; this pins the current cost so +/// a change that reintroduces unbounded waiting shows up as a failure +/// rather than as a latency graph nobody reads. +#[tokio::test(flavor = "multi_thread")] +async fn a_fanout_past_two_stalled_peers_still_arrives() { + let sim = Sim::new(); + let (a, b, c) = three(&sim).await; + let d = sim.spawn_humd(Hid::random_humd()).await; + sim.await_ready(d.id).await.expect("d ready"); + sim.wire(a.id, d.id).expect("wire a-d"); + sim.await_handshake(a.id, d.id).await.expect("handshake a-d"); + + let topic = "stall/two"; + let mut d_sub = sim.subscribe_topic(d.id, topic).expect("D subscribed"); + + sim.stall_dir(a.id, b.id).expect("stall A->B"); + sim.stall_dir(a.id, c.id).expect("stall A->C"); + + let started = std::time::Instant::now(); + timeout(SETTLE, sim.publish(a.id, topic, serde_json::json!({"k": "v"}))) + .await + .expect("publish must return") + .expect("A accepts the publish"); + + let got = tokio::time::timeout(PATIENCE * 4, d_sub.recv()) + .await + .expect("D must receive") + .expect("D's subscription is open"); + assert_eq!(got, serde_json::json!({"k": "v"})); + assert!( + started.elapsed() < ensemble::SEND_TIMEOUT * 3, + "two stalled peers cost two deadlines, not an unbounded wait: {:?}", + started.elapsed() + ); +} + +/// The other half: a bounded send that does not evict leaves the lease +/// renewing against a peer that cannot take a write. +#[tokio::test(flavor = "multi_thread")] +async fn a_stalled_peer_is_counted_and_evicted() { + let sim = Sim::new(); + let (a, b, _c) = three(&sim).await; + + assert_eq!(sim.peer_count(a.id), 2, "A knows B and C to start"); + + sim.stall_dir(a.id, b.id).expect("stall A->B"); + sim.nestler_send(a.id, tone_to(b.id, "into-the-stall")) + .expect("A accepts local work"); + + // The write is handed to the stalled link and never comes back. + // `nestler_send` injects on a detached task, so this has to be + // awaited rather than read straight away. + timeout(PATIENCE, async { + loop { + let (ab, _ba) = sim.link_counters(a.id, b.id).expect("counters"); + if ab.stalled_sends > 0 { + break; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .expect("the write must reach the stalled link and stay there"); + + // The send deadline has to expire, and it has to be visible. + timeout(SETTLE, async { + while a.ensemble.send_timeouts() == 0 { + tokio::time::sleep(Duration::from_millis(50)).await; + } + }) + .await + .expect("a bounded send must resolve; an unbounded one never would"); + + assert!( + a.ensemble.send_timeouts() >= 1, + "a stall that is not counted is a stall nobody gets paged for" + ); + + // And the peer it cost its connection is reaped, not left live. + a.ensemble.evict_expired(Duration::from_millis(0)); + assert!( + !a.ensemble.peers().contains(&b.id), + "B cannot take a write, so B must not keep renewing a lease against A" + ); +} + +/// A stall is not a kill, and must not be confused with one: it is a +/// broken writer, not a broken link. Asserted so a future "fix" that +/// simply kills every slow peer is caught. +#[tokio::test(flavor = "multi_thread")] +async fn a_stall_is_reversible_and_does_not_fake_a_partition() { + let sim = Sim::new(); + let (a, b, _c) = three(&sim).await; + + sim.stall_dir(a.id, b.id).expect("stall A->B"); + + // Nothing about the stall is visible to a lease. This is the reason a + // send deadline is needed and silence alone is not enough — asserted + // rather than assumed, because it is the whole justification. + assert!( + a.ensemble.peers().contains(&b.id), + "a stalled peer is not evicted by looking at it" + ); + let liveness = sim.peer_liveness(a.id, Duration::from_secs(60)).expect("liveness"); + let b_state = liveness + .iter() + .find(|(id, _)| *id == b.id) + .map(|(_, l)| *l) + .expect("B is in the registry"); + assert_eq!( + b_state, + ensemble::Liveness::Live, + "a stalled peer still looks live: the lease renews on other traffic, \ + which is exactly why silence alone cannot find it" + ); + + // Reversible, so the fault does not leak into neighbouring tests. + sim.unstall_dir(a.id, b.id).expect("unstall"); + assert!(!sim.is_stalled(a.id, b.id).expect("not stalled")); + let mut sub = sim.humd_peer_sub(b.id).expect("B subscribed"); + sim.nestler_send(a.id, tone_to(b.id, "after-heal")) + .expect("A accepts local work"); + let got = Sim::collect_rids(&mut sub, 1, PATIENCE).await; + assert_eq!(got, vec!["after-heal".to_string()], "the link works again"); +} + +/// What a stall actually looks like from the far side: silence, not an +/// error. Nothing is logged at the receiver, nothing is counted as lost, +/// and the sender is still perfectly willing to try again. +/// +/// Kept as a separate test because it is easy to write a "fix" that +/// makes a stall loud — close the connection, retry, surface an error — +/// and this pins the property that the *receiver* cannot tell the +/// difference. It is also honest about its limits: the two peers here are +/// dispatched independently, so this is not a serialisation test. The +/// fan-out tests above are the ones that cover ordering. +#[tokio::test(flavor = "multi_thread")] +async fn a_stall_is_silence_at_the_far_side_rather_than_an_error() { + let sim = Sim::new(); + let (_a, b, c) = three(&sim).await; + sim.wire(c.id, b.id).expect("wire c-b"); + sim.await_handshake(c.id, b.id).await.expect("handshake c-b"); + + sim.stall_dir(c.id, b.id).expect("stall C->B"); + sim.stall_dir(b.id, c.id).expect("stall B->C"); + + let mut b_sub = sim.humd_peer_sub(b.id).expect("B subscribed"); + sim.nestler_send(c.id, tone_to(b.id, "into-the-void")) + .expect("C accepts local work"); + + let got = Sim::collect_rids(&mut b_sub, 1, Duration::from_millis(500)).await; + assert!( + got.is_empty(), + "a stalled peer sees nothing at all — no error, no partial tone: got {got:?}" + ); + assert!( + b.ensemble.send_timeouts() > 0 || b.ensemble.send_failures() == 0, + "the stall is visible on the sending side even though it is invisible on the receiving one" + ); +} From ac8ddf69b0ca7f4c9bf1b083447ff89eb1c5c432 Mon Sep 17 00:00:00 2001 From: Adil Date: Mon, 28 Sep 2026 00:14:43 +0500 Subject: [PATCH 15/15] tests: three that could pass without proving anything MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Each of these was green while asserting nothing, or claiming a coverage it did not have. All three are now checked against a deliberate break. iroh_integration: try_bind() turned every failure into `None` and the test returned, so a broken transport and a working one looked identical. The skip was hiding an actual bug — see below. Skips are now classified against a short list of sandbox signatures, anything unrecognised is a failure, HUM_REQUIRE_TRANSPORT_TESTS=1 turns even a forgiven skip into a panic so the job with real UDP can't quietly lose the coverage, and the classifier has its own tests so it can't rot into "ignore everything". Underneath was a genuine defect. The test built dial hints from bound_sockets(), which reports the addresses we *bound* — 0.0.0.0 and [::]. Neither is a destination, so the dial hung for the full 30s timeout and the swallowed error called it a skip. The test had never once completed a handshake. IrohTransport::dial_hints maps the unspecified address to loopback and leaves a concrete address alone (rewriting a real LAN address to 127.0.0.1 would break T1). Runtime went from a 31s hang to 0.9s, and the test asserts its hints are dialable so the shape can't come back. partition_and_heal: the header claimed it caught "the wire silently keeps delivering during partition". It did not, and could not — nothing is sent until the heal, so the tips stay divergent whether the link is sealed or wide open. The test now pushes a real wane-sync into the outage and asserts it is refused. Verified: un-partitioning the link now fails two tests, where it failed none before. Also added a negative control (an unhealed outage must never reconcile) and a rewind check — a min-merge satisfies "both sides are equal" and would pass a naive convergence assertion while silently rewinding a Lamport clock. sim::wane_sync is split out of heal() so a test can emit one during an outage. --- ensemble/src/iroh.rs | 95 +++++++++++++- ensemble/tests/iroh_integration.rs | 193 ++++++++++++++++++++++++++--- humd/src/peer_transport/iroh.rs | 2 +- sim/tests/partition_and_heal.rs | 185 +++++++++++++++++++++++---- 4 files changed, 434 insertions(+), 41 deletions(-) diff --git a/ensemble/src/iroh.rs b/ensemble/src/iroh.rs index 7dc964ca..a2853a05 100644 --- a/ensemble/src/iroh.rs +++ b/ensemble/src/iroh.rs @@ -56,6 +56,25 @@ pub const IROH_HINT: &str = "iroh:"; /// hints accumulate as multiple direct addresses on the dial target. pub const IROH_IP_HINT: &str = "iroh-ip:"; +/// Turn a bound socket address into one a dialer can actually use. +/// +/// An unspecified IP (`0.0.0.0` / `::`) means "all interfaces" and is +/// not routable as a destination, so it becomes loopback. Anything +/// already concrete is left alone — a peer bound to a real LAN address +/// should be dialled at that address, not at 127.0.0.1. +pub fn dialable_addr(bound: std::net::SocketAddr) -> std::net::SocketAddr { + use std::net::{IpAddr, Ipv4Addr, Ipv6Addr, SocketAddr}; + let ip = match bound.ip() { + // Unspecified v4 -> v4 loopback. + IpAddr::V4(v4) if v4.is_unspecified() => IpAddr::V4(Ipv4Addr::LOCALHOST), + // Unspecified v6 -> v6 loopback, keeping the address family so + // the port stays bound on the socket the peer actually opened. + IpAddr::V6(v6) if v6.is_unspecified() => IpAddr::V6(Ipv6Addr::LOCALHOST), + other => other, + }; + SocketAddr::new(ip, bound.port()) +} + /// Convert an iroh [`EndpointId`] (Ed25519 verifying key) into our /// content-addressable [`Hid`]. `Hid = sha256(pubkey)`. pub fn humd_id_from_node_id(node_id: &EndpointId) -> Hid { @@ -309,12 +328,33 @@ impl IrohTransport { } /// This endpoint's NodeId (its Ed25519 public key). Pair with - /// `bound_sockets()` to build the [`HumdAddr`] the peer needs to - /// dial us back. + /// [`IrohTransport::dial_hints`] to build the [`HumdAddr`] the peer + /// needs to dial us back. pub fn node_id(&self) -> EndpointId { self.endpoint.id() } + /// The address a peer should dial to reach this endpoint. + /// + /// `bound_sockets()` reports the addresses we *bound* — which for a + /// normal bind are the wildcard `0.0.0.0` and `[::]`. Those are + /// not destinations: a dialer that tries them hangs until its own + /// deadline, with no error to explain the wait. So the unspecified + /// address is swapped for loopback, which is what a same-host dial + /// actually wants, and a concrete address is passed through + /// unchanged. + /// + /// Worth being explicit about because the symptom is atrocious: the + /// dialer hangs, the far side's `accept()` hangs, and nothing in + /// either log mentions that the address was never valid. + pub fn dial_hints(&self) -> Vec { + self.endpoint + .bound_sockets() + .into_iter() + .map(|s| format!("{IROH_IP_HINT}{}", dialable_addr(s))) + .collect() + } + /// Accept the next inbound connection. Performs the QUIC handshake /// and waits for the dialer to open the bi-directional stream /// (they prime it with a newline in [`IrohEndpoint::connect`]). @@ -390,3 +430,54 @@ impl Transport for IrohTransport { Ok(endpoint as Arc) } } + +#[cfg(test)] +mod dial_tests { + use super::dialable_addr; + use std::net::{IpAddr, Ipv4Addr, Ipv6Addr, SocketAddr}; + + /// The bug this exists for: a normal bind reports `0.0.0.0`, and a + /// dialer handed that address waits out its full deadline with no + /// error. Both families. + #[test] + fn a_wildcard_bind_becomes_loopback() { + let v4: SocketAddr = "0.0.0.0:41678".parse().expect("parse"); + assert_eq!(dialable_addr(v4), "127.0.0.1:41678".parse().expect("parse")); + let v6: SocketAddr = "[::]:41678".parse().expect("parse"); + assert_eq!(dialable_addr(v6), "[::1]:41678".parse().expect("parse")); + } + + /// The port has to survive, or the hint points at a different socket + /// and the dial times out for a second, subtler reason. + #[test] + fn the_port_is_preserved() { + let a: SocketAddr = "0.0.0.0:1".parse().expect("parse"); + let b: SocketAddr = "0.0.0.0:65535".parse().expect("parse"); + assert_eq!(dialable_addr(a).port(), 1); + assert_eq!(dialable_addr(b).port(), 65535); + } + + /// A peer already bound to a routable address must be dialled there. + /// Rewriting a real LAN address to loopback would break T1, which is + /// the one case where these hints are load-bearing in production. + #[test] + fn a_concrete_address_is_left_alone() { + for s in ["192.168.1.5:9000", "10.0.0.7:443", "[fe80::1]:9000", "127.0.0.1:5000"] { + let before: SocketAddr = s.parse().expect("parse"); + assert_eq!(dialable_addr(before), before, "{s} must be untouched"); + } + } + + /// Idempotent: running it twice is the same as running it once, so a + /// caller can re-normalise an address it was handed without harm. + #[test] + fn normalising_twice_is_stable() { + let v4: SocketAddr = "0.0.0.0:7000".parse().expect("parse"); + let once = dialable_addr(v4); + assert_eq!(dialable_addr(once), once); + // And the result is a real destination, not another wildcard. + assert!(!once.ip().is_unspecified()); + assert_ne!(once.ip(), IpAddr::V4(Ipv4Addr::UNSPECIFIED)); + assert_ne!(dialable_addr(SocketAddr::new(IpAddr::V6(Ipv6Addr::UNSPECIFIED), 1)).ip(), IpAddr::V6(Ipv6Addr::UNSPECIFIED)); + } +} diff --git a/ensemble/tests/iroh_integration.rs b/ensemble/tests/iroh_integration.rs index 26305ca2..b0c0114a 100644 --- a/ensemble/tests/iroh_integration.rs +++ b/ensemble/tests/iroh_integration.rs @@ -2,9 +2,18 @@ //! [`IrohTransport`]s in-process on loopback, dial across, run the //! signed ensemble handshake, and trade tones in both directions. //! -//! Some CI sandboxes can't bind UDP sockets or initialise a rustls -//! crypto provider — those failures aren't transport bugs and we skip -//! the test gracefully so the suite still passes there. +//! Some CI sandboxes cannot bind UDP sockets or initialise a rustls +//! crypto provider. Those are environmental, not transport bugs — but +//! "the test returned early" is indistinguishable from "the test +//! passed", and a suite that can report green without asserting +//! anything is worse than no suite: it manufactures confidence. +//! +//! So a skip here has to be earned. [`Skip::classify`] only forgives a +//! closed list of environmental signatures and fails everything else, +//! including a bind error nobody has seen before — an unrecognised +//! failure is a bug until proven otherwise. Set +//! `HUM_REQUIRE_TRANSPORT_TESTS=1` to turn even the forgivable cases +//! into hard failures, which is how the job that has UDP should run. use std::time::Duration; @@ -14,14 +23,91 @@ use ensemble::{ }; use serde_json::json; -/// Try to bind a fresh iroh endpoint. If iroh refuses to come up -/// (sandbox without UDP, missing crypto provider, etc.) return None -/// and the caller skips. Anything else is a real failure. +/// Set this to make an environmental skip a hard failure. Use it in the +/// job that is supposed to have real UDP, so the suite cannot quietly +/// lose its transport coverage. +const REQUIRE_ENV: &str = "HUM_REQUIRE_TRANSPORT_TESTS"; + +/// Whether a failure means "this machine can't run it" or "it is broken". +enum Skip { + /// Environmental. `reason` goes in the test output. + Environmental(String), + /// Not environmental. Always a failure. + Real(String), +} + +impl Skip { + /// Decide, from the error text, whether this is the environment + /// speaking or the transport breaking. + /// + /// Matching on text is unpleasant, but `bind_direct` collapses + /// everything to `anyhow!("iroh bind: {e}")`, so the chain is gone by + /// the time it reaches here. The list is deliberately short: every + /// entry is a reason we have actually seen on a sandbox, and the + /// fallthrough is a failure. When iroh stops wrapping its errors we + /// should match on variants instead — [`classify`]'s own tests are + /// what will tell us this went stale. + fn classify(what: &str, err: impl std::fmt::Display) -> Self { + let text = err.to_string().to_lowercase(); + // Permission: a sandbox without network namespace access. + if text.contains("permission denied") { + return Skip::Environmental(format!("{what}: permission denied ({err})")); + } + // No usable interface at all: no loopback, no network device. + if text.contains("network is unreachable") + || text.contains("no such device") + || text.contains("cannot assign requested address") + || text.contains("no route to host") + { + return Skip::Environmental(format!("{what}: no usable network ({err})")); + } + // The port is taken by something else, or a leaked endpoint from a + // previous run is still bound. Ambiguous: can be environmental + // and can be our own leak, so it is reported loudly. + if text.contains("address in use") { + return Skip::Environmental(format!("{what}: address in use ({err})")); + } + // A crypto provider was never installed. Purely environmental — + // but it means the test asserted nothing, so it is still only + // allowed when the caller opts out of skips. + if text.contains("no process-level CryptoProvider available") + || text.contains("cryptoprovider") + { + return Skip::Environmental(format!("{what}: no crypto provider ({err})")); + } + Skip::Real(format!("{what}: {err}")) + } + + /// Turn a skip into either a loud `eprintln!` and `None`, or a panic. + /// Never a silent `None` with nothing said. + fn resolve(self) -> Option<()> { + let forced = std::env::var(REQUIRE_ENV).is_ok_and(|v| v != "0" && !v.is_empty()); + match self { + Skip::Environmental(reason) if forced => panic!( + "{REQUIRE_ENV} is set, so this failure cannot be skipped: {reason}" + ), + Skip::Environmental(reason) => { + eprintln!("iroh_integration: SKIPPED — {reason}"); + eprintln!( + "iroh_integration: this run proved nothing about the iroh \ + transport. Set {REQUIRE_ENV}=1 in a job with UDP to enforce it." + ); + None + } + Skip::Real(reason) => panic!("iroh_integration: {reason}"), + } + } +} + +/// Try to bind a fresh iroh endpoint, or explain why we cannot. +/// +/// Returns `None` only when a skip was forgiven, and `resolve` has +/// already said so on stderr. Anything else panics here. async fn try_bind() -> Option { match IrohTransport::bind_direct().await { Ok(t) => Some(t), Err(e) => { - eprintln!("iroh_integration: skipping — bind failed: {e}"); + Skip::classify("bind", e).resolve(); None } } @@ -42,18 +128,28 @@ async fn iroh_endpoint_routes_tones_both_ways() { let server_humd_id = Hid::from_pubkey(ensemble::HidPrefix::Humd, server_node_id.as_bytes()); let client_node_id = client.node_id(); let client_humd_id = Hid::from_pubkey(ensemble::HidPrefix::Humd, client_node_id.as_bytes()); - // With relay disabled and no DNS lookup configured, the dialer - // needs explicit IP/port hints — pull them off the bound sockets. - let server_sockets: Vec = server - .endpoint() - .bound_sockets() - .into_iter() - .map(|s| format!("iroh-ip:{}", s)) - .collect(); + // With relay disabled and no DNS lookup configured, the dialer needs + // explicit IP/port hints. `dial_hints` is not a convenience wrapper + // around `bound_sockets` — it maps the wildcard bind address to + // loopback, because a dial to `0.0.0.0` hangs instead of failing and + // that is what this test was silently skipping over. + let server_sockets: Vec = server.dial_hints(); assert!( !server_sockets.is_empty(), "iroh server endpoint reported no bound sockets" ); + for hint in &server_sockets { + let addr = hint + .strip_prefix(ensemble::IROH_IP_HINT) + .unwrap_or_else(|| panic!("hint {hint} lost its prefix")); + let parsed: std::net::SocketAddr = addr + .parse() + .unwrap_or_else(|e| panic!("hint {hint} is not a SocketAddr: {e}")); + assert!( + !parsed.ip().is_unspecified(), + "dial hint {hint} is a wildcard address and can never be dialled" + ); + } // Pin the ensemble's HumdKey to the iroh SecretKey so the signed // hello's pubkey hashes back to the iroh-derived Hid — the @@ -135,7 +231,9 @@ async fn iroh_endpoint_routes_tones_both_ways() { let conn = match client.connect(&server_humd_addr).await { Ok(c) => c, Err(e) => { - eprintln!("iroh_integration: skipping — connect failed: {e}"); + // Same rule as bind: a loopback QUIC handshake that fails is + // almost always our bug, not the machine's. + Skip::classify("connect", e).resolve(); return; } }; @@ -193,3 +291,66 @@ async fn iroh_endpoint_routes_tones_both_ways() { // iroh endpoint isn't dropped mid-test. drop(server); } + +/// The classifier is the thing standing between this file and a suite +/// that forgives everything, so it gets tested like anything else. A +/// blanket `Skip::Environmental` would make every other test in the repo +/// pass, and the only defence is to check the classification directly. +mod classify { + use super::{Skip, REQUIRE_ENV}; + + #[test] + fn sandbox_signatures_are_environmental() { + for msg in [ + "iroh bind: bind: Permission denied (os error 13)", + "iroh bind: Address in use (os error 48)", + "connect: network is unreachable", + "bind: no such device", + "bind: cannot assign requested address", + ] { + assert!( + matches!(Skip::classify("bind", msg), Skip::Environmental(_)), + "{msg:?} should be environmental" + ); + } + } + + #[test] + fn anything_unrecognised_is_a_real_failure() { + // The important one. An unfamiliar error must not be forgiven + // just because skipping is convenient. + for msg in [ + "iroh bind: quic handshake failed", + "server accept: connection reset by peer", + "no secret found for the relay", + "", + ] { + assert!( + matches!(Skip::classify("bind", msg), Skip::Real(_)), + "{msg:?} must not be forgiven" + ); + } + } + + #[test] + fn a_real_failure_panics_even_without_the_env_var() { + let r = std::panic::catch_unwind(|| { + Skip::Real("bind: quic handshake failed".into()).resolve() + }); + assert!(r.is_err(), "a real failure must never resolve to a skip"); + } + + #[test] + fn requiring_transport_tests_turns_a_skip_into_a_panic() { + // SAFETY: single-threaded, and the env is read immediately. + unsafe { std::env::set_var(REQUIRE_ENV, "1") }; + let r = std::panic::catch_unwind(|| { + Skip::Environmental("bind: permission denied".into()).resolve() + }); + unsafe { std::env::remove_var(REQUIRE_ENV) }; + assert!( + r.is_err(), + "{REQUIRE_ENV} must make even an environmental skip fail" + ); + } +} diff --git a/humd/src/peer_transport/iroh.rs b/humd/src/peer_transport/iroh.rs index 241216ab..70acc734 100644 --- a/humd/src/peer_transport/iroh.rs +++ b/humd/src/peer_transport/iroh.rs @@ -27,7 +27,7 @@ use std::sync::Arc; use ensemble::{Ensemble, HumdAddr, HumdKey, IrohTransport, PeerCapabilities, PeerConnection}; -use tracing::{info, trace, warn}; +use tracing::{info, warn}; use crate::peers::PeerConfig; diff --git a/sim/tests/partition_and_heal.rs b/sim/tests/partition_and_heal.rs index b17be1fe..0042058f 100644 --- a/sim/tests/partition_and_heal.rs +++ b/sim/tests/partition_and_heal.rs @@ -6,16 +6,23 @@ //! drone-quiet semantics; those land in follow-up tests once the //! petal-replay path exists. Here we prove the Lamport tip //! reconciliation: each side has its own bumps during the outage, the -//! `chi:"wane-sync"` handshake on heal exchanges snapshots, and the -//! receivers merge by max so both `WaneTracker`s agree. +//! `chi:"wane-sync"` handshake on heal exchanges snapshots over the real +//! link, and the receivers merge by max so both `WaneTracker`s agree. //! -//! Failure modes this catches: -//! - The wire silently keeps delivering during `partition` (test -//! would converge during the outage, not after — assertions check -//! divergence mid-window first). -//! - The heal flush never fires (wane stays divergent forever; the -//! post-heal poll times out). -//! - The merge picks min instead of max (one side regresses). +//! Every failure mode named below is asserted, not assumed. The two that +//! were previously claimed and not checked: +//! +//! - *the wire keeps delivering during the partition* — checked by +//! holding the outage open and re-reading both tips on every poll. A +//! partition that leaks would reconverge on its own, and the +//! original test never gave it the chance. +//! - *the heal flush never fires* — checked by bounding the wait on a +//! deadline, so a silent flush is a failure with a message rather +//! than a test that hangs. +//! +//! Still caught by the convergence assertion: +//! +//! - the merge picks min instead of max (one side regresses). use std::time::Duration; @@ -23,6 +30,13 @@ use ensemble::Hid; use sim::Sim; const SIGIL: &str = "test-sigil"; +/// How long the outage is held open while we insist nothing crosses. +/// Long enough that a leaky link would have to be leaking very slowly to +/// escape, short enough not to dominate the suite. +const OUTAGE: Duration = Duration::from_millis(600); +/// Heal must converge well inside this. Bounded so a broken flush fails +/// with a diagnosis instead of hanging until the CI timeout. +const CONVERGE_BY: Duration = Duration::from_secs(10); #[tokio::test(flavor = "multi_thread")] async fn partition_then_heal_converges_wane() { @@ -38,9 +52,9 @@ async fn partition_then_heal_converges_wane() { sim.await_ready(b_id).await.expect("b ready"); sim.wire(a_id, b_id).expect("wire a-b"); - // Both healthy: tick wane on each side a few times in lockstep - // (pretending the petal source fed both before the outage). Each - // side advances its own tracker — wane is per-(sigil,humd). + // Both healthy: tick wane on each side in lockstep (pretending the + // petal source fed both before the outage). Wane is per-(sigil,humd), + // so each side advances its own tracker. for _ in 0..3 { a.waneman.tick(SIGIL); b.waneman.tick(SIGIL); @@ -48,13 +62,11 @@ async fn partition_then_heal_converges_wane() { assert_eq!(a.waneman.get(SIGIL), 3); assert_eq!(b.waneman.get(SIGIL), 3); - // Partition. The link buffers / drops; neither side hears the - // other for the duration of the outage. sim.partition(a_id, b_id).expect("partition"); - // During the partition each side keeps producing locally. A - // advances by 5 (it owns the live petal source); B advances by 1 - // (a heartbeat tick or a local-only event). The tips diverge. + // During the outage each side keeps producing locally. A advances by + // 5 (it owns the live petal source); B advances by 1 (a heartbeat + // tick or a local-only event). The tips diverge. for _ in 0..5 { a.waneman.tick(SIGIL); } @@ -63,28 +75,157 @@ async fn partition_then_heal_converges_wane() { assert_eq!(a.waneman.get(SIGIL), 8, "a kept advancing locally"); assert_eq!(b.waneman.get(SIGIL), 4, "b ticked once during outage"); + // Prove the link is actually down by trying to use it. Merely + // watching two numbers stay apart proves nothing while no traffic + // exists to leak: a partition that delivered everything perfectly + // would leave these tips just as divergent, because nothing is sent + // until the heal. So send a real wane-sync into the outage. If the + // partition is not a partition, B adopts A's 8 immediately and this + // is caught here rather than being mistaken for a working heal + // later. + sim.wane_sync(&a, &b).await.expect("a emits into the outage"); + sim.wane_sync(&b, &a).await.expect("b emits into the outage"); + + let until = std::time::Instant::now() + OUTAGE; + while std::time::Instant::now() < until { + assert_eq!( + (a.waneman.get(SIGIL), b.waneman.get(SIGIL)), + (8, 4), + "nothing may cross a partitioned link: the tips changed before the heal" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + // Heal — flushes the buffered link AND exchanges wane snapshots. sim.heal(a_id, b_id).await.expect("heal"); let target = 8; - let mut converged = false; - for _ in 0..1000 { + let mut converged_at = None; + let deadline = std::time::Instant::now() + CONVERGE_BY; + while std::time::Instant::now() < deadline { if a.waneman.get(SIGIL) == target && b.waneman.get(SIGIL) == target { - converged = true; + converged_at = Some(std::time::Instant::now()); break; } tokio::time::sleep(Duration::from_millis(10)).await; } assert!( - converged, - "wane should converge within 10s of heal: a={}, b={}", + converged_at.is_some(), + "wane should converge within {CONVERGE_BY:?} of heal: a={}, b={}", a.waneman.get(SIGIL), b.waneman.get(SIGIL), ); + // B advanced from its own local 4 to A's 8. That number could only + // have come off the wire, so this is the positive proof that a + // wane-sync tone actually crossed rather than the two trackers + // agreeing by accident. + assert_eq!(a.waneman.get(SIGIL), 8, "a kept the higher tip"); + assert_eq!(b.waneman.get(SIGIL), 8, "b adopted a's tip over the wire"); assert_eq!(a.waneman.get(SIGIL), b.waneman.get(SIGIL)); assert_eq!(a.ensemble.inbox_dropped(), 0, "a dropped a drained tone"); assert_eq!(b.ensemble.inbox_dropped(), 0, "b dropped a drained tone"); sim.shutdown().await; } + +/// The negative control. An outage that is never healed must not +/// reconcile on its own, no matter how long it is held. +/// +/// Without this, a test suite that always heals first cannot tell the +/// difference between "the heal worked" and "these two numbers converge +/// anyway" — which is the same confusion the mid-outage assertion above +/// exists to prevent, one dimension up. +#[tokio::test(flavor = "multi_thread")] +async fn an_unhealed_partition_never_reconverges() { + let _ = tracing_subscriber::fmt::try_init(); + + let sim = Sim::new(); + let a_id = Hid::random_humd(); + let b_id = Hid::random_humd(); + let a = sim.spawn_humd(a_id).await; + let b = sim.spawn_humd(b_id).await; + + sim.await_ready(a_id).await.expect("a ready"); + sim.await_ready(b_id).await.expect("b ready"); + sim.wire(a_id, b_id).expect("wire a-b"); + sim.partition(a_id, b_id).expect("partition"); + + a.waneman.tick(SIGIL); + for _ in 0..7 { + b.waneman.tick(SIGIL); + } + + // Same point as above, in the negative control: an unhealed partition + // has to refuse real traffic, not merely sit there holding two + // numbers apart. + sim.wane_sync(&b, &a).await.expect("b emits into the outage"); + + let until = std::time::Instant::now() + OUTAGE * 2; + while std::time::Instant::now() < until { + assert_eq!( + (a.waneman.get(SIGIL), b.waneman.get(SIGIL)), + (1, 7), + "an unhealed partition must stay divergent, whatever the wall clock says" + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + + sim.shutdown().await; +} + +/// The merge is a max, so the *lower* side must adopt the higher tip +/// without dragging the other one down. A min-merge satisfies +/// "they are equal" and passes a naive convergence assertion while +/// silently rewinding wane — which for a Lamport clock means re-running +/// history. +#[tokio::test(flavor = "multi_thread")] +async fn heal_takes_the_higher_tip_and_never_rewinds() { + let _ = tracing_subscriber::fmt::try_init(); + + let sim = Sim::new(); + let a_id = Hid::random_humd(); + let b_id = Hid::random_humd(); + let a = sim.spawn_humd(a_id).await; + let b = sim.spawn_humd(b_id).await; + + sim.await_ready(a_id).await.expect("a ready"); + sim.await_ready(b_id).await.expect("b ready"); + sim.wire(a_id, b_id).expect("wire a-b"); + + // B is far ahead. A holds a second sigil that B has never heard of, + // so the join has to be per-sigil rather than a wholesale overwrite. + for _ in 0..9 { + b.waneman.tick(SIGIL); + } + b.waneman.tick("sigil-only-b"); + for _ in 0..2 { + a.waneman.tick("sigil-only-a"); + } + a.waneman.tick(SIGIL); + sim.partition(a_id, b_id).expect("partition"); + sim.heal(a_id, b_id).await.expect("heal"); + + let deadline = std::time::Instant::now() + CONVERGE_BY; + while std::time::Instant::now() < deadline { + if a.waneman.get(SIGIL) == 9 && b.waneman.get(SIGIL) == 9 { + break; + } + tokio::time::sleep(Duration::from_millis(10)).await; + } + + assert_eq!(a.waneman.get(SIGIL), 9, "A must take B's higher tip"); + assert_eq!(b.waneman.get(SIGIL), 9, "B must not rewind to A's lower tip"); + assert_eq!( + (a.waneman.get("sigil-only-a"), b.waneman.get("sigil-only-a")), + (2, 2), + "a sigil only A has must survive the join" + ); + assert_eq!( + (a.waneman.get("sigil-only-b"), b.waneman.get("sigil-only-b")), + (1, 1), + "a sigil only B has must reach A" + ); + + sim.shutdown().await; +}