From b8a4449762a2f580d8a802ac59bcf55ce7402ff4 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:14:10 +0100 Subject: [PATCH 01/23] Refuse a ceiling frame counter and saturate the gap tracker An authenticated peer sending a frame counter of u64::MAX pinned its own replay window's high-water mark at the ceiling, after which every later counter it sent fell more than a window below `highest` and was dropped, wedging that peer's receive path until a rekey replaced the session. The same counter overflowed the MMP gap tracker's expected-counter add, which wraps to zero in a release build and aborts under a build with overflow checks on. Refuse the value in ReplayWindow::check, which covers the FSP session paths and the off-task FMP decrypt worker in one place, and advance the gap tracker with a saturating add. Neither send path in this tree can emit u64::MAX, so no conforming peer notices; u64::MAX - 1, the highest counter an honest peer can send, is unaffected. --- CHANGELOG.md | 18 ++++++++++++++++++ src/mmp/receiver.rs | 29 +++++++++++++++++++++++++++-- src/noise/replay.rs | 8 ++++++++ src/noise/tests.rs | 35 +++++++++++++++++++++++++++++++++++ 4 files changed, 88 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 51947372..d08a2786 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -529,6 +529,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 #### FMP/FSP session integrity +- A frame whose counter is `u64::MAX` is now refused by the replay window + instead of being accepted as a new high-water mark. Accepting it pinned + `highest` at the ceiling, after which every subsequent counter from that peer + fell more than a replay window below it and was rejected, wedging that peer's + own receive path until a rekey replaced the session. The send side already + refuses to emit that counter (`take_send_counter` and `advance_nonce` both + return a nonce-overflow error), so no conforming peer can produce it and the + refusal is invisible on the wire; the highest counter an honest peer can send, + `u64::MAX - 1`, is still accepted. Reaching this required an + already-authenticated peer running modified code, and the damage was confined + to that peer's own session. + +- The MMP gap tracker advances its expected-counter state with a saturating add, + so a received counter of `u64::MAX` no longer overflows it. The wrap silently + reset the expectation to zero in a release build and aborted the task under a + build with overflow checks on, such as the test harness. Behaviour is + unchanged for every counter an honest peer can emit. + - A session setup message naming an already-established peer no longer replaces that peer's session. The handler did this whenever `node.rekey.enabled` was false: it ran a fresh responder handshake and overwrote the entry, discarding diff --git a/src/mmp/receiver.rs b/src/mmp/receiver.rs index 3c3b5e0e..3cbfee1b 100644 --- a/src/mmp/receiver.rs +++ b/src/mmp/receiver.rs @@ -62,7 +62,7 @@ impl GapTracker { fn observe(&mut self, counter: u64) -> u64 { let Some(expected) = self.expected_next else { // First frame: initialize - self.expected_next = Some(counter + 1); + self.expected_next = Some(counter.saturating_add(1)); return 0; }; @@ -91,7 +91,7 @@ impl GapTracker { // Update expected (always advance to counter+1 or keep expected if // this was a late/reordered frame) if counter >= expected { - self.expected_next = Some(counter + 1); + self.expected_next = Some(counter.saturating_add(1)); } lost @@ -560,6 +560,31 @@ mod tests { assert_eq!(mean, 0); } + #[test] + fn test_gap_tracker_saturates_on_first_frame_at_max_counter() { + // Pre-fix this aborts the test process at the unchecked `counter + 1` + // under the dev profile's overflow checks rather than failing an + // assertion; the abort is the red, not a harness fault. + let mut g = GapTracker::new(); + assert_eq!(g.observe(u64::MAX), 0); + // A saturated expectation must simply stop the tracker advancing, and + // every later counter then takes the in-order branch. + assert_eq!(g.observe(u64::MAX), 0); + let (count, max, _mean) = g.take_interval_stats(); + assert_eq!(count, 0); + assert_eq!(max, 0); + } + + #[test] + fn test_gap_tracker_saturates_on_advance_at_max_counter() { + // Primes the tracker first so the advance branch, not the first-frame + // branch, is the site that would overflow. + let mut g = GapTracker::new(); + g.observe(10); + g.observe(u64::MAX); + assert_eq!(g.observe(u64::MAX), 0); + } + #[test] fn test_gap_tracker_single_burst() { let mut g = GapTracker::new(); diff --git a/src/noise/replay.rs b/src/noise/replay.rs index 2e4e6443..99217e05 100644 --- a/src/noise/replay.rs +++ b/src/noise/replay.rs @@ -31,6 +31,14 @@ impl ReplayWindow { /// Returns true if the counter is acceptable, false if it should be rejected. /// Does NOT update the window - call `accept` after successful decryption. pub fn check(&self, counter: u64) -> bool { + // The send side refuses to emit u64::MAX (`CipherState::advance_nonce`, + // `NoiseSession::take_send_counter`), so no conforming peer produces it. + // Refusing it here keeps `accept` from pinning `highest` at the ceiling, + // which would wedge the window against every later counter. + if counter == u64::MAX { + return false; + } + if counter > self.highest { // New highest - always acceptable return true; diff --git a/src/noise/tests.rs b/src/noise/tests.rs index 82cd7a93..baabfc04 100644 --- a/src/noise/tests.rs +++ b/src/noise/tests.rs @@ -387,6 +387,41 @@ fn test_replay_window_reset() { assert!(window.check(100)); } +#[test] +fn test_replay_window_max_counter_does_not_wedge_the_window() { + let mut window = ReplayWindow::new(); + + window.accept(100); + // Mirror the real receive path: check, and accept only if the check passed. + if window.check(u64::MAX) { + window.accept(u64::MAX); + } + + // The honest peer's next frame must still be acceptable. + assert!(window.check(101), "ceiling frame wedged the window"); +} + +#[test] +fn test_replay_window_rejects_max_counter() { + let window = ReplayWindow::new(); + assert!(!window.check(u64::MAX)); +} + +#[test] +fn test_replay_window_accepts_highest_counter_an_honest_peer_can_send() { + // take_send_counter refuses u64::MAX, so u64::MAX - 1 is the highest + // counter a conforming peer emits. The ceiling guard must be exactly one + // value wide and leave that one alone. + let mut window = ReplayWindow::new(); + assert!(window.check(u64::MAX - 1)); + + window.accept(u64::MAX - 1); + assert!( + !window.check(u64::MAX - 1), + "replay should still be rejected" + ); +} + #[test] fn test_session_replay_protection() { let keypair1 = generate_keypair(); From f48113a80857128d7eda937e545b6ab6bd48059c Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:21:59 +0100 Subject: [PATCH 02/23] Bound the UDP transport's DNS cache and make it evict The cache held one entry per distinct hostname string ever dialed. The TTL was applied only on the read, so a stale entry was overwritten on the next dial of the same name and otherwise stayed for the life of the process, and nothing in the tree removed from the map at all. Under a rendezvous policy that turns advertised endpoints into dial candidates the keys are strings a remote party chose, so the growth is theirs to drive. Move the two cache touches into testable free functions. A store now refreshes a name already present without evicting anything, otherwise sweeps every entry past its TTL and, if the map is still full, drops the oldest until it is not. The bound is 256 hostnames as a module constant, well above what any configured peer list produces. Eviction is by insertion time rather than last use: the timestamp is already there as the TTL clock, and tracking last use would mean writing to the map on the read path of every dial. The cost of a wrong eviction is one DNS lookup, not a failed dial. --- CHANGELOG.md | 12 +++ src/transport/udp/mod.rs | 182 +++++++++++++++++++++++++++++++++++++-- 2 files changed, 189 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d08a2786..fe06b2db 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -433,6 +433,18 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 #### Transports & config +- The UDP transport's DNS cache is now bounded and actually evicts. The map + held one entry per distinct hostname string ever dialed, and the TTL was + applied only on the read, so a stale entry was overwritten on the next dial + of the same name and otherwise stayed for the life of the process. Under a + rendezvous policy that accepts advertised endpoints the keys are strings a + remote party chose, which made the growth theirs to drive. A store now + sweeps entries past their TTL and, if the map is still full, drops the + oldest, holding it to 256 hostnames. Refreshing a name already cached + evicts nothing. Eviction is by insertion time rather than last use, so a + rarely dialed name in a very large peer list may re-resolve more often; the + cost of a wrong eviction is one DNS lookup, not a failed dial. + - A failed private-key write no longer leaves a node silently running an ephemeral identity. Six write results in the identity path were discarded, and the sharpest was in `persistent` mode: a failed write to `fips.key` fell diff --git a/src/transport/udp/mod.rs b/src/transport/udp/mod.rs index 83de32df..81c5b4eb 100644 --- a/src/transport/udp/mod.rs +++ b/src/transport/udp/mod.rs @@ -29,6 +29,18 @@ use tracing::{debug, info, trace, warn}; /// DNS cache TTL for hostname resolution (60 seconds). const DNS_CACHE_TTL: Duration = Duration::from_secs(60); +/// Upper bound on the number of hostnames the DNS cache holds at once. +/// +/// The cache is keyed by the address string a dial was asked for, and under a +/// rendezvous policy that accepts advertised endpoints those strings come from +/// remote parties, so without a bound the map grows for the life of the +/// process. 256 sits about two orders of magnitude above the number of +/// distinct hostnames a configured peer list produces, so no ordinary +/// deployment reaches it. Lowering it starts to be reachable by a large peer +/// list, and the only cost of an eviction is one extra DNS lookup on the next +/// dial of that name; raising it buys nothing but resident memory. +const DNS_CACHE_MAX_ENTRIES: usize = 256; + /// UDP transport for FIPS. /// /// Provides connectionless, unreliable packet delivery over UDP/IP. @@ -145,10 +157,8 @@ impl UdpTransport { // Check cache { let cache = self.dns_cache.lock().unwrap_or_else(|e| e.into_inner()); - if let Some((resolved, cached_at)) = cache.get(addr) - && cached_at.elapsed() < DNS_CACHE_TTL - { - return Ok(*resolved); + if let Some(resolved) = cache_lookup(&cache, addr, Instant::now()) { + return Ok(resolved); } } @@ -158,7 +168,13 @@ impl UdpTransport { // Store in cache { let mut cache = self.dns_cache.lock().unwrap_or_else(|e| e.into_inner()); - cache.insert(addr.clone(), (resolved, Instant::now())); + cache_store( + &mut cache, + addr.clone(), + resolved, + Instant::now(), + DNS_CACHE_MAX_ENTRIES, + ); } Ok(resolved) @@ -601,6 +617,56 @@ async fn udp_receive_loop( } } +/// A cached resolution for `key`, if one is present and still inside +/// `DNS_CACHE_TTL` at `now`. +fn cache_lookup( + cache: &HashMap, + key: &TransportAddr, + now: Instant, +) -> Option { + cache + .get(key) + .filter(|(_, cached_at)| now.duration_since(*cached_at) < DNS_CACHE_TTL) + .map(|(resolved, _)| *resolved) +} + +/// Record a resolution, keeping the cache at or below `cap` entries. +/// +/// Refreshing a name already present never evicts anything. Otherwise every +/// entry past its TTL is dropped first, and only if that leaves the map full +/// is the oldest remaining entry evicted. Eviction is by insertion time rather +/// than by last use: the timestamp is already there as the TTL clock, and +/// tracking last use would mean writing to the map on the read path of every +/// dial. The sweep is linear in `cap` and runs only on a resolution miss, so +/// at most once per TTL per name. +fn cache_store( + cache: &mut HashMap, + key: TransportAddr, + resolved: SocketAddr, + now: Instant, + cap: usize, +) { + if let Some(entry) = cache.get_mut(&key) { + *entry = (resolved, now); + return; + } + + cache.retain(|_, (_, cached_at)| now.duration_since(*cached_at) < DNS_CACHE_TTL); + + while cache.len() >= cap { + let Some(oldest) = cache + .iter() + .min_by_key(|(_, (_, cached_at))| *cached_at) + .map(|(key, _)| key.clone()) + else { + break; + }; + cache.remove(&oldest); + } + + cache.insert(key, (resolved, now)); +} + // ============================================================================ // Tests // ============================================================================ @@ -611,6 +677,112 @@ mod tests { use crate::transport::packet_channel; use tokio::time::{Duration, timeout}; + /// A distinct hostname key, so each store is a fresh entry. + fn dns_key(n: usize) -> TransportAddr { + TransportAddr::from(format!("host{n}.example:2121")) + } + + fn dns_value() -> SocketAddr { + "198.51.100.1:2121".parse().unwrap() + } + + /// The cache is keyed by strings a remote party can choose, so its size + /// has to be bounded no matter how many distinct names are dialed. + #[test] + fn dns_cache_store_refuses_to_exceed_the_cap() { + const CAP: usize = 8; + let now = Instant::now(); + let mut cache = HashMap::new(); + + for n in 0..CAP + 5 { + cache_store(&mut cache, dns_key(n), dns_value(), now, CAP); + assert!( + cache.len() <= CAP, + "cache grew to {} entries past a cap of {CAP}", + cache.len() + ); + } + } + + /// A stale entry used to be overwritten on the next dial of the same name + /// and otherwise never removed, so a name dialed once sat there forever. + #[test] + fn dns_cache_store_evicts_entries_past_their_ttl() { + let now = Instant::now(); + let expired_at = now.checked_sub(DNS_CACHE_TTL * 2).expect("monotonic clock"); + let mut cache = HashMap::new(); + cache.insert(dns_key(0), (dns_value(), expired_at)); + + cache_store( + &mut cache, + dns_key(1), + dns_value(), + now, + DNS_CACHE_MAX_ENTRIES, + ); + + assert!( + !cache.contains_key(&dns_key(0)), + "an entry past its TTL should be swept, not left to accumulate" + ); + assert!(cache_lookup(&cache, &dns_key(0), now).is_none()); + assert!(cache_lookup(&cache, &dns_key(1), now).is_some()); + } + + /// With nothing expired, the cap is enforced by dropping the oldest entry. + /// The ages here are all well inside the TTL, so the expiry sweep cannot + /// be what makes room and the eviction branch is the one under test. + #[test] + fn dns_cache_store_evicts_the_oldest_entry_when_every_entry_is_fresh() { + const CAP: usize = 4; + let now = Instant::now(); + let mut cache = HashMap::new(); + for n in 0..CAP { + let age = Duration::from_secs((CAP - n) as u64); + assert!(age < DNS_CACHE_TTL, "fixture must stay inside the TTL"); + let cached_at = now.checked_sub(age).expect("monotonic clock"); + cache.insert(dns_key(n), (dns_value(), cached_at)); + } + assert_eq!(cache.len(), CAP, "no entry should be expired going in"); + + cache_store(&mut cache, dns_key(CAP), dns_value(), now, CAP); + + assert_eq!(cache.len(), CAP); + assert!( + !cache.contains_key(&dns_key(0)), + "the oldest entry should be the one evicted" + ); + for n in 1..=CAP { + assert!( + cache.contains_key(&dns_key(n)), + "entry {n} should have survived" + ); + } + } + + /// Re-resolving a name already cached is the common case on a live node. + /// It must not cost another entry its place. + #[test] + fn dns_cache_store_refreshing_an_existing_key_evicts_nothing() { + const CAP: usize = 4; + let now = Instant::now(); + let mut cache = HashMap::new(); + for n in 0..CAP { + let cached_at = now + .checked_sub(Duration::from_secs((CAP - n) as u64)) + .expect("monotonic clock"); + cache.insert(dns_key(n), (dns_value(), cached_at)); + } + + cache_store(&mut cache, dns_key(0), dns_value(), now, CAP); + + assert_eq!(cache.len(), CAP); + for n in 0..CAP { + assert!(cache.contains_key(&dns_key(n)), "entry {n} should remain"); + } + assert_eq!(cache_lookup(&cache, &dns_key(0), now), Some(dns_value())); + } + fn make_config(port: u16) -> UdpConfig { UdpConfig { bind_addr: Some(format!("127.0.0.1:{}", port)), From 759fee50f569c3e6157aa10034cf4823b0a77c46 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:20:54 +0100 Subject: [PATCH 03/23] Bound the Ethernet discovery buffer and drop its quadratic dedup Beacons are unauthenticated broadcast frames, and the discovery buffer deduplicated them by scanning a Vec for the source MAC with no cap on how many it held. Anything on the segment could name a fresh MAC per frame and drive both quadratic CPU in the receive loop and unbounded memory. The once-per-tick drain does not bound it either: a transport that is still receiving but not operational is skipped before draining. Key the buffer on source MAC and cap it at 1024 distinct MACs between drains. Keep the drain order oldest sighting first, because the reconcile layer spends a finite connect budget in that order and it must not depend on hash iteration order. A MAC already buffered is refreshed whether or not the buffer is full, so a flood of new MACs cannot crowd out a neighbour already seen. Count refused beacons in the transport stats as beacons_dropped, and log on the first drop and then on each power of ten, so the flooder does not choose the log rate. --- CHANGELOG.md | 18 +++ src/transport/ethernet/discovery.rs | 191 ++++++++++++++++++++++++++-- src/transport/ethernet/mod.rs | 4 +- src/transport/ethernet/stats.rs | 9 ++ src/transport/mod.rs | 1 + 5 files changed, 210 insertions(+), 13 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index fe06b2db..9ac36e73 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -795,6 +795,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 #### Admission / peer caps +- The Ethernet transport's discovery buffer is now bounded and no longer costs + a linear scan per beacon. Beacons are unauthenticated broadcast frames, and + the buffer deduplicated by scanning a `Vec` for the source MAC and had no + cap, so anything on the segment could name a fresh MAC per frame and drive + both quadratic CPU in the receive loop and unbounded memory. It is drained + once per tick only while the transport is operational, so a transport that + is receiving but not operational was never drained at all. The buffer is now + a map keyed on source MAC, capped at 1024 distinct MACs between drains, with + the drain order still oldest sighting first so which neighbour gets dialed + under a connect budget does not depend on hash iteration order. A MAC already + buffered is always refreshed, so a flood of new MACs cannot crowd out a + neighbour already seen. Refused beacons are counted in the transport's stats + as `beacons_dropped` and reported in the log on the first drop and then on + each power-of-ten thereafter, so the flooder does not set the log rate. + **What this does not close**: a flood can still crowd out a neighbour not yet + seen in that tick, and anything able to flood raw frames on the segment can + already jam the beacon at L2 more cheaply. + - An accepted inbound TCP connection no longer holds a slot indefinitely without sending anything. The cap was tested at accept and the pool insert and counter bump followed with no read in between, while the frame reader's diff --git a/src/transport/ethernet/discovery.rs b/src/transport/ethernet/discovery.rs index 26ff75e9..d5528da7 100644 --- a/src/transport/ethernet/discovery.rs +++ b/src/transport/ethernet/discovery.rs @@ -7,7 +7,10 @@ use crate::transport::{DiscoveredPeer, TransportAddr, TransportId}; use secp256k1::XOnlyPublicKey; +use std::collections::HashMap; +use std::collections::hash_map::Entry; use std::sync::Mutex; +use tracing::warn; /// Discovery protocol version. pub const DISCOVERY_VERSION: u8 = 0x01; @@ -46,10 +49,33 @@ pub fn parse_beacon(data: &[u8]) -> Option { XOnlyPublicKey::from_slice(&data[2..34]).ok() } +/// Maximum distinct source MACs held between drains. +/// +/// Beacons are unauthenticated broadcast frames, so anything on the segment +/// can name as many source MACs as it likes; without a bound the buffer grows +/// with the flood rate, and it is not drained at all while the transport is +/// not operational. This caps it at roughly a thousand small structs, tens of +/// kilobytes. Raising it costs that much more memory per transport; lowering +/// it risks truncating discovery on a very large segment. A thousand distinct +/// beaconing FIPS neighbors within one tick is far outside anything a real +/// deployment produces. +const MAX_BUFFERED_PEERS: usize = 1024; + /// Buffer for discovered peers, drained by `discover()`. pub struct DiscoveryBuffer { transport_id: TransportId, - peers: Mutex>, + peers: Mutex, +} + +/// Peers keyed by source MAC, plus the sighting order `take()` restores. +#[derive(Default)] +struct Buffered { + by_mac: HashMap<[u8; 6], (u64, DiscoveredPeer)>, + seq: u64, + /// Beacons refused for want of room, cumulative and never reset. + dropped: u64, + /// Cumulative drop count that earns the next log record. + warn_at: u64, } impl DiscoveryBuffer { @@ -57,25 +83,82 @@ impl DiscoveryBuffer { pub fn new(transport_id: TransportId) -> Self { Self { transport_id, - peers: Mutex::new(Vec::new()), + peers: Mutex::new(Buffered::default()), } } /// Add a discovered peer from a received beacon. - pub fn add_peer(&self, src_mac: [u8; 6], pubkey: XOnlyPublicKey) { - let addr = TransportAddr::from_bytes(&src_mac); - let peer = DiscoveredPeer::with_hint(self.transport_id, addr, pubkey); - let mut peers = self.peers.lock().unwrap_or_else(|e| e.into_inner()); - // Deduplicate by MAC address — keep the latest - peers.retain(|p| p.addr.as_bytes() != src_mac); - peers.push(peer); + /// + /// Returns false when the beacon was refused because the buffer is full. + /// A MAC already buffered is always refreshed, so a flood of new MACs + /// cannot stop a known neighbor from being seen again. + pub fn add_peer(&self, src_mac: [u8; 6], pubkey: XOnlyPublicKey) -> bool { + let mut buffered = self.peers.lock().unwrap_or_else(|e| e.into_inner()); + buffered.seq += 1; + let seq = buffered.seq; + let full = buffered.by_mac.len() >= MAX_BUFFERED_PEERS; + let stored = match buffered.by_mac.entry(src_mac) { + // Refreshing moves the MAC to the end, as retain-then-push did. + Entry::Occupied(mut slot) => { + slot.insert((seq, self.peer(src_mac, pubkey))); + true + } + Entry::Vacant(slot) if !full => { + slot.insert((seq, self.peer(src_mac, pubkey))); + true + } + Entry::Vacant(_) => false, + }; + if !stored { + buffered.dropped += 1; + } + stored } - /// Drain all discovered peers since the last call. + /// Drain all discovered peers since the last call, oldest sighting first. pub fn take(&self) -> Vec { - let mut peers = self.peers.lock().unwrap_or_else(|e| e.into_inner()); - std::mem::take(&mut *peers) + let mut buffered = self.peers.lock().unwrap_or_else(|e| e.into_inner()); + let mut ordered: Vec<(u64, DiscoveredPeer)> = + buffered.by_mac.drain().map(|(_, entry)| entry).collect(); + // The reconcile layer spends a finite connect budget in this order, so + // which neighbor gets dialed must not depend on hash iteration order. + ordered.sort_unstable_by_key(|(seq, _)| *seq); + // Rate-limited: the drop rate is whatever the flooder chooses, and one + // record per drain would hand it the log volume too. + if buffered.dropped >= buffered.warn_at.max(1) { + warn!( + transport_id = %self.transport_id, + dropped = buffered.dropped, + cap = MAX_BUFFERED_PEERS, + "discovery buffer full, beacons from unseen neighbors refused" + ); + buffered.warn_at = next_decade(buffered.dropped); + } + ordered.into_iter().map(|(_, peer)| peer).collect() } + + /// Beacons refused for want of room since this buffer was created. + pub fn dropped(&self) -> u64 { + self.peers.lock().unwrap_or_else(|e| e.into_inner()).dropped + } + + /// Build the buffered peer record for one beacon. + fn peer(&self, src_mac: [u8; 6], pubkey: XOnlyPublicKey) -> DiscoveredPeer { + let addr = TransportAddr::from_bytes(&src_mac); + DiscoveredPeer::with_hint(self.transport_id, addr, pubkey) + } +} + +/// Smallest power of ten strictly greater than `n`, saturating at `u64::MAX`. +fn next_decade(n: u64) -> u64 { + let mut threshold = 1u64; + while threshold <= n { + match threshold.checked_mul(10) { + Some(next) => threshold = next, + None => return u64::MAX, + } + } + threshold } // ============================================================================ @@ -163,4 +246,88 @@ mod tests { let peers = buffer.take(); assert_eq!(peers.len(), 1); } + + /// Distinct MAC number `n`, for filling the buffer. + fn nth_mac(n: usize) -> [u8; 6] { + let bytes = (n as u64).to_be_bytes(); + [0x02, bytes[3], bytes[4], bytes[5], bytes[6], bytes[7]] + } + + #[test] + fn discovery_buffer_stops_buffering_past_the_cap() { + // The defect: an unauthenticated flood of source MACs grew the buffer + // without bound. Fails against the uncapped Vec, which returns all of + // them. + let buffer = DiscoveryBuffer::new(TransportId::new(1)); + let pubkey = test_pubkey(); + for n in 0..MAX_BUFFERED_PEERS + 50 { + buffer.add_peer(nth_mac(n), pubkey); + } + + let peers = buffer.take(); + assert_eq!(peers.len(), MAX_BUFFERED_PEERS); + // Drop-new keeps the earliest sightings. + assert_eq!(peers[0].addr.as_bytes(), &nth_mac(0)); + } + + #[test] + fn discovery_buffer_counts_dropped_beacons() { + let buffer = DiscoveryBuffer::new(TransportId::new(1)); + let pubkey = test_pubkey(); + for n in 0..MAX_BUFFERED_PEERS { + assert!(buffer.add_peer(nth_mac(n), pubkey)); + } + for n in MAX_BUFFERED_PEERS..MAX_BUFFERED_PEERS + 7 { + assert!(!buffer.add_peer(nth_mac(n), pubkey)); + } + + assert_eq!(buffer.dropped(), 7); + } + + #[test] + fn discovery_buffer_repeat_beacon_from_a_full_buffer_still_refreshes() { + let buffer = DiscoveryBuffer::new(TransportId::new(1)); + let pubkey = test_pubkey(); + for n in 0..MAX_BUFFERED_PEERS + 50 { + buffer.add_peer(nth_mac(n), pubkey); + } + // A neighbor already buffered must not be refused by a full buffer. + assert!(buffer.add_peer(nth_mac(0), pubkey)); + + let peers = buffer.take(); + assert_eq!(peers.len(), MAX_BUFFERED_PEERS); + assert_eq!(peers[peers.len() - 1].addr.as_bytes(), &nth_mac(0)); + } + + #[test] + fn discovery_buffer_drain_preserves_last_seen_order() { + // A regression pin on the map rewrite rather than a test of the + // defect: retain-then-push already produced this order. + let buffer = DiscoveryBuffer::new(TransportId::new(1)); + let pubkey = test_pubkey(); + let a = [0xaa; 6]; + let b = [0xbb; 6]; + let c = [0xcc; 6]; + + buffer.add_peer(a, pubkey); + buffer.add_peer(b, pubkey); + buffer.add_peer(c, pubkey); + buffer.add_peer(a, pubkey); + + let macs: Vec<_> = buffer + .take() + .iter() + .map(|p| p.addr.as_bytes().to_vec()) + .collect(); + assert_eq!(macs, vec![b.to_vec(), c.to_vec(), a.to_vec()]); + } + + #[test] + fn next_decade_steps_by_powers_of_ten() { + assert_eq!(next_decade(0), 1); + assert_eq!(next_decade(1), 10); + assert_eq!(next_decade(9), 10); + assert_eq!(next_decade(10), 100); + assert_eq!(next_decade(u64::MAX), u64::MAX); + } } diff --git a/src/transport/ethernet/mod.rs b/src/transport/ethernet/mod.rs index 60b0dbf1..ece1a804 100644 --- a/src/transport/ethernet/mod.rs +++ b/src/transport/ethernet/mod.rs @@ -446,7 +446,9 @@ async fn ethernet_receive_loop( stats.record_beacon_recv(); if discovery_enabled && let Some(pubkey) = parse_beacon(&buf[..len]) { - discovery_buffer.add_peer(src_mac, pubkey); + if !discovery_buffer.add_peer(src_mac, pubkey) { + stats.record_beacon_dropped(); + } trace!( transport_id = %transport_id, remote_mac = %format_mac(&src_mac), diff --git a/src/transport/ethernet/stats.rs b/src/transport/ethernet/stats.rs index c20492fd..02cb2d14 100644 --- a/src/transport/ethernet/stats.rs +++ b/src/transport/ethernet/stats.rs @@ -15,6 +15,7 @@ pub struct EthernetStats { pub recv_errors: AtomicU64, pub beacons_sent: AtomicU64, pub beacons_recv: AtomicU64, + pub beacons_dropped: AtomicU64, pub frames_too_short: AtomicU64, pub frames_too_long: AtomicU64, } @@ -31,6 +32,7 @@ impl EthernetStats { recv_errors: AtomicU64::new(0), beacons_sent: AtomicU64::new(0), beacons_recv: AtomicU64::new(0), + beacons_dropped: AtomicU64::new(0), frames_too_short: AtomicU64::new(0), frames_too_long: AtomicU64::new(0), } @@ -68,6 +70,11 @@ impl EthernetStats { self.beacons_recv.fetch_add(1, Ordering::Relaxed); } + /// Record a received beacon the discovery buffer had no room for. + pub fn record_beacon_dropped(&self) { + self.beacons_dropped.fetch_add(1, Ordering::Relaxed); + } + /// Take a snapshot of all counters. pub fn snapshot(&self) -> EthernetStatsSnapshot { EthernetStatsSnapshot { @@ -79,6 +86,7 @@ impl EthernetStats { recv_errors: self.recv_errors.load(Ordering::Relaxed), beacons_sent: self.beacons_sent.load(Ordering::Relaxed), beacons_recv: self.beacons_recv.load(Ordering::Relaxed), + beacons_dropped: self.beacons_dropped.load(Ordering::Relaxed), frames_too_short: self.frames_too_short.load(Ordering::Relaxed), frames_too_long: self.frames_too_long.load(Ordering::Relaxed), } @@ -102,6 +110,7 @@ pub struct EthernetStatsSnapshot { pub recv_errors: u64, pub beacons_sent: u64, pub beacons_recv: u64, + pub beacons_dropped: u64, pub frames_too_short: u64, pub frames_too_long: u64, } diff --git a/src/transport/mod.rs b/src/transport/mod.rs index a29cd34e..5fc7345a 100644 --- a/src/transport/mod.rs +++ b/src/transport/mod.rs @@ -1268,6 +1268,7 @@ impl TransportHandle { "recv_errors": snap.recv_errors, "beacons_sent": snap.beacons_sent, "beacons_recv": snap.beacons_recv, + "beacons_dropped": snap.beacons_dropped, "frames_too_short": snap.frames_too_short, "frames_too_long": snap.frames_too_long, }) From 7f5813f305409a0c9af5dfad6d221a411cbdf7bf Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:26:48 +0100 Subject: [PATCH 04/23] Stop the macOS BPF reader from parking past a shutdown request The reader thread handed each frame to the async consumer with blocking_send, which parks with no way to be woken. Stopping the transport aborts the consumer first, so nothing drains the 1024-frame channel, and the socket's Drop then joined a thread that could never return. On a busy interface the daemon had to be killed. Drop the receiver before joining. A send parked on a full channel then fails at once, which costs nothing in steady state. Do the same take in shutdown() where the lock is free, since stop_async calls it before the abort. Send through a helper that watches the same shutdown pipe the thread's select() already honours, so a send waiting for room cannot outlive a shutdown request even if the receiver is still held. The helper yields before it sleeps, so the saturated-path handoff rate is unchanged; that matters here because this thread exists to lift a throughput ceiling. It lives at module scope rather than inside the macOS-gated module so Linux CI compiles and tests it. --- CHANGELOG.md | 15 ++ src/transport/ethernet/socket.rs | 247 ++++++++++++++++++++++++++++++- 2 files changed, 256 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9ac36e73..3edbabf9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -445,6 +445,21 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 rarely dialed name in a very large peer list may re-resolve more often; the cost of a wrong eviction is one DNS lookup, not a failed dial. +- macOS: stopping an Ethernet transport under load no longer hangs the + process. The BPF reader thread handed each frame to the async consumer with + `blocking_send`, which parks with no way to be woken. Stopping the transport + aborts the consumer first, so nothing drains the 1024-frame channel, and the + socket's `Drop` then joined a thread that could never return: on a busy + interface the daemon had to be killed. The socket now drops the receiver + before joining, which releases a parked send at once, and the reader thread + sends through a helper that watches the same shutdown pipe its `select()` + already honours, so a send waiting for room cannot outlive a shutdown + request. The helper yields before it sleeps, so the saturated-path handoff + rate is unchanged. **Not covered by CI**: the reader thread is macOS-only + and Linux CI compiles none of it. What the tests prove is that the helper + the thread now waits in is cancellable; that a real BPF thread exits under + load still needs a manual check on a Mac. + - A failed private-key write no longer leaves a node silently running an ephemeral identity. Six write results in the identity path were discarded, and the sharpest was in `persistent` mode: a failed write to `fips.key` fell diff --git a/src/transport/ethernet/socket.rs b/src/transport/ethernet/socket.rs index 622d7f6c..a58e4f85 100644 --- a/src/transport/ethernet/socket.rs +++ b/src/transport/ethernet/socket.rs @@ -21,6 +21,90 @@ mod platform; #[cfg(unix)] pub use platform::PacketSocket; +/// Outcome of `send_frame`. +#[cfg(unix)] +#[cfg_attr(not(target_os = "macos"), allow(dead_code))] +pub(crate) enum SendOutcome { + Sent, + Stop, +} + +/// Retry iterations spent yielding before the send loop starts sleeping. +/// +/// A transiently full channel drains in microseconds, so yielding keeps the +/// saturated-path handoff rate uncapped, which is the whole reason this +/// module has a dedicated reader thread. Raising it burns more CPU against a +/// genuinely stuck consumer; lowering it puts a sleep in the common case. +#[cfg(unix)] +#[cfg_attr(not(target_os = "macos"), allow(dead_code))] +const SEND_YIELD_SPINS: u32 = 64; + +/// Longest the send loop sleeps between attempts on a full channel. +/// +/// This bounds only how quickly a parked send notices a shutdown request that +/// closing the receiver has not already covered. Raising it delays that +/// notice; lowering it costs more wakeups under sustained backpressure. +#[cfg(unix)] +#[cfg_attr(not(target_os = "macos"), allow(dead_code))] +const SEND_RETRY_MAX: std::time::Duration = std::time::Duration::from_millis(1); + +/// Send one item, waiting out a full channel but waking on `shutdown_fd`. +/// +/// Returns `Stop` when the receiver is gone or shutdown has been requested, +/// which is the reader thread's cue to exit. Unlike `blocking_send` this +/// cannot park past a shutdown request, so the `join()` in `Drop` always +/// returns. The caller must keep the socket owning `shutdown_fd` alive across +/// the call; `poll` on a closed fd reports `POLLNVAL` rather than `POLLIN`, so +/// even a lifetime mistake degrades to waiting rather than to a false stop. +/// +/// Compiled on every unix so Linux CI exercises the tests below; only the +/// macOS reader thread calls it. +#[cfg(unix)] +#[cfg_attr(not(target_os = "macos"), allow(dead_code))] +pub(crate) fn send_frame( + tx: &tokio::sync::mpsc::Sender, + item: T, + shutdown_fd: std::os::unix::io::RawFd, +) -> SendOutcome { + use tokio::sync::mpsc::error::TrySendError; + + let mut item = item; + let mut spins = 0u32; + let mut backoff = std::time::Duration::from_micros(50); + loop { + match tx.try_send(item) { + Ok(()) => return SendOutcome::Sent, + Err(TrySendError::Closed(_)) => return SendOutcome::Stop, + Err(TrySendError::Full(returned)) => { + if fd_is_readable(shutdown_fd) { + return SendOutcome::Stop; + } + item = returned; + if spins < SEND_YIELD_SPINS { + spins += 1; + std::thread::yield_now(); + } else { + std::thread::sleep(backoff); + backoff = (backoff * 2).min(SEND_RETRY_MAX); + } + } + } + } +} + +/// True if `fd` has data ready, tested without blocking. +#[cfg(unix)] +#[cfg_attr(not(target_os = "macos"), allow(dead_code))] +pub(crate) fn fd_is_readable(fd: std::os::unix::io::RawFd) -> bool { + let mut pfd = libc::pollfd { + fd, + events: libc::POLLIN, + revents: 0, + }; + let ret = unsafe { libc::poll(&mut pfd, 1, 0) }; + ret > 0 && (pfd.revents & libc::POLLIN) != 0 +} + // ============================================================================= // Linux: AsyncFd-based async wrapper // ============================================================================= @@ -111,7 +195,9 @@ mod async_impl { pub struct AsyncPacketSocket { inner: Arc, - rx: tokio::sync::Mutex>, + /// `None` once shutdown has taken the receiver, which is what makes + /// a reader thread parked on a full channel return at once. + rx: tokio::sync::Mutex>>, reader_thread: Option>, } @@ -146,7 +232,13 @@ mod async_impl { match result { Ok((n, mac)) => { let data = read_buf[..n].to_vec(); - if tx.blocking_send((data, mac)).is_err() { + // Not blocking_send: a send parked on a + // full channel must still notice shutdown, + // or Drop's join() never returns. + if matches!( + super::send_frame(&tx, (data, mac), shutdown_fd), + super::SendOutcome::Stop + ) { return; } } @@ -207,7 +299,7 @@ mod async_impl { Ok(Self { inner, - rx: tokio::sync::Mutex::new(rx), + rx: tokio::sync::Mutex::new(Some(rx)), reader_thread: Some(reader_thread), }) } @@ -230,7 +322,10 @@ mod async_impl { } pub async fn recv_from(&self, buf: &mut [u8]) -> Result<(usize, [u8; 6]), TransportError> { - let mut rx = self.rx.lock().await; + let mut guard = self.rx.lock().await; + let Some(rx) = guard.as_mut() else { + return Err(TransportError::RecvFailed("reader thread stopped".into())); + }; match rx.recv().await { Some((data, mac)) => { let n = data.len().min(buf.len()); @@ -247,15 +342,25 @@ mod async_impl { /// Signal the reader thread to stop. /// - /// Sets the shutdown flag; the reader thread checks it after - /// each BPF read timeout (~250ms) and exits. + /// Drops the receiver where it can, which makes a send parked on a + /// full channel fail immediately, then writes the shutdown pipe that + /// the thread's `select()` and `send_frame` both watch. The receiver + /// is unavailable while a `recv_from` holds the lock; `Drop` takes it + /// unconditionally, so the pipe is what covers that window. pub fn shutdown(&self) { + if let Ok(mut guard) = self.rx.try_lock() { + guard.take(); + } self.inner.request_shutdown(); } } impl Drop for AsyncPacketSocket { fn drop(&mut self) { + // Drop the receiver before joining: a send parked on a full + // channel then returns at once, with no polling and no latency + // added to the steady-state path. + self.rx.get_mut().take(); self.inner.request_shutdown(); if let Some(handle) = self.reader_thread.take() { let _ = handle.join(); @@ -284,3 +389,133 @@ pub struct PacketSocket; #[cfg(windows)] pub struct AsyncPacketSocket; + +// ============================================================================= +// Tests +// ============================================================================= + +#[cfg(all(test, unix))] +mod tests { + use super::{SendOutcome, fd_is_readable, send_frame}; + use std::sync::mpsc; + use std::time::Duration; + + /// A pipe, as the shutdown signal, returned as (read fd, write fd). + /// + /// Leaked deliberately: these live for the length of one test and closing + /// them mid-poll is exactly the confusion the test is meant to avoid. + fn shutdown_pipe() -> (std::os::unix::io::RawFd, std::os::unix::io::RawFd) { + let mut fds = [0i32; 2]; + let ret = unsafe { libc::pipe(fds.as_mut_ptr()) }; + assert_eq!(ret, 0, "pipe() failed"); + (fds[0], fds[1]) + } + + fn signal(write_fd: std::os::unix::io::RawFd) { + let byte = [1u8]; + let ret = unsafe { libc::write(write_fd, byte.as_ptr() as *const libc::c_void, 1) }; + assert_eq!(ret, 1, "write() to shutdown pipe failed"); + } + + #[test] + fn fd_is_readable_is_false_for_an_unwritten_pipe_and_true_after_a_write() { + let (read_fd, write_fd) = shutdown_pipe(); + assert!(!fd_is_readable(read_fd)); + signal(write_fd); + assert!(fd_is_readable(read_fd)); + } + + #[test] + fn send_frame_delivers_when_the_channel_has_room() { + let (tx, mut rx) = tokio::sync::mpsc::channel::>(1); + let (read_fd, _write_fd) = shutdown_pipe(); + + assert!(matches!( + send_frame(&tx, vec![1u8, 2, 3], read_fd), + SendOutcome::Sent + )); + assert_eq!(rx.try_recv().unwrap(), vec![1u8, 2, 3]); + } + + #[test] + fn send_frame_returns_stop_when_the_receiver_is_gone() { + let (tx, rx) = tokio::sync::mpsc::channel::>(1); + let (read_fd, _write_fd) = shutdown_pipe(); + drop(rx); + + assert!(matches!( + send_frame(&tx, vec![0u8], read_fd), + SendOutcome::Stop + )); + } + + #[test] + fn send_frame_returns_stop_when_the_receiver_is_dropped_while_the_channel_is_full() { + // The mechanism `Drop` relies on: closing the channel releases a + // sender that is waiting for room. + let (tx, rx) = tokio::sync::mpsc::channel::>(1); + let (read_fd, _write_fd) = shutdown_pipe(); + tx.try_send(vec![0u8]).unwrap(); + + let (done_tx, done_rx) = mpsc::channel(); + let sender = std::thread::spawn(move || { + let outcome = send_frame(&tx, vec![1u8], read_fd); + done_tx.send(matches!(outcome, SendOutcome::Stop)).unwrap(); + }); + // The send is parked on a full channel; only the drop frees it. + assert!(done_rx.recv_timeout(Duration::from_millis(50)).is_err()); + drop(rx); + + let stopped = done_rx + .recv_timeout(Duration::from_secs(5)) + .expect("send_frame did not return after the receiver was dropped"); + sender.join().unwrap(); + assert!(stopped); + } + + #[test] + fn send_frame_returns_stop_when_shutdown_is_requested_and_the_channel_is_full() { + // The defect: `blocking_send` on a full channel nobody is draining + // parks forever, so the reader thread never sees shutdown and the + // `join()` in `Drop` never returns. See the ignored test below for + // the same fixture against `blocking_send`. + let (tx, _rx) = tokio::sync::mpsc::channel::>(1); + let (read_fd, write_fd) = shutdown_pipe(); + tx.try_send(vec![0u8]).unwrap(); + signal(write_fd); + + let (done_tx, done_rx) = mpsc::channel(); + let sender = std::thread::spawn(move || { + let outcome = send_frame(&tx, vec![1u8], read_fd); + done_tx.send(matches!(outcome, SendOutcome::Stop)).unwrap(); + }); + + let stopped = done_rx + .recv_timeout(Duration::from_secs(5)) + .expect("send_frame parked past a shutdown request"); + sender.join().unwrap(); + assert!(stopped); + } + + #[test] + #[ignore = "demonstrates the defect: blocking_send never returns, so this hangs"] + fn blocking_send_parks_past_a_shutdown_request_when_the_channel_is_full() { + // Run with `--ignored` to watch the old send site hang. Kept as the + // observed red-before for the test above, which cannot itself fail + // against the old code because `send_frame` did not exist then. + let (tx, _rx) = tokio::sync::mpsc::channel::>(1); + let (_read_fd, write_fd) = shutdown_pipe(); + tx.try_send(vec![0u8]).unwrap(); + signal(write_fd); + + let (done_tx, done_rx) = mpsc::channel(); + std::thread::spawn(move || { + let _ = tx.blocking_send(vec![1u8]); + done_tx.send(()).unwrap(); + }); + + done_rx + .recv_timeout(Duration::from_secs(5)) + .expect("blocking_send returned, so the send site was already cancellable"); + } +} From 2517d207516d1219ffc595c3f33f8e0a2aa94f99 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:26:06 +0100 Subject: [PATCH 05/23] Create the control socket and its directory with a restrictive mode bind(2) creates the socket inode with 0777 & ~umask, so under a permissive umask the control socket was world-accessible for the window between the bind and the chmod to 0770 that followed it. The parent directory was worse than a window: create_dir_all takes the same 0777 & ~umask and nothing ever set a mode on it, so the directory holding the socket stayed world-writable for the life of the host, and a world-writable parent lets an unprivileged account plant an entry at the socket path. Add a small shared helper that binds under a umask masking the "other" bits and creates directories with an explicit 0750. The socket ends at the mode it always did, with the existing chmod and chown left as the authority on it, and 0750 is what the systemd unit and the FreeBSD rc script already apply to the runtime directory, so no packaged deployment sees a different mode. The umask is held across the bind alone, and it only clears bits, so anything else created in that window comes out more restrictive rather than less. Both the daemon and gateway control sockets go through the helper; the duplicated bind sequences stay as they are. The window between the stale-socket probe and the bind is documented at both sites rather than closed: reaching it needs write access to the socket's parent directory, which the packaged layouts give to root alone, and an account holding it can deny the daemon its socket more simply by squatting the path first. --- CHANGELOG.md | 24 +++++++ src/control/mod.rs | 15 +++- src/gateway/control.rs | 15 +++- src/utils/mod.rs | 5 +- src/utils/sockperm.rs | 157 +++++++++++++++++++++++++++++++++++++++++ 5 files changed, 211 insertions(+), 5 deletions(-) create mode 100644 src/utils/sockperm.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 3edbabf9..91f18198 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -866,6 +866,30 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 NXDOMAIN. Connecting the socket also means a dead upstream surfaces ECONNREFUSED immediately instead of stalling for five seconds. +#### Control socket + +- The control socket and the directory holding it are now created with a + restrictive mode rather than created wide and narrowed afterwards. `bind(2)` + makes the socket inode `0777 & ~umask`, so under a permissive umask the + socket was world-accessible for the window between the bind and the `chmod` + to 0770 that followed it; the bind now runs under a umask that masks the + "other" bits, so the inode is 0770 from creation and the chmod and chown stay + the authority on its final mode. The parent directory was worse than a + window: it was created with `create_dir_all`, which is also `0777 & ~umask`, + and nothing ever set a mode on it, so under a permissive umask the directory + holding the socket stayed world-writable for the life of the host, and a + world-writable parent lets an unprivileged account plant an entry at the + socket path. Directories this code creates now come out 0750, which is what + the systemd unit (`RuntimeDirectoryMode=0750`) and the FreeBSD rc script + (`install -d -m 0750`) already apply, so no packaged deployment sees a + different mode and no `fipsctl` user loses access. Both the daemon and the + gateway control sockets are covered. **What this does not close**: the window + between the stale-socket probe and the bind is documented at the site rather + than removed. Reaching it needs write access to the socket's parent + directory, which the packaged layouts give to root alone, and an account + holding it can deny the daemon its socket more simply by squatting the path + first. + #### Key material and identity files - Private key writes no longer follow a symlink, and the key file's mode is diff --git a/src/control/mod.rs b/src/control/mod.rs index e26de0c1..f334b516 100644 --- a/src/control/mod.rs +++ b/src/control/mod.rs @@ -151,7 +151,7 @@ mod unix_impl { if let Some(parent) = socket_path.parent() && !parent.exists() { - std::fs::create_dir_all(parent)?; + crate::utils::sockperm::make_parent(parent)?; debug!(path = %parent.display(), "Created control socket directory"); } @@ -160,7 +160,9 @@ mod unix_impl { Self::remove_stale_socket(&socket_path)?; } - let listener = UnixListener::bind(&socket_path)?; + // Bound under a tightened umask, so the inode is never + // world-accessible in the window before the chmod below. + let listener = crate::utils::sockperm::bind(&socket_path)?; // Make the socket and its parent directory group-accessible so // 'fips' group members can use fipsctl/fipstop without root. @@ -183,6 +185,15 @@ mod unix_impl { /// /// If the file exists but no one is listening, remove it so we can /// bind. This handles unclean daemon exits. + /// + /// The gap between the connect probe and the bind that follows is + /// accepted rather than closed. Reaching it needs write access to the + /// socket's parent directory, which the packaged layouts give to root + /// alone (0750 and root-owned under both systemd and the FreeBSD rc + /// script), and an account holding it can deny the daemon its socket + /// more simply by squatting the path before the daemon starts. The + /// removal itself unlinks a symlink rather than its target, so it is + /// not an arbitrary delete. fn remove_stale_socket(path: &Path) -> Result<(), std::io::Error> { // Try connecting to see if someone is listening match std::os::unix::net::UnixStream::connect(path) { diff --git a/src/gateway/control.rs b/src/gateway/control.rs index a77014dd..7a04316b 100644 --- a/src/gateway/control.rs +++ b/src/gateway/control.rs @@ -65,7 +65,7 @@ impl GatewayControlSocket { if let Some(parent) = socket_path.parent() && !parent.exists() { - std::fs::create_dir_all(parent)?; + crate::utils::sockperm::make_parent(parent)?; debug!(path = %parent.display(), "Created gateway control socket directory"); } @@ -74,7 +74,9 @@ impl GatewayControlSocket { Self::remove_stale_socket(&socket_path)?; } - let listener = UnixListener::bind(&socket_path)?; + // Bound under a tightened umask, so the inode is never + // world-accessible in the window before the chmod below. + let listener = crate::utils::sockperm::bind(&socket_path)?; // Set permissions to 0770 and chown to fips group use std::os::unix::fs::PermissionsExt; @@ -93,6 +95,15 @@ impl GatewayControlSocket { } /// Remove a stale socket file from a previous unclean exit. + /// + /// The gap between the connect probe and the bind that follows is + /// accepted rather than closed. Reaching it needs write access to the + /// socket's parent directory, which the packaged layouts give to root + /// alone (0750 and root-owned under both systemd and the FreeBSD rc + /// script), and an account holding it can deny the daemon its socket + /// more simply by squatting the path before the daemon starts. The + /// removal itself unlinks a symlink rather than its target, so it is + /// not an arbitrary delete. fn remove_stale_socket(path: &Path) -> Result<(), std::io::Error> { match std::os::unix::net::UnixStream::connect(path) { Ok(_) => Err(std::io::Error::new( diff --git a/src/utils/mod.rs b/src/utils/mod.rs index 1fd66c36..cb934c02 100644 --- a/src/utils/mod.rs +++ b/src/utils/mod.rs @@ -1,6 +1,9 @@ //! Utility modules. //! //! Shared infrastructure that doesn't belong to a specific protocol layer: -//! session index allocation and other cross-cutting concerns. +//! session index allocation, socket permission handling, and other +//! cross-cutting concerns. pub mod index; +#[cfg(unix)] +pub mod sockperm; diff --git a/src/utils/sockperm.rs b/src/utils/sockperm.rs new file mode 100644 index 00000000..2e1d1d08 --- /dev/null +++ b/src/utils/sockperm.rs @@ -0,0 +1,157 @@ +//! Permission-safe creation of Unix domain sockets and the directories +//! holding them. +//! +//! The socket inode and its parent directory are created with a mode the +//! process umask can only tighten, rather than created wide and narrowed +//! afterwards. The caller's own chmod and chown stay where they are and +//! remain the authority on the socket's final mode; this closes the window +//! between creation and that fix-up, and the case of an intermediate +//! directory that nothing fixes up at all. + +use std::path::Path; +use tokio::net::UnixListener; + +/// Mode for a directory this module creates to hold a control socket. +/// +/// Matches what the packaging already applies (systemd's +/// `RuntimeDirectoryMode=0750`, `install -d -m 0750` in the FreeBSD rc +/// script), so no packaged deployment sees a different directory mode than +/// it does today. Widening it would expose the socket path to accounts that +/// cannot reach it now; the umask can still tighten it further. +const SOCKET_DIR_MODE: u32 = 0o750; + +/// umask held across the socket bind. +/// +/// `bind(2)` creates the socket inode with `0777 & !umask`, so under a +/// permissive umask the socket is world-accessible until the chmod that +/// follows it. Masking the "other" bits makes the inode 0770 at creation, +/// which is the mode the caller applies a moment later anyway. Changing +/// this changes the mode the socket is created with, not the mode it ends +/// up with. +const BIND_UMASK: libc::mode_t = 0o007; + +/// Restores the process umask when dropped. +struct UmaskGuard(libc::mode_t); + +impl UmaskGuard { + /// Install `mask` as the process umask, remembering the previous one. + fn tighten(mask: libc::mode_t) -> Self { + // SAFETY: umask(2) cannot fail and touches only process state. + Self(unsafe { libc::umask(mask) }) + } +} + +impl Drop for UmaskGuard { + fn drop(&mut self) { + // SAFETY: as above; restoring the mask this guard replaced. + unsafe { + libc::umask(self.0); + } + } +} + +/// Create the directory that will hold a socket, and any missing ancestors. +/// +/// Directories come out 0750 rather than `0777 & !umask`. Nothing chmods an +/// intermediate directory afterwards, so one created under a permissive +/// umask would stay world-writable for the life of the host, and a +/// world-writable parent lets an unprivileged account plant an entry at the +/// socket path. +pub fn make_parent(parent: &Path) -> Result<(), std::io::Error> { + use std::os::unix::fs::DirBuilderExt; + + std::fs::DirBuilder::new() + .recursive(true) + .mode(SOCKET_DIR_MODE) + .create(parent) +} + +/// Bind a Unix listener whose inode is never world-accessible. +/// +/// The umask is process-global, so it is held across the bind alone. It +/// only clears bits, so anything else created inside that window comes out +/// more restrictive, never less. +pub fn bind(path: &Path) -> Result { + let _umask = UmaskGuard::tighten(BIND_UMASK); + UnixListener::bind(path) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::os::unix::fs::PermissionsExt; + use std::sync::Mutex; + + /// The umask is process-global, so the tests that set it run one at a + /// time. This does not serialize against the rest of the test binary; + /// the mask used is 0o022, the ordinary default, so a file another test + /// creates in the window is unaffected. + static UMASK_LOCK: Mutex<()> = Mutex::new(()); + + /// Take the umask lock, ignoring poisoning: a test that fails while + /// holding it must not turn its siblings red for an unrelated reason. + fn umask_lock() -> std::sync::MutexGuard<'static, ()> { + UMASK_LOCK.lock().unwrap_or_else(|e| e.into_inner()) + } + + /// Read the current umask, which is only observable by replacing it. + fn current_umask() -> libc::mode_t { + // SAFETY: umask(2) cannot fail; the value read is put straight back. + unsafe { + let old = libc::umask(0o022); + libc::umask(old); + old + } + } + + fn mode_of(path: &Path) -> u32 { + std::fs::symlink_metadata(path) + .unwrap() + .permissions() + .mode() + } + + #[tokio::test] + async fn socket_is_created_without_other_access_under_a_permissive_umask() { + let _lock = umask_lock(); + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("control.sock"); + + let restore = UmaskGuard::tighten(0o022); + let listener = bind(&path).unwrap(); + drop(restore); + + assert_eq!(mode_of(&path) & 0o007, 0); + drop(listener); + } + + #[tokio::test] + async fn socket_bind_leaves_the_process_umask_as_it_found_it() { + let _lock = umask_lock(); + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("control.sock"); + + let restore = UmaskGuard::tighten(0o022); + let listener = bind(&path).unwrap(); + let after = current_umask(); + drop(restore); + + assert_eq!(after, 0o022); + drop(listener); + } + + #[test] + fn socket_parent_and_its_ancestors_are_created_without_other_access() { + let _lock = umask_lock(); + let dir = tempfile::tempdir().unwrap(); + let intermediate = dir.path().join("run"); + let parent = intermediate.join("fips"); + + let restore = UmaskGuard::tighten(0o022); + make_parent(&parent).unwrap(); + drop(restore); + + assert_eq!(mode_of(&intermediate) & 0o007, 0); + assert_eq!(mode_of(&parent) & 0o007, 0); + } +} From f42194099ba08cb499886fd664bca7939686fe52 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:18:35 +0100 Subject: [PATCH 06/23] Hold the last loaded peer ACL when an input file cannot be read A read error on peers.allow, peers.deny or the hosts file was logged and swallowed, leaving that file's entries empty, and the reloader published the result with no test that the load had succeeded. An unreadable peers.deny therefore admitted the peers it named and an unreadable peers.allow took a strict allowlist node to admitting everyone, with one warning line as the only signal. The recorded modification times advanced before the load ran, so nothing retried until the file changed again and a persistent permission or I/O fault left the empty ACL in force indefinitely. Make the ACL and hosts loaders report read failures instead of returning an empty result. The reloader now keeps the ACL it last published when any input is present but unreadable, leaves the modification times alone, and arms a retry so the next tick reloads regardless of them, which also covers the hosts change that check_reload has already consumed. The fault is logged once on the transition into the held state rather than once per tick. An absent file remains a policy and still loads as an empty set. A NotFound that a successful stat contradicts is a file being rewritten under us and is held. A reload whose inputs all read cleanly but which empties an enforcing ACL while its files are still present is held for one tick, so a read that caught a non-atomic in-place edit mid-write does not publish a torn policy, and released on the next tick so a deliberate blanking still takes effect. ACL status gains a stale flag so an operator whose edit appears to have no effect can see that the policy in force is older than the files on disk. There is no last-good snapshot at startup, so an unreadable file at boot still yields no entries; it is now logged as an error and armed to retry. --- CHANGELOG.md | 26 +++ src/control/snapshot.rs | 1 + src/node/acl.rs | 345 ++++++++++++++++++++++++++++++++++++---- src/upper/hosts.rs | 78 +++++++-- 4 files changed, 407 insertions(+), 43 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 91f18198..4848f047 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -828,6 +828,32 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 seen in that tick, and anything able to flood raw frames on the segment can already jam the beacon at L2 more cheaply. +- A read failure on `peers.allow`, `peers.deny` or the `hosts` file no longer + turns the node into an open one. Every read error other than a steadily + absent file was logged and swallowed, leaving that file's entries empty, and + the reloader published the result unconditionally: an unreadable `peers.deny` + admitted the peers it named, and an unreadable `peers.allow` took a node from + admitting a named few to admitting everyone, with one warning line as the + only signal. Because the recorded modification times advanced before the + load, nothing retried until the file changed again, so a persistent + permission or I/O fault left the empty ACL in force indefinitely. The + reloader now keeps the last loaded ACL when any input is present but + unreadable, leaves the modification times alone, retries on the next tick + regardless of them, and logs the fault once on the transition rather than + once per tick. An absent file is still a policy and still loads as an empty + set; a `NotFound` that a successful stat contradicts is treated as a file + being rewritten under us and held. A reload whose inputs all read cleanly but + which empties an enforcing ACL while its files are still on disk is held for + one tick, which catches a read that caught a non-atomic in-place edit + mid-write, and released on the next so a deliberate blanking still takes + effect. `fipsctl` ACL status gains a `stale` flag reporting that the policy + in force is older than the files on disk. **What this does not close**: + there is no last-good snapshot at startup, so a node whose ACL file is + unreadable at boot still comes up with no entries, now logged as an error and + armed to retry on the first tick. Admission is also checked only at handshake + time, so a peer admitted during a window that has already happened keeps its + link. + - An accepted inbound TCP connection no longer holds a slot indefinitely without sending anything. The cap was tested at accept and the pool insert and counter bump followed with no read in between, while the frame reader's diff --git a/src/control/snapshot.rs b/src/control/snapshot.rs index 042f79b3..bc9db421 100644 --- a/src/control/snapshot.rs +++ b/src/control/snapshot.rs @@ -133,6 +133,7 @@ fn empty_acl_status() -> PeerAclStatus { deny_file_entries: Vec::new(), allow_entries: Vec::new(), deny_entries: Vec::new(), + stale: false, } } diff --git a/src/node/acl.rs b/src/node/acl.rs index dbccbb12..81f6eeb1 100644 --- a/src/node/acl.rs +++ b/src/node/acl.rs @@ -20,7 +20,7 @@ use std::fmt; use std::path::{Path, PathBuf}; use std::sync::Arc; use std::time::SystemTime; -use tracing::{debug, info, warn}; +use tracing::{debug, error, info, warn}; /// Default path for the peer allow list. /// @@ -105,6 +105,28 @@ pub enum PeerAclContext { OutboundHandshake, } +/// How many consecutive reloads the empty-snapshot guard may hold back. +/// +/// A reload whose files all read cleanly but which yields an empty ACL where +/// an enforcing one was in force is more likely a read that raced an in-place +/// rewrite than a policy change, so the previous snapshot is held. The guard +/// releases after this many holds, so an operator who deliberately blanks an +/// ACL file in place still converges, one tick late. Raising it widens the +/// window in which a genuine emptying is ignored; setting it to zero disables +/// the torn-read protection. +const EMPTY_ACL_HOLD_LIMIT: u32 = 1; + +/// A peer ACL input file exists but could not be read. +#[derive(Debug, thiserror::Error)] +#[error("failed to read {}: {source}", path.display())] +pub struct AclLoadError { + /// The file whose read failed. + pub path: PathBuf, + /// The underlying I/O failure. + #[source] + pub source: std::io::Error, +} + /// Snapshot of the currently loaded ACL state. #[derive(Debug, Clone, PartialEq, Eq, Serialize)] pub struct PeerAclStatus { @@ -119,6 +141,9 @@ pub struct PeerAclStatus { pub deny_file_entries: Vec, pub allow_entries: Vec, pub deny_entries: Vec, + /// Whether the ACL in force is older than the files on disk because a + /// reload input could not be read. + pub stale: bool, } impl fmt::Display for PeerAclContext { @@ -154,26 +179,38 @@ impl PeerAcl { #[cfg(test)] pub fn load_files(allow_path: &Path, deny_path: &Path) -> Self { let hosts = HostMap::new(); - Self::load_files_with_hosts(allow_path, deny_path, &hosts) + Self::try_load_files_with_hosts(allow_path, deny_path, &hosts).unwrap() } /// Load the allow/deny files into a new ACL using alias resolution. - pub fn load_files_with_hosts(allow_path: &Path, deny_path: &Path, hosts: &HostMap) -> Self { + /// + /// An absent file is a policy and contributes an empty set; a file that + /// is present and unreadable is a fault and is returned as an error, so + /// the caller can keep enforcing whatever it loaded last rather than + /// silently becoming an open node. + pub fn try_load_files_with_hosts( + allow_path: &Path, + deny_path: &Path, + hosts: &HostMap, + ) -> Result { let mut acl = Self::new(); - acl.load_file(allow_path, true, hosts); - acl.load_file(deny_path, false, hosts); + acl.load_file(allow_path, true, hosts)?; + acl.load_file(deny_path, false, hosts)?; + acl.log_loaded(); + Ok(acl) + } - if !acl.is_empty() { + /// Log the shape of a freshly loaded ACL, unless it has no entries. + fn log_loaded(&self) { + if !self.is_empty() { debug!( - allow_entries = acl.allow.len(), - deny_entries = acl.deny.len(), - allow_all = acl.allow_all, - deny_all = acl.deny_all, + allow_entries = self.allow.len(), + deny_entries = self.deny.len(), + allow_all = self.allow_all, + deny_all = self.deny_all, "Loaded peer ACL files" ); } - - acl } /// Evaluate whether a peer is allowed. @@ -244,16 +281,29 @@ impl PeerAcl { self.deny_file_entries.iter().cloned().collect() } - fn load_file(&mut self, path: &Path, is_allow: bool, hosts: &HostMap) { + /// Merge one ACL file into this ACL. + /// + /// An absent file is a policy and an unreadable one is a fault, and + /// `NotFound` alone does not say which: a stat that still finds the file + /// after the read missed it means the file is being rewritten under us, + /// which is transient and must not be published as an empty policy. + fn load_file( + &mut self, + path: &Path, + is_allow: bool, + hosts: &HostMap, + ) -> Result<(), AclLoadError> { let contents = match std::fs::read_to_string(path) { Ok(c) => c, - Err(e) if e.kind() == std::io::ErrorKind::NotFound => { + Err(e) if e.kind() == std::io::ErrorKind::NotFound && file_mtime(path).is_none() => { debug!(path = %path.display(), "No ACL file found, skipping"); - return; + return Ok(()); } Err(e) => { - warn!(path = %path.display(), error = %e, "Failed to read ACL file"); - return; + return Err(AclLoadError { + path: path.to_path_buf(), + source: e, + }); } }; @@ -309,6 +359,8 @@ impl PeerAcl { self.deny_npubs.insert(resolved_npub); } } + + Ok(()) } fn resolve_entry(entry: &str, hosts: &HostMap) -> Result<(PeerIdentity, String), String> { @@ -341,6 +393,13 @@ pub struct PeerAclReloader { deny_path: PathBuf, last_allow_mtime: Option, last_deny_mtime: Option, + /// Set while a reload input is unreadable. Forces the next reload + /// attempt regardless of mtimes, because the mtime comparison alone + /// cannot see a change the hosts reloader has already consumed, and + /// gates the fault log to the transition into the held state. + retry_pending: bool, + /// Consecutive reloads held back by the empty-snapshot guard. + empty_holds: u32, } impl PeerAclReloader { @@ -376,7 +435,23 @@ impl PeerAclReloader { let last_allow_mtime = file_mtime(&allow_path); let last_deny_mtime = file_mtime(&deny_path); let hosts = HostMapReloader::new(base_hosts, hosts_path); - let acl = PeerAcl::load_files_with_hosts(&allow_path, &deny_path, hosts.hosts()); + + // There is no last-good snapshot to hold at startup, so an + // unreadable file still comes up on an empty ACL, as it always has. + // It is logged as the fault it is and armed for retry, so the first + // tick after the file becomes readable enforces the real policy. + let (acl, retry_pending) = + match PeerAcl::try_load_files_with_hosts(&allow_path, &deny_path, hosts.hosts()) { + Ok(acl) => (acl, false), + Err(e) => { + error!( + path = %e.path.display(), + error = %e.source, + "Peer ACL file is present but unreadable; starting with no ACL entries" + ); + (PeerAcl::new(), true) + } + }; Self { acl: arc_swap::ArcSwap::from(Arc::new(acl)), @@ -385,9 +460,28 @@ impl PeerAclReloader { deny_path, last_allow_mtime, last_deny_mtime, + retry_pending, + empty_holds: 0, } } + /// Keep the published snapshot after a reload input failed to read. + /// + /// Leaves the recorded mtimes and the ACL in force untouched, arms the + /// retry so the next tick reloads regardless of mtimes, and logs the + /// fault once, on the transition into the held state, rather than once + /// per tick for as long as the fault lasts. + fn hold_snapshot(&mut self, path: &Path, error: &dyn fmt::Display) { + if !self.retry_pending { + error!( + path = %path.display(), + error = %error, + "Peer ACL input is unreadable; holding the last loaded ACL" + ); + } + self.retry_pending = true; + } + /// Acquire a lock-free guard over the current ACL snapshot. pub fn acl(&self) -> arc_swap::Guard> { self.load() @@ -408,6 +502,7 @@ impl PeerAclReloader { deny_file_entries: acl.deny_file_entries(), allow_entries: acl.allow_entries(), deny_entries: acl.deny_entries(), + stale: self.retry_pending, } } } @@ -418,19 +513,70 @@ impl Reloadable for PeerAclReloader { async fn reload(&mut self) -> bool { let allow_mtime = file_mtime(&self.allow_path); let deny_mtime = file_mtime(&self.deny_path); - let hosts_changed = self.hosts.check_reload(); + let hosts_changed = match self.hosts.try_check_reload() { + Ok(changed) => changed, + Err(e) => { + let path = self.hosts.path().to_path_buf(); + self.hold_snapshot(&path, &e); + return false; + } + }; if allow_mtime == self.last_allow_mtime && deny_mtime == self.last_deny_mtime && !hosts_changed + && !self.retry_pending { return false; } + let new_acl = match PeerAcl::try_load_files_with_hosts( + &self.allow_path, + &self.deny_path, + self.hosts.hosts(), + ) { + Ok(acl) => acl, + Err(e) => { + self.hold_snapshot(&e.path.clone(), &e.source); + return false; + } + }; + + // Every input read cleanly and the policy still evaporated. With the + // ACL files themselves freshly written and still on disk that is more + // likely a read that caught one mid-rewrite than an operator emptying + // both lists, so hold and look again next tick. Deleting a file, or + // dropping the aliases an entry resolved through, remains an + // unambiguous way to say "no policy" and is published immediately. + let acl_files_changed = + allow_mtime != self.last_allow_mtime || deny_mtime != self.last_deny_mtime; + if new_acl.is_empty() + && !self.acl.load().is_empty() + && acl_files_changed + && (allow_mtime.is_some() || deny_mtime.is_some()) + && self.empty_holds < EMPTY_ACL_HOLD_LIMIT + { + self.empty_holds += 1; + self.retry_pending = true; + warn!( + allow_file = %self.allow_path.display(), + deny_file = %self.deny_path.display(), + "Peer ACL reload emptied an enforcing ACL; holding the last loaded ACL" + ); + return false; + } + + if self.retry_pending { + info!( + allow_file = %self.allow_path.display(), + deny_file = %self.deny_path.display(), + "Peer ACL inputs read cleanly again; publishing the files on disk" + ); + } + self.retry_pending = false; + self.empty_holds = 0; self.last_allow_mtime = allow_mtime; self.last_deny_mtime = deny_mtime; - let new_acl = - PeerAcl::load_files_with_hosts(&self.allow_path, &self.deny_path, self.hosts.hosts()); info!( allow_file = %self.allow_path.display(), @@ -784,19 +930,15 @@ mod tests { } #[test] - fn test_acl_read_error_is_ignored() { + fn test_acl_read_error_is_reported_rather_than_yielding_an_empty_acl() { let dir = tempfile::tempdir().unwrap(); let allow = dir.path().join("peers.allow"); let deny = dir.path().join("peers.deny"); std::fs::create_dir(&allow).unwrap(); - let acl = PeerAcl::load_files(&allow, &deny); + let err = PeerAcl::try_load_files_with_hosts(&allow, &deny, &HostMap::new()).unwrap_err(); - assert!(acl.is_empty()); - assert_eq!( - acl.check(&test_peer(&test_npub())), - PeerAclDecision::DefaultAllow - ); + assert_eq!(err.path, allow); } #[test] @@ -810,7 +952,7 @@ mod tests { hosts.insert("node-a", &npub).unwrap(); write_file(&allow, "NODE-A\n"); - let acl = PeerAcl::load_files_with_hosts(&allow, &deny, &hosts); + let acl = PeerAcl::try_load_files_with_hosts(&allow, &deny, &hosts).unwrap(); assert_eq!(acl.allow_file_entries(), vec!["NODE-A".to_string()]); assert_eq!(acl.allow_entries(), vec![npub.clone()]); @@ -828,7 +970,7 @@ mod tests { hosts.insert("node-a", &npub).unwrap(); write_file(&allow, &format!("node-a\n{npub}\nnode-a\n")); - let acl = PeerAcl::load_files_with_hosts(&allow, &deny, &hosts); + let acl = PeerAcl::try_load_files_with_hosts(&allow, &deny, &hosts).unwrap(); assert_eq!( acl.allow_file_entries(), @@ -885,6 +1027,147 @@ mod tests { ); } + /// Make a file unreadable, returning false if the effective uid can read + /// it anyway. Root bypasses the mode bits, so the permission-fault tests + /// cannot run there and skip instead of passing vacuously; that leaves + /// the EACCES path unexercised in any root CI job. + fn make_unreadable(path: &Path) -> bool { + use std::os::unix::fs::PermissionsExt; + std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o000)).unwrap(); + std::fs::read_to_string(path).is_err() + } + + fn make_readable(path: &Path) { + use std::os::unix::fs::PermissionsExt; + std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o644)).unwrap(); + } + + #[tokio::test] + async fn test_acl_reload_holds_last_good_snapshot_when_deny_file_unreadable() { + let dir = tempfile::tempdir().unwrap(); + let allow = dir.path().join("peers.allow"); + let deny = dir.path().join("peers.deny"); + let denied = test_npub(); + + write_file(&deny, &format!("{denied}\n")); + let mut reloader = PeerAclReloader::with_paths(allow, deny.clone()); + assert_eq!( + reloader.acl().check(&test_peer(&denied)), + PeerAclDecision::DenyList + ); + + // Rewrite before revoking access so the mtime change makes the + // reloader actually attempt the read that then fails. + std::thread::sleep(std::time::Duration::from_millis(5)); + write_file(&deny, &format!("{denied}\n")); + if !make_unreadable(&deny) { + return; + } + + assert!(!reloader.reload().await); + assert_eq!( + reloader.acl().check(&test_peer(&denied)), + PeerAclDecision::DenyList + ); + assert_eq!(reloader.acl().effective_mode(), "denylist"); + assert!(reloader.status().stale); + + make_readable(&deny); + } + + #[tokio::test] + async fn test_acl_reload_retries_after_a_transient_read_error() { + let dir = tempfile::tempdir().unwrap(); + let allow = dir.path().join("peers.allow"); + let deny = dir.path().join("peers.deny"); + let denied = test_npub(); + + write_file(&deny, &format!("{denied}\n")); + let mut reloader = PeerAclReloader::with_paths(allow, deny.clone()); + std::thread::sleep(std::time::Duration::from_millis(5)); + write_file(&deny, &format!("{denied}\n")); + if !make_unreadable(&deny) { + return; + } + assert!(!reloader.reload().await); + + // No further mtime change: only the armed retry can pick this up. + make_readable(&deny); + assert!(reloader.reload().await); + assert_eq!( + reloader.acl().check(&test_peer(&denied)), + PeerAclDecision::DenyList + ); + assert!(!reloader.status().stale); + } + + #[tokio::test] + async fn test_acl_reload_holds_last_good_when_the_hosts_file_becomes_unreadable() { + let dir = tempfile::tempdir().unwrap(); + let allow = dir.path().join("peers.allow"); + let deny = dir.path().join("peers.deny"); + let hosts = dir.path().join("hosts"); + let npub = test_npub(); + + write_file(&allow, "node-a\n"); + write_file(&hosts, &format!("node-a {npub}\n")); + + let mut reloader = + PeerAclReloader::with_alias_sources(allow, deny, HostMap::new(), hosts.clone()); + assert_eq!( + reloader.acl().check(&test_peer(&npub)), + PeerAclDecision::AllowList + ); + + std::thread::sleep(std::time::Duration::from_millis(5)); + write_file(&hosts, &format!("node-a {npub}\n")); + if !make_unreadable(&hosts) { + return; + } + + assert!(!reloader.reload().await); + assert_eq!( + reloader.acl().check(&test_peer(&npub)), + PeerAclDecision::AllowList + ); + assert_eq!(reloader.acl().default_decision(), "allow"); + assert_eq!( + reloader.acl().allow_file_entries(), + vec!["node-a".to_string()] + ); + + make_readable(&hosts); + } + + #[tokio::test] + async fn test_acl_reload_does_not_publish_an_empty_acl_over_an_enforcing_one() { + let dir = tempfile::tempdir().unwrap(); + let allow = dir.path().join("peers.allow"); + let deny = dir.path().join("peers.deny"); + let allowed = test_npub(); + + write_file(&allow, &format!("{allowed}\n")); + let mut reloader = PeerAclReloader::with_paths(allow.clone(), deny); + assert_eq!( + reloader.acl().check(&test_peer(&allowed)), + PeerAclDecision::AllowList + ); + + std::thread::sleep(std::time::Duration::from_millis(5)); + write_file(&allow, ""); + + assert!(!reloader.reload().await); + assert_eq!( + reloader.acl().check(&test_peer(&allowed)), + PeerAclDecision::AllowList + ); + + // The hold is bounded: a file the operator really did blank in place + // is published on the following tick. + assert!(reloader.reload().await); + assert!(reloader.acl().is_empty()); + } + #[test] fn test_acl_status_reports_effective_state_and_entries() { let dir = tempfile::tempdir().unwrap(); @@ -975,7 +1258,7 @@ mod tests { hosts.insert("node-a", &npub).unwrap(); std::fs::write(&allow, "node-a\n").unwrap(); - let acl = PeerAcl::load_files_with_hosts(&allow, &deny, &hosts); + let acl = PeerAcl::try_load_files_with_hosts(&allow, &deny, &hosts).unwrap(); let peer = PeerIdentity::from_npub(&npub).unwrap(); assert_eq!(acl.allow_file_entries(), vec!["node-a".to_string()]); diff --git a/src/upper/hosts.rs b/src/upper/hosts.rs index 3ba6f568..468610d4 100644 --- a/src/upper/hosts.rs +++ b/src/upper/hosts.rs @@ -134,17 +134,34 @@ impl HostMap { /// /// If the file does not exist, returns an empty map (not an error). /// Parse errors on individual lines are logged as warnings and skipped. + /// A read failure is logged and also yields an empty map; a caller that + /// must not mistake an unreadable file for an empty one uses + /// [`Self::try_load_hosts_file`] instead. pub fn load_hosts_file(path: &Path) -> Self { - let contents = match std::fs::read_to_string(path) { - Ok(c) => c, - Err(e) if e.kind() == std::io::ErrorKind::NotFound => { - debug!(path = %path.display(), "No hosts file found, skipping"); - return Self::new(); - } + match Self::try_load_hosts_file(path) { + Ok(map) => map, Err(e) => { warn!(path = %path.display(), error = %e, "Failed to read hosts file"); - return Self::new(); + Self::new() } + } + } + + /// Load a host map from a hosts file, reporting read failures. + /// + /// An absent file is a policy, not a fault: it resolves to an empty map + /// and `Ok`. Anything else — no read permission, an I/O error, non-UTF-8 + /// content, or a `NotFound` that contradicts a successful stat and so + /// means the file is being rewritten under us — is returned as an error + /// so the caller can keep whatever it loaded last. + pub fn try_load_hosts_file(path: &Path) -> Result { + let contents = match std::fs::read_to_string(path) { + Ok(c) => c, + Err(e) if e.kind() == std::io::ErrorKind::NotFound && file_mtime(path).is_none() => { + debug!(path = %path.display(), "No hosts file found, skipping"); + return Ok(Self::new()); + } + Err(e) => return Err(e), }; let mut map = Self::new(); @@ -183,7 +200,7 @@ impl HostMap { if !map.is_empty() { info!(path = %path.display(), count = map.len(), "Loaded hosts file"); } - map + Ok(map) } /// Merge another host map into this one. The other map wins on conflicts. @@ -224,8 +241,16 @@ impl HostMapReloader { /// /// Performs the initial load of the hosts file and merges with the base map. pub fn new(base: HostMap, path: std::path::PathBuf) -> Self { - let last_mtime = file_mtime(&path); - let hosts_file = HostMap::load_hosts_file(&path); + // A failed initial read records no mtime, so the next check sees a + // change and retries rather than treating the unread file as empty + // for the lifetime of the process. + let (last_mtime, hosts_file) = match HostMap::try_load_hosts_file(&path) { + Ok(map) => (file_mtime(&path), map), + Err(e) => { + warn!(path = %path.display(), error = %e, "Failed to read hosts file"); + (None, HostMap::new()) + } + }; let mut effective = base.clone(); effective.merge(hosts_file); @@ -242,6 +267,11 @@ impl HostMapReloader { &self.effective } + /// Path of the hosts file this reloader tracks. + pub fn path(&self) -> &Path { + &self.path + } + /// Check if the hosts file has been modified and reload if so. /// /// Returns `true` if the map was reloaded. @@ -254,7 +284,32 @@ impl HostMapReloader { // File appeared, disappeared, or was modified self.last_mtime = current_mtime; - let hosts_file = HostMap::load_hosts_file(&self.path); + self.apply(HostMap::load_hosts_file(&self.path)); + true + } + + /// Check if the hosts file has been modified and reload if so, reporting + /// read failures. + /// + /// On failure neither the recorded mtime nor the effective map is + /// touched, so the caller keeps its last-good state and the next call + /// retries. Returns `true` if the map was reloaded. + pub fn try_check_reload(&mut self) -> Result { + let current_mtime = file_mtime(&self.path); + + if current_mtime == self.last_mtime { + return Ok(false); + } + + let hosts_file = HostMap::try_load_hosts_file(&self.path)?; + self.last_mtime = current_mtime; + self.apply(hosts_file); + Ok(true) + } + + /// Replace the effective map with the base merged with a freshly read + /// hosts file. + fn apply(&mut self, hosts_file: HostMap) { let mut new_effective = self.base.clone(); new_effective.merge(hosts_file); @@ -266,7 +321,6 @@ impl HostMapReloader { entries = count, "Reloaded hosts file" ); - true } } From fa9877dbc09a861717e9030e6beff63902d25888 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:30:33 +0100 Subject: [PATCH 07/23] Key the DNS mesh-interface filter on the live TUN device name The DNS responder drops .fips queries arriving over the mesh TUN, which is what keeps a widened dns.bind_addr from exposing the hosts file's alias space to every mesh peer. The interface index driving that filter was resolved from the configured TUN name, and macOS and FreeBSD assign the device a name of the kernel's own choosing, so the lookup found nothing, the index came back None, and None disables the filter. On both platforms it had therefore never run. Resolve the index from the name of the device the node actually created, which the TUN startup path already records before the DNS responder is spawned, and log a live device whose index will not resolve instead of letting it pass for "there is no mesh interface". Linux is unaffected because the configured name is the device's name there. On macOS and FreeBSD a node with a non-loopback DNS bind now stops answering .fips queries that arrive over the mesh interface, which is the point of the filter but is a visible change for anyone who was relying on the gap. An app-owned TUN leaves the device name unset, so the filter stays off in that configuration. --- CHANGELOG.md | 23 +++++++++++++++++++++++ src/node/lifecycle.rs | 41 +++++++++++++++++++++++++++++------------ src/node/tests/unit.rs | 28 ++++++++++++++++++++++++++++ 3 files changed, 80 insertions(+), 12 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4848f047..3450b5ac 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -732,6 +732,29 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 attributes the acceptance to clock skew, since a peer configured with a longer signalling TTL than ours now reaches it too. +#### DNS responder + +- The DNS responder's mesh-interface filter now works on macOS and FreeBSD, + where it had never run. The filter drops `.fips` queries that arrive over the + mesh TUN, which is what keeps a widened `dns.bind_addr` from exposing the + hosts file's alias space to every mesh peer. It was keyed on the interface + index resolved from the *configured* TUN name, but macOS and FreeBSD assign + the device a name of the kernel's choosing (`utunN`, `tunN`), so the lookup + found nothing, the index came back `None`, and `None` disables the filter. + The index is now resolved from the name of the device the node actually + created, which the TUN startup path already records, and a live device whose + index will not resolve is logged rather than passed off as "no mesh + interface". Linux is unaffected, since the configured name is the device's + name there. **Behaviour change on macOS and FreeBSD**: a node with a + non-loopback `dns.bind_addr` stops answering `.fips` queries that arrive over + the mesh interface. **What this does not close**: with an app-owned TUN the + node never learns a device name, so the filter stays off there. **Not + measured**: whether macOS and FreeBSD attribute a locally originated query + sent to the node's own mesh address to the TUN interface, as Linux does. If + they do, such a query is now dropped on those platforms; the shipped resolver + drop-in targets `[::1]` rather than the mesh address, so the packaged path is + not affected. + #### Data-plane / routing signals - The influence a remote party has over path MTU is now bounded, and the diff --git a/src/node/lifecycle.rs b/src/node/lifecycle.rs index e8a7b385..2b1204ce 100644 --- a/src/node/lifecycle.rs +++ b/src/node/lifecycle.rs @@ -1352,13 +1352,21 @@ impl Node { let reloader = crate::upper::hosts::HostMapReloader::new(base_hosts, hosts_path); // Resolve the TUN ifindex so the responder can - // drop queries arriving on the mesh interface - // (fips0). Without this, the `::` bind exposes - // /etc/fips/hosts alias probing to any mesh peer. - // When TUN isn't enabled or the name can't be - // resolved, `None` disables the filter (there - // is no mesh surface to defend anyway). - let mesh_ifindex = Self::lookup_mesh_ifindex(self.config().tun.name()); + // drop queries arriving on the mesh interface. + // Without this, the `::` bind exposes the hosts + // file's alias space to any mesh peer. The name + // comes from the device the TUN path actually + // created, not from the configured one: macOS and + // FreeBSD assign utunN/tunN of their own choosing + // and the configured name resolves to nothing + // there, which left the filter permanently off. + let mesh_ifindex = self.mesh_ifindex(); + if self.tun_name.is_some() && mesh_ifindex.is_none() { + warn!( + device = ?self.tun_name, + "Mesh interface index unresolved; DNS mesh filter disabled" + ); + } info!( bind = %bind, hosts = reloader.hosts().len(), @@ -1450,12 +1458,21 @@ impl Node { Ok(()) } - /// Resolve the mesh TUN interface index by name. + /// Resolve the index of the mesh TUN device this node actually created. /// - /// Returns `None` if the interface does not exist (e.g. TUN disabled - /// or not yet created). A `None` result disables the DNS responder's - /// mesh-interface filter — safe, because if there is no fips0 there - /// is no mesh exposure to defend against. + /// Reads the device name recorded when the TUN was brought up, which is + /// the kernel's name rather than the configured one. Returns `None` when + /// no TUN is up, which disables the DNS responder's mesh-interface + /// filter: with no mesh interface there is no mesh exposure to defend. + /// An app-owned TUN also leaves the name unset, so the filter stays off + /// there even though a mesh interface exists. + pub(crate) fn mesh_ifindex(&self) -> Option { + self.tun_name.as_deref().and_then(Self::lookup_mesh_ifindex) + } + + /// Resolve an interface index by name. + /// + /// Returns `None` if the interface does not exist. fn lookup_mesh_ifindex(name: &str) -> Option { #[cfg(unix)] { diff --git a/src/node/tests/unit.rs b/src/node/tests/unit.rs index 509aef5c..3f35f19d 100644 --- a/src/node/tests/unit.rs +++ b/src/node/tests/unit.rs @@ -2568,3 +2568,31 @@ fn test_peer_display_name_tracks_alias_change() { peer_identity.short_npub() ); } + +/// The DNS mesh-interface filter is keyed on the device the node actually +/// created, not on the configured name. macOS and FreeBSD hand out utunN and +/// tunN of the kernel's choosing, so a filter keyed on the configured name +/// resolved to nothing there and was permanently off. +#[cfg(unix)] +#[test] +fn mesh_filter_resolves_the_live_tun_device_rather_than_the_configured_name() { + let loopback = if cfg!(target_os = "macos") { + "lo0" + } else { + "lo" + }; + let c_name = std::ffi::CString::new(loopback).unwrap(); + let expected = unsafe { libc::if_nametoindex(c_name.as_ptr()) }; + if expected == 0 { + return; + } + + let mut config = Config::new(); + config.tun.name = Some("fips-absent-dev".to_string()); + let mut node = Node::new(config).unwrap(); + + assert_eq!(node.mesh_ifindex(), None); + + node.tun_name = Some(loopback.to_string()); + assert_eq!(node.mesh_ifindex(), Some(expected)); +} From b247fb6166dbf019c9fac2471af8efb755d8259a Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:17:24 +0100 Subject: [PATCH 08/23] Bind relay-returned adverts to the peer they claim to describe The relay pool verifies each event's signature but does not check a reply against the request filter, and neither option that would make it do so is enabled. The stale-advert refetch picked the newest created_at across everything a relay returned, with no author test, then cached it under the requested peer's npub, so one hostile advert relay could pin an endpoint set of its own choosing. Filter on the author before the timestamp contest, so a future-dated foreign event cannot suppress the genuine advert by winning it, and apply the same filter to the inbox-relay lookup, where the omission let an attacker-authored relay list steer our direct-message and signal traffic. A refetch that returns events, none of them signed by the peer, now leaves the cache alone rather than evicting: that is no evidence of withdrawal, and evicting on it hands the same relay a way to clear the entry. An empty answer still evicts. Clamp an advert's created_at forward to the 60s of clock skew the traversal signal path already tolerates, on the stored timestamp as well as the validity window, so a future-dated advert can neither outlive nor outrank a later genuine one. Clamp rather than refuse: a node whose own clock runs slow reads every honest advert as future-dated, and refusing would silently withdraw Nostr-mediated dialing for every peer at once. --- CHANGELOG.md | 31 +++++++++ src/discovery/nostr/runtime.rs | 83 ++++++++++++++++++----- src/discovery/nostr/tests.rs | 116 ++++++++++++++++++++++++++++++++- src/node/tests/unit.rs | 41 ++++++++++++ 4 files changed, 251 insertions(+), 20 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3450b5ac..b10913b7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -680,6 +680,37 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 #### NAT traversal / Nostr discovery +- An advert or inbox-relay list returned by a relay is now checked against + the peer it claims to describe before anything else looks at it. The relay + pool verifies every event's signature but does not check a reply against + the request filter, and neither of the two options that would make it do so + is enabled, so a relay may answer a request for one author's advert with an + event it signed itself. The stale-advert refetch picked the newest + `created_at` across everything returned, with no author test, and then wrote + the result into the advert cache under the requested peer's npub, so a + single hostile or compromised advert relay could pin an endpoint set of its + own choosing for that peer. The author test now runs before the timestamp + contest rather than after, so a future-dated foreign event cannot even + suppress the genuine advert by winning it. The same filter now applies to + the inbox-relay lookup, where the omission let an attacker-authored relay + list steer this node's direct-message and traversal-signal traffic. A + refetch that comes back with events, none of them signed by the peer, now + leaves the cached entry alone: that is no evidence the advert was + withdrawn, and evicting on it would hand the same relay a way to clear the + cache. A refetch that genuinely comes back empty still evicts. + +- An advert's `created_at` is now clamped forward to the same 60s of clock + skew the traversal-signal path already tolerates. An unbounded future + timestamp bought a cache entry a proportionally distant validity horizon + and an unbeatable position in every replacement comparison, so a later + genuine advert could never displace it and the size-cap eviction collected + it last. The clamp applies to the stored timestamp as well as the validity + window, at all three points where an advert is cached, so ordering and + expiry now agree. Clamping rather than refusing the event is deliberate: a + node whose own clock runs slow reads every peer's honest advert as + future-dated, and refusing would silently withdraw Nostr-mediated dialing + for every peer at once. + - Traversal punch targets taken from a peer's offer or answer are now filtered and bounded. A rendezvous-enabled node previously punched every address a signed offer named, including loopback, link-local, multicast, diff --git a/src/discovery/nostr/runtime.rs b/src/discovery/nostr/runtime.rs index 41704467..3ca47211 100644 --- a/src/discovery/nostr/runtime.rs +++ b/src/discovery/nostr/runtime.rs @@ -20,9 +20,9 @@ use zeroize::{Zeroize, Zeroizing}; use super::failure_state::FailureState; use super::offer_admission::{AdmissionReject, OfferAdmission}; use super::signal::{ - FreshnessOutcome, SignalEnvelope, build_signal_event, create_traversal_answer, - create_traversal_offer, estimate_clock_skew, unwrap_signal_event, validate_offer_freshness, - validate_traversal_answer_for_offer, + FRESHNESS_SKEW_TOLERANCE_MS, FreshnessOutcome, SignalEnvelope, build_signal_event, + create_traversal_answer, create_traversal_offer, estimate_clock_skew, unwrap_signal_event, + validate_offer_freshness, validate_traversal_answer_for_offer, }; use super::stun::observe_traversal_addresses; use super::traversal::{ @@ -594,21 +594,18 @@ impl NostrDiscovery { Err(_) => return NostrRefetchOutcome::Skipped, }; - let mut newest: Option<(u64, &Event)> = None; - for ev in events.iter() { - let ts = ev.created_at.as_secs(); - match newest { - Some((cur, _)) if ts <= cur => {} - _ => newest = Some((ts, ev)), + let Some(ev) = Self::newest_event_by_author(events.iter(), target_pubkey) else { + if !events.is_empty() { + // The relays answered, but nothing they returned was signed by + // this peer. That is no evidence of absence, so keep the entry. + return NostrRefetchOutcome::Skipped; } - } - - let Some((relay_created_at, ev)) = newest else { // Absent on relays. Evict any stale cache entry. self.advert_cache.write().await.remove(peer_npub); self.failure_state.reset_streak_after_refresh(peer_npub); return NostrRefetchOutcome::Evicted; }; + let relay_created_at = Self::effective_created_at_secs(ev.created_at.as_secs(), now_ms()); match cached_created_at { Some(cached) if relay_created_at <= cached => NostrRefetchOutcome::SameAdvert, @@ -761,6 +758,8 @@ impl NostrDiscovery { if let RelayPoolNotification::Event { event, .. } = notification { if event.kind == Kind::Custom(ADVERT_KIND) { let author_npub = event.pubkey.to_bech32().expect("infallible"); + let created_at = + Self::effective_created_at_secs(event.created_at.as_secs(), now_ms()); if let Some(valid_until_ms) = self.event_valid_until_ms(&event) && let Ok(advert) = Self::parse_overlay_advert_event(&event, &self.config.app) @@ -768,7 +767,7 @@ impl NostrDiscovery { let mut cache = self.advert_cache.write().await; let should_replace = cache .get(&author_npub) - .map(|existing| existing.created_at <= event.created_at.as_secs()) + .map(|existing| existing.created_at <= created_at) .unwrap_or(true); if should_replace && author_npub != self.npub { debug!( @@ -784,7 +783,7 @@ impl NostrDiscovery { CachedOverlayAdvert { author_npub, advert, - created_at: event.created_at.as_secs(), + created_at, valid_until_ms, }, ); @@ -1559,15 +1558,16 @@ impl NostrDiscovery { if author_npub != peer_npub { continue; } + let created_at = Self::effective_created_at_secs(event.created_at.as_secs(), now_ms()); let replace = best .as_ref() - .map(|current| event.created_at.as_secs() >= current.created_at) + .map(|current| created_at >= current.created_at) .unwrap_or(true); if replace { best = Some(CachedOverlayAdvert { author_npub, advert, - created_at: event.created_at.as_secs(), + created_at, valid_until_ms, }); } @@ -1640,7 +1640,7 @@ impl NostrDiscovery { return Ok(self.config.dm_relays.clone()); } }; - let newest = events.iter().max_by_key(|event| event.created_at.as_secs()); + let newest = Self::newest_event_by_author(events.iter(), target_pubkey); if let Some(event) = newest { let relays = nip17::extract_relay_list(event) .map(|relay| relay.to_string()) @@ -1764,6 +1764,42 @@ impl NostrDiscovery { Self::compute_advert_valid_until_ms(event, self.advert_max_age_ms(), now_ms()) } + /// Newest event in `events` that was actually signed by `author`. + /// + /// The relay pool verifies each event's signature but does not check a + /// reply against the REQ filter unless `verify_subscriptions` or + /// `ban_relay_on_mismatch` is set, and neither is. A relay may therefore + /// answer an author-filtered request with an event it signed itself, so + /// the author test happens here, before the timestamp contest, not after: + /// a future-dated foreign event must not be able to suppress the genuine + /// one by winning `created_at`. + pub(super) fn newest_event_by_author<'a>( + events: impl Iterator, + author: PublicKey, + ) -> Option<&'a Event> { + events + .filter(|event| event.pubkey == author) + .max_by_key(|event| event.created_at.as_secs()) + } + + /// A peer's advert `created_at`, in seconds, clamped so it can never read + /// more than `FRESHNESS_SKEW_TOLERANCE_MS` ahead of `now_ms`. + /// + /// An unbounded future `created_at` buys a cache entry two things it + /// should not have: a proportionally distant validity horizon, and an + /// unbeatable position in every replacement comparison, so a later genuine + /// advert can never displace it. Clamping rather than rejecting is + /// deliberate: a node whose own clock runs slow reads every peer's honest + /// advert as future-dated, and rejecting would take out Nostr-mediated + /// dialing for every peer at once with nothing but a cache miss to show + /// for it. Raising the tolerance widens the window in which a future-dated + /// advert outranks an honest one; lowering it makes an ordinary clock + /// difference look hostile. + pub(super) fn effective_created_at_secs(created_at_secs: u64, now_ms: u64) -> u64 { + let ceiling_secs = now_ms.saturating_add(FRESHNESS_SKEW_TOLERANCE_MS) / 1000; + created_at_secs.min(ceiling_secs) + } + pub(super) fn compute_advert_valid_until_ms( event: &Event, advert_max_age_ms: u64, @@ -1773,7 +1809,8 @@ impl NostrDiscovery { return None; } - let created_ms = event.created_at.as_secs().saturating_mul(1000); + let created_ms = Self::effective_created_at_secs(event.created_at.as_secs(), now_ms) + .saturating_mul(1000); let created_window_until = created_ms.saturating_add(advert_max_age_ms); if created_window_until <= now_ms { return None; @@ -2001,6 +2038,16 @@ impl NostrDiscovery { cache.insert(npub, advert); } + /// The cached `created_at` for `npub`, or `None` when nothing is cached. + /// Lets a unit test observe whether a refetch evicted an entry. + pub(crate) async fn cached_created_at_for_test(&self, npub: &str) -> Option { + self.advert_cache + .read() + .await + .get(npub) + .map(|c| c.created_at) + } + /// Queue a bootstrap event directly for lifecycle tests without live relays /// or a running traversal task. pub(crate) fn push_event_for_test(&self, event: BootstrapEvent) { diff --git a/src/discovery/nostr/tests.rs b/src/discovery/nostr/tests.rs index a42122ac..d972ed8d 100644 --- a/src/discovery/nostr/tests.rs +++ b/src/discovery/nostr/tests.rs @@ -1,6 +1,7 @@ use std::collections::HashSet; use std::net::{IpAddr, SocketAddr}; +use nostr::nips::nip17; use nostr::prelude::{EventBuilder, Kind, RelayUrl, Tag, Timestamp}; use super::runtime::{ @@ -46,14 +47,29 @@ fn can_reach(local_nat: NatType, remote_nat: NatType) -> bool { } fn signed_overlay_advert_event(created_at_secs: u64, expiration_secs: Option) -> nostr::Event { - let keys = nostr::Keys::generate(); + signed_overlay_advert_event_from(&nostr::Keys::generate(), created_at_secs, expiration_secs) +} + +fn signed_overlay_advert_event_from( + keys: &nostr::Keys, + created_at_secs: u64, + expiration_secs: Option, +) -> nostr::Event { let content = r#"{"identifier":"fips-overlay-v1","version":1,"endpoints":[{"transport":"tcp","addr":"8.8.8.8:443"}]}"#; let mut builder = EventBuilder::new(Kind::Custom(ADVERT_KIND), content) .custom_created_at(Timestamp::from(created_at_secs)); if let Some(expiration_secs) = expiration_secs { builder = builder.tags([Tag::expiration(Timestamp::from(expiration_secs))]); } - builder.sign_with_keys(&keys).unwrap() + builder.sign_with_keys(keys).unwrap() +} + +fn signed_inbox_relay_event(keys: &nostr::Keys, created_at_secs: u64, relay: &str) -> nostr::Event { + EventBuilder::new(Kind::InboxRelays, "") + .tags([Tag::relay(RelayUrl::parse(relay).unwrap())]) + .custom_created_at(Timestamp::from(created_at_secs)) + .sign_with_keys(keys) + .unwrap() } #[test] @@ -196,6 +212,102 @@ fn advert_freshness_rejects_stale_created_at_without_expiration() { assert!(valid_until.is_none()); } +/// A hostile advert relay may answer an author-filtered request with an event +/// it signed itself. Selection has to drop those before the newest-`created_at` +/// contest, or a future-dated foreign advert suppresses the genuine one. +#[test] +fn advert_selection_ignores_events_not_signed_by_the_target_peer() { + let now_secs = Timestamp::now().as_secs(); + let peer_keys = nostr::Keys::generate(); + let hostile_keys = nostr::Keys::generate(); + + let hostile = signed_overlay_advert_event_from(&hostile_keys, now_secs + 3_600, None); + let genuine = signed_overlay_advert_event_from(&peer_keys, now_secs.saturating_sub(10), None); + let events = [hostile, genuine]; + + let selected = NostrDiscovery::newest_event_by_author(events.iter(), peer_keys.public_key()) + .expect("the peer's own advert should be selected"); + assert_eq!(selected.pubkey, peer_keys.public_key()); +} + +/// Nothing signed by the peer means nothing to select, even though the relays +/// did answer. The caller reads this as "no evidence", not "withdrawn". +#[test] +fn advert_selection_returns_nothing_when_every_event_is_foreign() { + let now_secs = Timestamp::now().as_secs(); + let peer_keys = nostr::Keys::generate(); + let hostile_keys = nostr::Keys::generate(); + + let events = [signed_overlay_advert_event_from( + &hostile_keys, + now_secs + 3_600, + None, + )]; + + assert!( + NostrDiscovery::newest_event_by_author(events.iter(), peer_keys.public_key()).is_none() + ); +} + +/// The same omission on the inbox-relay lookup steers this node's DM and +/// signal traffic onto relays an attacker chose, so it gets the same filter. +#[test] +fn inbox_relay_selection_ignores_relay_lists_not_signed_by_the_target() { + let now_secs = Timestamp::now().as_secs(); + let peer_keys = nostr::Keys::generate(); + let hostile_keys = nostr::Keys::generate(); + + let events = [ + signed_inbox_relay_event(&hostile_keys, now_secs + 3_600, "wss://hostile.example/"), + signed_inbox_relay_event( + &peer_keys, + now_secs.saturating_sub(10), + "wss://genuine.example/", + ), + ]; + + let selected = NostrDiscovery::newest_event_by_author(events.iter(), peer_keys.public_key()) + .expect("the peer's own relay list should be selected"); + let relays = nip17::extract_relay_list(selected) + .map(|relay| relay.to_string()) + .collect::>(); + assert_eq!(relays, vec!["wss://genuine.example/".to_string()]); +} + +/// A far-future `created_at` must not buy a proportionally distant validity +/// horizon. The window is computed from the clamped timestamp instead, so the +/// entry expires on our clock rather than the publisher's. +#[test] +fn advert_freshness_clamps_created_at_beyond_the_forward_skew_tolerance() { + let now_secs = Timestamp::now().as_secs(); + let event = signed_overlay_advert_event(now_secs + 3_600, None); + let valid_until = + NostrDiscovery::compute_advert_valid_until_ms(&event, 600_000, now_secs * 1000) + .expect("a future-dated advert is still usable, just not for as long"); + assert_eq!(valid_until, (now_secs + 60) * 1000 + 600_000); +} + +/// Pins the forward bound to `FRESHNESS_SKEW_TOLERANCE_MS` exactly, mirroring +/// the signal path: 60s ahead is taken as published, 61s ahead is clamped. +/// This is the healthy-path half; an ordinary clock difference must not cost a +/// legitimate peer anything. +#[test] +fn advert_freshness_at_the_forward_skew_limit_is_untouched_and_one_second_beyond_is_clamped() { + let now_secs = Timestamp::now().as_secs(); + + let at_limit = signed_overlay_advert_event(now_secs + 60, None); + let valid_until = + NostrDiscovery::compute_advert_valid_until_ms(&at_limit, 600_000, now_secs * 1000) + .expect("an advert exactly at the forward tolerance should be accepted as published"); + assert_eq!(valid_until, (now_secs + 60) * 1000 + 600_000); + + let past_limit = signed_overlay_advert_event(now_secs + 61, None); + let clamped = + NostrDiscovery::compute_advert_valid_until_ms(&past_limit, 600_000, now_secs * 1000) + .expect("an advert one second past the tolerance is clamped, not refused"); + assert_eq!(clamped, (now_secs + 60) * 1000 + 600_000); +} + #[test] fn advert_freshness_uses_earliest_expiration_bound() { let now_secs = Timestamp::now().as_secs(); diff --git a/src/node/tests/unit.rs b/src/node/tests/unit.rs index 3f35f19d..3a9e2d84 100644 --- a/src/node/tests/unit.rs +++ b/src/node/tests/unit.rs @@ -1779,6 +1779,47 @@ fn spawn_blackhole_relay() -> String { format!("ws://127.0.0.1:{port}") } +/// The author filter on advert selection must not swallow a genuine eviction. +/// +/// Dropping foreign-authored events narrows what the selection can return, and +/// the eviction arm is guarded on the relays having answered with nothing at +/// all. If that guard is written too broadly it also suppresses the real case +/// this function exists for: the peer withdrew its advert and the cached entry +/// has to go. Discriminator: a seeded cache entry plus relays that return +/// nothing must still come back `Evicted` with the entry gone. +#[tokio::test] +async fn refetch_still_evicts_a_cached_advert_when_the_relays_return_nothing() { + let peer_npub = Identity::generate().npub(); + let mut bootstrap = NostrDiscovery::new_for_test(); + bootstrap + .set_advert_relays_for_test(vec![spawn_blackhole_relay()]) + .await; + + let endpoint = crate::discovery::nostr::OverlayEndpointAdvert { + transport: crate::discovery::nostr::OverlayTransportKind::Udp, + addr: "203.0.113.7:2121".to_string(), + }; + let advert = NostrDiscovery::cached_advert_for_test(peer_npub.clone(), endpoint, 1_000); + bootstrap + .insert_advert_for_test(peer_npub.clone(), advert) + .await; + + let outcome = bootstrap.refetch_advert_for_stale_check(&peer_npub).await; + + assert_eq!( + outcome, + crate::discovery::nostr::NostrRefetchOutcome::Evicted, + "an empty relay answer is still evidence the advert is gone" + ); + assert!( + bootstrap + .cached_created_at_for_test(&peer_npub) + .await + .is_none(), + "the stale entry should have been removed from the cache" + ); +} + /// The per-tick retry loop must not await the pre-dial advert refetch. /// /// `process_pending_retries` runs inline on the node's 1s rx-loop tick. Each From be961cb5b1c2581716f58580b2c54ff34ae7540b Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:31:59 +0100 Subject: [PATCH 09/23] Rate limit inbound traversal signals ahead of the decrypt A rendezvous-enabled node handed every kind-21059 event straight to the signal unwrap, which is two NIP-44 decrypts and a signature verify, inline on the single task that also routes traversal answers and maintains the advert cache. Nothing bounded how fast an unauthenticated stranger could schedule that work. The per-npub offer admission cannot: it keys on the sender's public key, which only exists once the first decrypt has already run, and it is a concurrency semaphore rather than a limit over time. A token bucket now sits between the kind check and the unwrap. What it can key on decides its shape: before decryption there is no sender identity at all, since the outer event is signed by a key generated per event, and the arrival relay is not an isolation boundary because an attacker publishes to the same relays an honest peer does. The shared allowance is therefore a single global bucket and is indiscriminate by construction. On its own that would shed our own traversals along with the attacker's, and since the attacker sets the rate every retry would land in the same shed, so a second smaller allowance is held in reserve and drawn only while this node has traversals of its own outstanding. A flood then denies a node its inbound offers, which nothing receiver-side can prevent without a pre-decrypt identity, rather than also denying it the answers to offers it sent. Shed signals are counted and recorded at debug level per event, with a warning each time the running total doubles, so a bucket sized below a busy node's real need appears in the log rather than as apparent relay flakiness. Two limits on what this buys, stated rather than left to be discovered. It bounds the crypto path and not the whole loop: the advert branch runs earlier in the same task and is not metered here. And the relay SDK verifies each event's outer signature on its own per-relay task before this loop sees it, which no receiver-side change short of dropping the subscription avoids. --- CHANGELOG.md | 36 +++++ src/discovery/nostr/mod.rs | 1 + src/discovery/nostr/runtime.rs | 45 +++++- src/discovery/nostr/signal_gate.rs | 252 +++++++++++++++++++++++++++++ 4 files changed, 333 insertions(+), 1 deletion(-) create mode 100644 src/discovery/nostr/signal_gate.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index b10913b7..aa4962c9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -711,6 +711,42 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 future-dated, and refusing would silently withdraw Nostr-mediated dialing for every peer at once. +- Inbound traversal signals are now rate limited before they are decrypted. A + rendezvous-enabled node handed every kind-21059 event straight to the unwrap, + which is two NIP-44 decrypts and a signature verify, inline on the single + task that also routes traversal answers and maintains the advert cache. + Nothing bounded how fast an unauthenticated stranger could schedule that + work: the per-npub offer admission cannot, because it keys on the sender's + public key, which only exists once the first decrypt has already run, and + because it is a concurrency semaphore rather than a limit over time. A token + bucket now sits ahead of the unwrap, so a flood costs a node its inbound + offers instead of the whole notify loop. + + What the limit can and cannot key on is worth stating, because it decides the + shape of the fix. Before decryption there is no sender identity at all: the + outer event is signed by a key generated per event, so bucketing on its + author would hand an attacker a fresh allowance for free, and the timestamp + and recipient tag are equally attacker-chosen. The arrival relay is drawn + from our own configured set but is not an isolation boundary either, since an + attacker publishes to the same relays an honest peer does. The shared + allowance is therefore a single global bucket and is indiscriminate by + construction, which on its own would shed our own traversals along with the + attacker's, and since the attacker sets the rate every retry would land in + the same shed. A second, smaller allowance is held in reserve and drawn only + while this node has traversals of its own outstanding, so a flood denies a + node its inbound offers, which nothing receiver-side can prevent without a + pre-decrypt identity, rather than also denying it the answers to offers it + sent. Shed signals are counted and reported at debug level per event with a + warning each time the running total doubles, so a bucket sized below a busy + node's real need shows up in the log rather than as apparent relay flakiness. + + Two limits on what this buys. It bounds the crypto path only: the advert + branch runs earlier in the same loop and is not metered here, so a stranger + can still put JSON parsing and a cache insert on the task per event. And the + relay SDK verifies each event's outer signature on its own per-relay task + before this loop ever sees it, which no receiver-side change short of + dropping the subscription can avoid. + - Traversal punch targets taken from a peer's offer or answer are now filtered and bounded. A rendezvous-enabled node previously punched every address a signed offer named, including loopback, link-local, multicast, diff --git a/src/discovery/nostr/mod.rs b/src/discovery/nostr/mod.rs index 6d0acd78..45e1053c 100644 --- a/src/discovery/nostr/mod.rs +++ b/src/discovery/nostr/mod.rs @@ -2,6 +2,7 @@ mod failure_state; mod offer_admission; mod runtime; mod signal; +mod signal_gate; mod stun; mod traversal; mod types; diff --git a/src/discovery/nostr/runtime.rs b/src/discovery/nostr/runtime.rs index 3ca47211..9f670a61 100644 --- a/src/discovery/nostr/runtime.rs +++ b/src/discovery/nostr/runtime.rs @@ -1,7 +1,7 @@ use std::collections::{HashMap, HashSet}; use std::net::SocketAddr; use std::sync::Arc; -use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; use std::time::{Duration, Instant}; use nostr::nips::nip17; @@ -24,6 +24,7 @@ use super::signal::{ create_traversal_answer, create_traversal_offer, estimate_clock_skew, unwrap_signal_event, validate_offer_freshness, validate_traversal_answer_for_offer, }; +use super::signal_gate::SignalGate; use super::stun::observe_traversal_addresses; use super::traversal::{ PunchTargetTally, is_doc_ip, is_never_punchable_ip, is_private_ip, nonce, now_ms, @@ -243,6 +244,9 @@ pub struct NostrDiscovery { active_initiators: Mutex>, seen_sessions: Mutex>, admission: OfferAdmission, + signal_gate: SignalGate, + /// Inbound traversal signals shed before decryption, since process start. + shed_signals: AtomicU64, event_tx: mpsc::UnboundedSender, event_rx: Mutex>, connect_task: Mutex>>, @@ -326,6 +330,8 @@ impl NostrDiscovery { active_initiators: Mutex::new(HashSet::new()), seen_sessions: Mutex::new(HashMap::new()), admission, + signal_gate: SignalGate::new(Instant::now()), + shed_signals: AtomicU64::new(0), event_tx, event_rx: Mutex::new(event_rx), connect_task: Mutex::new(None), @@ -797,6 +803,41 @@ impl NostrDiscovery { continue; } + // Ahead of the unwrap, which is two NIP-44 decrypts and a + // signature verify run inline on the single task that also + // routes answers and processes adverts. Nothing about the + // sender is known yet — the outer event is signed by a key + // generated per event — so the allowance is necessarily + // shared and indiscriminate, and the reserve is what keeps + // a flood from also shedding the answers to traversals + // this node started. + let awaiting_answers = match self.pending_answers.try_lock() { + Ok(pending) => !pending.is_empty(), + // Contended rather than known empty, so treat it as + // outstanding: the fail-open direction here spends the + // reserve, it does not shed. + Err(_) => true, + }; + if let Err(shed) = self.signal_gate.admit(awaiting_answers, Instant::now()) { + let total = self.shed_signals.fetch_add(1, Ordering::Relaxed) + 1; + // Debug, not warn, per event: the party that trips this + // is by definition sending faster than the node wants, + // so a record per drop turns the flood into log volume. + // The doubling summary below is the operator's signal. + debug!( + reason = ?shed, + total, + "shed inbound traversal signal before decrypt" + ); + if total.is_power_of_two() { + warn!( + shed = total, + "inbound traversal signals shed before decrypt" + ); + } + continue; + } + let unwrapped = match unwrap_signal_event(&self.keys, &event).await { Ok(unwrapped) => unwrapped, Err(err) => { @@ -1984,6 +2025,8 @@ impl NostrDiscovery { active_initiators: Mutex::new(HashSet::new()), seen_sessions: Mutex::new(HashMap::new()), admission, + signal_gate: SignalGate::new(Instant::now()), + shed_signals: AtomicU64::new(0), event_tx, event_rx: Mutex::new(event_rx), connect_task: Mutex::new(None), diff --git a/src/discovery/nostr/signal_gate.rs b/src/discovery/nostr/signal_gate.rs new file mode 100644 index 00000000..95b1ac55 --- /dev/null +++ b/src/discovery/nostr/signal_gate.rs @@ -0,0 +1,252 @@ +//! Rate limiting for inbound traversal signals, ahead of any cryptography. +//! +//! The notify loop used to hand every kind-21059 event straight to +//! `unwrap_signal_event`, which is two NIP-44 decrypts and a signature verify, +//! on the single task that also routes answers and processes adverts. Nothing +//! bounded how fast an unauthenticated stranger could schedule that work, and +//! the per-npub offer admission cannot: it keys on the sender's public key, +//! which only exists once the first decrypt has already run. +//! +//! **What can be keyed on, and what cannot.** Before decryption there is no +//! sender identity at all. The outer event is signed by a key generated per +//! event, so bucketing on its author hands an attacker a fresh allowance for +//! free, and `created_at` and the p-tag are equally attacker-chosen. The relay +//! the event arrived over is drawn from our own configured set, but it is not +//! an isolation boundary either: an attacker publishes to the same relays the +//! honest peer does, and a duplicate event is attributed to whichever relay +//! won the delivery race. So the shared allowance here is deliberately a +//! single global bucket, and it is indiscriminate by construction. +//! +//! **What the reserve is for.** An indiscriminate limit sheds our own +//! traversals along with the attacker's, and since the attacker sets the rate, +//! every retry lands in the same shed. The second bucket is drawn only while +//! this node has traversals of its own outstanding, so a flood costs a node +//! its inbound offers, which is irreducible without a pre-decrypt identity, +//! rather than also costing it the answers to offers it sent. +//! +//! **Lock discipline.** `admit` takes `now` as a parameter rather than reading +//! the clock, so the type is testable without sleeping and holds no state that +//! has to be advanced by a timer. It does its whole decision under one +//! `std::sync::Mutex` and never awaits inside it, as `offer_admission` does; +//! moving anything awaited inside that lock would hold it across a decrypt. + +use std::sync::Mutex; +use std::time::Instant; + +/// Sustained inbound traversal signals per second admitted for decryption, +/// across all senders and relays. +/// +/// A traversal exchange is a handful of events (one offer and one answer per +/// attempt), and the signal subscription is opened with `limit(0)` so relays +/// replay no stored backlog, which means there is no legitimate burst larger +/// than the number of peers bootstrapping in the same second. Raising this +/// buys a larger rendezvous hub headroom at the cost of handing an attacker +/// the same multiple of decrypt work; lowering it starts shedding honest +/// signals on a busy node, which shows up in the log as the shed counter +/// rather than as silence. +const SIGNAL_RATE_PER_SEC: f64 = 5.0; + +/// Burst capacity of the shared allowance, in signals. +const SIGNAL_BURST: f64 = 20.0; + +/// Sustained rate of the reserve, drawn only while this node has traversals +/// of its own outstanding. +/// +/// It is sized like the shared allowance rather than smaller because the case +/// it exists for is onboarding fanout: a node that has just sent offers to +/// tens of peers receives their answers back in a burst, and shedding those +/// looks to an operator like relay flakiness. Lowering it re-exposes that +/// case; raising it lets a flood arriving while we happen to be mid-traversal +/// buy more decrypt work than the shared allowance alone would. +const ANSWER_RESERVE_RATE_PER_SEC: f64 = 5.0; + +/// Burst capacity of the reserve, in signals. +const ANSWER_RESERVE_BURST: f64 = 20.0; + +/// Which allowance refused an inbound signal. +/// +/// The two are different operator stories: the first says the node shed a +/// signal while it had nothing of its own in flight, the second says a flood +/// is now deep enough to reach traversals this node started. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum SignalShed { + /// The shared allowance is exhausted. + Shared, + /// The shared allowance and the answer reserve are both exhausted. + Reserve, +} + +/// One token bucket, refilled from the caller's clock. +#[derive(Debug)] +struct Bucket { + tokens: f64, + burst: f64, + rate_per_sec: f64, + last: Instant, +} + +impl Bucket { + fn new(rate_per_sec: f64, burst: f64, now: Instant) -> Self { + Self { + tokens: burst, + burst, + rate_per_sec, + last: now, + } + } + + /// Advance the bucket to `now`, capped at its burst. + fn refill(&mut self, now: Instant) { + let elapsed = now.saturating_duration_since(self.last).as_secs_f64(); + if elapsed > 0.0 { + self.tokens = (self.tokens + elapsed * self.rate_per_sec).min(self.burst); + self.last = now; + } + } + + /// Whether a whole token is available, without taking it. + fn ready(&self) -> bool { + self.tokens >= 1.0 + } + + fn take(&mut self) { + self.tokens -= 1.0; + } +} + +/// The pre-decrypt allowance for inbound traversal signals. +pub(super) struct SignalGate { + inner: Mutex, +} + +#[derive(Debug)] +struct Inner { + shared: Bucket, + reserve: Bucket, +} + +impl SignalGate { + /// Build a gate whose buckets start full as of `now`. + pub(super) fn new(now: Instant) -> Self { + Self::with_limits( + SIGNAL_RATE_PER_SEC, + SIGNAL_BURST, + ANSWER_RESERVE_RATE_PER_SEC, + ANSWER_RESERVE_BURST, + now, + ) + } + + fn with_limits( + shared_rate: f64, + shared_burst: f64, + reserve_rate: f64, + reserve_burst: f64, + now: Instant, + ) -> Self { + Self { + inner: Mutex::new(Inner { + shared: Bucket::new(shared_rate, shared_burst, now), + reserve: Bucket::new(reserve_rate, reserve_burst, now), + }), + } + } + + /// Take one signal's worth of allowance, or say which bucket refused it. + /// + /// `awaiting_answers` says whether this node has traversals of its own + /// outstanding; only then may the reserve be drawn. The shared bucket is + /// always tried first, so the reserve is spent only on what the flood + /// would otherwise have shed. + pub(super) fn admit(&self, awaiting_answers: bool, now: Instant) -> Result<(), SignalShed> { + let mut inner = self.inner.lock().expect("signal-gate mutex poisoned"); + inner.shared.refill(now); + if inner.shared.ready() { + inner.shared.take(); + return Ok(()); + } + if !awaiting_answers { + return Err(SignalShed::Shared); + } + inner.reserve.refill(now); + if inner.reserve.ready() { + inner.reserve.take(); + return Ok(()); + } + Err(SignalShed::Reserve) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::time::Duration; + + #[test] + fn a_flood_is_admitted_up_to_the_burst_and_shed_after_it() { + let start = Instant::now(); + let gate = SignalGate::new(start); + + let admitted = (0..1000) + .filter(|_| gate.admit(false, start).is_ok()) + .count(); + + assert_eq!(admitted, SIGNAL_BURST as usize); + assert_eq!(gate.admit(false, start), Err(SignalShed::Shared)); + } + + #[test] + fn tokens_refill_so_a_steady_legitimate_signal_rate_is_never_shed() { + let start = Instant::now(); + let gate = SignalGate::new(start); + + // One second per iteration at exactly the sustained rate, for long + // enough that an arithmetic error in the refill drains the bucket. + for second in 0..100u64 { + let now = start + Duration::from_secs(second); + for signal in 0..SIGNAL_RATE_PER_SEC as usize { + assert_eq!( + gate.admit(false, now), + Ok(()), + "signal {signal} of second {second} should be admitted" + ); + } + } + } + + #[test] + fn a_flood_cannot_shed_the_answers_to_traversals_this_node_started() { + let start = Instant::now(); + let gate = SignalGate::new(start); + + // The flood arrives while this node has nothing outstanding, so it + // cannot reach the reserve at all. + for _ in 0..1000 { + let _ = gate.admit(false, start); + } + assert_eq!(gate.admit(false, start), Err(SignalShed::Shared)); + + let admitted = (0..1000) + .filter(|_| gate.admit(true, start).is_ok()) + .count(); + assert_eq!(admitted, ANSWER_RESERVE_BURST as usize); + assert_eq!(gate.admit(true, start), Err(SignalShed::Reserve)); + } + + #[test] + fn the_reserve_is_untouched_while_the_shared_allowance_still_has_tokens() { + let start = Instant::now(); + let gate = SignalGate::new(start); + + // Every one of these is inside the shared burst, so none of them may + // spend the reserve even though the caller is entitled to it. + for _ in 0..SIGNAL_BURST as usize { + assert_eq!(gate.admit(true, start), Ok(())); + } + + let reserve_admitted = (0..1000) + .filter(|_| gate.admit(true, start).is_ok()) + .count(); + assert_eq!(reserve_admitted, ANSWER_RESERVE_BURST as usize); + } +} From c07140cc6b76cd49363ef6debeb6be11b21cdb4a Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:14:41 +0100 Subject: [PATCH 10/23] Accept a STUN binding response only from the server we asked The STUN client discarded the source address recv_from returned and let parse_stun_binding_success decide, and that parser checks only the message type, the magic cookie and the 12-byte transaction id. An on-path attacker who could read our outbound binding request could copy the transaction id into a reply of their own, and the address they named became the reflexive candidate published in the traversal offer or answer, redirecting the peer's hole-punch packets. Compare the datagram's source against the resolved server address before parsing it. A mismatch is counted and the datagram dropped; the loop already continues past anything it cannot parse, so a normal exchange is unaffected. Report the rejections once per attempt rather than once per datagram, since a flooder controls that rate. Record the ordering invariant the traversal call sites depend on: this drains every datagram on the shared traversal socket until its deadline, so it must not run once punching can be in flight. --- CHANGELOG.md | 14 ++++ src/discovery/nostr/runtime.rs | 8 +++ src/discovery/nostr/stun.rs | 125 ++++++++++++++++++++++++++++++++- 3 files changed, 146 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index aa4962c9..a117f7ab 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -747,6 +747,20 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 before this loop ever sees it, which no receiver-side change short of dropping the subscription can avoid. +- A STUN binding response is now accepted only from the address the binding + request was sent to. The client discarded the source address `recv_from` + returned and let the parser decide, and the parser checks only the message + type, the magic cookie and the 12-byte transaction id. An on-path attacker + who could read the outbound request could therefore inject a reply carrying + a transaction id copied from it, and its chosen address became the reflexive + candidate the node published in a traversal offer or answer, redirecting the + peer's hole-punch packets. Datagrams from any other source are counted and + discarded, and one debug record per STUN attempt reports the count and the + last unexpected source, so a rejection is diagnosable without giving a + flooder control of the log rate. A server that answers from an address other + than the one dialed, which RFC 5389 forbids, now times out and the next + configured server is tried. + - Traversal punch targets taken from a peer's offer or answer are now filtered and bounded. A rendezvous-enabled node previously punched every address a signed offer named, including loopback, link-local, multicast, diff --git a/src/discovery/nostr/runtime.rs b/src/discovery/nostr/runtime.rs index 9f670a61..9f5b6338 100644 --- a/src/discovery/nostr/runtime.rs +++ b/src/discovery/nostr/runtime.rs @@ -1210,6 +1210,10 @@ impl NostrDiscovery { let base_socket = std::net::UdpSocket::bind(("0.0.0.0", 0))?; base_socket.set_nonblocking(true)?; + // This drains every datagram on the traversal socket until the STUN + // deadline, so it must complete before any punch can be in flight: a + // retry or re-observation once punching has started would swallow the + // peer's punch packets. let (reflexive_address, local_addresses, stun_server) = observe_traversal_addresses( &base_socket, &self.config.stun_servers, @@ -1461,6 +1465,10 @@ impl NostrDiscovery { let base_socket = std::net::UdpSocket::bind(("0.0.0.0", 0))?; base_socket.set_nonblocking(true)?; + // This drains every datagram on the traversal socket until the STUN + // deadline, so it must complete before any punch can be in flight: a + // retry or re-observation once punching has started would swallow the + // peer's punch packets. let (reflexive_address, local_addresses, stun_server) = observe_traversal_addresses( &base_socket, &self.config.stun_servers, diff --git a/src/discovery/nostr/stun.rs b/src/discovery/nostr/stun.rs index 3aa7a14a..777a2ec1 100644 --- a/src/discovery/nostr/stun.rs +++ b/src/discovery/nostr/stun.rs @@ -80,6 +80,20 @@ pub(super) async fn observe_traversal_addresses( Ok((None, local_addresses, None)) } +/// Send a STUN binding request on `socket` and return the reflexive address +/// the server reports. +/// +/// Only a datagram whose source is exactly the resolved server address is +/// parsed; anything else is consumed and discarded, so an on-path attacker +/// who has seen the transaction id cannot substitute a mapped address by +/// injecting a reply. The comparison is exact because every socket handed to +/// this function is bound to the IPv4 wildcard, so a source can never arrive +/// in v4-mapped IPv6 form. A dual-stack bind would require normalising both +/// sides with `IpAddr::to_canonical()` before comparing. +/// +/// Caller requirement: this drains and discards every datagram arriving on +/// `socket` until the deadline, so it must not be entered on a socket that +/// may concurrently carry other traffic the caller cares about. async fn perform_stun( socket: &std::net::UdpSocket, stun_server: &str, @@ -95,15 +109,32 @@ async fn perform_stun( udp.send_to(&request, addr).await?; let mut buf = [0u8; 2048]; let deadline = tokio::time::Instant::now() + response_timeout; + let mut rejected = 0u64; + let mut last_unexpected = None; loop { let result = tokio::time::timeout_at(deadline, udp.recv_from(&mut buf)).await; - let Ok(Ok((len, _remote))) = result else { + let Ok(Ok((len, remote))) = result else { break; }; + if remote != addr { + rejected += 1; + last_unexpected = Some(remote); + continue; + } if let Some(mapped) = parse_stun_binding_success(&buf[..len], &txn_id) { return Ok(Some(mapped)); } } + // One line per call rather than per datagram: a flooder controls the rate. + if rejected > 0 { + debug!( + stun_server = %stun_server, + expected = %addr, + rejected, + last_unexpected = ?last_unexpected, + "discarded STUN datagrams from unexpected sources" + ); + } Err(BootstrapError::Stun(format!( "timed out waiting for {}", stun_server @@ -521,4 +552,96 @@ mod tests { "2001:db8::1".parse::().unwrap() ))); } + + /// Build a complete Binding Success carrying one XOR-MAPPED-ADDRESS. + fn build_binding_success(mapped: std::net::SocketAddrV4, txn_id: &[u8; 12]) -> Vec { + let mut packet = build_success_header(0, txn_id); + let cookie = STUN_MAGIC_COOKIE.to_be_bytes(); + let octets = mapped.ip().octets(); + let xport = mapped.port() ^ ((STUN_MAGIC_COOKIE >> 16) as u16); + packet.extend_from_slice(&0x0020u16.to_be_bytes()); // XOR-MAPPED-ADDRESS + packet.extend_from_slice(&8u16.to_be_bytes()); + packet.push(0x00); // reserved + packet.push(0x01); // family IPv4 + packet.extend_from_slice(&xport.to_be_bytes()); + for (index, octet) in octets.iter().enumerate() { + packet.push(octet ^ cookie[index]); + } + let body_len = (packet.len() - 20) as u16; + packet[2..4].copy_from_slice(&body_len.to_be_bytes()); + packet + } + + /// Bind a loopback socket suitable for handing to `perform_stun`. + /// + /// `set_nonblocking` is mandatory rather than tidiness: `perform_stun` + /// passes the socket to `tokio::net::UdpSocket::from_std`, which requires + /// a non-blocking socket and does not make one. A blocking socket parks + /// the runtime thread and the deadline never fires. + fn bind_stun_caller() -> std::net::UdpSocket { + let socket = std::net::UdpSocket::bind("127.0.0.1:0").unwrap(); + socket.set_nonblocking(true).unwrap(); + socket + } + + #[tokio::test] + async fn stun_binding_response_from_an_unexpected_source_is_refused() { + let caller = bind_stun_caller(); + let server = std::net::UdpSocket::bind("127.0.0.1:0").unwrap(); + let attacker = std::net::UdpSocket::bind("127.0.0.1:0").unwrap(); + let server_addr = server.local_addr().unwrap(); + + // Stands in for an on-path attacker: it learns the transaction id the + // way a real one would, by reading the request, and answers from its + // own address while the server stays silent. + let forger = std::thread::spawn(move || { + let mut buf = [0u8; 2048]; + let (len, from) = server.recv_from(&mut buf).unwrap(); + assert!(len >= 20); + let mut txn_id = [0u8; 12]; + txn_id.copy_from_slice(&buf[8..20]); + let forged = build_binding_success("203.0.113.7:1".parse().unwrap(), &txn_id); + attacker.send_to(&forged, from).unwrap(); + }); + + let result = super::perform_stun( + &caller, + &server_addr.to_string(), + std::time::Duration::from_millis(300), + ) + .await; + forger.join().unwrap(); + assert!( + result.is_err(), + "a binding success from a host other than the server must not be believed, got {:?}", + result + ); + } + + #[tokio::test] + async fn stun_binding_response_from_the_server_is_accepted() { + let caller = bind_stun_caller(); + let server = std::net::UdpSocket::bind("127.0.0.1:0").unwrap(); + let server_addr = server.local_addr().unwrap(); + + let responder = std::thread::spawn(move || { + let mut buf = [0u8; 2048]; + let (_len, from) = server.recv_from(&mut buf).unwrap(); + let mut txn_id = [0u8; 12]; + txn_id.copy_from_slice(&buf[8..20]); + let reply = build_binding_success("198.51.100.9:4242".parse().unwrap(), &txn_id); + server.send_to(&reply, from).unwrap(); + }); + + let mapped = super::perform_stun( + &caller, + &server_addr.to_string(), + std::time::Duration::from_secs(2), + ) + .await + .unwrap() + .unwrap(); + responder.join().unwrap(); + assert_eq!(mapped.to_string(), "198.51.100.9:4242"); + } } From 9b8fe7b54f13d8bb08a5966ca0d339e3a2d9925a Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:20:06 +0100 Subject: [PATCH 11/23] Apply the private-address gate to a peer's reflexive address The exemption that lets a peer's reflexive address skip the private-address gate exists for the deployment whose STUN server sits inside the private network, so the observed reflexive address is legitimately private. It was applied unconditionally, so a node whose own STUN result was public still punched whatever private address a peer named as its reflexive one, and any sender whose offer or answer was accepted could aim a burst of UDP packets carrying this node's source address at a host inside the node's own private network. The gate now applies whenever our own reflexive address is public. It stays lifted when our own reflexive address is itself private, which is the LAN-STUN deployment it was written for, and also when we have no reflexive address at all, so a failed STUN probe cannot cost a node its same-LAN peering. An off-subnet refusal of a peer's reflexive address is a shape an honest asymmetric-STUN deployment now produces, so it no longer raises the refusal record to warning level on its own; the never-routable, port-0 and unparsable classes still do. A peer's candidate list is also bounded before it is walked rather than only after. The eight-target cap ran after both planning loops, so it bounded what a node punched but not what it spent deciding, and the deduplicating scan is quadratic in the plan those candidates feed. At most 32 candidates are vetted now, and the excess is recorded in the refusal tally and discarded rather than failing the offer, so an honest many-homed peer loses the tail of its list instead of its traversal. --- CHANGELOG.md | 36 +++++++++++++ src/discovery/nostr/runtime.rs | 4 +- src/discovery/nostr/tests.rs | 92 +++++++++++++++++++++++++++++++- src/discovery/nostr/traversal.rs | 71 +++++++++++++++++++----- 4 files changed, 189 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a117f7ab..ddce350c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -761,6 +761,42 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 than the one dialed, which RFC 5389 forbids, now times out and the next configured server is tried. +- The exemption that lets a peer's reflexive address skip the private-address + gate is now conditional on our own vantage point. That exemption exists for + the deployment whose STUN server sits inside the private network, so the + observed reflexive address is legitimately private; it was applied + unconditionally, so a node whose own STUN result was public still punched + whatever private address a peer named as its reflexive one. Any sender whose + offer or answer was accepted could therefore aim a burst of UDP packets, + carrying this node's source address, at a host inside the node's own private + network, which is the one place the candidate filter was written to keep it + out of. The gate now applies whenever our own reflexive address is public. + It stays lifted when our own reflexive address is itself private, which is + the LAN-STUN deployment the exemption was for, and also when we have no + reflexive address at all, so a failed STUN probe cannot cost a node its + same-LAN peering. Two consequences to state rather than discover: a peer + behind a private STUN server talking to a node with a public one loses its + reflexive candidate, which was never reachable from us in any case, and + because the /24 comparison is IPv4-only a unique-local IPv6 reflexive + address is refused unless our own reflexive address is unique-local too. + An off-subnet refusal of a peer's reflexive address is a shape an honest + deployment now produces, so it no longer raises the refusal record to + warning level on its own; the never-routable, port-0 and unparsable classes + still do. + +- A peer's candidate list is now bounded before it is walked rather than only + after. The eight-target cap ran after both planning loops had finished, so + it bounded what a node punched but not what it spent deciding: a signal + naming several thousand candidates had every one of them parsed and vetted, + and the deduplicating scan that follows is quadratic in the plan those + candidates feed. At most 32 candidates are now vetted, four times the + target cap and four times what the candidate generator produces on the + widest host, and the excess is discarded rather than failing the offer, so + an honest many-homed peer loses the tail of its list instead of its + traversal. The refusal record carries the discarded count as a new + `over_offered` field and treats a non-zero one as an attack shape, since + nothing honest reaches the bound. + - Traversal punch targets taken from a peer's offer or answer are now filtered and bounded. A rendezvous-enabled node previously punched every address a signed offer named, including loopback, link-local, multicast, diff --git a/src/discovery/nostr/runtime.rs b/src/discovery/nostr/runtime.rs index 9f5b6338..c541abcf 100644 --- a/src/discovery/nostr/runtime.rs +++ b/src/discovery/nostr/runtime.rs @@ -83,7 +83,7 @@ pub(super) fn short_id(id: &str) -> String { /// node collects by default and those refusals are the ones an operator needs /// to see; the routine off-subnet case stays at `debug`. fn log_refusals(tally: &PunchTargetTally, peer: &str, session: &str) { - if tally.offered <= tally.admitted && tally.capped == 0 { + if tally.offered <= tally.admitted && tally.capped == 0 && tally.over_offered == 0 { return; } let sample = tally.sample.as_deref().unwrap_or("-"); @@ -99,6 +99,7 @@ fn log_refusals(tally: &PunchTargetTally, peer: &str, session: &str) { unroutable = tally.unroutable, offsubnet = tally.offsubnet, capped = tally.capped, + over_offered = tally.over_offered, reflexive = %reflexive, sample = %sample, "traversal: punch candidates refused" @@ -114,6 +115,7 @@ fn log_refusals(tally: &PunchTargetTally, peer: &str, session: &str) { unroutable = tally.unroutable, offsubnet = tally.offsubnet, capped = tally.capped, + over_offered = tally.over_offered, reflexive = %reflexive, sample = %sample, "traversal: punch candidates refused" diff --git a/src/discovery/nostr/tests.rs b/src/discovery/nostr/tests.rs index d972ed8d..11ee567b 100644 --- a/src/discovery/nostr/tests.rs +++ b/src/discovery/nostr/tests.rs @@ -699,7 +699,11 @@ fn planned_remote_endpoints_bound_an_oversized_list_of_unroutable_candidates() { endpoints, vec!["198.51.100.20:63000".parse::().unwrap()] ); - assert_eq!(tally.unroutable, 300); + // Only the first MAX_OFFERED_CANDIDATES are vetted at all now, so the + // unroutable count is the bound rather than the whole list; the rest are + // recorded as never having been looked at. + assert_eq!(tally.unroutable, 32); + assert_eq!(tally.over_offered, 268); assert_eq!(tally.admitted, 1); assert!(tally.suspicious()); } @@ -708,6 +712,10 @@ fn planned_remote_endpoints_bound_an_oversized_list_of_unroutable_candidates() { /// so the observed reflexive address is itself private. Applying the /24 gate /// to a peer's reflexive address would drop it and remove the only branch /// that works across arbitrary NATs; this test reds if anyone does that. +/// +/// The exemption is conditional on exactly the vantage point this test sets +/// up: our own reflexive address is private here, so it still applies. The +/// two tests below cover the public and absent cases. #[test] fn planned_remote_endpoints_keep_private_reflexive_when_stun_is_on_the_lan() { let (endpoints, _tally) = planned_remote_endpoints( @@ -721,6 +729,88 @@ fn planned_remote_endpoints_keep_private_reflexive_when_stun_is_on_the_lan() { assert!(endpoints.contains(&"192.168.1.20:63000".parse().unwrap())); } +/// A node whose own STUN result is public shares no LAN with a private +/// address, so a peer's private reflexive address is only ever an address of +/// the peer's choosing. Admitting it made the reflexive branch a way to have +/// this node punch inside its own private network; the /24 gate now applies. +#[test] +fn a_peers_private_reflexive_address_is_refused_when_our_own_stun_result_is_public() { + let (endpoints, tally) = planned_remote_endpoints( + &[], + Some(&addr("203.0.113.10", 62000)), + &[], + Some(&addr("192.168.1.20", 63000)), + ) + .expect("endpoint planning should succeed"); + + assert!(endpoints.is_empty()); + assert_eq!(tally.reflexive, Some("off-subnet")); +} + +/// The conditional gate keys on our own reflexive address being private, and +/// a node with no reflexive address at all has to keep behaving as it did: +/// a failed STUN probe must not cost same-LAN peering. +#[test] +fn a_peers_private_reflexive_address_is_kept_when_we_have_no_stun_result_at_all() { + let (endpoints, tally) = planned_remote_endpoints( + &[addr("192.168.1.10", 62000)], + None, + &[], + Some(&addr("192.168.1.20", 63000)), + ) + .expect("endpoint planning should succeed"); + + assert!(endpoints.contains(&"192.168.1.20:63000".parse().unwrap())); + assert_eq!(tally.reflexive, None); +} + +/// Refusing a peer's private reflexive address is now something an honest +/// asymmetric-STUN deployment produces, so it must not warn on its own. The +/// never-routable case above still does. +#[test] +fn an_off_subnet_reflexive_refusal_alone_is_not_suspicious() { + let (_planned, tally) = plan_punch_targets( + &[], + Some(&addr("203.0.113.10", 62000)), + &[addr("203.0.113.5", 63000)], + Some(&addr("192.168.1.20", 63000)), + ); + + assert_eq!(tally.reflexive, Some("off-subnet")); + assert!( + tally.admitted > 0, + "the host-candidate path should still plan" + ); + assert!(!tally.suspicious()); +} + +/// The eight-target cap runs after both planning loops, so it bounds the +/// output and not the work. The discriminating assertion is `unroutable`: +/// vetting every candidate would count all thousand, so a count of exactly +/// `MAX_OFFERED_CANDIDATES` is what proves the excess was never walked. +#[test] +fn an_oversized_candidate_list_is_bounded_before_vetting() { + let mut remotes = Vec::new(); + for index in 0..1000u32 { + remotes.push(addr( + &format!("127.0.0.{}", 1 + (index % 254)), + 63000 + (index % 1000) as u16, + )); + } + + let (_planned, tally) = plan_punch_targets( + &[], + Some(&addr("203.0.113.10", 62000)), + &remotes, + Some(&addr("198.51.100.20", 63000)), + ); + + assert_eq!(tally.offered, 1001); + assert_eq!(tally.unroutable, 32); + assert_eq!(tally.over_offered, 968); + assert!(tally.suspicious()); +} + /// The four refusal classes tell four different operational stories, so a /// change that collapses them into one counter, or that makes the warning /// fire on the benign dual-homed shape, has to red here. diff --git a/src/discovery/nostr/traversal.rs b/src/discovery/nostr/traversal.rs index a3369311..53d1ec4e 100644 --- a/src/discovery/nostr/traversal.rs +++ b/src/discovery/nostr/traversal.rs @@ -32,6 +32,19 @@ pub(super) enum PunchStrategy { /// 400 packets, about 21 KB on the wire at 52 bytes each for IPv4. const MAX_PUNCH_TARGETS: usize = 8; +/// Upper bound on how many candidates one peer's signal may have vetted. +/// +/// Vetting is linear in this number and the `push_unique` scan that follows +/// is quadratic in the plan it feeds, so an unbounded candidate list lets one +/// signal buy an unbounded amount of our planning work regardless of the +/// eight-target cap, which only applies after both loops have run. Thirty-two +/// is four times `MAX_PUNCH_TARGETS` and four times what the candidate +/// generator produces on the widest host we have seen, so an honest peer +/// never reaches it. Raising it costs planning work per admitted signal; +/// lowering it costs an honest many-homed peer the tail of its candidate +/// list, which the tally records either way. +const MAX_OFFERED_CANDIDATES: usize = 32; + #[derive(Debug, Clone, PartialEq, Eq)] pub(super) struct PlannedPunchTarget { pub(super) strategy: PunchStrategy, @@ -44,6 +57,12 @@ pub(super) struct PlannedPunchTarget { pub(super) remote_ip: IpAddr, } +/// Whether a candidate's address text parses as a private or unique-local +/// address. +fn is_private_address(candidate: &TraversalAddress) -> bool { + candidate.ip.parse::().is_ok_and(is_private_ip) +} + fn same_subnet_24(left: &TraversalAddress, right: &TraversalAddress) -> bool { let left_parts = left.ip.split('.').collect::>(); let right_parts = right.ip.split('.').collect::>(); @@ -151,6 +170,8 @@ pub(super) struct PunchTargetTally { pub(super) offsubnet: usize, /// Planned targets discarded by the target cap. pub(super) capped: usize, + /// Candidates past `MAX_OFFERED_CANDIDATES` that were never vetted. + pub(super) over_offered: usize, /// The class label that refused the peer's reflexive address, if it was /// refused. Held apart from the candidate counts because losing the /// reflexive branch removes every path that works across arbitrary NATs, @@ -170,12 +191,21 @@ impl PunchTargetTally { /// entirely refused offer is the reflector case itself. An off-subnet-only /// refusal is the ordinary dual-homed shape and is not suspicious. /// - /// A refused reflexive address always counts: the /24 gate does not apply - /// to it, so the only ways it can be refused are the attacker-shaped ones. + /// A refused reflexive address counts unless the class is `OffSubnet`. + /// The /24 gate now applies to a peer's reflexive address whenever our own + /// reflexive address is public, so an off-subnet refusal of it is what an + /// honest peer behind a LAN STUN server produces against a node with a + /// public one. The other three classes still have no honest producer. + /// + /// A candidate list longer than `MAX_OFFERED_CANDIDATES` counts too: the + /// generator tops out near eight, so nothing honest reaches the bound. pub(super) fn suspicious(&self) -> bool { self.unroutable + self.zeroport + self.unparsable + self.capped > 0 + || self.over_offered > 0 || (self.offered > 0 && self.admitted == 0) - || self.reflexive.is_some() + || self + .reflexive + .is_some_and(|label| label != RejectClass::OffSubnet.label()) } /// Record one refused candidate against its class, keeping the first @@ -212,10 +242,13 @@ impl PunchTargetTally { /// /// Returns the parsed address, or the class of the check that refused it. /// `lan_refs` are our own addresses that a private candidate must share a /24 -/// with. `apply_private_gate` is false for the peer's reflexive address: a -/// STUN server inside the private network legitimately reports a private -/// reflexive address, and dropping it would remove the only branch that works -/// across arbitrary NATs. +/// with. `apply_private_gate` is conditionally false for the peer's reflexive +/// address: a STUN server inside the private network legitimately reports a +/// private reflexive address, and dropping it would remove the only branch +/// that works across arbitrary NATs. That exemption applies only when our own +/// reflexive address is itself private, or absent; a node whose own STUN +/// result is public has no LAN in common with a private reflexive address and +/// would only be punching an address of the peer's choosing. fn admit_remote( candidate: &TraversalAddress, lan_refs: &[TraversalAddress], @@ -261,28 +294,42 @@ pub(super) fn plan_punch_targets( ..PunchTargetTally::default() }; + // Whether our own vantage point is a LAN one: either STUN reported a + // private address for us, or it reported nothing at all. The second case + // is deliberately treated as a LAN vantage point rather than a public one, + // so a node whose STUN probe failed, or that runs without STUN, keeps + // admitting a same-LAN peer's private reflexive address as it always has. + let local_reflexive_on_lan = local_reflexive_address.is_none_or(is_private_address); + // Our own addresses a peer's private candidate has to share a /24 with. // The local reflexive address joins the set when it is itself private, // which is what keeps a LAN-STUN deployment able to match while the // shipped `share_local_candidates=false` leaves the local list empty. let mut lan_refs = local_addresses.to_vec(); if let Some(reflexive) = local_reflexive_address - && reflexive.ip.parse::().is_ok_and(is_private_ip) + && is_private_address(reflexive) { lan_refs.push(reflexive.clone()); } + // A peer names its own candidate list, so bound it before anything walks + // it. The excess is recorded and discarded rather than failing the whole + // offer, which would cost an honest many-homed peer its traversal. + let considered = &remote_addresses[..remote_addresses.len().min(MAX_OFFERED_CANDIDATES)]; + tally.over_offered = remote_addresses.len() - considered.len(); + // Everything on the remote side is peer-supplied, so it is vetted once // here and the branches below only ever see admitted candidates. - let remote_reflexive = - remote_reflexive_address.and_then(|remote| match admit_remote(remote, &lan_refs, false) { + let remote_reflexive = remote_reflexive_address.and_then(|remote| { + match admit_remote(remote, &lan_refs, !local_reflexive_on_lan) { Ok(ip) => Some((remote, ip)), Err(class) => { tally.refuse_reflexive(class, remote); None } - }); - let remote_candidates = remote_addresses + } + }); + let remote_candidates = considered .iter() .filter_map(|remote| match admit_remote(remote, &lan_refs, true) { Ok(ip) => Some((remote, ip)), From 957cd94bb0ca895e2146331a3fc2d8454491e7ac Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:26:39 +0100 Subject: [PATCH 12/23] Accept a punch packet only from an address we planned to probe The NAT-punch packet's discriminator is a plain digest of the session id, a value both peers already know, and it travels in the clear in every probe, so a matching packet proved only that its sender had seen one. The receive loop broke on the first packet whose digest matched, whatever its source, and returned that source as the peer address. Anyone who observed a probe could therefore have an arbitrary address adopted as the peer: the legitimate traversal was denied, the handshake and its retransmissions went to an address of the attacker's choosing, and the pair was charged a failure against its backoff state. The source address is now ranked against the planned target list before anything else happens with the packet. An unplanned source is dropped and is deliberately not acked either, since acking it is a reflection this node controls. A source matching a planned target exactly is adopted immediately, as before. A source matching a planned target's IP on a different port is what a symmetric NAT's fresh mapping toward us looks like, so it is still adopted rather than dropped, because that is the main class of NAT pairing punching exists to rescue; it is held as a candidate for a settle window first, so an exact match arriving inside that window supersedes it. The honest path returns as fast as it did. Two consequences to state rather than discover. A sender able to source packets from a planned target's IP on any port is still accepted, which is the residue that only an authenticated probe can close. And an attempt under a flood of spoofed matching packets now runs to its full timeout instead of ending on the first one, so refused sources are counted and reported once when the attempt ends rather than logged per packet. --- CHANGELOG.md | 28 ++++ src/discovery/nostr/tests.rs | 219 ++++++++++++++++++++++++++++++- src/discovery/nostr/traversal.rs | 87 +++++++++++- 3 files changed, 329 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ddce350c..c14ca4c8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -797,6 +797,34 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 `over_offered` field and treats a non-zero one as an attack shape, since nothing honest reaches the bound. +- A NAT-punch packet is now accepted only from an address this node planned to + probe. The punch packet's discriminator is a plain digest of the session id, + a value both peers already know, and it travels in the clear in every probe, + so acceptance proved only that the sender had seen one. The receive loop + broke on the first packet whose digest matched, whatever its source, and + returned that source as the peer address, so anyone who observed a probe, or + who could reach the node and guess the session id, could have an arbitrary + address adopted as the peer: the legitimate traversal was denied, the Noise + handshake and its retransmissions went to an address of the attacker's + choosing, and the pair was charged a failure against its backoff state. The + npub-pinned handshake still could not authenticate to the wrong host, so this + was a denial and a misdirection rather than an impersonation. The source + address is now ranked against the planned target list before anything else: + an unplanned source is dropped and, deliberately, is not acked either, since + acking it is a reflection the node controls. A source matching a planned + target exactly is adopted immediately, as before. A source matching a planned + target's IP on a different port is what a symmetric NAT's fresh mapping looks + like, and it is still adopted, because that is the main class of NAT pairing + punching exists to rescue; it is held as a candidate for 250 ms first, so an + exact match arriving inside that window supersedes it. The honest path's + latency is unchanged. Two consequences to state rather than discover: an + attacker that can source packets from a planned target's IP on any port is + still accepted, which is the residue only an authenticated probe can close; + and an attempt under a flood of spoofed matching packets now runs to its full + timeout instead of ending on the first one, so the refused sources are + counted and reported once when the attempt ends rather than logged per + packet. + - Traversal punch targets taken from a peer's offer or answer are now filtered and bounded. A rendezvous-enabled node previously punched every address a signed offer named, including loopback, link-local, multicast, diff --git a/src/discovery/nostr/tests.rs b/src/discovery/nostr/tests.rs index 11ee567b..8473dd79 100644 --- a/src/discovery/nostr/tests.rs +++ b/src/discovery/nostr/tests.rs @@ -1,5 +1,6 @@ use std::collections::HashSet; use std::net::{IpAddr, SocketAddr}; +use std::time::Duration; use nostr::nips::nip17; use nostr::prelude::{EventBuilder, Kind, RelayUrl, Tag, Timestamp}; @@ -14,8 +15,9 @@ use super::signal::{ }; use super::stun::{parse_stun_binding_success, parse_stun_url}; use super::traversal::{ - PunchStrategy, build_punch_packet, is_doc_ip, is_never_punchable_ip, is_private_ip, now_ms, - parse_punch_packet, plan_punch_targets, planned_remote_endpoints, session_hash, + PunchStrategy, SourceRank, build_punch_packet, is_doc_ip, is_never_punchable_ip, is_private_ip, + now_ms, parse_punch_packet, plan_punch_targets, planned_remote_endpoints, rank_punch_source, + run_punch_attempt, session_hash, }; use super::types::BootstrapError; use super::{ @@ -1373,6 +1375,219 @@ async fn signal_events_use_current_timestamps() { assert!(created_at <= after); } +/// A loopback socket bound on `host`, non-blocking as both production call +/// sites leave it, since `run_punch_attempt` hands it straight to +/// `UdpSocket::from_std`. +fn punch_socket(host: &str) -> std::net::UdpSocket { + let socket = std::net::UdpSocket::bind(format!("{host}:0")).expect("bind a loopback socket"); + socket + .set_nonblocking(true) + .expect("the punch socket must be non-blocking"); + socket +} + +/// A hint that starts punching immediately. `start_at_ms` is absolute wall +/// clock, so anything plausible-looking in the future would sleep out the test. +fn immediate_punch_hint(duration_ms: u64) -> PunchHint { + PunchHint { + start_at_ms: 0, + interval_ms: 20, + duration_ms, + } +} + +/// Send one well-formed probe carrying `session_id`'s hash from `from` to +/// `to`, which is what a replay of captured punch bytes looks like. +fn send_probe(from: &std::net::UdpSocket, to: SocketAddr, session_id: &str) { + let packet = build_punch_packet(PunchPacketKind::Probe, 1, session_id); + from.send_to(&packet, to).expect("probe should send"); +} + +/// Whether anything readable on `socket` is a punch ack. +fn received_an_ack(socket: &std::net::UdpSocket) -> bool { + let mut buf = [0u8; 2048]; + while let Ok((len, _)) = socket.recv_from(&mut buf) { + if parse_punch_packet(&buf[..len]) + .map(|packet| packet.kind == PunchPacketKind::Ack) + .unwrap_or(false) + { + return true; + } + } + false +} + +#[test] +fn rank_punch_source_accepts_a_planned_target() { + let target: SocketAddr = "198.51.100.20:63000".parse().unwrap(); + assert_eq!(rank_punch_source(target, &[target]), SourceRank::Planned); +} + +#[test] +fn rank_punch_source_reports_a_planned_targets_other_port_as_remapped() { + let target: SocketAddr = "198.51.100.20:63000".parse().unwrap(); + let remapped: SocketAddr = "198.51.100.20:41234".parse().unwrap(); + assert_eq!( + rank_punch_source(remapped, &[target]), + SourceRank::RemappedPort + ); +} + +#[test] +fn rank_punch_source_rejects_an_address_we_never_planned_to_probe() { + let target: SocketAddr = "198.51.100.20:63000".parse().unwrap(); + let stranger: SocketAddr = "203.0.113.9:63000".parse().unwrap(); + assert_eq!( + rank_punch_source(stranger, &[target]), + SourceRank::Unplanned + ); +} + +/// The regression test for the defect. The punch packet's discriminator is a +/// digest of a value both peers already know and it travels in the clear in +/// every probe, so anyone who has seen one can replay it. Acceptance is now +/// constrained to the targets this node planned; the spoofer is neither +/// adopted nor acked, and an ack would be a reflection we control. +#[tokio::test] +async fn a_matching_punch_packet_from_an_unplanned_source_is_neither_adopted_nor_acked() { + let victim = punch_socket("127.0.0.3"); + let peer = punch_socket("127.0.0.1"); + let spoofer = punch_socket("127.0.0.2"); + let victim_addr = victim.local_addr().expect("victim address"); + let targets = vec![peer.local_addr().expect("peer address")]; + + send_probe(&spoofer, victim_addr, "session-unplanned"); + let result = run_punch_attempt( + &victim, + "session-unplanned", + &targets, + immediate_punch_hint(400), + Duration::from_millis(700), + ) + .await; + + assert!( + matches!(result, Err(BootstrapError::PunchTimeout(_))), + "a spoofed source must not be adopted, got {result:?}" + ); + assert!( + !received_an_ack(&spoofer), + "an unplanned source must not be acked" + ); +} + +/// The spoofer wins the race on arrival order and still loses on address. +#[tokio::test] +async fn a_planned_source_is_adopted_even_when_a_spoofer_replies_first() { + let victim = punch_socket("127.0.0.3"); + let peer = punch_socket("127.0.0.1"); + let spoofer = punch_socket("127.0.0.2"); + let victim_addr = victim.local_addr().expect("victim address"); + let peer_addr = peer.local_addr().expect("peer address"); + + send_probe(&spoofer, victim_addr, "session-race"); + send_probe(&peer, victim_addr, "session-race"); + let result = run_punch_attempt( + &victim, + "session-race", + &[peer_addr], + immediate_punch_hint(400), + Duration::from_millis(700), + ) + .await; + + assert_eq!( + result.expect("the planned peer should be adopted"), + peer_addr + ); +} + +/// The healthy path, which is the check that the source constraint does not +/// red a legitimately clean run: one probe from the single planned target is +/// adopted immediately and acked. +#[tokio::test] +async fn the_ordinary_probe_from_a_planned_target_is_still_adopted_and_acked() { + let victim = punch_socket("127.0.0.3"); + let peer = punch_socket("127.0.0.1"); + let victim_addr = victim.local_addr().expect("victim address"); + let peer_addr = peer.local_addr().expect("peer address"); + + send_probe(&peer, victim_addr, "session-healthy"); + let result = run_punch_attempt( + &victim, + "session-healthy", + &[peer_addr], + immediate_punch_hint(400), + Duration::from_millis(700), + ) + .await; + + assert_eq!( + result.expect("the planned peer should be adopted"), + peer_addr + ); + assert!(received_an_ack(&peer), "a planned probe should be acked"); +} + +/// Peer-reflexive discovery: a symmetric NAT allocates a fresh port toward us, +/// so the peer's probe arrives from an address that is not in the plan but +/// shares a planned target's IP. Adopting it is the main class of NAT pairing +/// punching exists to rescue, and this test reds if the rule is ever tightened +/// to exact matching without that being reopened deliberately. +#[tokio::test] +async fn a_planned_targets_remapped_port_is_adopted_when_that_is_all_that_arrives() { + let victim = punch_socket("127.0.0.3"); + let peer = punch_socket("127.0.0.1"); + let victim_addr = victim.local_addr().expect("victim address"); + let peer_addr = peer.local_addr().expect("peer address"); + // The address the peer's own STUN observation named, before its NAT + // remapped the port: same host, a port nothing is bound to. + let stale_target = SocketAddr::new(peer_addr.ip(), peer_addr.port().wrapping_add(1).max(1)); + + send_probe(&peer, victim_addr, "session-remapped"); + let result = run_punch_attempt( + &victim, + "session-remapped", + &[stale_target], + immediate_punch_hint(400), + Duration::from_millis(2000), + ) + .await; + + assert_eq!( + result.expect("a remapped port on a planned target should be adopted"), + peer_addr + ); +} + +/// An exact match inside the settle window supersedes a remapped one that +/// arrived first, which is what the window is for. +#[tokio::test] +async fn an_exact_target_supersedes_a_remapped_port_inside_the_settle_window() { + let victim = punch_socket("127.0.0.3"); + let peer = punch_socket("127.0.0.1"); + let neighbour = punch_socket("127.0.0.1"); + let victim_addr = victim.local_addr().expect("victim address"); + let peer_addr = peer.local_addr().expect("peer address"); + + send_probe(&neighbour, victim_addr, "session-settle"); + send_probe(&peer, victim_addr, "session-settle"); + let result = run_punch_attempt( + &victim, + "session-settle", + &[peer_addr], + immediate_punch_hint(400), + Duration::from_millis(2000), + ) + .await; + + assert_eq!( + result.expect("the exact target should win"), + peer_addr, + "an exact match must supersede a source that only shares the IP" + ); +} + fn node_addr(first_byte: u8) -> NodeAddr { let mut bytes = [0u8; 16]; bytes[0] = first_byte; diff --git a/src/discovery/nostr/traversal.rs b/src/discovery/nostr/traversal.rs index 53d1ec4e..d0403349 100644 --- a/src/discovery/nostr/traversal.rs +++ b/src/discovery/nostr/traversal.rs @@ -3,6 +3,7 @@ use std::sync::Arc; use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; use tokio::net::UdpSocket; +use tracing::debug; use super::types::{ BootstrapError, PUNCH_ACK_MAGIC, PUNCH_MAGIC, PunchHint, PunchPacket, PunchPacketKind, @@ -45,6 +46,48 @@ const MAX_PUNCH_TARGETS: usize = 8; /// list, which the tally records either way. const MAX_OFFERED_CANDIDATES: usize = 32; +/// How long the punch loop keeps listening for an exact target match once it +/// has already accepted a planned target's address on a different port. +/// +/// A source that matches a planned target's IP but not its port is what a +/// symmetric NAT's fresh mapping toward us looks like, and it is worth +/// adopting; a source that matches a target exactly is worth more, so the +/// first remapped source does not end the attempt outright. Raising this +/// delays adoption on the remapped path only, never past the attempt timeout; +/// lowering it toward zero makes the first remapped source win. +const PUNCH_SETTLE_MS: u64 = 250; + +/// How much the source address of a punch packet is worth as a peer address. +/// +/// The packet's own discriminator is a plain digest of a value both peers +/// already know, so it proves only that the sender has seen a probe. What the +/// source address is checked against is the target list this node planned, +/// which is the difference between adopting a peer we chose to probe and +/// adopting whoever replayed those bytes first. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub(super) enum SourceRank { + /// Not an address we planned to probe, and so not adoptable. + Unplanned, + /// A planned target's address on a different port. + RemappedPort, + /// Exactly a target we planned to probe. + Planned, +} + +/// Rank one punch packet's source address against the targets we planned. +/// +/// `targets` holds at most `MAX_PUNCH_TARGETS` entries, so the scan is +/// bounded by construction. +pub(super) fn rank_punch_source(remote: SocketAddr, targets: &[SocketAddr]) -> SourceRank { + if targets.contains(&remote) { + SourceRank::Planned + } else if targets.iter().any(|target| target.ip() == remote.ip()) { + SourceRank::RemappedPort + } else { + SourceRank::Unplanned + } +} + #[derive(Debug, Clone, PartialEq, Eq)] pub(super) struct PlannedPunchTarget { pub(super) strategy: PunchStrategy, @@ -474,10 +517,21 @@ pub(super) async fn run_punch_attempt( let expected_hash = session_hash(session_id); let mut buf = [0u8; 2048]; + // Counted rather than logged per packet: an attacker sets how many of + // these arrive, so a record each would trade the adoption this closes for + // log volume. One record at the end of the attempt instead. + let mut unplanned = 0usize; + let mut superseded = 0usize; + let mut candidate: Option = None; + let mut settle_at: Option = None; let result = loop { - let recv = tokio::time::timeout_at(finish_at, udp.recv_from(&mut buf)).await; + let deadline = settle_at.map_or(finish_at, |settle| settle.min(finish_at)); + let recv = tokio::time::timeout_at(deadline, udp.recv_from(&mut buf)).await; let Ok(Ok((len, remote))) = recv else { - break Err(BootstrapError::PunchTimeout(session_id.to_string())); + break match candidate { + Some(remote) => Ok(remote), + None => Err(BootstrapError::PunchTimeout(session_id.to_string())), + }; }; let Ok(packet) = parse_punch_packet(&buf[..len]) else { continue; @@ -485,13 +539,40 @@ pub(super) async fn run_punch_attempt( if packet.session_hash != expected_hash { continue; } + // Ahead of the ack, not only ahead of the adoption: acking a source we + // never planned to probe is a reflection this node controls, and there + // is no reason to emit it. + let rank = rank_punch_source(remote, targets); + if rank == SourceRank::Unplanned { + unplanned += 1; + continue; + } if packet.kind == PunchPacketKind::Probe { let ack = build_punch_packet(PunchPacketKind::Ack, packet.sequence, session_id); let _ = udp.send_to(&ack, remote).await; } - break Ok(remote); + if rank == SourceRank::Planned { + break Ok(remote); + } + // A remapped port is adoptable, and on a symmetric NAT it is the only + // thing that ever arrives, so it is held rather than dropped. An exact + // match still supersedes it if one turns up inside the settle window. + if candidate.replace(remote).is_some() { + superseded += 1; + } + settle_at.get_or_insert_with(|| { + tokio::time::Instant::now() + Duration::from_millis(PUNCH_SETTLE_MS) + }); }; send_handle.abort(); + if unplanned > 0 || superseded > 0 { + debug!( + session = %super::runtime::short_id(session_id), + unplanned, + superseded, + "traversal: punch packets refused on their source address" + ); + } result } From b67835aeff1b7e7fb7b9fa1ca415f1e01b451b1f Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:33:23 +0100 Subject: [PATCH 13/23] Refuse an epoch-mismatch msg1 that would destroy a live peering The epoch in msg1 is sealed, so a msg1 announcing a different epoch than the one we have stored for that peer is authentic. It is also replayable: a captured one stays valid indefinitely, and accepting it removed the peer entry, its index registrations and the FSP session state behind it, from off the path and repeatedly. Gate the teardown on two receiver-local conditions. Refuse while the peering the msg1 would destroy has shown authenticated inbound traffic recently, which is the evidence that the old session is not in fact dead; last_seen moves only on a successful decrypt, so a peer that genuinely restarted clears the gate by having stopped sending, while a peering under replay is by construction still heartbeating. And refuse a second accepted epoch change for the same peer identity inside the same interval, which bounds the churn a peer can drive on its own. The stamp is written only on acceptance: a refusal that slid the window would let a sustained replay starve a genuinely restarting peer. The refusal drops the msg1 silently rather than resending msg2, which is bound to the original msg1's ephemeral. Both thresholds come from one constant at 15 seconds, sized so a restarting peer's msg1 resends still land inside its first handshake window and so the liveness gate cannot outlive the link reaper. One residual stays open: a msg1 captured before an accepted epoch change can still be replayed once per interval per peer. Closing it needs a seen-msg1 cache scoped to this arm, which is receiver-side too and is not part of this change. --- CHANGELOG.md | 17 +++ src/node/handlers/handshake.rs | 73 +++++++++++- src/node/mod.rs | 8 ++ src/node/tests/handshake.rs | 209 +++++++++++++++++++++++++++++++++ 4 files changed, 302 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index c14ca4c8..fb1d360a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -574,6 +574,23 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 build with overflow checks on, such as the test harness. Behaviour is unchanged for every counter an honest peer can emit. +- An epoch-mismatch msg1 no longer tears down a peering that is still + carrying authenticated traffic, and a second epoch change for the same peer + identity inside 15 seconds is refused. The epoch travels inside the AEAD, so + such a msg1 is authentic, but it stays authentic after capture: replaying + one destroyed a working peering, and with it the FSP session state that + peering carried, from off the path. The peering's last authenticated inbound + frame is the evidence that it is still alive, and nothing an unauthenticated + sender emits can refresh it, so a peer that genuinely restarted clears the + gate by having stopped sending. The refusal is a silent drop: no msg2 is + returned, since the stored msg2 is bound to the original msg1's ephemeral + and answering a sender-chosen address is free amplification. The interval is + stamped only when an epoch change is accepted, so a sustained replay cannot + starve a genuinely restarting peer. Both thresholds come from one constant, + sized so a restarting peer's msg1 resends still land inside its own first + handshake window and below `link_dead_timeout_secs`, and nothing changes on + the wire. + - A session setup message naming an already-established peer no longer replaces that peer's session. The handler did this whenever `node.rekey.enabled` was false: it ran a fresh responder handshake and overwrote the entry, discarding diff --git a/src/node/handlers/handshake.rs b/src/node/handlers/handshake.rs index f5d4ce52..62768bff 100644 --- a/src/node/handlers/handshake.rs +++ b/src/node/handlers/handshake.rs @@ -9,9 +9,33 @@ use crate::node::wire::{Msg1Header, Msg2Header, build_msg2}; use crate::node::{Node, NodeError}; use crate::peer::{ActivePeer, PeerConnection, PromotionResult, cross_connection_winner}; use crate::transport::{Link, LinkDirection, LinkId, ReceivedPacket}; -use std::time::Duration; +use std::time::{Duration, Instant}; use tracing::{debug, info, warn}; +/// Minimum interval between accepted epoch changes for one peer identity, +/// and the recency threshold at which the peering an epoch change would +/// destroy still counts as live. +/// +/// An epoch-mismatch msg1 is authentic but replayable: a captured one stays +/// valid indefinitely, and accepting it tears down a working peering. Both +/// conditions are receiver-local. The liveness half is the one that closes +/// the replay, since a peering under attack is by construction still +/// heartbeating; the interval half bounds the churn a peer can drive on its +/// own. +/// +/// Sized against the peer's own recovery rather than against a round number: +/// a genuinely restarting peer's msg1 resends fire at roughly t+1, t+3, t+7 +/// and t+15 seconds and its attempt is reaped at `handshake_timeout_secs` +/// (30), so 15 is the largest value at which a real restart still re-peers +/// inside its first handshake window with no reconnect backoff. It also sits +/// below `link_dead_timeout_secs` (30), so the liveness gate can never +/// outlive the reaper that would have removed the peering anyway. +/// +/// Raising it lengthens the outage an attacker's accepted replay causes, +/// because the genuine peer's recovery msg1 hits the same arm. Lowering it +/// weakens both halves and, below the resend ladder, buys nothing. +const EPOCH_RESTART_MIN_INTERVAL_SECS: u64 = 15; + /// Why an inbound msg1 got past the `accept_connections` gate, and against /// what identity the post-DH confirmation must check it. /// @@ -444,19 +468,58 @@ impl Node { if possible_restart && let Some(existing_peer) = self.peers.get(&peer_node_addr) { let new_epoch = conn.remote_epoch(); let existing_epoch = existing_peer.remote_epoch(); + let now_ms = Self::now_ms(); + // How long the peering this msg1 would destroy has gone without + // authenticated inbound traffic. `last_seen` moves only on a + // successful decrypt, so nothing an unauthenticated sender emits + // can refresh it. + let peering_idle_ms = existing_peer.idle_time(now_ms); match (existing_epoch, new_epoch) { (Some(existing), Some(new)) if existing != new => { // Epoch mismatch — peer restarted. Tear down stale session. + // + // Two receiver-local conditions have to hold first. The + // epoch is sealed, so this msg1 is authentic, but a + // captured one stays authentic forever and replaying it + // destroys a working peering. Refuse while the peering is + // still carrying authenticated traffic — a peer that has + // genuinely restarted stopped feeding `last_seen` when it + // died, so this self-clears — and refuse a second epoch + // change inside the dampening interval, which bounds the + // churn a peer can drive on its own. + let dampened = self + .restart_dampener + .get(&peer_node_addr) + .is_some_and(|t| t.elapsed().as_secs() < EPOCH_RESTART_MIN_INTERVAL_SECS); + let peering_is_live = peering_idle_ms < EPOCH_RESTART_MIN_INTERVAL_SECS * 1000; + if peering_is_live || dampened { + debug!( + peer = %self.peer_display_name(&peer_node_addr), + idle_ms = peering_idle_ms, + dampened, + "Epoch mismatch dampened, dropping msg1" + ); + // No msg2 is sent: the stored msg2 is bound to the + // original msg1's ephemeral, and answering an address + // the sender chose is free amplification. + self.connections.remove(&link_id); + self.links.remove(&link_id); + self.stats_mut() + .record_reject(RejectReason::Handshake(HandshakeReject::BadState)); + return; + } debug!( peer = %self.peer_display_name(&peer_node_addr), "Peer restart detected (epoch mismatch), removing stale session" ); self.remove_active_peer(&peer_node_addr); - let now_ms = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .map(|d| d.as_millis() as u64) - .unwrap_or(0); + // Stamped on acceptance only. A refusal that slid the + // window would let a sustained replay starve a genuinely + // restarting peer for as long as it kept sending. + let cutoff = Duration::from_secs(EPOCH_RESTART_MIN_INTERVAL_SECS); + self.restart_dampener.retain(|_, t| t.elapsed() < cutoff); + self.restart_dampener.insert(peer_node_addr, Instant::now()); self.schedule_reconnect(peer_node_addr, now_ms); // Fall through to process as new connection } diff --git a/src/node/mod.rs b/src/node/mod.rs index e5efad54..d88fcb6b 100644 --- a/src/node/mod.rs +++ b/src/node/mod.rs @@ -488,6 +488,12 @@ pub struct Node { /// Pending outbound handshakes by our sender_idx. /// Tracks which LinkId corresponds to which session index. pending_outbound: HashMap<(TransportId, u32), LinkId>, + /// When each peer identity's last ACCEPTED epoch change tore down its + /// peering. Keyed on identity rather than address, and held here rather + /// than on `ActivePeer`, because the teardown being dampened destroys + /// the peer entry itself. Pruned on insert; see + /// `EPOCH_RESTART_MIN_INTERVAL_SECS`. + restart_dampener: HashMap, // === Rate Limiting === /// Rate limiter for msg1 processing (DoS protection). @@ -788,6 +794,7 @@ impl Node { index_allocator: IndexAllocator::new(), peers_by_index: HashMap::new(), pending_outbound: HashMap::new(), + restart_dampener: HashMap::new(), msg1_rate_limiter, setup_rate_limiter, icmp_rate_limiter: IcmpRateLimiter::new(), @@ -947,6 +954,7 @@ impl Node { index_allocator: IndexAllocator::new(), peers_by_index: HashMap::new(), pending_outbound: HashMap::new(), + restart_dampener: HashMap::new(), msg1_rate_limiter, setup_rate_limiter, icmp_rate_limiter: IcmpRateLimiter::new(), diff --git a/src/node/tests/handshake.rs b/src/node/tests/handshake.rs index 78a8675c..506b1d33 100644 --- a/src/node/tests/handshake.rs +++ b/src/node/tests/handshake.rs @@ -1811,3 +1811,212 @@ async fn a_stale_reverse_address_entry_does_not_hide_a_peer_reachable_by_address returns above the insert that would overwrite the stale entry" ); } + +// ===== Epoch-restart dampening ===== +// +// An epoch-mismatch msg1 is authentic, because the epoch travels inside the +// AEAD, but it is replayable: a captured one stays valid forever and +// accepting it destroys a working peering. Two receiver-local conditions +// gate the teardown, and each of the first two cases below breaks one of +// them. + +/// Install a peering for `initiator` that carries an epoch its genuine msg1 +/// does not, and that has gone `idle_secs` without authenticated inbound +/// traffic. Returns the link the peering is bound to. +fn install_peering_at_a_different_epoch( + node: &mut Node, + initiator: &Node, + transport_id: TransportId, + source_addr: &TransportAddr, + idle_secs: u64, +) -> LinkId { + use crate::peer::ActivePeer; + + let identity = PeerIdentity::from_pubkey_full(initiator.identity().pubkey_full()); + let node_addr = *identity.node_addr(); + let link_id = node.allocate_link_id(); + let authenticated_at = Node::now_ms().saturating_sub(idle_secs * 1000); + let mut peer = ActivePeer::new(identity, link_id, authenticated_at); + peer.set_current_addr(transport_id, source_addr.clone()); + // Anything but the epoch the initiator's msg1 carries, so the msg1 reads + // as a restart. + peer.set_remote_epoch(Some([0xAA; 8])); + node.peers.insert(node_addr, peer); + node.addr_to_link + .insert((transport_id, source_addr.clone()), link_id); + link_id +} + +/// A peering long enough past its last authenticated inbound frame that the +/// liveness gate does not hold the restart back. +const IDLE_SECS: u64 = 60; + +#[tokio::test] +async fn an_epoch_mismatch_msg1_against_a_live_peering_leaves_it_intact() { + let transport_id = TransportId::new(1); + let mut node = make_node(); + let initiator = make_node(); + let initiator_addr = node_addr_of(&initiator); + let source_addr = TransportAddr::from_string("127.0.0.1:41001"); + + // The peering is carrying authenticated traffic: it decrypted a frame a + // moment ago. Under replay that is always the case, because the genuine + // peer is heartbeating. + let peer_link = + install_peering_at_a_different_epoch(&mut node, &initiator, transport_id, &source_addr, 0); + + let bad_state_before = node.stats().handshake.bad_state; + node.handle_msg1(ReceivedPacket::with_timestamp( + transport_id, + source_addr.clone(), + genuine_msg1(&initiator, &node), + Node::now_ms(), + )) + .await; + + let peer = node + .get_peer(&initiator_addr) + .expect("a live peering must survive an epoch-mismatch msg1"); + assert_eq!( + peer.link_id(), + peer_link, + "the peering must be the one that was already established, not a \ + replacement promoted from the msg1" + ); + assert_eq!( + peer.remote_epoch(), + Some([0xAA; 8]), + "the stored epoch must not have moved to the one the msg1 carried" + ); + assert_eq!( + node.connection_count(), + 0, + "the dropped msg1 must leave no connection behind" + ); + assert_eq!( + node.stats().handshake.bad_state - bad_state_before, + 1, + "the drop must be counted" + ); +} + +#[tokio::test] +async fn a_second_epoch_change_inside_the_dampening_interval_leaves_the_peering_intact() { + let transport_id = TransportId::new(1); + let mut node = make_node(); + let initiator = make_node(); + let initiator_addr = node_addr_of(&initiator); + let source_addr = TransportAddr::from_string("127.0.0.1:41002"); + + // First epoch change: the peering is genuinely idle, so it is accepted + // and stamps the dampener. + let first_link = install_peering_at_a_different_epoch( + &mut node, + &initiator, + transport_id, + &source_addr, + IDLE_SECS, + ); + node.handle_msg1(ReceivedPacket::with_timestamp( + transport_id, + source_addr.clone(), + genuine_msg1(&initiator, &node), + Node::now_ms(), + )) + .await; + let promoted = node + .get_peer(&initiator_addr) + .expect("the first epoch change must be accepted"); + assert_ne!( + promoted.link_id(), + first_link, + "the first epoch change must have replaced the peering" + ); + + // The peer moves epoch again straight away. Nothing about the second + // msg1 is distinguishable from the first, which is why the interval, + // not the message, has to be what refuses it. + let second_link = install_peering_at_a_different_epoch( + &mut node, + &initiator, + transport_id, + &source_addr, + IDLE_SECS, + ); + + let bad_state_before = node.stats().handshake.bad_state; + node.handle_msg1(ReceivedPacket::with_timestamp( + transport_id, + source_addr.clone(), + genuine_msg1(&initiator, &node), + Node::now_ms(), + )) + .await; + + let peer = node + .get_peer(&initiator_addr) + .expect("a second epoch change inside the interval must not tear the peering down"); + assert_eq!( + peer.link_id(), + second_link, + "the peering must be the one that was already established" + ); + assert_eq!( + peer.remote_epoch(), + Some([0xAA; 8]), + "the stored epoch must not have moved to the one the msg1 carried" + ); + assert_eq!( + node.connection_count(), + 0, + "the dropped msg1 must leave no connection behind" + ); + assert_eq!( + node.stats().handshake.bad_state - bad_state_before, + 1, + "the drop must be counted" + ); +} + +/// Healthy path, and NOT discriminating: this passes with or without the +/// gates. It is here so that tightening either one, or a bug that stamps the +/// dampener on a refusal, reds the suite instead of silently refusing every +/// genuine restart. +#[tokio::test] +async fn a_first_epoch_change_against_a_silent_peering_still_restarts_it() { + let transport_id = TransportId::new(1); + let mut node = make_node(); + let initiator = make_node(); + let initiator_addr = node_addr_of(&initiator); + let source_addr = TransportAddr::from_string("127.0.0.1:41003"); + + let stale_link = install_peering_at_a_different_epoch( + &mut node, + &initiator, + transport_id, + &source_addr, + IDLE_SECS, + ); + + node.handle_msg1(ReceivedPacket::with_timestamp( + transport_id, + source_addr.clone(), + genuine_msg1(&initiator, &node), + Node::now_ms(), + )) + .await; + + let peer = node + .get_peer(&initiator_addr) + .expect("a restart with no prior epoch change must be promoted"); + assert_ne!( + peer.link_id(), + stale_link, + "the stale peering must have been torn down and replaced" + ); + assert_eq!( + peer.remote_epoch(), + Some(initiator.startup_epoch()), + "the replacement must carry the epoch the msg1 announced" + ); +} From d95fc708e0369ac8b665e749b2eafe20f6b8cb37 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:24:16 +0100 Subject: [PATCH 14/23] Cap how long a superseded FSP key epoch can be retained The drain deadline for the `previous` slot slides forward on every inbound frame that authenticates against it. That is deliberate: it stops the old epoch being erased out from under a peer that lost msg3 and is still sealing in it. But the only party that can push the deadline out is the authenticated peer holding that key, so a peer that keeps using the old epoch keeps the retired key resident indefinitely. Add an absolute ceiling measured from the cutover, so the sliding grace delays erasure by a bounded amount rather than preventing it. The ceiling has to clear the worst-case legitimate recovery of a peer that lost msg3, which is the msg3 resend ladder plus the responder's handshake timeout plus the rekey dampening window, about 90 seconds at stock settings. It defaults to 120 seconds and is raised to the budget the configured handshake timers actually imply, so tightening a timer cannot push the ceiling under the recovery it has to leave room for. This does not close the related gap where an FSP rekey we initiate and the peer never answers is never abandoned, which leaves the session's current epoch pinned and not rotating. The cap erases the old epoch on schedule regardless, which is a strict improvement, but a session read afterwards can show no drain alongside a stale current key for that reason rather than because of this change. --- CHANGELOG.md | 20 +++++++ src/node/handlers/mod.rs | 2 +- src/node/handlers/rekey.rs | 45 +++++++++++++- src/node/session.rs | 117 +++++++++++++++++++++++++++++++++++-- 4 files changed, 176 insertions(+), 8 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index fb1d360a..f09f5eb3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -591,6 +591,26 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 handshake window and below `link_dead_timeout_secs`, and nothing changes on the wire. +- Retention of a superseded FSP key epoch is now capped at an absolute + ceiling measured from the cutover, defaulting to 120 seconds against the + 10-second drain window. The drain deadline slides forward on every inbound + frame that authenticates against the `previous` slot, which is what keeps a + peer that lost msg3 from having the old epoch erased out from under it, but + it also meant the authenticated peer holding that key could keep the retired + key resident for as long as it kept sealing frames in the old epoch. The + sliding grace is unchanged; it now delays erasure by a bounded amount rather + than preventing it. The ceiling is set to clear the worst-case legitimate + recovery of a peer that lost msg3 (the msg3 resend ladder, then + `handshake_timeout_secs` before the responder abandons, then the rekey + dampening window before it may re-initiate, about 90 seconds at stock + settings), and it is raised automatically if the configured handshake timers + imply a longer budget, so shortening a timer cannot push the ceiling under + the recovery it has to leave room for. Nothing changes on the wire; each side + runs its own drain. A peer that has still not recovered when the ceiling + fires is left with undecryptable frames until its own rekey retry + re-converges the epochs, since nothing tears an established session down on + repeated decrypt failure. + - A session setup message naming an already-established peer no longer replaces that peer's session. The handler did this whenever `node.rekey.enabled` was false: it ran a fresh responder handshake and overwrote the entry, discarding diff --git a/src/node/handlers/mod.rs b/src/node/handlers/mod.rs index 5270cad3..d5feb846 100644 --- a/src/node/handlers/mod.rs +++ b/src/node/handlers/mod.rs @@ -8,7 +8,7 @@ mod encrypted; mod forwarding; pub(in crate::node) mod handshake; mod mmp; -mod rekey; +pub(in crate::node) mod rekey; mod rx_loop; pub(in crate::node) mod session; mod timeout; diff --git a/src/node/handlers/rekey.rs b/src/node/handlers/rekey.rs index bac07b8a..4c957e1c 100644 --- a/src/node/handlers/rekey.rs +++ b/src/node/handlers/rekey.rs @@ -19,6 +19,48 @@ const DRAIN_WINDOW_SECS: u64 = 10; /// a peer's rekey msg1. const REKEY_DAMPENING_SECS: u64 = 30; +/// Floor on the absolute ceiling for `previous`-slot retention after a +/// cutover, in seconds. +/// +/// The drain deadline is peer-progress-aware: it slides forward on every +/// inbound frame that authenticates against the old epoch, so a peer that +/// keeps sealing in that epoch holds the retired key for as long as it +/// likes. This bounds that. It has to stay longer than the worst-case +/// recovery of a legitimate peer that lost msg3, which at stock defaults +/// is the msg3 resend ladder (about 31 s) plus `handshake_timeout_secs` +/// (30 s) before the responder abandons plus `REKEY_DAMPENING_SECS` +/// (30 s) before it may re-initiate, so about 90 s. 120 s clears that +/// with margin and still bounds retention to roughly one +/// `node.rekey.after_secs` period. `drain_max_retention_ms` takes the +/// larger of this floor and the budget the running configuration +/// actually implies, so a shortened handshake timer cannot push the +/// ceiling under the recovery it has to clear. +/// +/// Lowering it below that budget cuts off legitimate slow peers: their +/// frames go silently undecryptable until their own rekey retry +/// re-converges the epochs, because nothing tears an established session +/// down on repeated decrypt failure. Raising it lengthens the window in +/// which a retired key stays resident. +const DRAIN_MAX_RETENTION_SECS: u64 = DRAIN_WINDOW_SECS * 12; + +/// Effective ceiling on total `previous`-slot retention, in milliseconds. +/// +/// The larger of `DRAIN_MAX_RETENTION_SECS` and the msg3 recovery budget +/// the configured handshake timers imply, so the ceiling always clears +/// the recovery it is supposed to leave room for. +pub(in crate::node) fn drain_max_retention_ms(rate_limit: &crate::config::RateLimitConfig) -> u64 { + let mut ladder_ms: u64 = 0; + let mut interval = rate_limit.handshake_resend_interval_ms as f64; + for _ in 0..rate_limit.handshake_max_resends { + ladder_ms = ladder_ms.saturating_add(interval as u64); + interval *= rate_limit.handshake_resend_backoff; + } + let recovery_budget_ms = ladder_ms + .saturating_add(rate_limit.handshake_timeout_secs.saturating_mul(1000)) + .saturating_add(REKEY_DAMPENING_SECS * 1000); + (DRAIN_MAX_RETENTION_SECS * 1000).max(recovery_budget_ms) +} + /// Liveness bound on how long the FSP rekey initiator holds the /// `current` + `pending` state before cutting over to the new epoch. /// @@ -443,6 +485,7 @@ impl Node { let rekey_after_messages = self.config().node.rekey.after_messages; let now_ms = Self::now_ms(); let drain_ms = DRAIN_WINDOW_SECS * 1000; + let drain_max_ms = drain_max_retention_ms(&self.config().node.rate_limit); let dampening_ms = REKEY_DAMPENING_SECS * 1000; let mut sessions_to_cutover: Vec = Vec::new(); @@ -492,7 +535,7 @@ impl Node { } // 2. Drain window expiry - if entry.is_draining() && entry.drain_expired(now_ms, drain_ms) { + if entry.is_draining() && entry.drain_expired(now_ms, drain_ms, drain_max_ms) { sessions_to_drain.push(*node_addr); } diff --git a/src/node/session.rs b/src/node/session.rs index 4af0d463..4c036e59 100644 --- a/src/node/session.rs +++ b/src/node/session.rs @@ -726,10 +726,24 @@ impl SessionEntry { /// permanent silent decrypt failure. A peer that never catches up /// is instead handled by the FSP session liveness path (fresh /// handshake / teardown of a genuinely dead link). - pub(crate) fn drain_expired(&self, now_ms: u64, drain_ms: u64) -> bool { + /// + /// `max_drain_ms` is an absolute ceiling measured from the cutover + /// alone, so the sliding deadline delays erasure by a bounded amount + /// rather than preventing it: the only party that can push the + /// deadline out is the authenticated peer holding the old key, and + /// without a ceiling it holds that key for as long as it keeps using + /// it. What the ceiling costs is that a peer which has still not + /// recovered by then is cut off deliberately, and its frames are + /// undecryptable until its own rekey retry re-converges the epochs. + /// It must therefore stay above the worst-case legitimate recovery; + /// see `DRAIN_MAX_RETENTION_SECS`. + pub(crate) fn drain_expired(&self, now_ms: u64, drain_ms: u64, max_drain_ms: u64) -> bool { if self.drain_started_ms == 0 { return false; } + if now_ms.saturating_sub(self.drain_started_ms) >= max_drain_ms { + return true; + } let deadline_anchor = self.drain_started_ms.max(self.previous_last_used_ms); now_ms.saturating_sub(deadline_anchor) >= drain_ms } @@ -1226,6 +1240,9 @@ mod overlapping_epoch_tests { #[test] fn drain_expiry_is_peer_progress_aware() { const DRAIN_MS: u64 = 10_000; + // Well clear of the shipped ceiling, so this case still exercises + // the sliding deadline and nothing else. + const MAX_MS: u64 = 120_000; let cutover_ms = 1_000; // Build the post-cutover state via the production cutover path: @@ -1254,7 +1271,7 @@ mod overlapping_epoch_tests { // Even though `now - drain_started_ms` exceeds DRAIN_MS, the // window is NOT expired: the peer just used `previous`. assert!( - !entry.drain_expired(t, DRAIN_MS), + !entry.drain_expired(t, DRAIN_MS, MAX_MS), "previous slot must not be retired while peer keeps using it (t={t})" ); assert!( @@ -1267,11 +1284,11 @@ mod overlapping_epoch_tests { // last `previous`-slot use was at t=25_000; the window now // elapses DRAIN_MS after that, NOT DRAIN_MS after the cutover. assert!( - !entry.drain_expired(34_999, DRAIN_MS), + !entry.drain_expired(34_999, DRAIN_MS, MAX_MS), "window must not expire before DRAIN_MS past the last previous use" ); assert!( - entry.drain_expired(35_000, DRAIN_MS), + entry.drain_expired(35_000, DRAIN_MS, MAX_MS), "window must expire DRAIN_MS after the last previous-slot decrypt" ); @@ -1289,6 +1306,7 @@ mod overlapping_epoch_tests { #[test] fn drain_expiry_unaffected_when_peer_off_old_epoch() { const DRAIN_MS: u64 = 10_000; + const MAX_MS: u64 = 120_000; let cutover_ms = 1_000; let (_old_send, old_recv) = xk_pair(1, 2); @@ -1300,12 +1318,99 @@ mod overlapping_epoch_tests { // No old-epoch frames ever arrive: `previous_last_used_ms` stays // 0, the deadline anchor is the cutover time. assert!( - !entry.drain_expired(cutover_ms + DRAIN_MS - 1, DRAIN_MS), + !entry.drain_expired(cutover_ms + DRAIN_MS - 1, DRAIN_MS, MAX_MS), "window must not expire early" ); assert!( - entry.drain_expired(cutover_ms + DRAIN_MS, DRAIN_MS), + entry.drain_expired(cutover_ms + DRAIN_MS, DRAIN_MS, MAX_MS), "window must expire on the plain wall-clock timer when peer is off the old epoch" ); } + // 12. A peer that keeps exercising the old epoch delays erasure by a + // bounded amount rather than preventing it. The refreshes must + // continue past the ceiling: a case that stops refreshing at the + // boundary passes without the ceiling and proves nothing. + #[test] + fn drain_retention_is_capped_against_a_peer_pinning_the_old_epoch() { + const DRAIN_MS: u64 = 10_000; + const MAX_MS: u64 = 120_000; + + let (_old_send, old_recv) = xk_pair(1, 2); + let (_new_send, new_recv) = xk_pair(3, 4); + let mut entry = entry_with_current(old_recv); + entry.set_pending_session(new_recv); + assert!(entry.cutover_to_new_session(1)); + + // One old-epoch frame every half window, which is what a peer + // pinning the drain deadline actually does. + let mut t = 1u64; + while t < MAX_MS { + entry.refresh_previous_use(t); + assert!( + !entry.drain_expired(t, DRAIN_MS, MAX_MS), + "ceiling fired before the peer's grace ran out (t={t})" + ); + t += DRAIN_MS / 2; + } + + // Still refreshing, so the sliding deadline is nowhere near due. + entry.refresh_previous_use(MAX_MS + 1); + assert!( + entry.drain_expired(MAX_MS + 1, DRAIN_MS, MAX_MS), + "a peer pinning the old epoch retained the retired key past the ceiling" + ); + } + + // 13. The ceiling must not shorten the grace the sliding deadline + // exists to give a peer that lost msg3 and is still catching up. + #[test] + fn drain_retention_cap_does_not_shorten_the_ordinary_grace() { + const DRAIN_MS: u64 = 10_000; + const MAX_MS: u64 = 120_000; + const CUTOVER_MS: u64 = 1_000; + + let (_old_send, old_recv) = xk_pair(1, 2); + let (_new_send, new_recv) = xk_pair(3, 4); + let mut entry = entry_with_current(old_recv); + entry.set_pending_session(new_recv); + assert!(entry.cutover_to_new_session(CUTOVER_MS)); + assert!(entry.is_draining()); + + // A peer that lost msg3 and is still catching up sends one + // old-epoch frame part-way through the window. + let last_use = CUTOVER_MS + DRAIN_MS / 2; + entry.refresh_previous_use(last_use); + assert!(!entry.drain_expired(CUTOVER_MS + DRAIN_MS, DRAIN_MS, MAX_MS)); + assert!(!entry.drain_expired(last_use + DRAIN_MS - 1, DRAIN_MS, MAX_MS)); + assert!(entry.drain_expired(last_use + DRAIN_MS, DRAIN_MS, MAX_MS)); + } + + // 14. The shipped ceiling has to clear the worst-case legitimate + // recovery of a peer that lost msg3: the msg3 resend ladder, the + // responder's handshake timeout, and the rekey dampening window + // before it may re-initiate. A later tightening of the ceiling + // reds this rather than silently amputating that recovery. + #[test] + fn drain_retention_cap_clears_the_msg3_recovery_budget() { + const DRAIN_MS: u64 = 10_000; + // 31 s resend ladder + 30 s handshake timeout + 30 s dampening. + const RECOVERY_MS: u64 = 91_000; + const CUTOVER_MS: u64 = 1_000; + let max_ms = crate::node::handlers::rekey::drain_max_retention_ms( + &crate::config::RateLimitConfig::default(), + ); + + let (_old_send, old_recv) = xk_pair(1, 2); + let (_new_send, new_recv) = xk_pair(3, 4); + let mut entry = entry_with_current(old_recv); + entry.set_pending_session(new_recv); + assert!(entry.cutover_to_new_session(CUTOVER_MS)); + assert!(entry.is_draining()); + + entry.refresh_previous_use(CUTOVER_MS + RECOVERY_MS); + assert!( + !entry.drain_expired(CUTOVER_MS + RECOVERY_MS, DRAIN_MS, max_ms), + "ceiling fires inside the msg3 recovery budget" + ); + } } From 42622c8efe9fc6d8aa3568808130d7d648c95370 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:28:24 +0100 Subject: [PATCH 15/23] Bound induced routing errors by the peer that induced them The 100 ms suppression gate on CoordsRequired, PathBroken and MtuExceeded was keyed on the failed datagram's destination address. That field is chosen by whoever sent the datagram, so a fresh random destination per packet was always a first sighting and the gate admitted every one. Each admission inserted a key and then walked the whole map, making per-packet cost grow with the flood rate while the sender's cost stayed flat, and the emitted error is addressed to the datagram's source address, which nothing binds to the sender either. Add a per-link-peer token bucket, keyed on the AEAD-authenticated peer the frame arrived over, and consult it ahead of the destination gate. That peer is the only value at the emission point a sender cannot mint, so it is the only one that can bound the emission or the growth of the address-keyed map behind it. Default 20 signals a second sustained with a burst of 50, both named constants. The token is peeked and spent only once the destination gate has also admitted, so one unroutable destination behind a high-fanout peer cannot burn that peer's whole budget on signals nothing sends. The destination map gets a hard 4096-entry ceiling and its expiry sweep is amortized to once per eviction interval instead of running on every admission; at capacity it admits without recording rather than refusing, because refusing would turn a full map into node-wide silence during partition healing, which is exactly when the map is largest and the signals are most needed. Also stop attaching the reporter's cached coordinates to a transit- emitted PathBroken. The error goes to an unverified source address, so the field answered a coordinate-cache read to anyone naming any address. It is optional on the wire and no receiver reads it, so an unmodified peer parses the frame unchanged. Three counters make each outcome visible on the fipstop Routing tab. --- CHANGELOG.md | 39 ++++ src/bin/fipstop/ui/routing.rs | 3 + src/bin/fipstop/ui/snapshots.rs | 2 +- src/control/snapshots/show_routing.json | 3 + src/node/handlers/forwarding.rs | 97 +++++++--- src/node/metrics.rs | 16 ++ src/node/mod.rs | 8 + src/node/peer_error_budget.rs | 239 ++++++++++++++++++++++++ src/node/routing_error_rate_limit.rs | 169 ++++++++++++++++- src/node/stats.rs | 3 + src/node/tests/forwarding.rs | 88 +++++++++ 11 files changed, 637 insertions(+), 30 deletions(-) create mode 100644 src/node/peer_error_budget.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index f09f5eb3..a1bf8244 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -939,6 +939,45 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 #### Data-plane / routing signals +- A transit node's induced routing errors are now bounded by the authenticated + link peer that induced them. The 100 ms suppression gate on + `CoordsRequired`, `PathBroken` and `MtuExceeded` was keyed on the failed + datagram's destination address, which is an envelope field the sender picks, + so a fresh random destination on every packet was always a first sighting and + every packet was admitted. Each admission also inserted a key and then walked + the whole map, so per-packet cost grew with the flood rate while the sender's + cost stayed flat, and the error itself is addressed to the datagram's source + address, which nothing binds to the sender either. A new per-link-peer token + bucket, 20 signals a second sustained with a burst of 50, is now consulted + first, keyed on the AEAD-authenticated peer the frame arrived over: the one + value at that point a sender cannot mint. The per-destination interval is + kept unchanged behind it, because it still does the aggregate suppression a + genuine outage needs, and no gate was added on the address the error is + returned to, which would have handed a sender a way to silence honest errors + toward a victim it names by keeping that victim's key hot. + + Two ordering choices in there rather than left to be discovered. The peer's + token is peeked and only spent once the destination gate has also admitted, + so a single unroutable destination behind a high-fanout peer cannot burn that + peer's whole budget on signals nothing sends and silence every other + destination behind it. And the destination map now carries a hard ceiling of + 4096 entries with its expiry sweep amortized to once per eviction interval + rather than run on every admission; when it is full it admits without + recording rather than refusing, because refusing would turn a full map into + node-wide silence exactly during partition healing, when many destinations + are legitimately unroutable at once. Emission stays bounded by the peer + budget in that state. Three counters, rendered on the fipstop Routing tab, + make each of the three outcomes visible instead of silent. + +- A transit-emitted `PathBroken` no longer carries the reporter's cached + coordinates for the unreachable destination. The signal is returned to the + datagram's source address, so anyone able to reach the node could name any + address and have the node's coordinate cache read back to them, one entry per + packet. The field is optional on the wire and no receiver reads it, so this + is an emission change only: an unmodified peer parses the frame exactly as + before. Which of the two signals is emitted still discloses whether the entry + exists. + - The influence a remote party has over path MTU is now bounded, and the per-destination path MTU cache has a way back. The `path_mtu` field is an unsigned per-hop transit annotation carried outside the signed proof, and the diff --git a/src/bin/fipstop/ui/routing.rs b/src/bin/fipstop/ui/routing.rs index 819541e4..015234d6 100644 --- a/src/bin/fipstop/ui/routing.rs +++ b/src/bin/fipstop/ui/routing.rs @@ -298,6 +298,9 @@ fn draw_routing_stats( ("Path Broken Refused", err("unbound_broken")), ("MTU Exceeded Refused", err("unbound_mtu")), ("Forged Pairing", err("unbound_forged")), + ("Emit Over Peer Budget", err("emit_over_peer_budget")), + ("Emit Over Dest Interval", err("emit_over_dest_interval")), + ("Emit Limiter At Capacity", err("emit_limiter_at_capacity")), ], )); right.push(Line::from("")); diff --git a/src/bin/fipstop/ui/snapshots.rs b/src/bin/fipstop/ui/snapshots.rs index b108cc66..a91611b4 100644 --- a/src/bin/fipstop/ui/snapshots.rs +++ b/src/bin/fipstop/ui/snapshots.rs @@ -1151,7 +1151,7 @@ fn routing_focused_pane_scrolls() { // column is the taller of the two, so scrolling fully to the bottom would // over-scroll the right column past Congestion; this offset lands the // Congestion region inside the short window instead. - app1.scroll_offsets.insert((Tab::Routing, 2), 28); + app1.scroll_offsets.insert((Tab::Routing, 2), 31); let buf1 = testkit::render(100, 20, |frame, area| { super::routing::draw(frame, &app1, area); }); diff --git a/src/control/snapshots/show_routing.json b/src/control/snapshots/show_routing.json index 9741572e..1e5328e2 100644 --- a/src/control/snapshots/show_routing.json +++ b/src/control/snapshots/show_routing.json @@ -33,6 +33,9 @@ }, "error_signals": { "coords_required": 0, + "emit_limiter_at_capacity": 0, + "emit_over_dest_interval": 0, + "emit_over_peer_budget": 0, "lookup_resp_mtu_below_floor": 0, "mtu_exceeded": 0, "mtu_exceeded_below_floor": 0, diff --git a/src/node/handlers/forwarding.rs b/src/node/handlers/forwarding.rs index c0b4dd42..3124231f 100644 --- a/src/node/handlers/forwarding.rs +++ b/src/node/handlers/forwarding.rs @@ -8,6 +8,7 @@ use crate::NodeAddr; use crate::node::reject::ForwardingReject; +use crate::node::routing_error_rate_limit::LimitVerdict; use crate::node::session_wire::{ FSP_COMMON_PREFIX_SIZE, FSP_HEADER_SIZE, FSP_PHASE_ESTABLISHED, FSP_PHASE_MSG1, FSP_PHASE_MSG2, FspCommonPrefix, FspEncryptedHeader, parse_encrypted_coords, @@ -101,7 +102,7 @@ impl Node { bytes = payload.len(), "Dropping transit SessionDatagram: no route to destination" ); - self.send_routing_error(&datagram).await; + self.send_routing_error(from, &datagram).await; return; } }; @@ -145,7 +146,7 @@ impl Node { self.metrics() .forwarding .record_reject_bytes(ForwardingReject::MtuExceeded, payload.len()); - self.send_mtu_exceeded_error(&datagram, mtu).await; + self.send_mtu_exceeded_error(from, &datagram, mtu).await; } _ => { self.metrics() @@ -283,6 +284,43 @@ impl Node { } } + /// Decide whether this node may emit a routing error induced by `from` + /// about `dest`. + /// + /// Two gates, in this order. The per-link-peer budget bounds what one + /// admitted peer can induce, and is the only one keyed on something the + /// sender cannot mint; it is consulted first so the per-destination map + /// only grows at the budget rate. The per-destination interval is the + /// aggregate suppressor during a genuine outage. + /// + /// The budget token is peeked and only committed once the destination gate + /// has also admitted. Charging it on a suppressed signal would let a single + /// unroutable destination behind a high-fanout peer spend that peer's whole + /// budget on emissions nothing sends, silencing every other destination + /// behind it. + fn admit_error_emission(&mut self, from: &NodeAddr, dest: &NodeAddr) -> bool { + let now = Instant::now(); + + if !self.peer_error_budget.has_token(from, now) { + self.metrics().errors.emit_over_peer_budget.inc(); + return false; + } + + match self.routing_error_rate_limiter.check(dest, now) { + LimitVerdict::Suppress => { + self.metrics().errors.emit_over_dest_interval.inc(); + return false; + } + LimitVerdict::AdmitAtCapacity => { + self.metrics().errors.emit_limiter_at_capacity.inc(); + } + LimitVerdict::Admit => {} + } + + self.peer_error_budget.commit(from, now); + true + } + /// Generate and send a routing error signal back to the datagram's source. /// /// If we have cached coords for the destination, send PathBroken (we know @@ -291,12 +329,11 @@ impl Node { /// /// If we can't route the error back to the source either, drop silently. /// No cascading errors. - async fn send_routing_error(&mut self, original: &SessionDatagram) { - // Rate limit: one error signal per destination per 100ms - if !self - .routing_error_rate_limiter - .should_send(&original.dest_addr) - { + /// + /// `from` is the authenticated link peer the original datagram arrived + /// from, and is what the emission is charged against. + async fn send_routing_error(&mut self, from: &NodeAddr, original: &SessionDatagram) { + if !self.admit_error_emission(from, &original.dest_addr) { return; } @@ -307,15 +344,21 @@ impl Node { .map(|d| d.as_millis() as u64) .unwrap_or(0); - let error_payload = - if let Some(coords) = self.coord_cache().get(&original.dest_addr, now_ms) { - let coords = coords.clone(); - PathBroken::new(original.dest_addr, my_addr) - .with_last_coords(coords) - .encode() - } else { - CoordsRequired::new(original.dest_addr, my_addr).encode() - }; + // The choice between the two signals still leaks whether this node + // holds coords for the destination, but the coordinates themselves + // are not attached: the address the error is returned to is the + // datagram's own src_addr, which nothing binds to the peer that sent + // it, so attaching them would answer a cache read to whoever names an + // address. No receiver reads the field. + let error_payload = if self + .coord_cache() + .get(&original.dest_addr, now_ms) + .is_some() + { + PathBroken::new(original.dest_addr, my_addr).encode() + } else { + CoordsRequired::new(original.dest_addr, my_addr).encode() + }; let error_dg = SessionDatagram::new(my_addr, original.src_addr, error_payload) .with_ttl(self.config().node.session.default_ttl); @@ -356,12 +399,20 @@ impl Node { /// Called when `send_encrypted_link_message()` fails with /// `NodeError::MtuExceeded` during forwarding. The signal tells the /// source the bottleneck MTU so it can immediately reduce its path MTU. - async fn send_mtu_exceeded_error(&mut self, original: &SessionDatagram, bottleneck_mtu: u16) { - // Rate limit: reuse routing_error_rate_limiter keyed on dest_addr - if !self - .routing_error_rate_limiter - .should_send(&original.dest_addr) - { + /// + /// `from` is the authenticated link peer the original datagram arrived + /// from, and is what the emission is charged against. MtuExceeded shares + /// the link peer's budget with the routing errors rather than holding its + /// own: a separate bucket would insulate path-MTU discovery from + /// routing-error pressure, at the cost of a second knob and of letting one + /// peer induce twice the total emission. + async fn send_mtu_exceeded_error( + &mut self, + from: &NodeAddr, + original: &SessionDatagram, + bottleneck_mtu: u16, + ) { + if !self.admit_error_emission(from, &original.dest_addr) { return; } diff --git a/src/node/metrics.rs b/src/node/metrics.rs index e81457b7..6dfc0bb3 100644 --- a/src/node/metrics.rs +++ b/src/node/metrics.rs @@ -508,6 +508,19 @@ pub struct ErrorMetrics { /// annotation. pub lookup_resp_mtu_below_floor: Counter, pub unbound: UnboundSignals, + /// Routing errors this node declined to emit because the authenticated + /// link peer that induced them had spent its budget. A rising count is + /// either a peer flooding unroutable traffic or a hub relaying more + /// simultaneously-broken destinations than the budget allows. + pub emit_over_peer_budget: Counter, + /// Routing errors this node declined to emit because one for the same + /// destination went out within the per-destination interval. This is the + /// aggregate suppression a real outage produces. + pub emit_over_dest_interval: Counter, + /// Routing errors emitted without recording their destination, because + /// the per-destination limiter's map was full. The signal was still sent; + /// what was lost is interval suppression for that destination. + pub emit_limiter_at_capacity: Counter, } impl ErrorMetrics { @@ -524,6 +537,9 @@ impl ErrorMetrics { unbound_broken: self.unbound.broken.get(), unbound_mtu: self.unbound.mtu.get(), unbound_forged: self.unbound.forged.get(), + emit_over_peer_budget: self.emit_over_peer_budget.get(), + emit_over_dest_interval: self.emit_over_dest_interval.get(), + emit_limiter_at_capacity: self.emit_limiter_at_capacity.get(), } } } diff --git a/src/node/mod.rs b/src/node/mod.rs index d88fcb6b..3a52a44f 100644 --- a/src/node/mod.rs +++ b/src/node/mod.rs @@ -17,6 +17,7 @@ pub(crate) mod encrypt_worker; mod handlers; mod lifecycle; pub(crate) mod metrics; +mod peer_error_budget; mod rate_limit; pub(crate) mod reject; mod reloadable; @@ -32,6 +33,7 @@ mod tree; pub(crate) mod wire; use self::discovery_rate_limit::{DiscoveryBackoff, DiscoveryForwardRateLimiter}; +use self::peer_error_budget::PeerErrorBudget; use self::rate_limit::{HandshakeRateLimiter, SessionSetupRateLimiter}; use self::reloadable::Reloadable; use self::routing_error_rate_limit::RoutingErrorRateLimiter; @@ -504,6 +506,10 @@ pub struct Node { icmp_rate_limiter: IcmpRateLimiter, /// Rate limiter for routing error signals (CoordsRequired / PathBroken). routing_error_rate_limiter: RoutingErrorRateLimiter, + /// Budget for routing errors this node may be induced to emit, charged to + /// the authenticated link peer whose datagram induced them. The only + /// bound on the emission that a sender cannot escape by varying a field. + peer_error_budget: PeerErrorBudget, /// Rate limiter for source-side CoordsRequired/PathBroken responses. coords_response_rate_limiter: RoutingErrorRateLimiter, /// Backoff for failed discovery lookups (originator-side). @@ -799,6 +805,7 @@ impl Node { setup_rate_limiter, icmp_rate_limiter: IcmpRateLimiter::new(), routing_error_rate_limiter: RoutingErrorRateLimiter::new(), + peer_error_budget: PeerErrorBudget::new(), coords_response_rate_limiter: RoutingErrorRateLimiter::with_interval( std::time::Duration::from_millis(coords_response_interval_ms), ), @@ -959,6 +966,7 @@ impl Node { setup_rate_limiter, icmp_rate_limiter: IcmpRateLimiter::new(), routing_error_rate_limiter: RoutingErrorRateLimiter::new(), + peer_error_budget: PeerErrorBudget::new(), coords_response_rate_limiter: RoutingErrorRateLimiter::with_interval( std::time::Duration::from_millis(coords_response_interval_ms), ), diff --git a/src/node/peer_error_budget.rs b/src/node/peer_error_budget.rs new file mode 100644 index 00000000..785337bc --- /dev/null +++ b/src/node/peer_error_budget.rs @@ -0,0 +1,239 @@ +//! Per-link-peer budget for induced routing-error emissions. +//! +//! A transit node synthesizes a routing error (CoordsRequired, PathBroken or +//! MtuExceeded) in response to a datagram it could not forward. Every field of +//! that datagram is chosen by whoever sent it, so a per-destination or +//! per-source gate can be escaped by varying the field it is keyed on. The one +//! value at the emission point an attacker cannot mint is the authenticated +//! link peer the frame arrived from, whose cardinality is bounded by the peer +//! table and by admission. This budget is keyed on it, and is consulted ahead +//! of any address-keyed structure so that those structures only grow at the +//! budget rate. + +use crate::NodeAddr; +use std::collections::HashMap; +use std::time::{Duration, Instant}; + +/// Sustained rate, in signals per second, at which one authenticated link peer +/// may induce this node to emit routing errors. +/// +/// Bounds the reflection an admitted peer can aim at a victim it names, and +/// bounds how fast that peer can grow the per-destination limiter's map. +/// Raising it costs proportionally more reflected traffic per peer; lowering it +/// silences a hub peer that relays many sources through a genuine outage +/// sooner, which costs those sources their CoordsRequired and their +/// path-MTU feedback. +pub const PEER_ERROR_RATE_PER_SEC: u32 = 20; + +/// Number of routing errors one link peer may induce back to back before the +/// sustained rate applies. +/// +/// Sized so that an ordinary burst of unroutable traffic behind one peer still +/// signals promptly. Raising it lets a peer front-load a larger reflection; +/// lowering it makes a legitimate convergence burst arrive as a trickle. +pub const PEER_ERROR_BURST: u32 = 50; + +/// Tokens are carried in thousandths so the refill of a sub-millisecond +/// interval is not rounded away. +const MILLI: u64 = 1000; + +/// A peer whose bucket has refilled to full carries no state worth keeping, so +/// entries are dropped once per this interval to bound the map across peer +/// churn. +const SWEEP_INTERVAL: Duration = Duration::from_secs(30); + +/// One peer's token bucket. +struct Bucket { + /// Remaining tokens, in thousandths of a signal. + milli_tokens: u64, + /// When `milli_tokens` was last brought up to date. + last_refill: Instant, +} + +/// Token-bucket budget for routing errors, keyed on the authenticated link +/// peer that induced them. +pub struct PeerErrorBudget { + buckets: HashMap, + /// Refill rate in thousandths of a token per millisecond. + milli_per_ms: u64, + /// Bucket ceiling, in thousandths of a token. + capacity: u64, + last_sweep: Instant, +} + +impl PeerErrorBudget { + /// Create a budget at the shipped rate and burst. + pub fn new() -> Self { + Self::with_rate(PEER_ERROR_RATE_PER_SEC, PEER_ERROR_BURST) + } + + /// Create a budget with an explicit sustained rate and burst. + pub fn with_rate(per_sec: u32, burst: u32) -> Self { + Self { + buckets: HashMap::new(), + milli_per_ms: u64::from(per_sec), + capacity: u64::from(burst) * MILLI, + last_sweep: Instant::now(), + } + } + + /// Whether `peer` has a token to spend, without spending it. + /// + /// Separate from [`Self::commit`] so a signal that a later gate suppresses + /// does not consume budget: an outage behind a high-fanout peer would + /// otherwise spend that peer's whole budget on emissions the + /// per-destination interval discards, silencing every other destination + /// behind it. + pub fn has_token(&mut self, peer: &NodeAddr, now: Instant) -> bool { + self.refill(peer, now); + self.buckets + .get(peer) + .is_some_and(|b| b.milli_tokens >= MILLI) + } + + /// Spend one token for `peer`. Call only on the path that actually emits. + pub fn commit(&mut self, peer: &NodeAddr, now: Instant) { + self.refill(peer, now); + if let Some(bucket) = self.buckets.get_mut(peer) { + bucket.milli_tokens = bucket.milli_tokens.saturating_sub(MILLI); + } + self.sweep(now); + } + + /// Bring `peer`'s bucket up to date, creating a full one on first sighting. + fn refill(&mut self, peer: &NodeAddr, now: Instant) { + let capacity = self.capacity; + let milli_per_ms = self.milli_per_ms; + let bucket = self.buckets.entry(*peer).or_insert(Bucket { + milli_tokens: capacity, + last_refill: now, + }); + let elapsed_ms = now + .saturating_duration_since(bucket.last_refill) + .as_millis() as u64; + if elapsed_ms > 0 { + bucket.milli_tokens = (bucket.milli_tokens + elapsed_ms * milli_per_ms).min(capacity); + bucket.last_refill = now; + } + } + + /// Drop full buckets, at most once per [`SWEEP_INTERVAL`]. + fn sweep(&mut self, now: Instant) { + if now.saturating_duration_since(self.last_sweep) < SWEEP_INTERVAL { + return; + } + self.last_sweep = now; + let capacity = self.capacity; + let milli_per_ms = self.milli_per_ms; + self.buckets.retain(|_, b| { + let elapsed_ms = now.saturating_duration_since(b.last_refill).as_millis() as u64; + b.milli_tokens + elapsed_ms * milli_per_ms < capacity + }); + } + + #[cfg(test)] + pub fn len(&self) -> usize { + self.buckets.len() + } +} + +impl Default for PeerErrorBudget { + fn default() -> Self { + Self::new() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn addr(val: u8) -> NodeAddr { + let mut bytes = [0u8; 16]; + bytes[0] = val; + NodeAddr::from_bytes(bytes) + } + + /// Spend one token per admitted emission. + fn spend(budget: &mut PeerErrorBudget, peer: &NodeAddr, now: Instant) -> bool { + if !budget.has_token(peer, now) { + return false; + } + budget.commit(peer, now); + true + } + + #[test] + fn a_peer_may_emit_its_full_burst_then_is_suppressed() { + let mut budget = PeerErrorBudget::new(); + let now = Instant::now(); + let peer = addr(1); + + for i in 0..PEER_ERROR_BURST { + assert!(spend(&mut budget, &peer, now), "burst signal {i} refused"); + } + assert!(!spend(&mut budget, &peer, now)); + } + + #[test] + fn an_exhausted_budget_refills_at_the_sustained_rate() { + let mut budget = PeerErrorBudget::new(); + let start = Instant::now(); + let peer = addr(1); + + for _ in 0..PEER_ERROR_BURST { + assert!(spend(&mut budget, &peer, start)); + } + assert!(!spend(&mut budget, &peer, start)); + + // One second of refill buys exactly the sustained rate back. + let later = start + Duration::from_secs(1); + for i in 0..PEER_ERROR_RATE_PER_SEC { + assert!( + spend(&mut budget, &peer, later), + "refilled signal {i} refused" + ); + } + assert!(!spend(&mut budget, &peer, later)); + } + + #[test] + fn one_peer_exhausting_its_budget_does_not_silence_another() { + let mut budget = PeerErrorBudget::new(); + let now = Instant::now(); + let noisy = addr(1); + let quiet = addr(2); + + for _ in 0..PEER_ERROR_BURST { + assert!(spend(&mut budget, &noisy, now)); + } + assert!(!spend(&mut budget, &noisy, now)); + assert!(spend(&mut budget, &quiet, now)); + } + + #[test] + fn peeking_does_not_spend_a_token() { + let mut budget = PeerErrorBudget::with_rate(1, 1); + let now = Instant::now(); + let peer = addr(1); + + assert!(budget.has_token(&peer, now)); + assert!(budget.has_token(&peer, now)); + budget.commit(&peer, now); + assert!(!budget.has_token(&peer, now)); + } + + #[test] + fn full_buckets_are_dropped_by_the_sweep() { + let mut budget = PeerErrorBudget::new(); + let start = Instant::now(); + for i in 0..50u8 { + assert!(spend(&mut budget, &addr(i), start)); + } + assert_eq!(budget.len(), 50); + + // Long enough for every bucket to have refilled to full. + let later = start + SWEEP_INTERVAL + Duration::from_secs(1); + assert!(spend(&mut budget, &addr(200), later)); + assert_eq!(budget.len(), 1); + } +} diff --git a/src/node/routing_error_rate_limit.rs b/src/node/routing_error_rate_limit.rs index 32cdab48..be8fff32 100644 --- a/src/node/routing_error_rate_limit.rs +++ b/src/node/routing_error_rate_limit.rs @@ -2,11 +2,56 @@ //! //! Prevents routing error floods (CoordsRequired / PathBroken) by //! rate-limiting error signals per destination address at transit nodes. +//! +//! The destination address is chosen by whoever sent the datagram, so this +//! gate is an aggregate suppressor during a real outage and never a bound on +//! what one sender can induce: a fresh destination is always a first sighting. +//! The bound is `PeerErrorBudget`, keyed on the authenticated link peer and +//! consulted first. What this module owes on top of its interval is that its +//! own map stays bounded and its per-admit cost stays sub-linear whatever the +//! sender does with the key. use crate::NodeAddr; use std::collections::HashMap; use std::time::{Duration, Instant}; +/// Maximum number of destinations this limiter remembers at once. +/// +/// A hard ceiling on the map an attacker can grow by varying the destination +/// address. Raising it costs one `NodeAddr` plus one `Instant` per entry and +/// buys interval suppression across more simultaneously-unroutable +/// destinations; lowering it makes admission-without-recording (see +/// [`LimitVerdict::AdmitAtCapacity`]) the common case sooner, which weakens +/// the interval gate but never the peer budget. +const MAX_ENTRIES: usize = 4096; + +/// Fraction of `max_age` between amortized sweeps. +/// +/// The sweep is a full-map `retain`, so running it on every admit made +/// per-packet cost linear in a map the sender sizes. Eight sweeps per entry +/// lifetime keeps expired entries from accumulating without putting the scan +/// on the per-packet path. +const SWEEPS_PER_MAX_AGE: u32 = 8; + +/// What the limiter decided about one candidate error signal. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum LimitVerdict { + /// Send it; the destination was recorded. + Admit, + /// Send it, but the map was full so the destination was not recorded and + /// the interval will not suppress its successor. + /// + /// This gate fails open deliberately. Failing closed would turn a full map + /// into node-wide silence on error signalling, and the map is fullest + /// exactly during partition healing, when many destinations are + /// legitimately unroutable at once and sources most need the signal. The + /// bound on emission is the per-peer budget, not this map. + AdmitAtCapacity, + /// Suppress it; an error for this destination went out within the + /// interval. + Suppress, +} + /// Rate limiter for routing error signals (CoordsRequired / PathBroken). /// /// Tracks the last time a routing error was sent for each destination @@ -18,6 +63,12 @@ pub struct RoutingErrorRateLimiter { min_interval: Duration, /// Maximum age of entries before cleanup. max_age: Duration, + /// When `cleanup` last ran. + last_sweep: Instant, + /// Sweeps run since construction. Read by the tests that hold the + /// amortization property: a full-map scan per admit is the denial-of- + /// service multiplier this counter exists to catch coming back. + sweeps: u64, } impl RoutingErrorRateLimiter { @@ -29,6 +80,8 @@ impl RoutingErrorRateLimiter { last_sent: HashMap::new(), min_interval: Duration::from_millis(100), max_age: Duration::from_secs(10), + last_sweep: Instant::now(), + sweeps: 0, } } @@ -38,6 +91,8 @@ impl RoutingErrorRateLimiter { last_sent: HashMap::new(), min_interval, max_age: Duration::from_secs(10), + last_sweep: Instant::now(), + sweeps: 0, } } @@ -47,29 +102,57 @@ impl RoutingErrorRateLimiter { /// this destination, or if this is the first error. Updates internal /// state when returning true. pub fn should_send(&mut self, dest_addr: &NodeAddr) -> bool { - let now = Instant::now(); + self.check(dest_addr, Instant::now()) != LimitVerdict::Suppress + } + /// Decide about one error signal at an explicit `now`, reporting whether + /// the destination could be recorded. + /// + /// Callers that distinguish the at-capacity admission use this; callers + /// that only need a yes or no use [`Self::should_send`]. + pub fn check(&mut self, dest_addr: &NodeAddr, now: Instant) -> LimitVerdict { if let Some(&last) = self.last_sent.get(dest_addr) - && now.duration_since(last) < self.min_interval + && now.saturating_duration_since(last) < self.min_interval { - return false; + return LimitVerdict::Suppress; + } + + if self.last_sent.len() >= MAX_ENTRIES && !self.last_sent.contains_key(dest_addr) { + self.maybe_cleanup(now); + if self.last_sent.len() >= MAX_ENTRIES { + return LimitVerdict::AdmitAtCapacity; + } } self.last_sent.insert(*dest_addr, now); - self.cleanup(now); - true + self.maybe_cleanup(now); + LimitVerdict::Admit + } + + /// Run the sweep if one is due. + fn maybe_cleanup(&mut self, now: Instant) { + if now.saturating_duration_since(self.last_sweep) >= self.max_age / SWEEPS_PER_MAX_AGE { + self.cleanup(now); + } } /// Remove entries older than max_age. fn cleanup(&mut self, now: Instant) { + self.last_sweep = now; + self.sweeps += 1; self.last_sent - .retain(|_, &mut last| now.duration_since(last) < self.max_age); + .retain(|_, &mut last| now.saturating_duration_since(last) < self.max_age); } #[cfg(test)] pub fn len(&self) -> usize { self.last_sent.len() } + + #[cfg(test)] + pub fn sweeps(&self) -> u64 { + self.sweeps + } } impl Default for RoutingErrorRateLimiter { @@ -89,6 +172,15 @@ mod tests { NodeAddr::from_bytes(bytes) } + /// A distinct destination address per index, standing for the fresh + /// `dest_addr` a flooding sender puts on every datagram. + fn minted_addr(val: u32) -> NodeAddr { + let mut bytes = [0u8; 16]; + bytes[..4].copy_from_slice(&val.to_le_bytes()); + bytes[15] = 0xff; + NodeAddr::from_bytes(bytes) + } + #[test] fn test_first_send_allowed() { let mut limiter = RoutingErrorRateLimiter::new(); @@ -144,6 +236,71 @@ mod tests { assert_eq!(limiter.len(), 1); } + #[test] + fn the_map_stays_bounded_when_a_sender_mints_distinct_destination_keys() { + let mut limiter = RoutingErrorRateLimiter::new(); + let now = Instant::now(); + + for i in 0..100_000u32 { + limiter.check(&minted_addr(i), now); + } + + assert!( + limiter.len() <= MAX_ENTRIES, + "limiter held {} entries, above the {MAX_ENTRIES} ceiling", + limiter.len() + ); + } + + #[test] + fn an_admission_at_capacity_still_sends_rather_than_going_silent() { + let mut limiter = RoutingErrorRateLimiter::new(); + let now = Instant::now(); + + for i in 0..MAX_ENTRIES as u32 { + assert_eq!(limiter.check(&minted_addr(i), now), LimitVerdict::Admit); + } + + // The map is full and nothing in it is old enough to evict, so the + // next distinct destination cannot be recorded. It must still be sent. + assert_eq!( + limiter.check(&minted_addr(MAX_ENTRIES as u32), now), + LimitVerdict::AdmitAtCapacity + ); + } + + #[test] + fn the_map_scan_does_not_run_once_per_admitted_destination() { + let mut limiter = RoutingErrorRateLimiter::new(); + let now = Instant::now(); + let before = limiter.sweeps(); + + for i in 0..1_000u32 { + limiter.check(&minted_addr(i), now); + } + + assert_eq!( + limiter.sweeps() - before, + 0, + "the full-map scan ran inside a single sweep interval" + ); + } + + #[test] + fn the_map_scan_still_runs_once_a_sweep_interval_has_passed() { + let mut limiter = RoutingErrorRateLimiter::new(); + let start = Instant::now(); + limiter.check(&minted_addr(0), start); + let before = limiter.sweeps(); + + let later = start + Duration::from_secs(11); + limiter.check(&minted_addr(1), later); + + assert_eq!(limiter.sweeps() - before, 1); + // The first destination aged out, so the sweep did its job. + assert_eq!(limiter.len(), 1); + } + #[test] fn test_with_interval_custom_rate() { let mut limiter = RoutingErrorRateLimiter::with_interval(Duration::from_millis(500)); diff --git a/src/node/stats.rs b/src/node/stats.rs index acf5f7c9..d92b838e 100644 --- a/src/node/stats.rs +++ b/src/node/stats.rs @@ -421,6 +421,9 @@ pub struct ErrorSignalStatsSnapshot { pub unbound_broken: u64, pub unbound_mtu: u64, pub unbound_forged: u64, + pub emit_over_peer_budget: u64, + pub emit_over_dest_interval: u64, + pub emit_limiter_at_capacity: u64, } #[derive(Clone, Debug, Default, Serialize)] diff --git a/src/node/tests/forwarding.rs b/src/node/tests/forwarding.rs index 4ba0db08..c71bd9e7 100644 --- a/src/node/tests/forwarding.rs +++ b/src/node/tests/forwarding.rs @@ -5,6 +5,7 @@ //! multi-hop forwarding through live node topologies. use super::*; +use crate::node::peer_error_budget::PEER_ERROR_BURST; use crate::node::session_wire::{FSP_FLAG_CP, build_fsp_header}; use crate::protocol::{SessionAck, SessionDatagram, SessionSetup, encode_coords}; use crate::tree::TreeCoordinate; @@ -1111,3 +1112,90 @@ fn test_sample_transport_congestion() { node.sample_transport_congestion(); assert!(!node.transport_drops[&tid].dropping); } + +// --- Emission bounds on induced routing errors --- + +/// A distinct destination per index, standing for the fresh `dest_addr` a +/// flooding sender puts on every datagram to escape the per-destination gate. +fn minted_dest(val: u32) -> NodeAddr { + let mut bytes = [0u8; 16]; + bytes[..4].copy_from_slice(&val.to_le_bytes()); + bytes[15] = 0xfe; + NodeAddr::from_bytes(bytes) +} + +/// Feed one transit datagram whose destination this node cannot route. +async fn inject_unroutable(node: &mut Node, from: &NodeAddr, src: NodeAddr, dest: NodeAddr) { + let dg = SessionDatagram::new(src, dest, vec![0x10, 0x00, 0x00, 0x00]).with_ttl(8); + let encoded = dg.encode(); + node.handle_session_datagram(from, &encoded[1..], false) + .await; +} + +#[tokio::test] +async fn one_link_peer_cannot_induce_unbounded_errors_by_varying_the_destination() { + let mut node = make_node(); + let attacker = make_node_addr(0xAA); + let overshoot = 10u32; + + for i in 0..PEER_ERROR_BURST + overshoot { + // Fresh destination and fresh spoofed source per packet: neither + // address-keyed gate sees a repeat. + inject_unroutable( + &mut node, + &attacker, + minted_dest(i + 1_000_000), + minted_dest(i), + ) + .await; + } + + let errors = &node.metrics().errors; + assert_eq!( + errors.emit_over_dest_interval.get(), + 0, + "the per-destination gate cannot bound a sender that varies the destination" + ); + assert_eq!( + errors.emit_over_peer_budget.get(), + u64::from(overshoot), + "everything past the link peer's burst must be refused" + ); +} + +#[tokio::test] +async fn a_destination_suppressed_error_does_not_spend_the_link_peer_budget() { + let mut node = make_node(); + let peer = make_node_addr(0xAA); + let src = make_node_addr(0x01); + let dest = make_node_addr(0x02); + let injected = PEER_ERROR_BURST * 4; + + for _ in 0..injected { + inject_unroutable(&mut node, &peer, src, dest).await; + } + + let errors = &node.metrics().errors; + assert_eq!( + errors.emit_over_peer_budget.get(), + 0, + "an outage on one destination must not spend the peer's budget for the others" + ); + assert_eq!( + errors.emit_over_dest_interval.get(), + u64::from(injected - 1), + "only the first error for a destination goes out within the interval" + ); +} + +#[tokio::test] +async fn a_single_unroutable_datagram_still_produces_its_error() { + let mut node = make_node(); + let peer = make_node_addr(0xAA); + + inject_unroutable(&mut node, &peer, make_node_addr(0x01), make_node_addr(0x02)).await; + + let errors = &node.metrics().errors; + assert_eq!(errors.emit_over_peer_budget.get(), 0); + assert_eq!(errors.emit_over_dest_interval.get(), 0); +} From cbe35f1cace836d6987d78985e62620b6b85bbe2 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:18:14 +0100 Subject: [PATCH 16/23] Act on a lookup response only when it answers a lookup we issued The originator path took any LookupResponse whose request_id was not in the transit dedup map, so an admitted peer could harvest one genuine signed response for a target and re-inject it whenever it liked. Each injection cleared our in-flight lookup, recorded a reachability success for a target that might be unreachable, refreshed the cached coordinates for a further full TTL, and flushed our queued packets onto that route at a moment the sender chose. The signature verify ran before any check that the response was wanted, so it was the first cost gate on the path. Record the request_id of every lookup request we send on that target's pending entry, and drop a response unless it names a target with a lookup outstanding and carries one of the ids issued for it. The id is fresh 64-bit randomness drawn per attempt and the target signs over it, so a harvested response is bound to the request it answered and cannot be redirected or replayed. The check runs before the identity-cache resolve and before the verify, so a response nobody asked for costs nothing. initiate_lookup now establishes the pending entry itself rather than relying on its callers, which keeps "if a request went out, its id is recorded" true everywhere. The recorded set is capped at eight ids and evicts the oldest rather than refusing the newest, so a retry ladder longer than the cap cannot discard the attempt most likely to be answered; replies to earlier attempts of an outstanding lookup are still accepted, which is the common case on a link whose round trip exceeds the first rung. Drops are counted as resp_unsolicited, in show routing, show metrics and the fipstop routing pane. The counter has a nonzero floor in healthy operation: a request is flooded to every qualifying tree peer, so the duplicate replies land there once the first has been accepted. --- CHANGELOG.md | 25 +++ src/bin/fipstop/ui/routing.rs | 1 + src/control/snapshots/show_routing.json | 3 +- src/node/handlers/discovery.rs | 71 ++++++ src/node/metrics.rs | 3 + src/node/reject.rs | 9 + src/node/stats.rs | 1 + src/node/tests/discovery.rs | 280 ++++++++++++++++++++++++ 8 files changed, 392 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a1bf8244..b208700b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1052,6 +1052,31 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 claimed source and destination pairing no honest forwarder could produce. The drop log line now carries the signal type and the refusal class. +- A discovery lookup response is now acted on only when it answers a lookup + this node actually has outstanding. The originator path took any response + whose `request_id` was not in the transit dedup map, so an admitted peer + could harvest one genuine signed response for a target and re-inject it at + will: each injection cleared the victim's in-flight lookup, recorded a + reachability success for a target that might be unreachable, refreshed the + cached coordinates for a further full TTL, and flushed the victim's queued + packets onto a route at a moment the sender chose. It also reached the + signature verify before any check that the response was wanted, so the + verify was the first cost gate on the path. The node now records the + `request_id` of every lookup request it sends on that target's pending + entry, and a response is dropped unless it names a target with a lookup + outstanding and carries one of the ids issued for it. Because the id is + fresh 64-bit randomness drawn per attempt and the target signs over it, a + harvested response is bound to the request it answered and cannot be + redirected or replayed. The check runs before the identity-cache resolve + and before the signature verify, so a response nobody asked for costs + nothing. Replies to earlier attempts of a still-outstanding lookup are + still accepted, which is the common case on a link whose round trip + exceeds the first rung of the retry ladder. Drops are counted as + `resp_unsolicited`, visible through `show routing`, `show metrics` and the + fipstop routing pane; the counter has a nonzero floor in healthy operation, + because a request is flooded to every qualifying tree peer and the + duplicate replies land there once the first has been accepted. + #### Admission / peer caps - The Ethernet transport's discovery buffer is now bounded and no longer costs diff --git a/src/bin/fipstop/ui/routing.rs b/src/bin/fipstop/ui/routing.rs index 015234d6..d79f64a3 100644 --- a/src/bin/fipstop/ui/routing.rs +++ b/src/bin/fipstop/ui/routing.rs @@ -206,6 +206,7 @@ fn draw_routing_stats( ("Timed Out", disc("resp_timed_out")), ("Identity Miss", disc("resp_identity_miss")), ("Proof Failed", disc("resp_proof_failed")), + ("Unsolicited", disc("resp_unsolicited")), ("Decode Error", disc("resp_decode_error")), ], )); diff --git a/src/control/snapshots/show_routing.json b/src/control/snapshots/show_routing.json index 1e5328e2..11fcd3ef 100644 --- a/src/control/snapshots/show_routing.json +++ b/src/control/snapshots/show_routing.json @@ -29,7 +29,8 @@ "resp_no_route": 0, "resp_proof_failed": 0, "resp_received": 0, - "resp_timed_out": 0 + "resp_timed_out": 0, + "resp_unsolicited": 0 }, "error_signals": { "coords_required": 0, diff --git a/src/node/handlers/discovery.rs b/src/node/handlers/discovery.rs index 24fe5703..32aac19e 100644 --- a/src/node/handlers/discovery.rs +++ b/src/node/handlers/discovery.rs @@ -185,6 +185,32 @@ impl Node { let target = response.target; let path_mtu = response.path_mtu; + // Correlate against our own outstanding lookups first. The + // request_id is fresh 64-bit randomness we drew per attempt and + // the target signs over it, so requiring the response to carry + // one we issued for this target is what makes this path + // solicited: an unsolicited or replayed response is dropped + // here, before the identity resolve and before the signature + // verify, and so cannot clear pending state, record a backoff + // success, refresh the coordinate cache, or flush queued + // packets. A duplicate of a response already accepted lands + // here too, which is ordinary and is why this is debug level. + let solicited = self + .pending_lookups + .get(&target) + .is_some_and(|pending| pending.matches(response.request_id)); + if !solicited { + self.metrics() + .discovery + .record_reject(DiscoveryReject::RespUnsolicited); + debug!( + request_id = response.request_id, + target = %self.peer_display_name(&target), + "LookupResponse does not match an outstanding request, dropping" + ); + return; + } + // Look up the target's public key from identity_cache let mut prefix = [0u8; 15]; prefix.copy_from_slice(&target.as_bytes()[0..15]); @@ -487,6 +513,10 @@ impl Node { /// filters contain the target. Returns the number of peers sent to. /// The originator does NOT record the request_id in recent_requests, /// so when the response arrives, it's recognized as "our request". + /// It records the id on the target's pending entry instead, which is + /// what the response path correlates against; recording it here rather + /// than in the callers keeps "if a request went out, its id is + /// recorded" true for every caller. pub(in crate::node) async fn initiate_lookup(&mut self, target: &NodeAddr, ttl: u8) -> usize { self.metrics().discovery.req_initiated.inc(); @@ -494,6 +524,12 @@ impl Node { let origin_coords = self.tree_state().my_coords().clone(); let request = LookupRequest::generate(*target, origin, origin_coords, ttl, 0); + let now_ms = Self::now_ms(); + self.pending_lookups + .entry(*target) + .or_insert_with(|| PendingLookup::new(now_ms)) + .record(request.request_id); + // Send only to tree peers whose bloom filter contains the target let peer_addrs: Vec = self .peers @@ -790,6 +826,18 @@ impl Node { } } +/// How many outstanding `request_id`s one pending lookup remembers. +/// +/// Bounds the per-target correlator at eight u64s. The retry ladder +/// (`node.discovery.attempt_timeouts_secs`) is operator configuration and +/// can be longer than this, so the recorder evicts the oldest id rather +/// than refusing the newest: dropping the newest would discard the id most +/// likely to be answered and fail a healthy lookup. Raising this costs +/// eight bytes per extra attempt on every pending target and widens the +/// set of ids a late response may still match; lowering it means a reply +/// to an early attempt on a long ladder is dropped as unsolicited. +const MAX_RECORDED_IDS: usize = 8; + /// Tracks a pending discovery lookup with retry state. pub struct PendingLookup { /// When the lookup was first initiated. @@ -798,6 +846,12 @@ pub struct PendingLookup { pub last_sent_ms: u64, /// Current attempt number (1 = initial, 2 = first retry, ...). pub attempt: u8, + /// `request_id`s issued for this target, oldest first, capped at + /// [`MAX_RECORDED_IDS`]. A response is only acted on when it carries + /// one of these, which is what makes the accept path solicited. The + /// entry itself is dropped at ladder timeout, so this set needs no + /// expiry of its own. + pub ids: Vec, } impl PendingLookup { @@ -806,6 +860,23 @@ impl PendingLookup { initiated_ms: now_ms, last_sent_ms: now_ms, attempt: 1, + ids: Vec::new(), } } + + /// Remember a `request_id` we just put on the wire for this target. + pub fn record(&mut self, request_id: u64) { + if self.ids.contains(&request_id) { + return; + } + if self.ids.len() >= MAX_RECORDED_IDS { + self.ids.remove(0); + } + self.ids.push(request_id); + } + + /// Whether `request_id` is one this node issued for this target. + pub fn matches(&self, request_id: u64) -> bool { + self.ids.contains(&request_id) + } } diff --git a/src/node/metrics.rs b/src/node/metrics.rs index 6dfc0bb3..73f8c1c5 100644 --- a/src/node/metrics.rs +++ b/src/node/metrics.rs @@ -281,6 +281,7 @@ pub struct DiscoveryMetrics { pub resp_forwarded: Counter, pub resp_identity_miss: Counter, pub resp_proof_failed: Counter, + pub resp_unsolicited: Counter, pub resp_no_route: Counter, pub resp_accepted: Counter, pub resp_timed_out: Counter, @@ -299,6 +300,7 @@ impl DiscoveryMetrics { DiscoveryReject::RespDecodeError => self.resp_decode_error.inc(), DiscoveryReject::RespIdentityMiss => self.resp_identity_miss.inc(), DiscoveryReject::RespProofFailed => self.resp_proof_failed.inc(), + DiscoveryReject::RespUnsolicited => self.resp_unsolicited.inc(), DiscoveryReject::RespNoRoute => self.resp_no_route.inc(), } } @@ -325,6 +327,7 @@ impl DiscoveryMetrics { resp_forwarded: self.resp_forwarded.get(), resp_identity_miss: self.resp_identity_miss.get(), resp_proof_failed: self.resp_proof_failed.get(), + resp_unsolicited: self.resp_unsolicited.get(), resp_no_route: self.resp_no_route.get(), resp_accepted: self.resp_accepted.get(), resp_timed_out: self.resp_timed_out.get(), diff --git a/src/node/reject.rs b/src/node/reject.rs index c270c637..0b3bccbc 100644 --- a/src/node/reject.rs +++ b/src/node/reject.rs @@ -136,6 +136,14 @@ pub enum DiscoveryReject { /// Response proof signature failed verification. Tracked via /// [`DiscoveryStats::resp_proof_failed`](crate::node::stats::DiscoveryStats). RespProofFailed, + /// Response arrived on the originator path but carries no + /// `request_id` this node has outstanding for the named target, so + /// it answers no lookup of ours. Expected to be nonzero in healthy + /// operation: the request is flooded to every qualifying tree peer, + /// so duplicate replies land here after the first is accepted. + /// Tracked via + /// [`DiscoveryStats::resp_unsolicited`](crate::node::stats::DiscoveryStats). + RespUnsolicited, /// Response could not be routed toward the origin: no reverse-path /// entry for the `request_id` and no greedy tree route to the /// origin. Tracked via @@ -371,6 +379,7 @@ mod tests { DiscoveryReject::RespDecodeError, DiscoveryReject::RespIdentityMiss, DiscoveryReject::RespProofFailed, + DiscoveryReject::RespUnsolicited, ]; for v in variants { let r = RejectReason::Discovery(v); diff --git a/src/node/stats.rs b/src/node/stats.rs index d92b838e..9664870c 100644 --- a/src/node/stats.rs +++ b/src/node/stats.rs @@ -334,6 +334,7 @@ pub struct DiscoveryStatsSnapshot { pub resp_forwarded: u64, pub resp_identity_miss: u64, pub resp_proof_failed: u64, + pub resp_unsolicited: u64, pub resp_no_route: u64, pub resp_accepted: u64, pub resp_timed_out: u64, diff --git a/src/node/tests/discovery.rs b/src/node/tests/discovery.rs index 449aa155..126bdb08 100644 --- a/src/node/tests/discovery.rs +++ b/src/node/tests/discovery.rs @@ -84,6 +84,18 @@ async fn test_request_ttl_zero_not_forwarded() { // Unit Tests — LookupResponse Handler // ============================================================================ +/// Record `request_id` as outstanding for `target`, exactly as +/// `initiate_lookup` does when it puts a request on the wire. The response +/// handler correlates against this, so a unit test that hands the handler a +/// response without it is testing the correlation gate rather than whatever +/// it names. +fn seed_pending_lookup(node: &mut Node, target: crate::NodeAddr, request_id: u64) { + node.pending_lookups + .entry(target) + .or_insert_with(|| handlers::discovery::PendingLookup::new(Node::now_ms())) + .record(request_id); +} + #[tokio::test] async fn test_response_decode_error() { let mut node = make_node(); @@ -107,6 +119,8 @@ async fn test_response_originator_caches_route() { // Register target identity in cache so verification can find it node.register_identity(target, target_identity.pubkey_full()); + seed_pending_lookup(&mut node, target, 555); + // Create a valid response with a real proof signature (includes coords) let proof_data = LookupResponse::proof_bytes(555, &target, &coords); let proof = target_identity.sign(&proof_data); @@ -183,6 +197,8 @@ async fn test_response_proof_verification_success() { // Register target in identity_cache node.register_identity(target, target_identity.pubkey_full()); + seed_pending_lookup(&mut node, target, 700); + // Sign with correct proof_bytes (including coords) let proof_data = LookupResponse::proof_bytes(700, &target, &coords); let proof = target_identity.sign(&proof_data); @@ -218,6 +234,8 @@ async fn test_response_proof_verification_failure() { node.register_identity(target, target_identity.pubkey_full()); // Sign with a DIFFERENT identity (wrong key) + seed_pending_lookup(&mut node, target, 701); + let wrong_identity = Identity::generate(); let proof_data = LookupResponse::proof_bytes(701, &target, &coords); let proof = wrong_identity.sign(&proof_data); @@ -251,6 +269,8 @@ async fn test_response_identity_cache_miss() { // Do NOT register target in identity_cache + seed_pending_lookup(&mut node, target, 702); + let proof_data = LookupResponse::proof_bytes(702, &target, &coords); let proof = target_identity.sign(&proof_data); @@ -285,6 +305,8 @@ async fn test_response_coord_substitution_detected() { // Register target in identity_cache node.register_identity(target, target_identity.pubkey_full()); + seed_pending_lookup(&mut node, target, 703); + // Sign proof with real coords let proof_data = LookupResponse::proof_bytes(703, &target, &real_coords); let proof = target_identity.sign(&proof_data); @@ -305,6 +327,255 @@ async fn test_response_coord_substitution_detected() { ); } +/// Build a signed LookupResponse body for `target_identity` over +/// `request_id`, ready to hand to `handle_lookup_response`. +fn signed_response_body( + target_identity: &Identity, + request_id: u64, + coords: &TreeCoordinate, +) -> Vec { + let target = *target_identity.node_addr(); + let proof_data = LookupResponse::proof_bytes(request_id, &target, coords); + let proof = target_identity.sign(&proof_data); + LookupResponse::new(request_id, target, coords.clone(), proof).encode()[1..].to_vec() +} + +/// Register `target_identity` and return its address and a plausible +/// coordinate for it, the shared preamble of the correlation tests. +fn register_lookup_target(node: &mut Node, target_identity: &Identity) -> TreeCoordinate { + let target = *target_identity.node_addr(); + node.register_identity(target, target_identity.pubkey_full()); + TreeCoordinate::from_addrs(vec![target, make_node_addr(0xF0)]).unwrap() +} + +fn wall_clock_ms() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_millis() as u64) + .unwrap_or(0) +} + +#[tokio::test] +async fn test_unsolicited_lookup_response_is_dropped_before_proof_verification() { + // Any admitted peer can hand us a correctly signed response for a target + // we never asked about. Accepting it lets that peer clear our pending + // state, refresh a cache entry's TTL and flush our queued packets at a + // moment it picks, so the response must not be acted on at all. + let mut node = make_node(); + let from = make_node_addr(0xAA); + + let target_identity = Identity::generate(); + let target = *target_identity.node_addr(); + let coords = register_lookup_target(&mut node, &target_identity); + let body = signed_response_body(&target_identity, 900, &coords); + + assert!( + node.pending_lookups.is_empty(), + "precondition: this node has no lookup outstanding for anything" + ); + + node.handle_lookup_response(&from, &body).await; + + assert!( + !node.coord_cache().contains(&target, wall_clock_ms()), + "a response answering no request of ours must not reach the coordinate cache" + ); + assert_eq!( + node.metrics().discovery.resp_accepted.get(), + 0, + "an unsolicited response must not count as accepted" + ); + assert_eq!( + node.metrics().discovery.resp_unsolicited.get(), + 1, + "the drop must be visible on a counter, not only in a log" + ); + assert_eq!( + node.metrics().discovery.resp_proof_failed.get(), + 0, + "the drop must happen before the signature verify, so the verify is not a cost gate" + ); +} + +#[tokio::test] +async fn test_lookup_response_with_a_request_id_we_never_issued_is_dropped() { + // Correlating on the target alone would leave the attack open: there is + // no inbound limiter on responses, so a peer can spray a harvested one + // and land inside any window in which we happen to be looking that + // target up. The id must match too. + let mut node = make_node(); + let from = make_node_addr(0xAA); + + let target_identity = Identity::generate(); + let target = *target_identity.node_addr(); + let coords = register_lookup_target(&mut node, &target_identity); + + node.initiate_lookup(&target, 5).await; + let issued = node.pending_lookups.get(&target).unwrap().ids.clone(); + assert_eq!(issued.len(), 1, "precondition: one attempt went out"); + + let body = signed_response_body(&target_identity, issued[0] ^ 1, &coords); + node.handle_lookup_response(&from, &body).await; + + assert!( + !node.coord_cache().contains(&target, wall_clock_ms()), + "a response bearing an id we never issued must not reach the coordinate cache" + ); + assert!( + node.pending_lookups.contains_key(&target), + "it must not cancel the lookup that is genuinely outstanding" + ); + assert_eq!(node.metrics().discovery.resp_unsolicited.get(), 1); +} + +#[tokio::test] +async fn test_a_response_matching_a_pending_attempt_is_accepted_and_clears_the_pending_lookup() { + // The healthy path. A fix that reds a legitimate lookup is no use, and + // this is the test that catches it. + let mut node = make_node(); + let from = make_node_addr(0xAA); + + let target_identity = Identity::generate(); + let target = *target_identity.node_addr(); + let coords = register_lookup_target(&mut node, &target_identity); + + node.initiate_lookup(&target, 5).await; + let issued = node.pending_lookups.get(&target).unwrap().ids[0]; + + let body = signed_response_body(&target_identity, issued, &coords); + node.handle_lookup_response(&from, &body).await; + + assert_eq!( + node.coord_cache().get(&target, wall_clock_ms()), + Some(&coords), + "a response to our own outstanding request must be cached" + ); + assert!( + !node.pending_lookups.contains_key(&target), + "accepting it must clear the pending lookup" + ); + assert_eq!(node.metrics().discovery.resp_accepted.get(), 1); + assert_eq!(node.metrics().discovery.resp_unsolicited.get(), 0); +} + +#[tokio::test] +async fn test_a_late_response_for_an_earlier_retry_attempt_is_still_accepted() { + // Each retry draws a fresh id, and on any link with more than a second + // of round trip the reply to an earlier attempt is the common case. A + // correlator that remembered only the newest id would drop it. + let mut node = make_node(); + let from = make_node_addr(0xAA); + + let target_identity = Identity::generate(); + let target = *target_identity.node_addr(); + let coords = register_lookup_target(&mut node, &target_identity); + + node.initiate_lookup(&target, 5).await; + node.initiate_lookup(&target, 5).await; + let issued = node.pending_lookups.get(&target).unwrap().ids.clone(); + assert_eq!(issued.len(), 2, "precondition: two attempts, two ids"); + + let body = signed_response_body(&target_identity, issued[0], &coords); + node.handle_lookup_response(&from, &body).await; + + assert_eq!( + node.coord_cache().get(&target, wall_clock_ms()), + Some(&coords), + "the first attempt's id is still ours and its answer must be accepted" + ); +} + +#[tokio::test] +async fn test_a_second_genuine_response_after_the_first_is_accepted_is_dropped() { + // The request is flooded to every qualifying tree peer, so duplicate + // replies are routine. They are dropped at the correlation gate, which + // gives the unsolicited counter a nonzero floor in healthy operation: + // it is not by itself a sign of attack traffic. + let mut node = make_node(); + let from = make_node_addr(0xAA); + + let target_identity = Identity::generate(); + let target = *target_identity.node_addr(); + let coords = register_lookup_target(&mut node, &target_identity); + + node.initiate_lookup(&target, 5).await; + let issued = node.pending_lookups.get(&target).unwrap().ids[0]; + let body = signed_response_body(&target_identity, issued, &coords); + + node.handle_lookup_response(&from, &body).await; + node.handle_lookup_response(&from, &body).await; + + assert_eq!( + node.coord_cache().get(&target, wall_clock_ms()), + Some(&coords), + "the value written by the first response must still be there" + ); + assert_eq!( + node.metrics().discovery.resp_accepted.get(), + 1, + "only the first of the two answers our request" + ); + assert_eq!( + node.metrics().discovery.resp_unsolicited.get(), + 1, + "the duplicate is counted, which is why the counter has a healthy floor" + ); +} + +#[tokio::test] +async fn test_a_validly_signed_response_for_a_retired_lookup_is_dropped() { + // The pending entry's lifetime is what bounds how stale an accepted + // coordinate can be. Once the retry ladder is exhausted and the entry + // goes, a transit node holding the genuine reply can no longer deliver + // it late. + let mut node = make_node(); + let from = make_node_addr(0xAA); + + let target_identity = Identity::generate(); + let target = *target_identity.node_addr(); + let coords = register_lookup_target(&mut node, &target_identity); + + node.initiate_lookup(&target, 5).await; + let issued = node.pending_lookups.get(&target).unwrap().ids[0]; + + // Drive the whole ladder: three retries, then the final timeout. + let mut now_ms = Node::now_ms(); + for _ in 0..4 { + now_ms += 100_000; + node.check_pending_lookups(now_ms).await; + } + assert!( + !node.pending_lookups.contains_key(&target), + "precondition: the ladder retired the lookup" + ); + + let body = signed_response_body(&target_identity, issued, &coords); + node.handle_lookup_response(&from, &body).await; + + assert!( + !node.coord_cache().contains(&target, wall_clock_ms()), + "a reply to a retired lookup must not install a coordinate" + ); + assert_eq!(node.metrics().discovery.resp_unsolicited.get(), 1); +} + +#[test] +fn pending_lookup_id_set_evicts_the_oldest_id_rather_than_refusing_the_newest() { + // The retry ladder is operator configuration and can be longer than the + // recorded-id cap. Refusing the newest id would discard the attempt most + // likely to be answered and fail a healthy lookup. + let mut pending = handlers::discovery::PendingLookup::new(0); + for id in 0..12u64 { + pending.record(id); + } + assert!( + pending.matches(11), + "the newest attempt's id must always be remembered" + ); + assert!(!pending.matches(0), "the oldest id is the one evicted"); + assert_eq!(pending.ids.len(), 8, "the set stays bounded"); +} + // ============================================================================ // Unit Tests — RecentRequest Expiry // ============================================================================ @@ -886,6 +1157,8 @@ async fn test_originator_stores_path_mtu_in_cache() { node.register_identity(target, target_identity.pubkey_full()); + seed_pending_lookup(&mut node, target, 800); + let proof_data = LookupResponse::proof_bytes(800, &target, &coords); let proof = target_identity.sign(&proof_data); @@ -930,6 +1203,8 @@ async fn test_originator_ignores_sub_floor_path_mtu_but_still_caches_coords() { node.register_identity(target, target_identity.pubkey_full()); + seed_pending_lookup(&mut node, target, 801); + let proof_data = LookupResponse::proof_bytes(801, &target, &coords); let proof = target_identity.sign(&proof_data); @@ -980,6 +1255,8 @@ async fn test_actionable_lookup_response_path_mtu_does_not_bump_below_floor_coun node.register_identity(target, target_identity.pubkey_full()); + seed_pending_lookup(&mut node, target, 802); + let proof_data = LookupResponse::proof_bytes(802, &target, &coords); let proof = target_identity.sign(&proof_data); @@ -1021,6 +1298,8 @@ async fn test_originator_lookup_response_keeps_tighter_path_mtu_lookup() { let target_fips = crate::FipsAddress::from_node_addr(&target); node.path_mtu_lookup_insert(target_fips, 1280); + seed_pending_lookup(&mut node, target, 800); + let proof_data = LookupResponse::proof_bytes(800, &target, &coords); let proof = target_identity.sign(&proof_data); @@ -1055,6 +1334,7 @@ fn make_verified_lookup_response( let coords = TreeCoordinate::from_addrs(vec![target, root]).unwrap(); node.register_identity(target, target_identity.pubkey_full()); + seed_pending_lookup(node, target, request_id); let proof_data = LookupResponse::proof_bytes(request_id, &target, &coords); let proof = target_identity.sign(&proof_data); From 65321617ae7733594c7bef3dc8f18086eacf010a Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:29:36 +0100 Subject: [PATCH 17/23] Evict from the lookup dedup cache instead of refusing the request The discovery dedup cache is also the reverse-path table for responses in flight, and at its 4096-entry bound it dropped the arriving request. That drop sat ahead of both the check for whether the request names this node and the forwarding path, so one link peer emitting fresh request_ids could stop the node answering lookups for itself and stop it carrying anyone else's, for as long as it kept the cache full. Make room instead of refusing. An arrival-order index partitioned by the link peer the request came from says who pays: a peer over its own share loses its oldest entry, and at global capacity the peer holding the most entries loses its oldest. A light peer's reverse path is therefore never taken to admit a heavy one, and extra identities buy a flooder proportionally less. A share is the cache divided by the current link-peer count with a floor of 64, so it tracks the peer count rather than being pinned to a number a many-peer node outgrows. The loosening this accepts is that an evicted request_id arriving again inside the window is forwarded a second time rather than recognised as a duplicate; the per-target forward limiter and TTL already bound that. Meter answering lookups for ourselves per link peer, in the same change, because the cache filling up was the only thing bounding it. The response proof is signed over the requester's request_id, so every request addressed to this node costs a fresh Schnorr signature that cannot be cached or served twice. A token bucket of 256 signatures refilling at 32 per second per link peer absorbs the legitimate burst that follows a topology change, when many correspondents re-look-up at once through the few links that lead here, while capping what one neighbour can make the node sign. A refused request keeps its dedup entry, and retries carry fresh request_ids, so a refusal cannot suppress the retry. Evictions count as req_dedup_evicted and signing refusals as req_sign_rate_limited, both in show routing, show metrics and the fipstop routing pane. The old req_dedup_cache_full counter stays in place, frozen at zero, so a dashboard carried across versions does not lose the series. --- CHANGELOG.md | 32 ++++ src/bin/fipstop/ui/routing.rs | 2 + src/control/snapshots/show_routing.json | 2 + src/node/discovery_rate_limit.rs | 138 ++++++++++++++++ src/node/handlers/discovery.rs | 115 ++++++++++++-- src/node/metrics.rs | 5 + src/node/mod.rs | 26 ++- src/node/reject.rs | 13 ++ src/node/stats.rs | 2 + src/node/tests/discovery.rs | 200 ++++++++++++++++++++++++ 10 files changed, 516 insertions(+), 19 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index b208700b..ea316156 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1077,6 +1077,38 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 because a request is flooded to every qualifying tree peer and the duplicate replies land there once the first has been accepted. +- A flooded discovery dedup cache no longer makes a node unresolvable. The + cache is both the duplicate filter and the reverse-path table for lookup + responses, and at its 4096-entry bound it dropped the arriving request. + That drop sat ahead of both the check for whether the request names this + node and the forwarding path, so one link peer emitting fresh request_ids + could stop the node answering lookups for itself and stop it carrying + anyone else's, for as long as it kept the cache full. The cache now makes + room instead of refusing: over a peer's own share it drops that peer's + oldest entry, and at global capacity it drops the oldest entry of whichever + peer holds the most, so a light peer's reverse path is never taken to admit + a heavy one and extra identities buy a flooder proportionally less. A + peer's share is the cache divided by the current link-peer count, with a + floor of 64. The loosening this accepts is that an evicted request_id + arriving again inside the window is forwarded a second time rather than + recognised as a duplicate, which the per-target forward limiter and TTL + already bound. Evictions are counted as `req_dedup_evicted`; the old + `req_dedup_cache_full` counter stays in place, frozen at zero, so a + dashboard carried across versions does not lose the series. + +- Answering a lookup for ourselves is now metered per link peer. The response + proof is signed over the requester's `request_id`, so every request + addressed to this node costs a fresh Schnorr signature that cannot be + cached or served twice, and until now the only thing bounding that rate was + the dedup cache filling up, which is the defect above. A token bucket per + link peer, 256 signatures of burst refilling at 32 per second, absorbs the + legitimate burst that follows a topology change, when many correspondents + re-look-up at once through the few links that lead here, while capping what + one neighbour can make the node sign. Refusals are counted as + `req_sign_rate_limited` and visible in `show routing`, `show metrics` and + the fipstop routing pane. A refused request keeps its dedup entry, and + retries carry fresh request_ids, so a refusal cannot suppress the retry. + #### Admission / peer caps - The Ethernet transport's discovery buffer is now bounded and no longer costs diff --git a/src/bin/fipstop/ui/routing.rs b/src/bin/fipstop/ui/routing.rs index d79f64a3..19014763 100644 --- a/src/bin/fipstop/ui/routing.rs +++ b/src/bin/fipstop/ui/routing.rs @@ -192,6 +192,8 @@ fn draw_routing_stats( ("Bloom Miss", disc("req_bloom_miss")), ("Backoff Suppressed", disc("req_backoff_suppressed")), ("Fwd Rate Limited", disc("req_forward_rate_limited")), + ("Sign Rate Limited", disc("req_sign_rate_limited")), + ("Dedup Evicted", disc("req_dedup_evicted")), ("TTL Exhausted", disc("req_ttl_exhausted")), ("Decode Error", disc("req_decode_error")), ], diff --git a/src/control/snapshots/show_routing.json b/src/control/snapshots/show_routing.json index 11fcd3ef..95cbb229 100644 --- a/src/control/snapshots/show_routing.json +++ b/src/control/snapshots/show_routing.json @@ -12,6 +12,7 @@ "req_bloom_miss": 0, "req_decode_error": 0, "req_dedup_cache_full": 0, + "req_dedup_evicted": 0, "req_deduplicated": 0, "req_duplicate": 0, "req_fallback_forwarded": 0, @@ -20,6 +21,7 @@ "req_initiated": 0, "req_no_tree_peer": 0, "req_received": 0, + "req_sign_rate_limited": 0, "req_target_is_us": 0, "req_ttl_exhausted": 0, "resp_accepted": 0, diff --git a/src/node/discovery_rate_limit.rs b/src/node/discovery_rate_limit.rs index ba1106d8..0ac379fe 100644 --- a/src/node/discovery_rate_limit.rs +++ b/src/node/discovery_rate_limit.rs @@ -12,6 +12,11 @@ //! - **`DiscoveryForwardRateLimiter`** (transit-side): Per-target minimum //! interval for forwarded requests. Defense-in-depth against misbehaving //! nodes generating fresh request_ids at high rate. +//! +//! - **`LookupSignRateLimiter`** (target-side): Per-link-peer token bucket +//! on answering lookups for ourselves. Every such answer costs a fresh +//! Schnorr signature, because the proof is bound to the requester's +//! `request_id` and so cannot be cached or reused. use crate::NodeAddr; use std::collections::HashMap; @@ -222,6 +227,113 @@ impl Default for DiscoveryForwardRateLimiter { } } +// ============================================================================ +// Target-side: Lookup Signing Budget +// ============================================================================ + +/// Signatures one link peer may buy in a burst before the refill paces it. +/// +/// Sized for the case that actually produces a burst: a topology change +/// flushes correspondents' coordinate caches and they all look this node up +/// at once, through whichever few link peers lead here, each retrying on the +/// `node.discovery.attempt_timeouts_secs` ladder. Lowering this makes a +/// genuinely popular node intermittently unresolvable, which is the same +/// symptom as the flood it defends against; raising it raises the worst-case +/// signing burst one neighbour can force. +const DEFAULT_SIGN_BURST: f64 = 256.0; + +/// Sustained signatures per second per link peer. +/// +/// At the default eight or so link peers this caps the node near 256 +/// signatures per second in the sustained case. The real cost of one +/// `Identity::sign` on this codebase has not been measured, so this number +/// is a bound rather than a tuned value; it is the one line to change if a +/// measurement says otherwise. +const DEFAULT_SIGN_RATE: f64 = 32.0; + +/// Maximum age of an idle bucket before cleanup. +const SIGN_MAX_AGE: Duration = Duration::from_secs(300); + +/// Token bucket per link peer for lookups this node answers about itself. +/// +/// A min-interval limiter is the wrong shape here: a popular node receives +/// legitimate bursts of lookups for itself through the few link peers that +/// lead to it, and a min interval refuses all but the first of each burst. +/// A bucket absorbs the burst and paces the sustained rate. +pub struct LookupSignRateLimiter { + buckets: HashMap, + burst: f64, + rate: f64, +} + +struct SignBucket { + /// Tokens remaining, at most `burst`. + tokens: f64, + /// When `tokens` was last refilled. + updated: Instant, +} + +impl LookupSignRateLimiter { + /// Create with default burst and refill rate. + pub fn new() -> Self { + Self::with_params(DEFAULT_SIGN_BURST, DEFAULT_SIGN_RATE) + } + + /// Create with a custom burst and refill rate. + pub fn with_params(burst: f64, rate: f64) -> Self { + Self { + buckets: HashMap::new(), + burst, + rate, + } + } + + /// Spend one token for `from`, or report that its budget is exhausted. + /// + /// Returns true when the signature may be produced. A zero burst is + /// read as "unlimited" rather than "refuse everything", so a + /// misconfiguration cannot make this node unresolvable. + pub fn should_sign(&mut self, from: &NodeAddr) -> bool { + if self.burst <= 0.0 { + return true; + } + let now = Instant::now(); + let burst = self.burst; + let rate = self.rate; + let bucket = self.buckets.entry(*from).or_insert(SignBucket { + tokens: burst, + updated: now, + }); + let elapsed = now.duration_since(bucket.updated).as_secs_f64(); + bucket.tokens = (bucket.tokens + elapsed * rate).min(burst); + bucket.updated = now; + if bucket.tokens < 1.0 { + return false; + } + bucket.tokens -= 1.0; + self.cleanup(now); + true + } + + /// Drop buckets untouched for longer than [`SIGN_MAX_AGE`]; a full + /// bucket carries no state worth keeping. + fn cleanup(&mut self, now: Instant) { + self.buckets + .retain(|_, b| now.duration_since(b.updated) < SIGN_MAX_AGE); + } + + #[cfg(test)] + pub fn len(&self) -> usize { + self.buckets.len() + } +} + +impl Default for LookupSignRateLimiter { + fn default() -> Self { + Self::new() + } +} + // ============================================================================ // Tests // ============================================================================ @@ -373,4 +485,30 @@ mod tests { limiter.cleanup(Instant::now()); assert_eq!(limiter.len(), 1); } + + #[test] + fn test_sign_budget_is_spent_per_peer_and_does_not_touch_another_peer() { + let mut limiter = LookupSignRateLimiter::with_params(4.0, 0.0); + for _ in 0..4 { + assert!(limiter.should_sign(&addr(1))); + } + assert!( + !limiter.should_sign(&addr(1)), + "the burst is the whole budget when nothing refills it" + ); + assert!( + limiter.should_sign(&addr(2)), + "one peer spending its budget must not spend another's" + ); + assert_eq!(limiter.len(), 2); + } + + #[test] + fn test_sign_budget_of_zero_burst_is_read_as_unlimited() { + let mut limiter = LookupSignRateLimiter::with_params(0.0, 0.0); + for _ in 0..1000 { + assert!(limiter.should_sign(&addr(1))); + } + assert_eq!(limiter.len(), 0, "unlimited keeps no per-peer state"); + } } diff --git a/src/node/handlers/discovery.rs b/src/node/handlers/discovery.rs index 32aac19e..7d055550 100644 --- a/src/node/handlers/discovery.rs +++ b/src/node/handlers/discovery.rs @@ -12,7 +12,19 @@ use crate::transport::{TransportAddr, TransportId}; use crate::{NodeAddr, PeerIdentity}; use tracing::{debug, info, trace, warn}; -const MAX_RECENT_DISCOVERY_REQUESTS: usize = 4096; +/// Cap on the discovery request dedup cache, which is also the reverse-path +/// table for responses in flight. +pub(in crate::node) const MAX_RECENT_DISCOVERY_REQUESTS: usize = 4096; + +/// Floor under one link peer's share of the dedup cache. +/// +/// A peer's share is the cache divided by the current link-peer count, and +/// this is what stops that share collapsing to nothing on a node with very +/// many links. It is a cap and not a reservation: shares can sum past the +/// cache size, in which case the peer holding the most entries pays for the +/// next admission. Raising it lets one busy neighbour hold more of the +/// cache; lowering it clips a genuine transit burst. +pub(in crate::node) const MIN_RECENT_PER_PEER: usize = 64; impl Node { /// Handle an incoming LookupRequest from a peer. @@ -56,26 +68,44 @@ impl Node { return; } - if self.recent_requests.len() >= MAX_RECENT_DISCOVERY_REQUESTS { - self.metrics() - .discovery - .record_reject(DiscoveryReject::ReqDedupCacheFull); - debug!( - request_id = request.request_id, - from = %self.peer_display_name(from), - recent_requests = self.recent_requests.len(), - max_recent_requests = MAX_RECENT_DISCOVERY_REQUESTS, - "Discovery request dedup cache full, dropping LookupRequest" - ); - return; - } + // A full cache evicts rather than refuses. Refusing meant one peer + // could fill the cache with fresh request_ids and stop this node + // answering lookups for itself and forwarding anyone else's, which + // is a denial of the service the cache exists to protect. The + // eviction is charged to the peer that filled the cache: over its + // own share it pays for itself, and at global capacity the peer + // holding the most entries pays, so extra identities buy a flooder + // proportionally less and a light peer's reverse path survives. + self.make_room_for_request(from); // Record for reverse-path forwarding and dedup self.recent_requests .insert(request.request_id, RecentRequest::new(*from, now_ms)); + self.recent_by_peer + .entry(*from) + .or_default() + .push_back(request.request_id); // Are we the target? if request.target == *self.node_addr() { + // Answering costs a fresh Schnorr signature every time: the + // proof is bound to the requester's request_id, so it cannot be + // cached or served twice. Meter that per link peer, or a + // neighbour generating request_ids sets this node's signing + // rate. The dedup entry above stays regardless, so a refused + // request still occupies its id and a retry, which carries a + // fresh id, is unaffected. + if !self.discovery_sign_limiter.should_sign(from) { + self.metrics() + .discovery + .record_reject(DiscoveryReject::ReqSignRateLimited); + debug!( + request_id = request.request_id, + from = %self.peer_display_name(from), + "Lookup signing budget spent for this peer, not answering" + ); + return; + } self.metrics().discovery.req_target_is_us.inc(); debug!( request_id = request.request_id, @@ -715,10 +745,61 @@ impl Node { } /// Remove expired entries from the recent_requests cache. - fn purge_expired_requests(&mut self, current_time_ms: u64) { + pub(in crate::node) fn purge_expired_requests(&mut self, current_time_ms: u64) { let expiry_ms = self.config().node.discovery.recent_expiry_secs * 1000; - self.recent_requests - .retain(|_, entry| !entry.is_expired(current_time_ms, expiry_ms)); + let recent = &mut self.recent_requests; + recent.retain(|_, entry| !entry.is_expired(current_time_ms, expiry_ms)); + self.recent_by_peer.retain(|_, ids| { + ids.retain(|id| recent.contains_key(id)); + !ids.is_empty() + }); + } + + /// Evict from the dedup cache if admitting one more request would put + /// this peer over its share, or the cache over its capacity. + /// + /// The share is the cache divided by the current link-peer count, with + /// [`MIN_RECENT_PER_PEER`] as a floor, so it tracks the peer count + /// instead of being pinned to a number that a many-peer node outgrows. + fn make_room_for_request(&mut self, from: &NodeAddr) { + let share = + (MAX_RECENT_DISCOVERY_REQUESTS / self.peers.len().max(1)).max(MIN_RECENT_PER_PEER); + + let over_share = self + .recent_by_peer + .get(from) + .is_some_and(|ids| ids.len() >= share); + let victim = if over_share { + Some(*from) + } else if self.recent_requests.len() >= MAX_RECENT_DISCOVERY_REQUESTS { + // Never take from a peer under its share: charge the fattest. + self.recent_by_peer + .iter() + .max_by_key(|(_, ids)| ids.len()) + .map(|(peer, _)| *peer) + } else { + return; + }; + + let Some(victim) = victim else { return }; + let Some(ids) = self.recent_by_peer.get_mut(&victim) else { + return; + }; + let Some(evicted) = ids.pop_front() else { + return; + }; + if ids.is_empty() { + self.recent_by_peer.remove(&victim); + } + self.recent_requests.remove(&evicted); + self.metrics().discovery.req_dedup_evicted.inc(); + debug!( + request_id = evicted, + evicted_from = %self.peer_display_name(&victim), + admitting = %self.peer_display_name(from), + share = share, + "Discovery dedup cache full, evicting the oldest entry to make room" + ); } /// Min-fold our outgoing-link MTU into a LookupResponse's `path_mtu`. diff --git a/src/node/metrics.rs b/src/node/metrics.rs index 73f8c1c5..c32dfcef 100644 --- a/src/node/metrics.rs +++ b/src/node/metrics.rs @@ -266,6 +266,8 @@ pub struct DiscoveryMetrics { pub req_decode_error: Counter, pub req_duplicate: Counter, pub req_dedup_cache_full: Counter, + pub req_dedup_evicted: Counter, + pub req_sign_rate_limited: Counter, pub req_target_is_us: Counter, pub req_forwarded: Counter, pub req_ttl_exhausted: Counter, @@ -296,6 +298,7 @@ impl DiscoveryMetrics { DiscoveryReject::ReqDecodeError => self.req_decode_error.inc(), DiscoveryReject::ReqDuplicate => self.req_duplicate.inc(), DiscoveryReject::ReqDedupCacheFull => self.req_dedup_cache_full.inc(), + DiscoveryReject::ReqSignRateLimited => self.req_sign_rate_limited.inc(), DiscoveryReject::ReqTtlExhausted => self.req_ttl_exhausted.inc(), DiscoveryReject::RespDecodeError => self.resp_decode_error.inc(), DiscoveryReject::RespIdentityMiss => self.resp_identity_miss.inc(), @@ -312,6 +315,8 @@ impl DiscoveryMetrics { req_decode_error: self.req_decode_error.get(), req_duplicate: self.req_duplicate.get(), req_dedup_cache_full: self.req_dedup_cache_full.get(), + req_dedup_evicted: self.req_dedup_evicted.get(), + req_sign_rate_limited: self.req_sign_rate_limited.get(), req_target_is_us: self.req_target_is_us.get(), req_forwarded: self.req_forwarded.get(), req_ttl_exhausted: self.req_ttl_exhausted.get(), diff --git a/src/node/mod.rs b/src/node/mod.rs index 3a52a44f..35b4a600 100644 --- a/src/node/mod.rs +++ b/src/node/mod.rs @@ -32,7 +32,9 @@ mod tests; mod tree; pub(crate) mod wire; -use self::discovery_rate_limit::{DiscoveryBackoff, DiscoveryForwardRateLimiter}; +use self::discovery_rate_limit::{ + DiscoveryBackoff, DiscoveryForwardRateLimiter, LookupSignRateLimiter, +}; use self::peer_error_budget::PeerErrorBudget; use self::rate_limit::{HandshakeRateLimiter, SessionSetupRateLimiter}; use self::reloadable::Reloadable; @@ -69,7 +71,7 @@ use crate::upper::tun::{TunError, TunOutboundRx, TunState, TunTx}; use crate::utils::index::IndexAllocator; use crate::{Config, ConfigError, Identity, IdentityError, NodeAddr, PeerIdentity}; use rand::Rng; -use std::collections::{HashMap, HashSet, VecDeque}; +use std::collections::{BTreeMap, HashMap, HashSet, VecDeque}; use std::fmt; use std::sync::Arc; use std::thread::JoinHandle; @@ -367,6 +369,13 @@ pub struct Node { /// Recent discovery requests (dedup + reverse-path forwarding). /// Maps request_id → RecentRequest. recent_requests: HashMap, + /// Arrival-order index over `recent_requests`, partitioned by the link + /// peer each request arrived from. The cache is full-then-evict rather + /// than full-then-refuse, and this is what lets an eviction be charged + /// to the peer that filled the cache instead of to whoever happens to + /// be oldest. Timestamps are nondecreasing across inserts, so each + /// deque is in arrival order and the front is the oldest. + recent_by_peer: BTreeMap>, /// Per-destination path MTU lookup, keyed by FipsAddress (mirrors /// `coord_cache.entries[*].path_mtu`). Sync read-only access from /// the TUN reader/writer threads at TCP MSS clamp time so the @@ -516,6 +525,8 @@ pub struct Node { discovery_backoff: DiscoveryBackoff, /// Rate limiter for forwarded discovery requests (transit-side). discovery_forward_limiter: DiscoveryForwardRateLimiter, + /// Signing budget for lookups we answer about ourselves (target-side). + discovery_sign_limiter: LookupSignRateLimiter, // === Pending Transport Connects === /// Links waiting for transport-level connection establishment before @@ -761,6 +772,7 @@ impl Node { bloom_state, coord_cache, recent_requests: HashMap::new(), + recent_by_peer: BTreeMap::new(), transports: HashMap::new(), transport_drops: HashMap::new(), links: HashMap::new(), @@ -813,6 +825,7 @@ impl Node { discovery_forward_limiter: DiscoveryForwardRateLimiter::with_interval( std::time::Duration::from_secs(forward_min_interval_secs), ), + discovery_sign_limiter: LookupSignRateLimiter::new(), pending_connects: Vec::new(), retry_pending: HashMap::new(), nostr_discovery: None, @@ -922,6 +935,7 @@ impl Node { bloom_state, coord_cache, recent_requests: HashMap::new(), + recent_by_peer: BTreeMap::new(), transports: HashMap::new(), transport_drops: HashMap::new(), links: HashMap::new(), @@ -972,6 +986,7 @@ impl Node { ), discovery_backoff: DiscoveryBackoff::new(), discovery_forward_limiter: DiscoveryForwardRateLimiter::new(), + discovery_sign_limiter: LookupSignRateLimiter::new(), pending_connects: Vec::new(), retry_pending: HashMap::new(), nostr_discovery: None, @@ -2547,6 +2562,13 @@ impl Node { // === End-to-End Sessions === /// Get a session by remote NodeAddr. + /// Set the per-link-peer lookup signing budget (for tests). + #[cfg(test)] + pub(crate) fn set_discovery_sign_budget(&mut self, burst: f64, rate: f64) { + self.discovery_sign_limiter = + discovery_rate_limit::LookupSignRateLimiter::with_params(burst, rate); + } + /// Disable the discovery forward rate limiter (for tests). #[cfg(test)] pub(crate) fn disable_discovery_forward_rate_limit(&mut self) { diff --git a/src/node/reject.rs b/src/node/reject.rs index 0b3bccbc..40499f32 100644 --- a/src/node/reject.rs +++ b/src/node/reject.rs @@ -120,7 +120,19 @@ pub enum DiscoveryReject { /// Request dedup cache (`recent_requests`) is at capacity, so the /// `LookupRequest` is dropped without being forwarded. Tracked via /// [`DiscoveryStats::req_dedup_cache_full`](crate::node::stats::DiscoveryStats). + /// + /// Frozen at zero: a full cache now evicts its oldest entry and admits + /// the request, counted as + /// [`DiscoveryStats::req_dedup_evicted`](crate::node::stats::DiscoveryStats). + /// The variant and its counter stay so an operator reading a dashboard + /// across versions does not find the series missing. ReqDedupCacheFull, + /// This node is the lookup target, but the link peer the request + /// arrived from has spent its signing budget. Answering costs a fresh + /// Schnorr signature per request, so the budget bounds what one + /// neighbour can make this node sign. Tracked via + /// [`DiscoveryStats::req_sign_rate_limited`](crate::node::stats::DiscoveryStats). + ReqSignRateLimited, /// Request arrived with TTL=0 — no more forwarding hops allowed. /// Tracked via /// [`DiscoveryStats::req_ttl_exhausted`](crate::node::stats::DiscoveryStats). @@ -376,6 +388,7 @@ mod tests { DiscoveryReject::ReqDuplicate, DiscoveryReject::ReqDedupCacheFull, DiscoveryReject::ReqTtlExhausted, + DiscoveryReject::ReqSignRateLimited, DiscoveryReject::RespDecodeError, DiscoveryReject::RespIdentityMiss, DiscoveryReject::RespProofFailed, diff --git a/src/node/stats.rs b/src/node/stats.rs index 9664870c..957371df 100644 --- a/src/node/stats.rs +++ b/src/node/stats.rs @@ -319,6 +319,8 @@ pub struct DiscoveryStatsSnapshot { pub req_decode_error: u64, pub req_duplicate: u64, pub req_dedup_cache_full: u64, + pub req_dedup_evicted: u64, + pub req_sign_rate_limited: u64, pub req_target_is_us: u64, pub req_forwarded: u64, pub req_ttl_exhausted: u64, diff --git a/src/node/tests/discovery.rs b/src/node/tests/discovery.rs index 126bdb08..13624d44 100644 --- a/src/node/tests/discovery.rs +++ b/src/node/tests/discovery.rs @@ -614,6 +614,206 @@ async fn test_recent_request_expiry() { assert!(node.recent_requests.contains_key(&789)); } +// ============================================================================ +// Unit Tests — dedup cache capacity policy +// ============================================================================ + +use crate::node::handlers::discovery::{MAX_RECENT_DISCOVERY_REQUESTS, MIN_RECENT_PER_PEER}; + +/// Encode a LookupRequest for `target` carrying `request_id`, ready for +/// `handle_lookup_request` (which is handed the payload without the +/// msg_type byte). +fn lookup_request_payload(request_id: u64, target: &crate::NodeAddr) -> Vec { + let origin = make_node_addr(0xCC); + let coords = TreeCoordinate::from_addrs(vec![origin, make_node_addr(0)]).unwrap(); + LookupRequest::new(request_id, *target, origin, coords, 5, 0).encode()[1..].to_vec() +} + +/// Deliver `count` distinct requests from `from`, ids starting at `first_id`. +async fn flood_requests(node: &mut Node, from: &crate::NodeAddr, first_id: u64, count: u64) { + let target = make_node_addr(0xBB); + for i in 0..count { + let payload = lookup_request_payload(first_id + i, &target); + node.handle_lookup_request(from, &payload).await; + } +} + +/// Register `count` peers so the per-peer share of the dedup cache is the +/// floor rather than the whole cache, and return their addresses. +fn register_peers(node: &mut Node, count: usize) -> Vec { + (0..count) + .map(|i| { + let identity = Identity::generate(); + let addr = *identity.node_addr(); + let peer_identity = crate::PeerIdentity::from_pubkey(identity.pubkey()); + node.peers.insert( + addr, + ActivePeer::new(peer_identity, LinkId::new(i as u64), 0), + ); + addr + }) + .collect() +} + +#[tokio::test] +async fn test_a_full_dedup_cache_admits_the_new_request_by_evicting_the_oldest() { + // A full cache used to drop the arriving request, which let one peer + // spend 4096 fresh request_ids and stop the node forwarding anyone + // else's lookups until the entries aged out. + let mut node = make_node(); + let from = make_node_addr(0xAA); + + flood_requests(&mut node, &from, 1, MAX_RECENT_DISCOVERY_REQUESTS as u64).await; + assert_eq!( + node.recent_requests.len(), + MAX_RECENT_DISCOVERY_REQUESTS, + "precondition: the cache is full, or the rest observes nothing" + ); + + let payload = lookup_request_payload(u64::MAX, &make_node_addr(0xBB)); + node.handle_lookup_request(&from, &payload).await; + + assert!( + node.recent_requests.contains_key(&u64::MAX), + "the arriving request must be recorded, so its response can be routed back" + ); + assert!( + !node.recent_requests.contains_key(&1), + "room is made by dropping the oldest entry" + ); + assert_eq!( + node.recent_requests.len(), + MAX_RECENT_DISCOVERY_REQUESTS, + "the cache stays at its bound" + ); + assert_eq!(node.metrics().discovery.req_dedup_evicted.get(), 1); + assert_eq!( + node.metrics().discovery.req_dedup_cache_full.get(), + 0, + "the cache-full drop is gone, and its counter stays frozen at zero" + ); +} + +#[tokio::test] +async fn test_a_flooding_peer_evicts_only_its_own_dedup_entries() { + // The whole point of partitioning the cache by link peer: one peer + // filling its share must not cost another peer the reverse path its own + // lookup depends on. + let mut node = make_node(); + let peers = register_peers(&mut node, 64); + let flooder = peers[0]; + let light = peers[1]; + + let payload = lookup_request_payload(7, &make_node_addr(0xBB)); + node.handle_lookup_request(&light, &payload).await; + + // One over the share, so the flooder pays for its own admission. + flood_requests(&mut node, &flooder, 1000, MIN_RECENT_PER_PEER as u64 + 1).await; + + assert!( + node.recent_requests.contains_key(&7), + "a light peer's reverse-path entry must survive a neighbour's flood" + ); + assert!( + !node.recent_requests.contains_key(&1000), + "the flooder's own oldest entry is what pays for its newest" + ); + assert!( + node.recent_requests + .contains_key(&(1000 + MIN_RECENT_PER_PEER as u64)), + "and its newest is admitted rather than dropped" + ); +} + +#[tokio::test] +async fn test_a_node_whose_dedup_cache_is_flooded_still_answers_a_lookup_for_itself() { + // The availability claim. Filling the cache used to make the node + // unresolvable, because the cache-full drop sat ahead of the check for + // whether the request names us. + let mut node = make_node(); + let flooder = make_node_addr(0xAA); + let other = make_node_addr(0xAB); + + flood_requests(&mut node, &flooder, 1, MAX_RECENT_DISCOVERY_REQUESTS as u64).await; + + let my_addr = *node.node_addr(); + let payload = lookup_request_payload(u64::MAX, &my_addr); + node.handle_lookup_request(&other, &payload).await; + + assert_eq!( + node.metrics().discovery.req_target_is_us.get(), + 1, + "a flooded cache must not stop the node answering lookups for itself" + ); +} + +#[tokio::test] +async fn test_the_dedup_index_stays_level_with_the_cache_across_insert_duplicate_and_purge() { + // Two containers where there was one, so the desync is the maintenance + // risk. Everything the eviction policy decides reads the index, so an + // index that has drifted evicts the wrong entry or none at all. + let mut node = make_node(); + let peers = register_peers(&mut node, 64); + + flood_requests(&mut node, &peers[0], 1, 70).await; + flood_requests(&mut node, &peers[1], 500, 5).await; + // Duplicates, which must not be indexed twice. + flood_requests(&mut node, &peers[1], 500, 5).await; + + let indexed: usize = node.recent_by_peer.values().map(|ids| ids.len()).sum(); + assert_eq!( + indexed, + node.recent_requests.len(), + "every cached request is indexed exactly once" + ); + + // Age everything out and purge through the ordinary request path. + let expiry_ms = node.config().node.discovery.recent_expiry_secs * 1000; + let future = Node::now_ms() + expiry_ms + 1; + node.purge_expired_requests(future); + + assert!( + node.recent_requests.is_empty(), + "precondition: the purge removed everything" + ); + assert!( + node.recent_by_peer.is_empty(), + "the index must not keep entries the cache no longer holds" + ); +} + +#[tokio::test] +async fn test_answering_lookups_for_ourselves_stops_at_the_per_peer_signing_budget() { + // Each answer costs a fresh Schnorr signature, because the proof is + // bound to the requester's request_id and cannot be reused. Without a + // budget, one neighbour sets this node's signing rate. + let mut node = make_node(); + node.set_discovery_sign_budget(3.0, 0.0); + let from = make_node_addr(0xAA); + let other = make_node_addr(0xAB); + let my_addr = *node.node_addr(); + + for id in 0..4u64 { + let payload = lookup_request_payload(id, &my_addr); + node.handle_lookup_request(&from, &payload).await; + } + + assert_eq!( + node.metrics().discovery.req_target_is_us.get(), + 3, + "the burst is answered and the fourth request is not" + ); + assert_eq!(node.metrics().discovery.req_sign_rate_limited.get(), 1); + + let payload = lookup_request_payload(100, &my_addr); + node.handle_lookup_request(&other, &payload).await; + assert_eq!( + node.metrics().discovery.req_target_is_us.get(), + 4, + "one peer spending its budget must not make the node unresolvable through another" + ); +} + // ============================================================================ // Integration Tests — Multi-Node Forwarding // ============================================================================ From f5976443dd3a1df29deaede31a159ad38a494cc7 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:52:16 +0100 Subject: [PATCH 18/23] Believe a reactive MtuExceeded only when our own traffic corroborates it MtuExceeded arrives unauthenticated. The admission gate narrows which destination a signal may name but cannot say who named it, so a value at the 256-byte floor was a legal value from anyone able to route a datagram here. One such datagram drove a bound session's path MTU to the floor and pinned the address-keyed entry the SYN-time MSS clamp reads; recovery costs three consecutive higher notifications across two notification intervals. Raising the floor would only have set the outcome of a forgery rather than preventing it, and would have cost reactive feedback to any hop deliberately configured below the new value. Carry the largest frame this node has put on the wire toward each session since its last accepted path-MTU decrease, and refuse a report that does not name something smaller. An honest report exists only because a frame we sent did not fit some hop, so honest discovery satisfies this by construction, while a forgery must wait for us to emit something larger than the value it wants to claim. Every accepted claim is therefore bounded from below by our own traffic. The evidence is cleared on each accepted decrease and on release so one early large send cannot vouch for a whole session. Move the floor check ahead of the apply, where it now governs the session's own path MTU as well as the lookup table rather than only the latter, and give the reactive carrier its own floor constant held equal to the actionable one, so changing it later is a one-line edit. Separately, rate limit the path-MTU release an unauthenticated PathBroken drives, per destination and on its own limiter instance rather than the one the coordinate warmup send uses: a budget another signal can spend is not a bound. Deferring a release keeps the tighter value, which is the safe direction. The path-MTU handler tests that installed a session without sending anything now state the corroborating send explicitly. The end-to-end multi-hop discovery test is unchanged and still passes, which is what shows honest discovery is unaffected. --- CHANGELOG.md | 34 ++++ src/bin/fipstop/ui/routing.rs | 4 + src/bin/fipstop/ui/snapshots.rs | 2 +- src/control/snapshots/show_routing.json | 1 + src/node/handlers/session.rs | 113 +++++++++-- src/node/metrics.rs | 6 + src/node/mod.rs | 16 ++ src/node/session.rs | 31 +++ src/node/stats.rs | 1 + src/node/tests/session.rs | 241 +++++++++++++++++++++++- src/upper/icmp.rs | 16 ++ 11 files changed, 447 insertions(+), 18 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ea316156..a9298f1d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -977,6 +977,40 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 is an emission change only: an unmodified peer parses the frame exactly as before. Which of the two signals is emitted still discloses whether the entry exists. +- A reactive `MtuExceeded` is now believed only when this node has actually + sent a frame larger than the bottleneck it reports. The signal is + unauthenticated: the admission gate narrows which destination may be named + but cannot say who named it, so a value at the floor was a legal value from + anyone, and one datagram drove a bound session's path MTU to 256 and pinned + the address-keyed entry the SYN-time MSS clamp reads, with recovery costing + three consecutive higher notifications across two notification intervals. + Each session now carries the largest frame this node has put on the wire + toward it since the last accepted decrease, and a report is refused unless it + names something smaller. Honest path-MTU discovery satisfies that by + construction, because the report exists only because a frame we sent did not + fit; a forgery has to wait for us to emit something bigger than the value it + wants to claim, which bounds every accepted claim from below by our own + traffic. The evidence is cleared on each accepted decrease and on release, so + one large send early in a session cannot vouch for the rest of it. + + The guard sits ahead of both effects rather than between them, which is also + where the existing floor check moved to: the floor previously ran after the + session's own path MTU had already been changed and so governed only the + lookup table. The reactive carrier now names its own floor constant, held + equal to the actionable floor so no hop legitimately configured with a small + transport MTU loses its feedback; corroboration, not the floor's value, is + what stops a legal-but-forged claim. A separate counter, rendered on the + fipstop Routing tab, distinguishes an uncorroborated refusal from a + below-floor one. + +- The path-MTU release a `PathBroken` drives is now rate limited per + destination on a budget of its own. That signal is unauthenticated too, and + the release discards a bottleneck this node learned by having a packet + dropped, so repeating the claim discarded a genuine value as fast as it could + be relearned. The limiter is a separate instance rather than the one the + coordinate warmup send already uses: a budget another signal can spend is not + a bound. Deferring a release is the safe direction, since the value kept is + the tighter one. - The influence a remote party has over path MTU is now bounded, and the per-destination path MTU cache has a way back. The `path_mtu` field is an diff --git a/src/bin/fipstop/ui/routing.rs b/src/bin/fipstop/ui/routing.rs index 19014763..3afe933e 100644 --- a/src/bin/fipstop/ui/routing.rs +++ b/src/bin/fipstop/ui/routing.rs @@ -304,6 +304,10 @@ fn draw_routing_stats( ("Emit Over Peer Budget", err("emit_over_peer_budget")), ("Emit Over Dest Interval", err("emit_over_dest_interval")), ("Emit Limiter At Capacity", err("emit_limiter_at_capacity")), + ( + "MTU Exceeded Uncorroborated", + err("mtu_exceeded_uncorroborated"), + ), ], )); right.push(Line::from("")); diff --git a/src/bin/fipstop/ui/snapshots.rs b/src/bin/fipstop/ui/snapshots.rs index a91611b4..5aa93f72 100644 --- a/src/bin/fipstop/ui/snapshots.rs +++ b/src/bin/fipstop/ui/snapshots.rs @@ -1151,7 +1151,7 @@ fn routing_focused_pane_scrolls() { // column is the taller of the two, so scrolling fully to the bottom would // over-scroll the right column past Congestion; this offset lands the // Congestion region inside the short window instead. - app1.scroll_offsets.insert((Tab::Routing, 2), 31); + app1.scroll_offsets.insert((Tab::Routing, 2), 32); let buf1 = testkit::render(100, 20, |frame, area| { super::routing::draw(frame, &app1, area); }); diff --git a/src/control/snapshots/show_routing.json b/src/control/snapshots/show_routing.json index 95cbb229..efbdf1f0 100644 --- a/src/control/snapshots/show_routing.json +++ b/src/control/snapshots/show_routing.json @@ -42,6 +42,7 @@ "lookup_resp_mtu_below_floor": 0, "mtu_exceeded": 0, "mtu_exceeded_below_floor": 0, + "mtu_exceeded_uncorroborated": 0, "path_broken": 0, "path_mtu_notif_below_floor": 0, "unbound_broken": 0, diff --git a/src/node/handlers/session.rs b/src/node/handlers/session.rs index c8e11bcb..3601dd71 100644 --- a/src/node/handlers/session.rs +++ b/src/node/handlers/session.rs @@ -41,6 +41,29 @@ use crate::upper::icmp::FIPS_OVERHEAD; use secp256k1::PublicKey; use tracing::{debug, info, trace, warn}; +/// Minimum interval between path-MTU releases driven by `PathBroken` for one +/// destination. +/// +/// `PathBroken` is unauthenticated, so a release is a remote party's claim +/// that the path a tightened MTU described is gone. Without an interval the +/// claim can be repeated at line rate, discarding a genuinely learned +/// bottleneck as fast as it is relearned. Raising it defers a legitimate +/// release after a second real break, which costs throughput on the new path +/// but never a blackhole, since the deferred value is the tighter one. +pub(in crate::node) const PATH_MTU_RELEASE_MIN_INTERVAL: std::time::Duration = + std::time::Duration::from_millis(1000); + +/// Bytes the link layer adds to an encoded `SessionDatagram` on its way to the +/// wire: the established FMP header, the 4-byte session-relative timestamp and +/// the AEAD tag. Mirrors the buffer `send_encrypted_link_message_with_ce` +/// builds. +const LINK_FRAME_OVERHEAD: usize = ESTABLISHED_HEADER_SIZE + 4 + crate::noise::TAG_SIZE; + +/// Wire size of an encoded `SessionDatagram` of `encoded_len` bytes. +fn link_wire_len(encoded_len: usize) -> usize { + encoded_len + LINK_FRAME_OVERHEAD +} + /// Inputs to `try_send_session_data_pipelined` — the FSP+FMP pipelined /// fast path that hands both AEAD operations to the encrypt worker /// in a single dispatch. @@ -1560,8 +1583,17 @@ impl Node { self.coord_cache.remove(&msg.dest_addr); // The path this destination's stored MTU described is gone, so release - // it rather than carrying it onto whatever path replaces it. - self.path_mtu_lookup_release(&msg.dest_addr); + // it rather than carrying it onto whatever path replaces it. Rate + // limited per destination on its own budget: PathBroken is + // unauthenticated, and an unlimited release discards a genuinely + // learned bottleneck as fast as it is relearned. The budget is not + // shared with any other signal, so nothing else can spend it. + if self.path_mtu_release_limiter.should_send(&msg.dest_addr) { + self.path_mtu_lookup_release(&msg.dest_addr); + } else { + trace!(dest = %msg.dest_addr, + "PathBroken path MTU release rate-limited, keeping the stored value"); + } // Trigger re-discovery to get fresh coordinates, but only if we have // the target's identity cached — otherwise we can't verify the @@ -1628,6 +1660,55 @@ impl Node { "MtuExceeded: transit router reports oversized packet" ); + // Both effects below — the session's own path MTU and the + // FipsAddress-keyed lookup the TUN MSS clamp reads — are refused from + // here, so one return covers both. The guards sit ahead of the apply + // rather than between the two effects, which is what makes the floor + // govern `current_mtu` and not only the lookup table. + + // Refuse a bottleneck too small to describe a usable path; a stored + // value that low drives the SYN-time MSS clamp into single digits or + // zero. The reactive carrier is unauthenticated, so it has its own + // floor constant, currently equal to the actionable one. + if msg.mtu < crate::upper::icmp::MIN_REACTIVE_PATH_MTU { + warn!( + dest = %peer_name, + reporter = %msg.reporter, + bottleneck_mtu = msg.mtu, + floor = crate::upper::icmp::MIN_REACTIVE_PATH_MTU, + "MtuExceeded reports a path MTU below the actionable floor; ignoring" + ); + self.metrics().errors.mtu_exceeded_below_floor.inc(); + return; + } + + // Corroboration. The admission gate narrows which destination may be + // named; it cannot authenticate the reporter, so a legal value is a + // legal value from anyone and the floor alone only sets the outcome of + // a forgery rather than preventing it. An honest report exists only + // because a frame this node emitted did not fit some hop, so require + // that this node has actually sent something larger than the value + // being claimed since the last accepted decrease. Honest path-MTU + // discovery satisfies this by construction; a forgery has to wait for + // us to emit a frame bigger than the value it wants to claim, which + // bounds every accepted claim from below by our own traffic. + let sent_wire_len = self + .sessions + .get(&msg.dest_addr) + .map(|e| e.max_sent_wire_len()) + .unwrap_or(0); + if msg.mtu >= sent_wire_len { + debug!( + dest = %peer_name, + reporter = %msg.reporter, + bottleneck_mtu = msg.mtu, + max_sent_wire_len = sent_wire_len, + "MtuExceeded reports a bottleneck no smaller than anything this node has sent; ignoring" + ); + self.metrics().errors.mtu_exceeded_uncorroborated.inc(); + return; + } + // Apply to PathMtuState: immediate decrease via apply_notification() if let Some(entry) = self.sessions.get_mut(&msg.dest_addr) && let Some(mmp) = entry.mmp_mut() @@ -1646,20 +1727,12 @@ impl Node { } } - // The lookup write below is not gated on a session existing, so an - // unencrypted MtuExceeded from anyone reaches it. Refuse to store a - // bottleneck too small to describe a usable path; a stored value that - // low drives the SYN-time MSS clamp into single digits or zero. - if msg.mtu < crate::upper::icmp::MIN_ACTIONABLE_PATH_MTU { - warn!( - dest = %peer_name, - reporter = %msg.reporter, - bottleneck_mtu = msg.mtu, - floor = crate::upper::icmp::MIN_ACTIONABLE_PATH_MTU, - "MtuExceeded reports a path MTU below the actionable floor; ignoring" - ); - self.metrics().errors.mtu_exceeded_below_floor.inc(); - return; + // Spent: the evidence vouched for this decrease and does not vouch for + // the next one. An initiating session has no `mmp` and so reaches this + // with the apply above skipped; the reset belongs to the acceptance, + // not to the apply. + if let Some(entry) = self.sessions.get_mut(&msg.dest_addr) { + entry.clear_sent_wire_len(); } // Mirror the bottleneck into the FipsAddress-keyed lookup used by @@ -2192,6 +2265,7 @@ impl Node { if let Some(entry) = self.sessions.get_mut(dest_addr) { entry.record_sent(send.payload.len()); + entry.record_sent_wire_len(wire_capacity); if let Some(mmp) = entry.mmp_mut() { mmp.sender.record_sent( fsp_counter, @@ -2467,6 +2541,13 @@ impl Node { self.send_encrypted_link_message(&next_hop_addr, &encoded) .await?; self.metrics().forwarding.record_originated(encoded.len()); + + // Evidence for the reactive path-MTU carrier. A transit hop + // re-encapsulates what it forwards, so the frame that overflows a + // downstream link is the size this frame is here. + if let Some(entry) = self.sessions.get_mut(&datagram.dest_addr) { + entry.record_sent_wire_len(link_wire_len(encoded.len())); + } Ok(()) } diff --git a/src/node/metrics.rs b/src/node/metrics.rs index c32dfcef..896a3621 100644 --- a/src/node/metrics.rs +++ b/src/node/metrics.rs @@ -515,6 +515,11 @@ pub struct ErrorMetrics { /// count means a forwarder on the reverse path is mangling the unsigned /// annotation. pub lookup_resp_mtu_below_floor: Counter, + /// `MtuExceeded` signals ignored because this node has not sent a frame + /// larger than the bottleneck they report since the last accepted + /// decrease. An honest report cannot arise without such a frame, so a + /// rising count is a forged or stale reactive signal. + pub mtu_exceeded_uncorroborated: Counter, pub unbound: UnboundSignals, /// Routing errors this node declined to emit because the authenticated /// link peer that induced them had spent its budget. A rising count is @@ -548,6 +553,7 @@ impl ErrorMetrics { emit_over_peer_budget: self.emit_over_peer_budget.get(), emit_over_dest_interval: self.emit_over_dest_interval.get(), emit_limiter_at_capacity: self.emit_limiter_at_capacity.get(), + mtu_exceeded_uncorroborated: self.mtu_exceeded_uncorroborated.get(), } } } diff --git a/src/node/mod.rs b/src/node/mod.rs index 35b4a600..c18c610e 100644 --- a/src/node/mod.rs +++ b/src/node/mod.rs @@ -521,6 +521,10 @@ pub struct Node { peer_error_budget: PeerErrorBudget, /// Rate limiter for source-side CoordsRequired/PathBroken responses. coords_response_rate_limiter: RoutingErrorRateLimiter, + /// Rate limiter for PathBroken-driven path-MTU releases, per destination. + /// Deliberately its own instance: a budget another signal can spend is + /// not a bound on this one. + path_mtu_release_limiter: RoutingErrorRateLimiter, /// Backoff for failed discovery lookups (originator-side). discovery_backoff: DiscoveryBackoff, /// Rate limiter for forwarded discovery requests (transit-side). @@ -821,6 +825,9 @@ impl Node { coords_response_rate_limiter: RoutingErrorRateLimiter::with_interval( std::time::Duration::from_millis(coords_response_interval_ms), ), + path_mtu_release_limiter: RoutingErrorRateLimiter::with_interval( + crate::node::handlers::session::PATH_MTU_RELEASE_MIN_INTERVAL, + ), discovery_backoff: DiscoveryBackoff::with_params(backoff_base_secs, backoff_max_secs), discovery_forward_limiter: DiscoveryForwardRateLimiter::with_interval( std::time::Duration::from_secs(forward_min_interval_secs), @@ -984,6 +991,9 @@ impl Node { coords_response_rate_limiter: RoutingErrorRateLimiter::with_interval( std::time::Duration::from_millis(coords_response_interval_ms), ), + path_mtu_release_limiter: RoutingErrorRateLimiter::with_interval( + crate::node::handlers::session::PATH_MTU_RELEASE_MIN_INTERVAL, + ), discovery_backoff: DiscoveryBackoff::new(), discovery_forward_limiter: DiscoveryForwardRateLimiter::new(), discovery_sign_limiter: LookupSignRateLimiter::new(), @@ -2662,6 +2672,12 @@ impl Node { /// `FipsAddress`-keyed map the TCP MSS clamp reads, and the session's own /// source-side path MTU estimate. fn path_mtu_lookup_release(&mut self, addr: &NodeAddr) { + // The evidence that corroborates a reactive MtuExceeded described the + // path being released, so it does not vouch for whatever replaces it. + if let Some(entry) = self.sessions.get_mut(addr) { + entry.clear_sent_wire_len(); + } + // The session's own source-side estimate described the same dead path, // and the increase ladder is the only thing that would ever raise it // again. Reset it here so the two halves of "this path is gone" stay diff --git a/src/node/session.rs b/src/node/session.rs index 4c036e59..29f9ac07 100644 --- a/src/node/session.rs +++ b/src/node/session.rs @@ -102,6 +102,15 @@ pub(crate) struct SessionEntry { /// Whether this node initiated the Noise handshake. /// Used for spin bit role assignment in session-layer MMP. is_initiator: bool, + /// Largest on-the-wire frame this node has sent toward the remote since + /// the last accepted path-MTU decrease or release, in bytes. + /// + /// Corroborates a reactive `MtuExceeded`, which is unauthenticated: an + /// honest report exists only because a frame this node emitted did not + /// fit some hop, so an honest report always names a value below this. + /// Reset on each accepted decrease and on release so one historical + /// large send cannot vouch for a session's whole lifetime. + max_sent_wire_len: u16, /// Session-layer MMP state. Initialized on Established transition. mmp: Option, @@ -205,6 +214,7 @@ impl SessionEntry { session_start_ms: 0, coords_warmup_remaining: 0, is_initiator, + max_sent_wire_len: 0, mmp: None, packets_sent: 0, packets_recv: 0, @@ -308,6 +318,27 @@ impl SessionEntry { self.coords_warmup_remaining = value; } + /// Largest wire frame sent toward the remote since the last accepted + /// path-MTU decrease or release. + pub(crate) fn max_sent_wire_len(&self) -> u16 { + self.max_sent_wire_len + } + + /// Note a frame of `wire_len` bytes sent toward the remote, keeping the + /// largest. Frames beyond `u16::MAX` saturate, which only ever makes the + /// corroboration more permissive and cannot exceed what a path MTU can + /// name. + pub(crate) fn record_sent_wire_len(&mut self, wire_len: usize) { + let wire_len = u16::try_from(wire_len).unwrap_or(u16::MAX); + self.max_sent_wire_len = self.max_sent_wire_len.max(wire_len); + } + + /// Forget what has been sent, so the next reactive report needs fresh + /// evidence of its own. + pub(crate) fn clear_sent_wire_len(&mut self) { + self.max_sent_wire_len = 0; + } + /// Mark the session as started (transition to Established). /// /// Records the current time as the session start for computing diff --git a/src/node/stats.rs b/src/node/stats.rs index 957371df..52d0f399 100644 --- a/src/node/stats.rs +++ b/src/node/stats.rs @@ -427,6 +427,7 @@ pub struct ErrorSignalStatsSnapshot { pub emit_over_peer_budget: u64, pub emit_over_dest_interval: u64, pub emit_limiter_at_capacity: u64, + pub mtu_exceeded_uncorroborated: u64, } #[derive(Clone, Debug, Default, Serialize)] diff --git a/src/node/tests/session.rs b/src/node/tests/session.rs index 44f800c7..2fe5eedd 100644 --- a/src/node/tests/session.rs +++ b/src/node/tests/session.rs @@ -2228,6 +2228,17 @@ fn install_halfopen(node: &mut Node, claimed: NodeAddr) { node.sessions.insert(claimed, entry); } +/// Record that this node put a frame of `wire_len` bytes on the wire toward +/// `dest`, which is what corroborates a reactive `MtuExceeded` reporting a +/// smaller bottleneck. Honest path-MTU discovery produces this by sending; +/// a handler test that installs a session without sending has to state it. +fn note_sent_wire_len(node: &mut Node, dest: &NodeAddr, wire_len: usize) { + node.sessions + .get_mut(dest) + .expect("session must exist to corroborate a report") + .record_sent_wire_len(wire_len); +} + /// Install the entry `initiate_session` creates: an address this node chose /// itself, with the handshake still in flight and MMP not yet initialized. fn install_initiating(node: &mut Node, remote: &Identity) { @@ -2263,6 +2274,7 @@ async fn test_handle_mtu_exceeded_writes_path_mtu_lookup_when_empty() { "lookup should start empty for this destination" ); + note_sent_wire_len(&mut tn.node, &dest, 1400); let inner = build_mtu_exceeded_inner(&dest, &reporter, 1280); tn.node.handle_mtu_exceeded(&reporter, &inner).await; @@ -2289,6 +2301,7 @@ async fn test_handle_mtu_exceeded_tightens_existing_path_mtu_lookup() { // response that didn't reflect the forward-path bottleneck). tn.node.path_mtu_lookup_insert(dest_fips, 1500); + note_sent_wire_len(&mut tn.node, &dest, 1400); let inner = build_mtu_exceeded_inner(&dest, &reporter, 1280); tn.node.handle_mtu_exceeded(&reporter, &inner).await; @@ -2420,8 +2433,9 @@ async fn test_handle_mtu_exceeded_at_the_floor_still_writes_path_mtu_lookup() { let dest = *remote.node_addr(); let reporter = NodeAddr::from_bytes([0xBB; 16]); let dest_fips = crate::FipsAddress::from_node_addr(&dest); - let floor = crate::upper::icmp::MIN_ACTIONABLE_PATH_MTU; + let floor = crate::upper::icmp::MIN_REACTIVE_PATH_MTU; + note_sent_wire_len(&mut tn.node, &dest, 1400); let inner = build_mtu_exceeded_inner(&dest, &reporter, floor); tn.node.handle_mtu_exceeded(&reporter, &inner).await; @@ -2944,6 +2958,7 @@ async fn test_mtu_exceeded_for_a_session_we_initiated_seeds_path_mtu_lookup_befo let reporter = NodeAddr::from_bytes([0xBB; 16]); let dest_fips = crate::FipsAddress::from_node_addr(&dest); + note_sent_wire_len(&mut node, &dest, 1400); let inner = build_mtu_exceeded_inner(&dest, &reporter, 1280); node.handle_mtu_exceeded(&reporter, &inner).await; @@ -2970,6 +2985,7 @@ async fn test_mtu_exceeded_from_a_third_party_forwarder_still_tightens_an_active let reporter = NodeAddr::from_bytes([0xBB; 16]); let dest_fips = crate::FipsAddress::from_node_addr(&dest); + note_sent_wire_len(&mut node, &dest, 1400); let inner = build_mtu_exceeded_inner(&dest, &reporter, 1280); node.handle_mtu_exceeded(&reporter, &inner).await; @@ -4959,3 +4975,226 @@ async fn test_peer_restart_reestablishes_through_a_pending_session_that_waited_o cleanup_nodes(&mut nodes).await; } + +// --------------------------------------------------------------------------- +// Reactive MtuExceeded: corroboration against what this node actually sent +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn a_reactive_mtu_exceeded_at_the_floor_no_longer_pins_a_session_this_node_has_not_overfilled() + { + // The defect itself. A report of exactly the floor is a legal value, and + // the admission gate cannot tell an honest forwarder from anyone else, so + // one packet drove a bound session's path MTU to the floor and pinned the + // FipsAddress-keyed entry the SYN-time MSS clamp reads. Nothing this node + // sent could have overflowed a hop at that size, so no honest report of it + // exists. + let mut node = make_node(); + + let remote = Identity::generate(); + install_established_session_with_mmp(&mut node, &remote); + let dest = *remote.node_addr(); + let reporter = NodeAddr::from_bytes([0xBB; 16]); + let dest_fips = crate::FipsAddress::from_node_addr(&dest); + + let before = node + .sessions + .get(&dest) + .and_then(|e| e.mmp()) + .map(|m| m.path_mtu.current_mtu()); + + let inner = + build_mtu_exceeded_inner(&dest, &reporter, crate::upper::icmp::MIN_REACTIVE_PATH_MTU); + node.handle_mtu_exceeded(&reporter, &inner).await; + + assert_eq!( + node.sessions + .get(&dest) + .and_then(|e| e.mmp()) + .map(|m| m.path_mtu.current_mtu()), + before, + "an uncorroborated report must leave the session path MTU alone" + ); + assert_eq!( + node.path_mtu_lookup_get(&dest_fips), + None, + "an uncorroborated report must leave no clamp entry behind" + ); + assert_eq!( + node.metrics().errors.mtu_exceeded_uncorroborated.get(), + 1, + "the refusal must be counted apart from the below-floor refusal" + ); + assert_eq!( + node.metrics().errors.mtu_exceeded_below_floor.get(), + 0, + "the floor is not what refused this; the value is exactly at it" + ); +} + +#[tokio::test] +async fn an_initiating_session_refuses_an_uncorroborated_report_and_accepts_a_corroborated_one() { + // The lookup write is the effect that survives on an initiating session, + // which has no MMP state at all, so this branch needs its own coverage: + // a guard placed on the apply rather than ahead of it would miss it. + let mut node = make_node(); + + let remote = Identity::generate(); + install_initiating(&mut node, &remote); + let dest = *remote.node_addr(); + let reporter = NodeAddr::from_bytes([0xBB; 16]); + let dest_fips = crate::FipsAddress::from_node_addr(&dest); + + let inner = build_mtu_exceeded_inner(&dest, &reporter, 800); + node.handle_mtu_exceeded(&reporter, &inner).await; + assert_eq!( + node.path_mtu_lookup_get(&dest_fips), + None, + "nothing this node sent could have overflowed a hop at 800 bytes" + ); + + // A SessionSetup can itself be the datagram that overflows a hop, so an + // initiating session must still be able to act on a real report. + note_sent_wire_len(&mut node, &dest, 1400); + node.handle_mtu_exceeded(&reporter, &inner).await; + assert_eq!( + node.path_mtu_lookup_get(&dest_fips), + Some(800), + "a report corroborated by an oversized send must still be applied" + ); +} + +#[tokio::test] +async fn a_second_reactive_decrease_needs_its_own_corroborating_send() { + // The evidence is spent on the decrease it vouched for. Otherwise one + // large send early in a session would vouch for every forged report for + // the rest of that session's life. + let mut node = make_node(); + + let remote = Identity::generate(); + install_established_session_with_mmp(&mut node, &remote); + let dest = *remote.node_addr(); + let reporter = NodeAddr::from_bytes([0xBB; 16]); + + note_sent_wire_len(&mut node, &dest, 1400); + let first = build_mtu_exceeded_inner(&dest, &reporter, 1200); + node.handle_mtu_exceeded(&reporter, &first).await; + assert_eq!( + node.sessions + .get(&dest) + .and_then(|e| e.mmp()) + .map(|m| m.path_mtu.current_mtu()), + Some(1200), + "the corroborated first decrease is accepted" + ); + + let second = build_mtu_exceeded_inner(&dest, &reporter, 600); + node.handle_mtu_exceeded(&reporter, &second).await; + assert_eq!( + node.sessions + .get(&dest) + .and_then(|e| e.mmp()) + .map(|m| m.path_mtu.current_mtu()), + Some(1200), + "a further decrease needs evidence of its own" + ); + + // A genuine re-route onto a smaller hop is preceded by a send that hop + // drops, so the honest sequence still converges. + note_sent_wire_len(&mut node, &dest, 900); + node.handle_mtu_exceeded(&reporter, &second).await; + assert_eq!( + node.sessions + .get(&dest) + .and_then(|e| e.mmp()) + .map(|m| m.path_mtu.current_mtu()), + Some(600), + "once this node has again sent something that does not fit, the report applies" + ); +} + +#[tokio::test] +async fn a_corroborated_report_below_the_reactive_floor_is_still_refused() { + // Corroboration and the floor are independent refusals. A hop that really + // is tiny still cannot drive the clamp into the band where the derived + // MSS degenerates. + let mut node = make_node(); + + let remote = Identity::generate(); + install_established_session_with_mmp(&mut node, &remote); + let dest = *remote.node_addr(); + let reporter = NodeAddr::from_bytes([0xBB; 16]); + let dest_fips = crate::FipsAddress::from_node_addr(&dest); + + note_sent_wire_len(&mut node, &dest, 1400); + let inner = build_mtu_exceeded_inner( + &dest, + &reporter, + crate::upper::icmp::MIN_REACTIVE_PATH_MTU - 1, + ); + node.handle_mtu_exceeded(&reporter, &inner).await; + + assert_eq!(node.path_mtu_lookup_get(&dest_fips), None); + assert_eq!(node.metrics().errors.mtu_exceeded_below_floor.get(), 1); + assert_eq!(node.metrics().errors.mtu_exceeded_uncorroborated.get(), 0); +} + +#[tokio::test] +async fn the_authenticated_path_mtu_notification_still_applies_at_the_actionable_floor() { + // The reactive guards must not leak onto the carrier that arrives inside + // an established session on the decrypted path, which is authenticated and + // needs no corroboration. + let mut node = make_node(); + + let remote = Identity::generate(); + install_established_session_with_mmp(&mut node, &remote); + let dest = *remote.node_addr(); + + let floor = crate::upper::icmp::MIN_ACTIONABLE_PATH_MTU; + let body = build_path_mtu_notification_body(floor); + node.handle_session_path_mtu_notification(&dest, &body); + + assert_eq!( + node.sessions + .get(&dest) + .and_then(|e| e.mmp()) + .map(|m| m.path_mtu.current_mtu()), + Some(floor), + "the authenticated carrier still applies a value at the actionable floor" + ); +} + +#[tokio::test] +async fn a_path_broken_flood_releases_the_stored_path_mtu_only_once_per_interval() { + use crate::protocol::PathBroken; + + // PathBroken is unauthenticated and its release discards a bottleneck this + // node learned the hard way. Unlimited, the claim can be repeated as fast + // as it can be sent, so a genuinely learned value never survives. + let mut node = make_node(); + + let remote = Identity::generate(); + install_initiating(&mut node, &remote); + let dest = *remote.node_addr(); + let reporter = NodeAddr::from_bytes([0xBB; 16]); + let dest_fips = crate::FipsAddress::from_node_addr(&dest); + + let encoded = PathBroken::new(dest, reporter).encode(); + let inner = &encoded[5..]; + + node.path_mtu_lookup_insert(dest_fips, 700); + node.handle_path_broken(&reporter, inner).await; + assert_eq!( + node.path_mtu_lookup_get(&dest_fips), + None, + "the first PathBroken still releases" + ); + + node.path_mtu_lookup_insert(dest_fips, 700); + node.handle_path_broken(&reporter, inner).await; + assert_eq!( + node.path_mtu_lookup_get(&dest_fips), + Some(700), + "a second release for the same destination inside the interval is refused" + ); +} diff --git a/src/upper/icmp.rs b/src/upper/icmp.rs index f866867d..d4f6d30e 100644 --- a/src/upper/icmp.rs +++ b/src/upper/icmp.rs @@ -122,6 +122,22 @@ pub const FIPS_IPV6_OVERHEAD: u16 = 77; /// small and refuses only the zero cliff, which no provenance makes usable. pub const MIN_ACTIONABLE_PATH_MTU: u16 = 256; +/// Smallest path MTU this node will act on when the claim arrives on the +/// unauthenticated reactive carrier, `MtuExceeded`. +/// +/// Held equal to [`MIN_ACTIONABLE_PATH_MTU`] so no hop legitimately configured +/// with a small transport MTU loses reactive feedback. It is a separate +/// constant because the two carriers differ in what they prove: the +/// authenticated `PathMtuNotification` and the proof-carrying discovery +/// response come from a party this node has verified, whereas this one comes +/// from whoever could route a datagram here. What keeps a legal-but-forged +/// claim from pinning a session is corroboration against what this node has +/// actually sent, not this floor. Raising it (576 is the value the original +/// path-MTU floor design proposed, and derives an inner IPv6 MTU of 499) +/// bounds the outcome of an uncorroborated claim further, at the cost of +/// ignoring an honest report from any hop configured between the two values. +pub const MIN_REACTIVE_PATH_MTU: u16 = MIN_ACTIONABLE_PATH_MTU; + /// Calculate the effective IPv6 MTU for FIPS-encapsulated traffic. /// /// Given a transport MTU (e.g., UDP payload size), returns the maximum From d37c556e2f68482fc8b680d0ea38c5e98dc71c9c Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:44:41 +0100 Subject: [PATCH 19/23] Bound the end-to-end session table The session table was the one remotely-grown map with no bound: an inbound SessionSetup naming an address nobody had seen inserted an entry, and neither existing limit reached it, the setup limiter governing arrival rate rather than population and the idle purge only reaching entries a peer stops using. One neighbour sending setups at the permitted rate could hold roughly 1440 half-open entries at any moment and grow the table without limit by keeping them warm. Add node.limits.max_sessions, default 1024, and refuse a setup that would grow the table past it. The check sits ahead of the setup limiter, so a full table costs no token, no responder handshake and no ack; a refused setup emits nothing, which is indistinguishable from loss to the sender and is already covered by its own msg1 resend schedule. The predicate is whether admitting would grow the table, not whether the sender is a stranger, so a resent setup for an entry already present is still served and an in-flight handshake is not broken. Zero means unlimited, which restores the previous behaviour exactly and is how to back this out on a running node. Hold unauthenticated half-open entries to half the table on top of that, so a handshake flood cannot deny the whole of it to peers that complete. Half rather than a tighter share because a reconnect storm, where every peer initiates at once after a restart or a healed partition, has to fit; those entries are reaped at handshake_timeout_secs while established ones survive idle_timeout_secs, so they turn over faster than the share suggests. Cap the locally originated path at the same ceiling, answering the application with ICMPv6 destination unreachable rather than returning an error, since the caller reads an error as "no route" and would respond with a discovery lookup and a queued packet on a node already at its limit. Refuse rather than evict: the setup that triggers the decision is unauthenticated at that point, so evicting would hand a stranger a way to tear down sessions it has nothing to do with. Refusals count as table_full and half_open_full. What stays open is per-neighbour fairness among established sessions: one hostile neighbour that completes handshakes and keeps each session warm can occupy the table and hold new establishment closed for as long as it keeps doing so, which is a denial of new sessions rather than the unbounded memory growth it replaces. --- CHANGELOG.md | 39 ++++++ src/config/node.rs | 18 +++ src/node/handlers/session.rs | 95 +++++++++++++ src/node/reject.rs | 12 ++ src/node/stats.rs | 14 ++ src/node/tests/session.rs | 252 +++++++++++++++++++++++++++++++++++ 6 files changed, 430 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index a9298f1d..f07f7d6a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -23,6 +23,15 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 non-positive rate is rejected at config validation rather than silently refusing every session. +- `node.limits.max_sessions`, defaulting to 1024, which bounds the end-to-end + session table. Zero means unlimited, which restores the previous behaviour + exactly and is the way to back the change out on a running node. The default + is four times the adjacent `node.session.pending_max_destinations`. A + session entry measures 6608 bytes of inline state plus heap, so the table + holds to roughly 7 MB, and a test pins that per-entry figure so the + arithmetic behind the default fails loudly if an entry grows. Existing + configurations parse unchanged, the key being optional. + - `node.rate_limit.established_handshake_burst` and `node.rate_limit.established_handshake_rate`, the parameters of the new established-link msg1 token bucket, which meters link-layer msg1 rather @@ -1189,6 +1198,36 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 time, so a peer admitted during a window that has already happened keeps its link. +- The end-to-end session table now has a bound. It was the one remotely-grown + map with none: an inbound SessionSetup naming an address nobody had seen + inserted an entry, and the two existing limits did not reach it, the setup + limiter governing the arrival rate rather than the population and the idle + purge only reaching entries a peer stops using. One neighbour sending setups + at the permitted rate could hold roughly 1440 half-open entries at any + moment and grow the table without limit by keeping them warm. Setups that + would grow the table past `node.limits.max_sessions` are now refused, ahead + of the setup limiter, so a full table costs no token, no responder handshake + and no ack; a refused setup emits nothing at all, which is indistinguishable + from loss to the sender and is already covered by its own msg1 resend + schedule. The test is whether admitting would grow the table, not whether + the sender is a stranger, so a resent setup for an entry already present is + still served and an in-flight handshake is not broken. Unauthenticated + half-open entries are additionally held to half the table, so a handshake + flood cannot deny the whole of it to peers that complete; that share is sized + to leave a reconnect storm, where every peer initiates at once after a + restart or a healed partition, room to land. Locally originated sessions are + capped at the same ceiling, answered with ICMPv6 destination unreachable so + the application gets an immediate error rather than a silent drop. The cap + refuses rather than evicts: the setup that triggers the decision is + unauthenticated at that point, so evicting would hand a stranger a way to + tear down sessions it has nothing to do with. Refusals are counted as + `table_full` and `half_open_full` in the session reject family. What stays + open is per-neighbour fairness among established sessions: one hostile + neighbour that completes handshakes and keeps each session warm can occupy + the table and hold new session establishment closed for as long as it keeps + doing so, which is a denial of new sessions rather than the unbounded memory + growth it replaces. + - An accepted inbound TCP connection no longer holds a slot indefinitely without sending anything. The cap was tested at accept and the pool insert and counter bump followed with no read in between, while the frame reader's diff --git a/src/config/node.rs b/src/config/node.rs index eead8709..38b614c6 100644 --- a/src/config/node.rs +++ b/src/config/node.rs @@ -28,6 +28,20 @@ pub struct LimitsConfig { /// Max pending inbound handshakes (`node.limits.max_pending_inbound`). #[serde(default = "LimitsConfig::default_max_pending_inbound")] pub max_pending_inbound: usize, + /// Max end-to-end sessions (`node.limits.max_sessions`), `0` = unlimited. + /// + /// The session table is the only remotely-grown map with no bound: an + /// inbound SessionSetup from an address nobody has seen inserts an + /// entry, and the idle purge only reaches entries a peer stops using. + /// The default of 1024 is four times the adjacent + /// `node.session.pending_max_destinations`. One entry measures 6608 + /// bytes of inline state plus heap, so the table holds to roughly 7 MB + /// and a test pins the per-entry figure the default rests on. Raising it + /// raises the memory an attacker can make this node hold; lowering it + /// refuses new sessions sooner on a node that legitimately talks + /// end-to-end to many others, such as a gateway. + #[serde(default = "LimitsConfig::default_max_sessions")] + pub max_sessions: usize, } impl Default for LimitsConfig { @@ -37,6 +51,7 @@ impl Default for LimitsConfig { max_peers: 128, max_links: 256, max_pending_inbound: 1000, + max_sessions: 1024, } } } @@ -54,6 +69,9 @@ impl LimitsConfig { fn default_max_pending_inbound() -> usize { 1000 } + fn default_max_sessions() -> usize { + 1024 + } } /// Rate limiting (`node.rate_limit.*`). diff --git a/src/node/handlers/session.rs b/src/node/handlers/session.rs index 3601dd71..6e5e2ec9 100644 --- a/src/node/handlers/session.rs +++ b/src/node/handlers/session.rs @@ -64,6 +64,20 @@ fn link_wire_len(encoded_len: usize) -> usize { encoded_len + LINK_FRAME_OVERHEAD } +/// Divisor giving the share of the session table that unauthenticated +/// half-open entries may hold, as `max_sessions / DIVISOR`. +/// +/// Two means a reconnect storm, where every peer that had a session +/// initiates at once after a restart or a healed partition, still fits in +/// half the table; a tighter share bites four times sooner and is felt by +/// a hub before it is felt by an attacker. Half-open entries are reaped +/// after `handshake_timeout_secs` while established ones survive +/// `idle_timeout_secs`, so they turn over faster than the share suggests. +/// Lowering the divisor raises the share, which lets a handshake flood +/// crowd out peers that complete; raising it refuses legitimate initiators +/// sooner in a storm. +const HALF_OPEN_SHARE_DIVISOR: usize = 2; + /// Inputs to `try_send_session_data_pipelined` — the FSP+FMP pipelined /// fast path that hands both AEAD operations to the encrypt worker /// in a single dispatch. @@ -517,6 +531,19 @@ impl Node { // no limit at all. A setup naming an established peer cannot grow the // table and is metered separately, so that a stranger flood over a // shared link cannot stop that peer's rekey from arming. + // Population cap, ahead of the limiter so a full table costs no + // token, no responder handshake and no ack. The predicate is "would + // admitting this grow the table", not "is this a stranger": `class` + // is Stranger for an existing Initiating or AwaitingMsg3 entry too, + // and refusing those would break in-flight legitimate handshakes and + // the duplicate-ack resend. Same shape as the pending-destination cap + // in `queue_pending_packet`. Refuse rather than evict: msg1 is + // unauthenticated here, so evicting would hand a stranger a teardown + // primitive it does not have. + if !self.admit_new_session(src_addr) { + return; + } + let class = if self .sessions .get(src_addr) @@ -1798,6 +1825,59 @@ impl Node { /// Creates a Noise XK handshake as initiator, wraps msg1 in a /// SessionSetup, encapsulates in a SessionDatagram, and routes /// toward the destination. + /// Whether a session for `addr` may be created, given the table cap. + /// + /// Returns true when an entry already exists, since admitting it cannot + /// grow the table. Counts its own refusals, so the two reasons are + /// distinguishable without turning on debug logging. + pub(in crate::node) fn admit_new_session(&mut self, addr: &NodeAddr) -> bool { + let max_sessions = self.config().node.limits.max_sessions; + if max_sessions == 0 || self.sessions.contains_key(addr) { + return true; + } + + if self.sessions.len() >= max_sessions { + debug!( + src = %self.peer_display_name(addr), + sessions = self.sessions.len(), + max_sessions = max_sessions, + "Session table full, refusing to create a session" + ); + self.stats_mut() + .record_reject(RejectReason::Session(SessionReject::TableFull)); + return false; + } + + // Half-open entries are unauthenticated and are reaped after + // `handshake_timeout_secs`, so they are the cheap half of the table + // to fill. Holding them to a share keeps room for peers that + // complete. The outer length test makes the scan unreachable below + // the share, and the table is itself bounded by the cap above. + // At least one, or a table capped at one would admit no inbound + // session at all rather than one. + let half_open_share = (max_sessions / HALF_OPEN_SHARE_DIVISOR).max(1); + if self.sessions.len() >= half_open_share { + let half_open = self + .sessions + .values() + .filter(|e| e.is_awaiting_msg3()) + .count(); + if half_open >= half_open_share { + debug!( + src = %self.peer_display_name(addr), + half_open = half_open, + half_open_share = half_open_share, + "Half-open session share exhausted, refusing to create a session" + ); + self.stats_mut() + .record_reject(RejectReason::Session(SessionReject::HalfOpenFull)); + return false; + } + } + + true + } + pub(in crate::node) async fn initiate_session( &mut self, dest_addr: NodeAddr, @@ -2639,6 +2719,17 @@ impl Node { return; } + // No session, so this one would grow the table. Answer the local + // application the way an unroutable destination is answered rather + // than returning an error from `initiate_session`: the caller reads + // an error as "no route" and responds with a discovery lookup and a + // queued packet, which is outbound traffic on a node already at its + // limit. + if !self.admit_new_session(&dest_addr) { + self.send_icmpv6_dest_unreachable(&ipv6_packet); + return; + } + // No session: initiate one and queue the packet. // If session initiation fails (no route), trigger discovery and // queue the packet for retry when discovery completes. @@ -2769,6 +2860,10 @@ impl Node { return; } + if !self.admit_new_session(&dest_addr) { + return; + } + match self.initiate_session(dest_addr, dest_pubkey).await { Ok(()) => { debug!(dest = %self.peer_display_name(&dest_addr), "Session initiated after discovery"); diff --git a/src/node/reject.rs b/src/node/reject.rs index 40499f32..ea58e467 100644 --- a/src/node/reject.rs +++ b/src/node/reject.rs @@ -258,6 +258,18 @@ pub enum SessionReject { /// before any handshake state was created or any ack sent. Tracked via /// [`SessionStats::setup_rate_limited`](crate::node::stats::SessionStats). SetupRateLimited, + /// A session would have been created but the table is at + /// `node.limits.max_sessions`. Refused rather than evicted: the + /// deciding message is unauthenticated at this point, so evicting + /// would hand a stranger a way to tear down established sessions. + /// Tracked via + /// [`SessionStats::table_full`](crate::node::stats::SessionStats). + TableFull, + /// A session would have been created but unauthenticated half-open + /// entries already hold their share of the table. Bounds what a + /// handshake flood can deny an established peer. Tracked via + /// [`SessionStats::half_open_full`](crate::node::stats::SessionStats). + HalfOpenFull, } /// MMP rejection reasons. diff --git a/src/node/stats.rs b/src/node/stats.rs index 52d0f399..d6028f62 100644 --- a/src/node/stats.rs +++ b/src/node/stats.rs @@ -79,6 +79,14 @@ pub struct SessionStats { /// A setup message was refused by the per-link-peer setup limiter, /// before any handshake state was created or any ack sent. pub setup_rate_limited: u64, + /// A session would have been created but the table is at + /// `node.limits.max_sessions`. A sustained rate means either the cap + /// is sized below what this node legitimately carries, or something is + /// holding the table full. + pub table_full: u64, + /// A session would have been created but unauthenticated half-open + /// entries already hold their share of the table. + pub half_open_full: u64, } impl SessionStats { @@ -96,6 +104,8 @@ impl SessionStats { pending_replaced: self.pending_replaced, ack_handshake_failed: self.ack_handshake_failed, setup_rate_limited: self.setup_rate_limited, + table_full: self.table_full, + half_open_full: self.half_open_full, } } @@ -110,6 +120,8 @@ impl SessionStats { SessionReject::RekeyPending => self.rekey_pending += 1, SessionReject::AckHandshakeFailed => self.ack_handshake_failed += 1, SessionReject::SetupRateLimited => self.setup_rate_limited += 1, + SessionReject::TableFull => self.table_full += 1, + SessionReject::HalfOpenFull => self.half_open_full += 1, } } } @@ -392,6 +404,8 @@ pub struct SessionStatsSnapshot { pub pending_replaced: u64, pub ack_handshake_failed: u64, pub setup_rate_limited: u64, + pub table_full: u64, + pub half_open_full: u64, } #[derive(Clone, Debug, Default, Serialize)] diff --git a/src/node/tests/session.rs b/src/node/tests/session.rs index 2fe5eedd..78676c67 100644 --- a/src/node/tests/session.rs +++ b/src/node/tests/session.rs @@ -4206,6 +4206,258 @@ async fn test_a_drained_stranger_bucket_still_admits_a_setup_naming_an_establish cleanup_nodes(&mut nodes).await; } +// ============================================================================ +// Integration tests: the session-table population cap +// ============================================================================ + +/// Build a two-node routable mesh with the session table capped for a test +/// and the setup limiter opened wide, so the cap is the only thing refusing. +async fn make_session_capped_pair(max_sessions: usize) -> Vec { + let configs = (0..2) + .map(|_| { + let mut config = Config::new(); + config.node.rekey.enabled = false; + config.node.limits.max_sessions = max_sessions; + config.node.rate_limit.session_setup_burst = 10_000; + config.node.rate_limit.session_setup_rate = 10_000.0; + config + }) + .collect(); + let mut nodes = run_tree_test_with_configs(configs, &[(0, 1)]).await; + verify_tree_convergence(&nodes); + populate_all_coord_caches(&mut nodes); + nodes +} + +#[tokio::test] +async fn test_forged_setups_stop_growing_the_session_table_once_the_cap_is_reached() { + // The table was the one remotely-grown map with no bound: each setup from + // an address nobody has seen inserted an entry, and neither existing limit + // reached it, the setup limiter governing arrival rate rather than + // population and the idle purge only reaching entries a peer stops using. + const MAX: usize = 8; + let share = MAX / 2; + let mut nodes = make_session_capped_pair(MAX).await; + + for _ in 0..share { + deliver_forged_setup_over_link(&mut nodes).await; + } + assert_eq!( + nodes[1].node.sessions.len(), + share, + "the admissible entries must be admitted, or this test would pass for \ + the wrong reason" + ); + assert_eq!(nodes[1].node.stats().session.half_open_full, 0); + + // Every SessionAck goes out through `send_session_datagram`, the only + // thing bumping this counter on a node with no transit traffic. A refused + // setup must not move it. + let originated = nodes[1].node.metrics().forwarding.originated_packets.get(); + + for _ in 0..4 { + deliver_forged_setup_over_link(&mut nodes).await; + } + + assert_eq!( + nodes[1].node.sessions.len(), + share, + "a table at its bound must stop growing" + ); + assert_eq!( + nodes[1].node.stats().session.half_open_full, + 4, + "each refusal must be counted; the DEBUG line is invisible by default" + ); + assert_eq!( + nodes[1].node.metrics().forwarding.originated_packets.get(), + originated, + "a refused setup must emit nothing at all" + ); + + cleanup_nodes(&mut nodes).await; +} + +#[tokio::test] +async fn test_a_setup_that_would_grow_a_full_table_is_refused_and_counted() { + // The table-full arm specifically: one established entry against a cap of + // one, so the half-open share is not what refuses. + let mut nodes = make_session_capped_pair(1).await; + establish_pair_session(&mut nodes).await; + assert_eq!( + nodes[1].node.sessions.len(), + 1, + "precondition: the table is full with the established peer" + ); + + let originated = nodes[1].node.metrics().forwarding.originated_packets.get(); + deliver_forged_setup_over_link(&mut nodes).await; + + assert_eq!( + nodes[1].node.sessions.len(), + 1, + "a full table must not grow for a stranger" + ); + assert_eq!(nodes[1].node.stats().session.table_full, 1); + assert_eq!( + nodes[1].node.metrics().forwarding.originated_packets.get(), + originated, + "a refused setup must cost no ack" + ); + + cleanup_nodes(&mut nodes).await; +} + +#[tokio::test] +async fn test_a_full_session_table_still_serves_a_setup_naming_an_existing_entry() { + // The guard against writing the cap as "refuse strangers". A setup for an + // entry already present cannot grow the table, and refusing it would break + // the duplicate-ack resend an initiator depends on. + let mut nodes = make_session_capped_pair(1).await; + establish_pair_session(&mut nodes).await; + + let node0_addr = *nodes[0].node.node_addr(); + let node1_addr = *nodes[1].node.node_addr(); + let refused_before = nodes[1].node.stats().session.table_full; + let originated = nodes[1].node.metrics().forwarding.originated_packets.get(); + + // A setup naming the established peer: the shape an inbound rekey has. + let setup = forge_setup_from_stranger(&nodes); + let datagram = SessionDatagram::new(node0_addr, node1_addr, setup).with_ttl(64); + let encoded = datagram.encode(); + nodes[1] + .node + .handle_session_datagram(&node0_addr, &encoded[1..], false) + .await; + + assert_eq!( + nodes[1].node.stats().session.table_full, + refused_before, + "a setup that cannot grow the table must not be refused by the cap" + ); + assert!( + nodes[1].node.metrics().forwarding.originated_packets.get() > originated, + "and it must still be answered" + ); + + cleanup_nodes(&mut nodes).await; +} + +#[tokio::test] +async fn test_a_full_session_table_does_not_evict_an_established_session() { + // Pins refuse-not-evict. The setup that triggers the decision is + // unauthenticated at that point, so evicting would hand a stranger a way + // to tear down a session it has nothing to do with. + let mut nodes = make_session_capped_pair(1).await; + establish_pair_session(&mut nodes).await; + let node0_addr = *nodes[0].node.node_addr(); + + for _ in 0..4 { + deliver_forged_setup_over_link(&mut nodes).await; + } + + assert!( + nodes[1] + .node + .get_session(&node0_addr) + .expect("the established session must survive a flood at the cap") + .is_established(), + "a stranger's setup must never cost an established peer its session" + ); + + cleanup_nodes(&mut nodes).await; +} + +#[tokio::test] +async fn test_the_session_table_admits_again_after_the_handshake_reaper_drains_it() { + // The cap is a ceiling, not a latch: half-open entries are reaped after + // `handshake_timeout_secs` and the room they free must be usable. + const MAX: usize = 8; + let share = MAX / 2; + let mut nodes = make_session_capped_pair(MAX).await; + + for _ in 0..(share + 2) { + deliver_forged_setup_over_link(&mut nodes).await; + } + assert!( + nodes[1].node.stats().session.half_open_full > 0, + "precondition: the table is refusing before the reaper runs" + ); + + let timeout_ms = nodes[1] + .node + .config() + .node + .rate_limit + .handshake_timeout_secs + * 1000; + let now_ms = Node::now_ms(); + nodes[1] + .node + .resend_pending_session_handshakes(now_ms + timeout_ms + 1) + .await; + assert_eq!( + nodes[1].node.sessions.len(), + 0, + "precondition: the reaper freed the half-open entries" + ); + + deliver_forged_setup_over_link(&mut nodes).await; + assert_eq!( + nodes[1].node.sessions.len(), + 1, + "room freed by the reaper must be usable, or the cap is a latch" + ); + + cleanup_nodes(&mut nodes).await; +} + +#[tokio::test] +async fn test_half_open_setups_cannot_consume_more_than_their_share_of_the_table() { + // Half-open entries are unauthenticated and cheap to create, so they are + // held to a share of the table rather than being allowed to fill it and + // deny it to every peer that would complete a handshake. + const MAX: usize = 16; + let share = MAX / 2; + let mut nodes = make_session_capped_pair(MAX).await; + + for _ in 0..(share + 2) { + deliver_forged_setup_over_link(&mut nodes).await; + } + + assert_eq!( + nodes[1].node.sessions.len(), + share, + "half-open entries must stop at their share, well below the table cap" + ); + assert_eq!(nodes[1].node.stats().session.half_open_full, 2); + assert_eq!( + nodes[1].node.stats().session.table_full, + 0, + "the table itself is not full, so the refusals must be attributed to \ + the share rather than to the cap" + ); + + cleanup_nodes(&mut nodes).await; +} + +#[test] +fn test_session_entry_size_stays_within_the_budget_the_cap_is_derived_from() { + // The default `max_sessions` is derived from what one entry costs. + // Measured at 6608 bytes of inline state when the cap was written, plus + // heap for the MMP window and handshake payloads, so 1024 sessions is + // roughly 7 MB. This is what fires if a large field is added later and + // the arithmetic behind that default stops holding. + const BUDGET: usize = 8192; + assert!( + std::mem::size_of::() <= BUDGET, + "SessionEntry is {} bytes, over the {} the max_sessions default \ + assumes; re-derive the default or shrink the entry", + std::mem::size_of::(), + BUDGET + ); +} + // ============================================================================ // Integration tests: a forged SessionAck against an in-flight initiation // ============================================================================ From e0fb8d363db9653d78ea49d6dc451df8af753572 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sun, 23 Aug 2026 12:25:00 +0100 Subject: [PATCH 20/23] Take the link frame overhead constant off a unix-only import LINK_FRAME_OVERHEAD read ESTABLISHED_HEADER_SIZE through the crate::node::wire import, which is #[cfg(unix)]. The constant feeds link_wire_len, whose only caller send_session_datagram is compiled on every platform, so the Windows build could not resolve the name and the library failed with E0425. Spell the path out in full instead of widening the import, which would also pull FLAG_KEY_EPOCH, FLAG_SP and build_established_header onto platforms that have no use for them. The module and the constant are both unconditional, so only the use statement was ever the problem. Found by GitHub CI, which builds Windows; the local gate is Linux-only and cannot see this class of break at all. Of the eight symbols this file imports only under cfg(unix), this was the one the security batch referenced from ungated code. --- src/node/handlers/session.rs | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/src/node/handlers/session.rs b/src/node/handlers/session.rs index 6e5e2ec9..5d0ed76c 100644 --- a/src/node/handlers/session.rs +++ b/src/node/handlers/session.rs @@ -57,7 +57,13 @@ pub(in crate::node) const PATH_MTU_RELEASE_MIN_INTERVAL: std::time::Duration = /// wire: the established FMP header, the 4-byte session-relative timestamp and /// the AEAD tag. Mirrors the buffer `send_encrypted_link_message_with_ce` /// builds. -const LINK_FRAME_OVERHEAD: usize = ESTABLISHED_HEADER_SIZE + 4 + crate::noise::TAG_SIZE; +/// +/// Spelled out in full rather than through the `crate::node::wire` import +/// above, which is `#[cfg(unix)]`. This constant feeds `link_wire_len`, whose +/// caller `send_session_datagram` is compiled on every platform, so taking the +/// name from that import fails to build on Windows. +const LINK_FRAME_OVERHEAD: usize = + crate::node::wire::ESTABLISHED_HEADER_SIZE + 4 + crate::noise::TAG_SIZE; /// Wire size of an encoded `SessionDatagram` of `encoded_len` bytes. fn link_wire_len(encoded_len: usize) -> usize { From 97d64f5d568904cb867c9e79f2fe677c93421a32 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sun, 23 Aug 2026 13:58:14 +0100 Subject: [PATCH 21/23] Gate the punch and ACL permission tests to the platforms that can run them GitHub CI runs unit tests on Windows and macOS; the local gate is Linux only and saw neither failure. The five punch tests distinguish a spoofer from a planned target by source IP, so each needs its own loopback address. Only Linux treats all of 127/8 as local, so the bind panicked elsewhere. Collapsing them onto 127.0.0.1 would make every source rank RemappedPort and the tests would stop testing what they exist for, so they are gated to Linux instead. The ranking decision itself stays covered everywhere by the rank_punch_source unit tests, which take no sockets. The three ACL permission-fault tests make a file unreadable through the unix mode bits, which Windows has no equivalent for: a read-only NTFS file is still readable, so the fault cannot be produced there. Both coverage gaps are recorded at the site rather than treated as discharged. --- src/discovery/nostr/tests.rs | 19 +++++++++++++++++++ src/node/acl.rs | 13 +++++++++++++ 2 files changed, 32 insertions(+) diff --git a/src/discovery/nostr/tests.rs b/src/discovery/nostr/tests.rs index 8473dd79..ff2be248 100644 --- a/src/discovery/nostr/tests.rs +++ b/src/discovery/nostr/tests.rs @@ -1375,9 +1375,21 @@ async fn signal_events_use_current_timestamps() { assert!(created_at <= after); } +/// These punch tests distinguish a spoofer from a planned target by source +/// **IP**, so each needs its own loopback address. Only Linux treats the whole +/// of 127/8 as local; macOS and Windows bind 127.0.0.1 alone unless an alias is +/// added, so the bind panics there. They are gated to Linux rather than +/// rewritten onto one address, because collapsing them onto 127.0.0.1 would +/// make every source rank `RemappedPort` and the tests would stop testing what +/// they are for. +/// +/// **Coverage gap**: on macOS and Windows nothing exercises `run_punch_attempt` +/// end to end. The ranking decision itself is covered on every platform by the +/// `rank_punch_source_*` unit tests above, which take no sockets. /// A loopback socket bound on `host`, non-blocking as both production call /// sites leave it, since `run_punch_attempt` hands it straight to /// `UdpSocket::from_std`. +#[cfg(target_os = "linux")] fn punch_socket(host: &str) -> std::net::UdpSocket { let socket = std::net::UdpSocket::bind(format!("{host}:0")).expect("bind a loopback socket"); socket @@ -1398,12 +1410,14 @@ fn immediate_punch_hint(duration_ms: u64) -> PunchHint { /// Send one well-formed probe carrying `session_id`'s hash from `from` to /// `to`, which is what a replay of captured punch bytes looks like. +#[cfg(target_os = "linux")] fn send_probe(from: &std::net::UdpSocket, to: SocketAddr, session_id: &str) { let packet = build_punch_packet(PunchPacketKind::Probe, 1, session_id); from.send_to(&packet, to).expect("probe should send"); } /// Whether anything readable on `socket` is a punch ack. +#[cfg(target_os = "linux")] fn received_an_ack(socket: &std::net::UdpSocket) -> bool { let mut buf = [0u8; 2048]; while let Ok((len, _)) = socket.recv_from(&mut buf) { @@ -1448,6 +1462,7 @@ fn rank_punch_source_rejects_an_address_we_never_planned_to_probe() { /// every probe, so anyone who has seen one can replay it. Acceptance is now /// constrained to the targets this node planned; the spoofer is neither /// adopted nor acked, and an ack would be a reflection we control. +#[cfg(target_os = "linux")] #[tokio::test] async fn a_matching_punch_packet_from_an_unplanned_source_is_neither_adopted_nor_acked() { let victim = punch_socket("127.0.0.3"); @@ -1477,6 +1492,7 @@ async fn a_matching_punch_packet_from_an_unplanned_source_is_neither_adopted_nor } /// The spoofer wins the race on arrival order and still loses on address. +#[cfg(target_os = "linux")] #[tokio::test] async fn a_planned_source_is_adopted_even_when_a_spoofer_replies_first() { let victim = punch_socket("127.0.0.3"); @@ -1505,6 +1521,7 @@ async fn a_planned_source_is_adopted_even_when_a_spoofer_replies_first() { /// The healthy path, which is the check that the source constraint does not /// red a legitimately clean run: one probe from the single planned target is /// adopted immediately and acked. +#[cfg(target_os = "linux")] #[tokio::test] async fn the_ordinary_probe_from_a_planned_target_is_still_adopted_and_acked() { let victim = punch_socket("127.0.0.3"); @@ -1534,6 +1551,7 @@ async fn the_ordinary_probe_from_a_planned_target_is_still_adopted_and_acked() { /// shares a planned target's IP. Adopting it is the main class of NAT pairing /// punching exists to rescue, and this test reds if the rule is ever tightened /// to exact matching without that being reopened deliberately. +#[cfg(target_os = "linux")] #[tokio::test] async fn a_planned_targets_remapped_port_is_adopted_when_that_is_all_that_arrives() { let victim = punch_socket("127.0.0.3"); @@ -1562,6 +1580,7 @@ async fn a_planned_targets_remapped_port_is_adopted_when_that_is_all_that_arrive /// An exact match inside the settle window supersedes a remapped one that /// arrived first, which is what the window is for. +#[cfg(target_os = "linux")] #[tokio::test] async fn an_exact_target_supersedes_a_remapped_port_inside_the_settle_window() { let victim = punch_socket("127.0.0.3"); diff --git a/src/node/acl.rs b/src/node/acl.rs index 81f6eeb1..8610ba00 100644 --- a/src/node/acl.rs +++ b/src/node/acl.rs @@ -1027,21 +1027,32 @@ mod tests { ); } + /// The three permission-fault tests below make a file unreadable through + /// the unix mode bits, which Windows has no equivalent for: a read-only + /// NTFS file is still readable, so the fault they need cannot be produced. + /// They are gated to unix rather than made to pass vacuously elsewhere. + /// + /// **Coverage gap**: on Windows nothing exercises the reloader's + /// unreadable-input path, so the fail-open defect this fix closes is + /// unverified there. /// Make a file unreadable, returning false if the effective uid can read /// it anyway. Root bypasses the mode bits, so the permission-fault tests /// cannot run there and skip instead of passing vacuously; that leaves /// the EACCES path unexercised in any root CI job. + #[cfg(unix)] fn make_unreadable(path: &Path) -> bool { use std::os::unix::fs::PermissionsExt; std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o000)).unwrap(); std::fs::read_to_string(path).is_err() } + #[cfg(unix)] fn make_readable(path: &Path) { use std::os::unix::fs::PermissionsExt; std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o644)).unwrap(); } + #[cfg(unix)] #[tokio::test] async fn test_acl_reload_holds_last_good_snapshot_when_deny_file_unreadable() { let dir = tempfile::tempdir().unwrap(); @@ -1075,6 +1086,7 @@ mod tests { make_readable(&deny); } + #[cfg(unix)] #[tokio::test] async fn test_acl_reload_retries_after_a_transient_read_error() { let dir = tempfile::tempdir().unwrap(); @@ -1101,6 +1113,7 @@ mod tests { assert!(!reloader.status().stale); } + #[cfg(unix)] #[tokio::test] async fn test_acl_reload_holds_last_good_when_the_hosts_file_becomes_unreadable() { let dir = tempfile::tempdir().unwrap(); From 298528a39c24209894521d02e21078ebceb5d71b Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:18:00 +0100 Subject: [PATCH 22/23] Hold the onion receive task until its pool entry is counted The Tor accept loop spawned the per-connection receive task, then inserted the pool entry, then bumped the inbound counter. A remote that reset immediately let the receive task reach its cleanup first: the removal found nothing, the conditional teardown that decrements never fired, and the increment landed afterwards with nothing left to undo it. Enough of those and the max_inbound gate rejects every further onion connection while the pool is visibly empty. Restore the readiness barrier the TCP accept loop already uses. The shared proxied receive loop takes an optional oneshot receiver and waits on it before its read loop; a dropped sender means the accept loop went away, so it falls through to the cleanup rather than returning and stranding the entry it was meant to release. The tor accept loop signals after both the insert and the counter bump. Outbound tor connections and nym pass None: neither holds a counted inbound slot, and nym binds no listener at all. The same accept path now also releases the slot of an entry it evicts. A reused ephemeral forward port can collide with an entry whose receive task has not finished cleaning up; left alone that task later removes the entry the new connection just inserted, leaking one slot and orphaning a live connection. --- CHANGELOG.md | 16 +++ src/transport/nym/mod.rs | 4 + src/transport/socks5/pool.rs | 110 ++++++++------- src/transport/tor/mod.rs | 266 ++++++++++++++++++++++++++++++++++- 4 files changed, 346 insertions(+), 50 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4e7b6ef6..14010373 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -491,6 +491,22 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 #### Data-plane / transports +- An inbound onion connection no longer leaks its inbound slot when the peer + goes away before the accept loop has pooled it. The Tor accept loop spawned + the per-connection receive task, then inserted the pool entry, then bumped + the inbound counter; a remote that reset immediately let the receive task + run its cleanup first, find nothing to remove, skip the teardown that + decrements, and leave the increment behind for the life of the process. Once + enough of those accumulated the `max_inbound` gate rejected every further + onion connection, with the pool visibly empty. The accept loop now holds the + receive task on a readiness barrier until both the pool entry and the + counter are in place, as the TCP accept loop already did, and an aborted + accept still falls through to the cleanup rather than stranding the entry. + The same accept path also releases the slot of an entry it evicts, which a + reused ephemeral forward port could otherwise leave orphaned. Outbound Tor + connections and the whole of the Nym transport are unaffected: neither holds + a counted inbound slot. + - A path MTU measured on one link no longer clamps a peer that has moved to another. Every writer of the per-destination path-MTU cache keeps the smaller of the existing and incoming value, which is right while a peer diff --git a/src/transport/nym/mod.rs b/src/transport/nym/mod.rs index f9eba88a..243d09a9 100644 --- a/src/transport/nym/mod.rs +++ b/src/transport/nym/mod.rs @@ -670,6 +670,10 @@ async fn nym_receive_loop( stats, "Nym", None, + // Nym binds no listener (`accept_connections()` is `false`), its pool + // metadata is `()` and its teardown hook decrements nothing, so there + // is no admission ordering for a readiness barrier to protect. + None, |_stats, _meta| {}, ) .await; diff --git a/src/transport/socks5/pool.rs b/src/transport/socks5/pool.rs index 97d815fc..7e2ab7b3 100644 --- a/src/transport/socks5/pool.rs +++ b/src/transport/socks5/pool.rs @@ -149,6 +149,11 @@ pub(crate) trait ProxiedStats: Send + Sync + 'static { /// covers every outbound connection and the whole of the nym transport (nym /// is outbound-only and keeps no counted slots). A deadline expiry is not a /// receive error and is deliberately not recorded as one. +/// +/// `ready_rx`, when present, is the accept loop's readiness barrier: the loop +/// must not run its cleanup before the accept loop has inserted the pool entry +/// and bumped its counter, or the removal finds nothing, `on_remove` never +/// fires, and the increment is stranded for the life of the process. #[allow(clippy::too_many_arguments)] pub(crate) async fn proxied_receive_loop( mut reader: OwnedReadHalf, @@ -160,6 +165,7 @@ pub(crate) async fn proxied_receive_loop( stats: Arc, label: &'static str, first_frame_timeout: Option, + ready_rx: Option>, on_remove: impl Fn(&S, &M), ) { debug!( @@ -169,67 +175,77 @@ pub(crate) async fn proxied_receive_loop( label ); - let mut first = true; - loop { - let read = match first_frame_timeout { - // Bound the first read only. A silent remote otherwise holds its - // inbound slot for as long as it keeps the socket open. - Some(d) if first => { - match tokio::time::timeout(d, read_fmp_packet(&mut reader, mtu)).await { - Ok(result) => result, - Err(_) => { - // Not a recv error: `record_recv_error` means framing - // or I/O failure, and folding deadline expiries into - // it corrupts that counter. + // An `Err` here means the accept loop went away between the insert and + // the signal. Fall through to the cleanup below rather than returning, + // so a pooled entry cannot be stranded with the counter incremented. + let admitted = match ready_rx { + Some(rx) => rx.await.is_ok(), + None => true, + }; + + if admitted { + let mut first = true; + loop { + let read = match first_frame_timeout { + // Bound the first read only. A silent remote otherwise holds its + // inbound slot for as long as it keeps the socket open. + Some(d) if first => { + match tokio::time::timeout(d, read_fmp_packet(&mut reader, mtu)).await { + Ok(result) => result, + Err(_) => { + // Not a recv error: `record_recv_error` means framing + // or I/O failure, and folding deadline expiries into + // it corrupts that counter. + debug!( + transport_id = %transport_id, + remote_addr = %remote_addr, + timeout_secs = d.as_secs_f64(), + "No complete frame within the first-frame deadline, dropping inbound {} connection", + label + ); + break; + } + } + } + _ => read_fmp_packet(&mut reader, mtu).await, + }; + first = false; + + match read { + Ok(data) => { + stats.record_recv(data.len()); + + trace!( + transport_id = %transport_id, + remote_addr = %remote_addr, + bytes = data.len(), + "{} packet received", + label + ); + + let packet = ReceivedPacket::new(transport_id, remote_addr.clone(), data); + + if packet_tx.send(packet).await.is_err() { debug!( transport_id = %transport_id, - remote_addr = %remote_addr, - timeout_secs = d.as_secs_f64(), - "No complete frame within the first-frame deadline, dropping inbound {} connection", + "Packet channel closed, stopping {} receive loop", label ); break; } } - } - _ => read_fmp_packet(&mut reader, mtu).await, - }; - first = false; - - match read { - Ok(data) => { - stats.record_recv(data.len()); - - trace!( - transport_id = %transport_id, - remote_addr = %remote_addr, - bytes = data.len(), - "{} packet received", - label - ); - - let packet = ReceivedPacket::new(transport_id, remote_addr.clone(), data); - - if packet_tx.send(packet).await.is_err() { + Err(e) => { + stats.record_recv_error(); debug!( transport_id = %transport_id, - "Packet channel closed, stopping {} receive loop", + remote_addr = %remote_addr, + error = %e, + "{} receive error, removing connection", label ); break; } } - Err(e) => { - stats.record_recv_error(); - debug!( - transport_id = %transport_id, - remote_addr = %remote_addr, - error = %e, - "{} receive error, removing connection", - label - ); - break; - } } } diff --git a/src/transport/tor/mod.rs b/src/transport/tor/mod.rs index 26b16ca9..53fa65ae 100644 --- a/src/transport/tor/mod.rs +++ b/src/transport/tor/mod.rs @@ -771,6 +771,9 @@ impl TorTransport { mtu, recv_stats, Direction::Outbound, + // An outbound connection holds no capped inbound slot and is + // not gated on an accept-loop insert. + None, None, ) .await; @@ -937,6 +940,9 @@ impl TorTransport { mtu, recv_stats, Direction::Outbound, + // An outbound connection holds no capped inbound slot and is + // not gated on an accept-loop insert. + None, None, ) .await; @@ -1053,7 +1059,9 @@ impl Transport for TorTransport { /// /// `first_frame_timeout` is `Some` for an inbound connection, which holds a /// capped pool slot from the moment it is accepted, and `None` for an -/// outbound one, which holds no such slot. +/// outbound one, which holds no such slot. `ready_rx`, when present, is the +/// accept loop's readiness barrier: the loop must not run its cleanup before +/// the accept loop has inserted the pool entry and bumped its counter. #[allow(clippy::too_many_arguments)] async fn tor_receive_loop( reader: tokio::net::tcp::OwnedReadHalf, @@ -1065,6 +1073,7 @@ async fn tor_receive_loop( stats: Arc, direction: Direction, first_frame_timeout: Option, + ready_rx: Option>, ) { proxied_receive_loop( reader, @@ -1076,6 +1085,7 @@ async fn tor_receive_loop( stats, "Tor", first_frame_timeout, + ready_rx, |stats, meta| match meta { Direction::Inbound => stats.record_pool_inbound_removed(), Direction::Outbound => stats.record_pool_outbound_removed(), @@ -1193,6 +1203,12 @@ async fn tor_accept_loop( let recv_addr = remote_addr.clone(); let recv_tx = packet_tx.clone(); + // Readiness barrier: the receive task must not reach its cleanup path + // before the pool insert and counter bump below, or it would remove + // nothing and leave an orphaned entry with a permanently incremented + // inbound counter. + let (ready_tx, ready_rx) = tokio::sync::oneshot::channel(); + let recv_task = tokio::spawn(async move { tor_receive_loop( read_half, @@ -1204,6 +1220,7 @@ async fn tor_accept_loop( recv_stats, Direction::Inbound, Some(first_frame_timeout), + Some(ready_rx), ) .await; }); @@ -1216,14 +1233,31 @@ async fn tor_accept_loop( meta: Direction::Inbound, }; - { + let evicted = { let mut pool_guard = pool.lock().await; - pool_guard.insert(remote_addr.clone(), conn); + pool_guard.insert(remote_addr.clone(), conn) + }; + + if let Some(old) = evicted { + // A reused ephemeral forward port can collide with an entry whose + // receive task has not finished cleaning up. Abort it and release + // its slot here: left alone it would later remove the entry we + // just inserted and decrement for it, leaking one slot and + // orphaning a live connection. + old.recv_task.abort(); + match old.meta { + Direction::Inbound => stats.record_pool_inbound_removed(), + Direction::Outbound => stats.record_pool_outbound_removed(), + } } stats.record_connection_accepted(); stats.record_pool_inbound_added(); + // Release the receive task now that both the pool entry and the + // inbound counter are in place. + let _ = ready_tx.send(()); + debug!( transport_id = %transport_id, peer_addr = %peer_addr, @@ -2051,4 +2085,230 @@ mod tests { drop(peer); accept.abort(); } + + // ======================================================================== + // Accept-loop readiness barrier + // ======================================================================== + + /// Build a throwaway `OwnedWriteHalf` for a hand-planted pool entry. + async fn spare_write_half() -> OwnedWriteHalf { + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let addr = listener.local_addr().unwrap(); + let client = TcpStream::connect(addr).await.unwrap(); + let (_server, _) = listener.accept().await.unwrap(); + let (_read, write) = client.into_split(); + write + } + + /// The accept loop can be torn down between the pool insert and the + /// `ready_tx.send()`: the sender is dropped, so `ready_rx.await` returns + /// `Err`. The receive loop must still fall through to its cleanup, or the + /// pooled entry and its inbound-counter increment are stranded with no + /// task left to undo them. A bare `return` on the error path fails both + /// assertions below. + #[tokio::test] + async fn onion_receive_loop_cleans_up_when_readiness_signal_is_dropped() { + let (tx, _rx) = packet_channel(10); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let listen = listener.local_addr().unwrap(); + let client = TcpStream::connect(listen).await.unwrap(); + let (server, peer_addr) = listener.accept().await.unwrap(); + let remote = TransportAddr::from_string(&peer_addr.to_string()); + let (read_half, write_half) = server.into_split(); + + let pool: ProxiedPool = Arc::new(Mutex::new(HashMap::new())); + let stats = Arc::new(TorStats::new()); + pool.lock().await.insert( + remote.clone(), + ProxiedConnection { + writer: Arc::new(Mutex::new(write_half)), + recv_task: tokio::spawn(async {}), + mtu: 1400, + established_at: Instant::now(), + meta: Direction::Inbound, + }, + ); + stats.record_pool_inbound_added(); + assert_eq!(stats.pool_inbound_count(), 1); + + let (ready_tx, ready_rx) = tokio::sync::oneshot::channel::<()>(); + drop(ready_tx); + + tor_receive_loop( + read_half, + TransportId::new(1), + remote.clone(), + tx, + pool.clone(), + 1400, + stats.clone(), + Direction::Inbound, + Some(Duration::from_millis(50)), + Some(ready_rx), + ) + .await; + + assert!( + pool.lock().await.is_empty(), + "an aborted accept must not strand a pool entry" + ); + assert_eq!( + stats.pool_inbound_count(), + 0, + "an aborted accept must not strand an inbound-counter increment" + ); + drop(client); + } + + /// The losing interleaving, constructed rather than raced for: the peer is + /// already gone when the receive task starts, so without the barrier the + /// task runs its cleanup against an empty pool, removes nothing, and the + /// accept loop's increment lands afterwards and is never undone. + /// + /// Break-check: pass `None` for `ready_rx`, or delete the `.await` on the + /// barrier in the shared loop, and the spawned task completes immediately. + /// The first assertion to go red is the "still parked" timeout below; + /// the pool-empty and count-zero assertions red after it. + #[tokio::test] + async fn inbound_slot_is_released_when_the_peer_dies_before_the_pool_insert() { + let (tx, _rx) = packet_channel(10); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let listen = listener.local_addr().unwrap(); + let client = TcpStream::connect(listen).await.unwrap(); + let (server, peer_addr) = listener.accept().await.unwrap(); + let remote = TransportAddr::from_string(&peer_addr.to_string()); + drop(client); + let (read_half, write_half) = server.into_split(); + + let pool: ProxiedPool = Arc::new(Mutex::new(HashMap::new())); + let stats = Arc::new(TorStats::new()); + let (ready_tx, ready_rx) = tokio::sync::oneshot::channel::<()>(); + + let recv_pool = pool.clone(); + let recv_stats = stats.clone(); + let recv_addr = remote.clone(); + let mut handle = tokio::spawn(async move { + tor_receive_loop( + read_half, + TransportId::new(1), + recv_addr, + tx, + recv_pool, + 1400, + recv_stats, + Direction::Inbound, + Some(Duration::from_secs(5)), + Some(ready_rx), + ) + .await; + }); + + assert!( + tokio::time::timeout(Duration::from_millis(100), &mut handle) + .await + .is_err(), + "the receive loop must stay parked on the barrier until the accept tail runs" + ); + + // The accept loop's tail, in order: insert, count, release. + pool.lock().await.insert( + remote.clone(), + ProxiedConnection { + writer: Arc::new(Mutex::new(write_half)), + recv_task: tokio::spawn(async {}), + mtu: 1400, + established_at: Instant::now(), + meta: Direction::Inbound, + }, + ); + stats.record_pool_inbound_added(); + let _ = ready_tx.send(()); + + handle.await.unwrap(); + + assert!( + pool.lock().await.is_empty(), + "the receive loop must remove the entry the accept loop inserted" + ); + assert_eq!( + stats.pool_inbound_count(), + 0, + "a peer that dies before the pool insert must not leak its inbound slot" + ); + } + + /// A reused ephemeral forward port can land a second accept on an address + /// whose previous entry has not finished cleaning up. The accept loop must + /// release the evicted entry's slot: left alone, the old receive task + /// later removes the entry the new one just inserted and decrements once, + /// leaking a slot and orphaning a live connection. + /// + /// Break-check: drop the eviction arm in `tor_accept_loop` and the final + /// count is 2 rather than 1. + #[tokio::test] + async fn colliding_pool_key_releases_the_slot_of_the_entry_it_evicts() { + use socket2::{Domain, Socket, Type}; + + let (tx, _rx) = packet_channel(10); + let listener = TcpListener::bind("127.0.0.1:0").await.unwrap(); + let listen = listener.local_addr().unwrap(); + + // Bind the client socket first so its address is known before the + // accept loop ever sees it: that makes the collision deterministic + // instead of waiting for the kernel to reuse a port. + let sock = Socket::new(Domain::IPV4, Type::STREAM, None).unwrap(); + sock.bind(&"127.0.0.1:0".parse::().unwrap().into()) + .unwrap(); + let client_addr = sock.local_addr().unwrap().as_socket().unwrap(); + let remote = TransportAddr::from_string(&client_addr.to_string()); + + let pool: ProxiedPool = Arc::new(Mutex::new(HashMap::new())); + let stats = Arc::new(TorStats::new()); + + // The stale entry: a receive task that never finishes, so nothing + // removes it before the colliding accept arrives. + pool.lock().await.insert( + remote.clone(), + ProxiedConnection { + writer: Arc::new(Mutex::new(spare_write_half().await)), + recv_task: tokio::spawn(std::future::pending::<()>()), + mtu: 1400, + established_at: Instant::now(), + meta: Direction::Inbound, + }, + ); + stats.record_pool_inbound_added(); + + let accept = tokio::spawn(tor_accept_loop( + listener, + TransportId::new(1), + tx, + pool.clone(), + 1400, + 64, + Duration::from_secs(5), + stats.clone(), + )); + + sock.connect(&listen.into()).unwrap(); + + assert!( + wait_until( + || stats.snapshot().connections_accepted == 1, + Duration::from_secs(2) + ) + .await, + "the colliding connection should have been accepted" + ); + + assert_eq!( + stats.pool_inbound_count(), + 1, + "evicting a stale entry must release its slot, not stack a second one" + ); + assert_eq!(pool.lock().await.len(), 1); + + accept.abort(); + drop(sock); + } } From 6eff0accce74c0302a9d28734d00de41e259a859 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 22 Aug 2026 20:29:53 +0100 Subject: [PATCH 23/23] Confine the profiler capture directory to the log root The directory given to `profile tick on` travelled from the control socket straight into a root create_dir_all with no validation. The socket is reachable by the fips group, which docs/reference/security.md writes down as strictly weaker than root, so a group member could create a root-owned directory anywhere on the filesystem, including a path a later privileged component reads. A privileged daemon now resolves --dir against /var/log/fips: absolute, no `..` component, and still under the root once every existing ancestor has been followed through its symlinks, so a symlinked parent cannot launder a lexically clean path. Plain canonicalize cannot do this on its own, since the directory being created does not exist yet; the resolver canonicalizes the deepest existing ancestor and re-appends the tail. An unprivileged daemon crosses no boundary and takes --dir as given, which keeps the documented non-root `cargo run` capture working. Whether the process is privileged is a parameter rather than a geteuid() call inside the resolver, so the rules are exercised without depending on the uid of whoever ran the tests, and the lifecycle tests use a start_in that takes an already-resolved directory for the same reason. The sink itself is no longer written over whatever is at its path. The name is a one-second UTC stamp and therefore predictable, so File::create would have followed a symlink pre-planted at the next name and truncated any real file it found. The file is created only if it does not exist, and a capture started in the same second as a previous one takes the next free suffix rather than failing: stop-then-start inside one second is an ordinary sequence that used to succeed by truncating. Capture files are created private to their owner and the capture directory no longer inherits a permissive umask, since a capture carries the node npub, build, platform and a timing series. Confining a root daemon to a different log root is no longer possible and cli-fipsctl.md drops that use of --dir. This affects only a --features profiling build; the subcommand is absent from a stock package. --- CHANGELOG.md | 29 +++ docs/reference/cli-fipsctl.md | 17 +- src/instr/capture.rs | 371 ++++++++++++++++++++++++++++++++-- 3 files changed, 399 insertions(+), 18 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 14010373..9f3285da 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -540,6 +540,35 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 other packaging paths did, so the contact of record in the new `.pkg` artifact would have been unreachable at its first release. +### Security + +#### Tick profiler + +- The `--dir` given to `profile tick on` is now confined to `/var/log/fips` + when the daemon runs as root. The control socket is reachable by the `fips` + group, which the security model writes down as strictly weaker than root, yet + the directory travelled from the socket straight into a root `create_dir_all` + with no validation: a group member could create a root-owned directory + anywhere on the filesystem, including a path a later privileged component + reads. The path must now be absolute, must contain no `..`, and must still + resolve under the root once every existing ancestor has been followed through + its symlinks, so a symlinked parent does not launder a lexically clean path. + A daemon that is not running as root crosses no such boundary and takes + `--dir` as given, which keeps the documented non-root `cargo run` capture + working; the flag can no longer point a root daemon at a different log root, + which was its other documented use. This only ever affected a + `--features profiling` build: the subcommand is absent from a stock package. + +- A capture file is no longer written over whatever is already at its path. The + name is a one-second UTC stamp and therefore predictable, so `File::create` + would have followed a symlink pre-planted at the next name, and it truncated + any real file it found. The file is created only if it does not exist, and a + capture started in the same second as a previous one takes the next free + `-N` suffix instead of failing. Capture files are created private to their + owner and the capture directory is no longer left world-accessible by a + permissive umask; a capture carries the node npub, build, platform and a + timing series. + ## [0.4.2] - unreleased ### Added diff --git a/docs/reference/cli-fipsctl.md b/docs/reference/cli-fipsctl.md index f6f5a195..ffa26936 100644 --- a/docs/reference/cli-fipsctl.md +++ b/docs/reference/cli-fipsctl.md @@ -241,9 +241,22 @@ package does not carry this subcommand and a stock daemon reports | Flag | Argument | Default | Description | | ---- | -------- | ------- | ----------- | -| `--dir` | directory path | `/var/log/fips` | Where to write the capture. Created if absent. Use it to profile a non-root `cargo run`, or on a platform whose log root differs. | +| `--dir` | directory path | `/var/log/fips` | Where to write the capture. Created if absent. Use it to profile a non-root `cargo run`. | -One file is written per capture, named `profile-.tsv`. +Against a daemon running as root, `--dir` must name an absolute path +under `/var/log/fips`, and a path that resolves outside it through a +symlinked parent is refused. The control socket is reachable by the +`fips` group, which the security model treats as strictly weaker than +root, so a group member cannot steer a root directory creation at an +arbitrary path. A daemon that is not running as root crosses no such +boundary and takes `--dir` as given, which is what keeps a non-root +`cargo run` capture working. Confining a root daemon's captures to a +different log root is not currently supported. + +One file is written per capture, named `profile-.tsv`, +with `-1`, `-2` and so on appended if a capture in the same second +already claimed the name. The capture file is created private to its +owner and is never written over an existing file. It opens with a `#`-prefixed header block (node npub, build version, platform, configured tick period, flush interval, byte cap, start time), then a tab-separated column header, then one row per measured diff --git a/src/instr/capture.rs b/src/instr/capture.rs index b07d3467..71b27e1e 100644 --- a/src/instr/capture.rs +++ b/src/instr/capture.rs @@ -12,7 +12,7 @@ use std::fs::File; use std::io::Write; -use std::path::PathBuf; +use std::path::{Component, Path, PathBuf}; use std::sync::Mutex; use std::sync::atomic::{AtomicBool, AtomicU8, AtomicU64, Ordering}; use std::time::{Duration, SystemTime, UNIX_EPOCH}; @@ -20,9 +20,27 @@ use std::time::{Duration, SystemTime, UNIX_EPOCH}; use super::recorder; use super::writer; -/// Default sink directory. Overridable per capture with `--dir`. +/// Default sink directory, and the root a privileged daemon confines `--dir` +/// to. Overridable per capture with `--dir`, within that root. pub(crate) const DEFAULT_DIR: &str = "/var/log/fips"; +/// How many same-second filename collisions a capture steps around before +/// giving up. Stop-then-start inside one second is an ordinary operator +/// sequence and the stamp has one-second granularity, so a bare failure there +/// would be a regression. Raising this only widens how many restarts one +/// second can hold; lowering it turns a fast restart back into an error. +const NAME_COLLISION_RETRIES: u32 = 16; + +/// Mode for a created capture file. A capture carries the node npub, build, +/// platform and a timing series, so it is not world-readable. +#[cfg(unix)] +const FILE_MODE: u32 = 0o600; + +/// Mode for a created capture directory. `mkdir` can only tighten this with +/// the umask, never loosen it. +#[cfg(unix)] +const DIR_MODE: u32 = 0o750; + /// Writer flush interval. pub(crate) const INTERVAL: Duration = Duration::from_secs(10); @@ -102,14 +120,133 @@ fn reap() { } } -/// Arm a capture. +/// True when the process runs with root's privileges. +/// +/// This is what makes an unconstrained `--dir` a privilege crossing: the +/// control socket is reachable by the `fips` group, which the security model +/// writes down as strictly weaker than root, so a group member must not be +/// able to steer a root `create_dir_all` at an arbitrary path. +#[cfg(unix)] +fn running_as_root() -> bool { + unsafe { libc::geteuid() == 0 } +} + +/// No control socket and no `fips` group off unix, so there is no weaker +/// principal to confine and the parameter keeps its original meaning. +#[cfg(not(unix))] +fn running_as_root() -> bool { + false +} + +/// Resolve a requested capture directory against the root it must stay under. +/// +/// `privileged` is a parameter rather than a `geteuid()` call so the rules can +/// be exercised without depending on the uid of whoever ran the tests. +/// +/// - No `--dir` keeps today's default, `root`. +/// - An unprivileged daemon crosses no boundary: it can already write wherever +/// the invoking user can, so the path is taken as given. This is what keeps +/// the documented non-root `cargo run` capture working. +/// - Otherwise the path must be absolute, must contain no `..` component, and +/// must resolve under `root` once every existing ancestor has been followed +/// through its symlinks. +/// +/// Accepted: resolution is check-then-act. A symlink planted between the check +/// here and the `create_dir_all` in `open_sink` would escape the root. Inside +/// `/var/log/fips` only root can plant one, which is the principal already +/// being trusted. +fn resolve_dir_under(dir: Option<&str>, root: &Path, privileged: bool) -> Result { + let Some(dir) = dir else { + return Ok(root.to_path_buf()); + }; + if !privileged { + return Ok(PathBuf::from(dir)); + } + + let requested = Path::new(dir); + if !requested.is_absolute() { + return Err(format!( + "profile directory must be an absolute path under {}: {dir}", + root.display() + )); + } + if requested + .components() + .any(|c| matches!(c, Component::ParentDir)) + { + return Err(format!( + "profile directory must not contain `..` components: {dir}" + )); + } + + let resolved = resolve_through_existing(requested)?; + let root = resolve_through_existing(root)?; + if !resolved.starts_with(&root) { + return Err(format!( + "profile directory must be under {}: {dir}", + root.display() + )); + } + Ok(resolved) +} + +/// Canonicalize the deepest existing ancestor of `path` and re-append the tail +/// that does not exist yet. +/// +/// Plain `canonicalize` returns `NotFound` for the directory a capture is +/// being asked to create, which is the ordinary case, so it cannot be used on +/// its own. Following the existing ancestors is what catches a lexically clean +/// path whose parent is a symlink out of the root. +fn resolve_through_existing(path: &Path) -> Result { + let mut tail: Vec = Vec::new(); + let mut probe = path.to_path_buf(); + loop { + if let Ok(base) = probe.canonicalize() { + let mut out = base; + for part in tail.iter().rev() { + out.push(part); + } + return Ok(out); + } + let Some(name) = probe.file_name().map(|n| n.to_os_string()) else { + return Err(format!( + "cannot resolve profile directory {}", + path.display() + )); + }; + tail.push(name); + if !probe.pop() { + return Err(format!( + "cannot resolve profile directory {}", + path.display() + )); + } + } +} + +/// Arm a capture, confining `--dir` to the capture root. /// /// Opens the sink first and only then starts the writer, so a bad `--dir` is -/// reported to the caller rather than logged into the void. +/// reported to the caller rather than logged into the void. The directory is +/// resolved before the capture slot is claimed, so a rejected path leaves the +/// slot free. pub(crate) fn start( dir: Option<&str>, node_npub: &str, tick_period_secs: u64, +) -> Result { + let dir = resolve_dir_under(dir, Path::new(DEFAULT_DIR), running_as_root())?; + start_in(&dir, node_npub, tick_period_secs) +} + +/// Arm a capture in an already-resolved directory. +/// +/// Split out from `start` so the confinement rules live in one place and the +/// lifecycle can be exercised without them. +pub(crate) fn start_in( + dir: &Path, + node_npub: &str, + tick_period_secs: u64, ) -> Result { claim()?; @@ -210,25 +347,80 @@ pub(crate) fn shutdown() { } } +/// Create the capture directory, giving it a mode rather than inheriting +/// whatever a permissive umask allows. +fn create_capture_dir(dir: &Path) -> std::io::Result<()> { + let mut builder = std::fs::DirBuilder::new(); + builder.recursive(true); + #[cfg(unix)] + { + use std::os::unix::fs::DirBuilderExt; + builder.mode(DIR_MODE); + } + builder.create(dir) +} + +/// Create the capture file, refusing to follow or truncate anything already at +/// the path. +/// +/// The basename is a one-second UTC stamp and therefore predictable, so +/// `File::create` would follow a symlink pre-planted at the next name. +/// `create_new` refuses any existing entry; a same-second restart, which used +/// to succeed by silently truncating the previous capture, steps to the next +/// free suffix instead of failing. +fn create_capture_file(dir: &Path, stamp: &str) -> Result<(File, PathBuf), String> { + let mut opts = std::fs::OpenOptions::new(); + opts.write(true).create_new(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt; + opts.mode(FILE_MODE); + } + + let mut last = None; + for n in 0..=NAME_COLLISION_RETRIES { + let name = if n == 0 { + format!("profile-{stamp}.tsv") + } else { + format!("profile-{stamp}-{n}.tsv") + }; + let path = dir.join(name); + match opts.open(&path) { + Ok(file) => return Ok((file, path)), + Err(e) if e.kind() == std::io::ErrorKind::AlreadyExists => { + last = Some((path, e)); + } + Err(e) => { + return Err(format!( + "cannot create profile file {}: {e}", + path.display() + )); + } + } + } + + let (path, e) = last.expect("the loop runs at least once"); + Err(format!( + "cannot create profile file {}: {e}", + path.display() + )) +} + /// Create the sink file and write its header block. Returns the open file, its /// path, and the number of header bytes written. fn open_sink( - dir: Option<&str>, + dir: &Path, node_npub: &str, tick_period_secs: u64, ) -> Result<(File, PathBuf, u64), String> { - let dir = PathBuf::from(dir.unwrap_or(DEFAULT_DIR)); - std::fs::create_dir_all(&dir) + create_capture_dir(dir) .map_err(|e| format!("cannot use profile directory {}: {e}", dir.display()))?; let start_unix = SystemTime::now() .duration_since(UNIX_EPOCH) .map(|d| d.as_secs()) .unwrap_or(0); - let path = dir.join(format!("profile-{}.tsv", compact_utc(start_unix))); - - let mut file = File::create(&path) - .map_err(|e| format!("cannot create profile file {}: {e}", path.display()))?; + let (mut file, path) = create_capture_file(dir, &compact_utc(start_unix))?; let header = format!( "# fips tick profile\n\ @@ -330,15 +522,14 @@ mod tests { fn capture_round_trip_writes_header_and_rows() { let _guard = crate::instr::test_serial(); let dir = tempfile::tempdir().expect("tempdir"); - let dir_str = dir.path().to_str().unwrap().to_string(); - let started = start(Some(&dir_str), "npub1test", 1).expect("start"); + let started = start_in(dir.path(), "npub1test", 1).expect("start"); assert_eq!(started["state"], "running"); assert!(gate(), "gate must be armed while running"); let path = PathBuf::from(started["path"].as_str().unwrap()); // A second `on` is refused while one is running, and names the file. - let refused = start(Some(&dir_str), "npub1test", 1).unwrap_err(); + let refused = start_in(dir.path(), "npub1test", 1).unwrap_err(); assert!(refused.contains(&path.display().to_string()), "{refused}"); // Feed one observation so the drained rows are not all zero. @@ -400,14 +591,162 @@ mod tests { #[test] fn start_fails_loudly_on_an_unwritable_directory() { let _guard = crate::instr::test_serial(); - let err = start(Some("/proc/fips-profile-should-not-exist"), "npub1test", 1) - .expect_err("must fail"); + let err = start_in( + Path::new("/proc/fips-profile-should-not-exist"), + "npub1test", + 1, + ) + .expect_err("must fail"); assert!(err.contains("profile directory"), "{err}"); // The failed attempt must leave the slot free for the next try. assert_eq!(STATE.load(Ordering::Acquire), IDLE); assert!(!gate()); } + // ======================================================================== + // `--dir` confinement + // + // All of these force `privileged = true` rather than reading the uid, so + // their verdict does not depend on who ran the suite. + // ======================================================================== + + #[test] + fn profile_dir_rejects_a_parent_traversal_escape() { + let root = tempfile::tempdir().expect("tempdir"); + let escape = root.path().join("..").join("..").join("etc"); + let err = resolve_dir_under(Some(escape.to_str().unwrap()), root.path(), true) + .expect_err("a `..` escape must be refused"); + assert!(err.contains(".."), "{err}"); + } + + #[test] + fn profile_dir_rejects_a_relative_path() { + let root = tempfile::tempdir().expect("tempdir"); + let err = resolve_dir_under(Some("sub"), root.path(), true) + .expect_err("a relative path must be refused"); + assert!(err.contains("absolute"), "{err}"); + } + + #[test] + fn profile_dir_rejects_an_absolute_path_outside_the_root() { + let root = tempfile::tempdir().expect("tempdir"); + let err = resolve_dir_under(Some("/etc/cron.d"), root.path(), true) + .expect_err("a path outside the root must be refused"); + assert!(err.contains("must be under"), "{err}"); + } + + /// The lexically clean escape: every component is innocent and an existing + /// ancestor is a symlink pointing out of the root. + #[test] + #[cfg(unix)] + fn profile_dir_rejects_a_symlinked_ancestor_pointing_out_of_the_root() { + let root = tempfile::tempdir().expect("tempdir"); + let outside = tempfile::tempdir().expect("tempdir"); + let link = root.path().join("link"); + std::os::unix::fs::symlink(outside.path(), &link).expect("symlink"); + + let asked = link.join("x"); + let err = resolve_dir_under(Some(asked.to_str().unwrap()), root.path(), true) + .expect_err("a symlinked ancestor must not escape the root"); + assert!(err.contains("must be under"), "{err}"); + } + + #[test] + fn profile_dir_accepts_the_root_and_a_subdirectory_that_does_not_exist_yet() { + let root = tempfile::tempdir().expect("tempdir"); + resolve_dir_under(Some(root.path().to_str().unwrap()), root.path(), true) + .expect("the root itself must be allowed"); + + // Not yet created: the resolver must not depend on the path existing. + let fresh = root.path().join("run-1"); + let got = resolve_dir_under(Some(fresh.to_str().unwrap()), root.path(), true) + .expect("a subdirectory that does not exist yet must be allowed"); + assert!(got.ends_with("run-1"), "{}", got.display()); + } + + #[test] + fn profile_dir_defaults_to_the_root_when_none_is_given() { + let root = tempfile::tempdir().expect("tempdir"); + let got = resolve_dir_under(None, root.path(), true).expect("the default must be allowed"); + assert_eq!(got, root.path()); + } + + #[test] + fn profile_dir_takes_the_unprivileged_bypass() { + let root = tempfile::tempdir().expect("tempdir"); + let got = resolve_dir_under(Some("/anywhere/at/all"), root.path(), false) + .expect("an unprivileged daemon crosses no boundary"); + assert_eq!(got, PathBuf::from("/anywhere/at/all")); + } + + /// A pre-planted entry at the predicted capture name must not be followed + /// or truncated. The stamp is one-second granular, so the name is + /// guessable to within a second. + #[test] + fn profile_sink_does_not_truncate_a_preexisting_path_at_the_capture_name() { + let _guard = crate::instr::test_serial(); + let dir = tempfile::tempdir().expect("tempdir"); + let start_unix = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map(|d| d.as_secs()) + .unwrap_or(0); + let squatted = dir.path().join(format!( + "profile-{}.tsv", + compact_utc(start_unix.saturating_sub(1)) + )); + std::fs::write(&squatted, b"do not truncate me").expect("plant"); + + // Name the same stamp the squatter used, so the sink collides with it. + let (_file, path) = + create_capture_file(dir.path(), &compact_utc(start_unix.saturating_sub(1))) + .expect("the sink must step around the collision"); + + assert_ne!(path, squatted, "the sink must not reuse the existing name"); + assert_eq!( + std::fs::read_to_string(&squatted).expect("read"), + "do not truncate me", + "an existing file at the capture name must survive" + ); + } + + /// Stop-then-start inside one second is an ordinary operator sequence and + /// must produce a second capture, not an error. + #[test] + fn profile_sink_uniquifies_a_same_second_restart() { + let _guard = crate::instr::test_serial(); + let dir = tempfile::tempdir().expect("tempdir"); + let (_first, first_path) = create_capture_file(dir.path(), "20260727T191500Z").unwrap(); + let (_second, second_path) = create_capture_file(dir.path(), "20260727T191500Z") + .expect("a same-second restart must not fail"); + assert_ne!(first_path, second_path); + assert!(first_path.exists() && second_path.exists()); + } + + #[test] + #[cfg(unix)] + fn profile_sink_creates_a_private_file_and_directory() { + use std::os::unix::fs::PermissionsExt; + + let _guard = crate::instr::test_serial(); + let root = tempfile::tempdir().expect("tempdir"); + let dir = root.path().join("captures"); + create_capture_dir(&dir).expect("create dir"); + let (_file, path) = create_capture_file(&dir, "20260727T191500Z").expect("create file"); + + // The umask can only tighten what `mkdir`/`open` were given, so the + // assertion is on the bits that must be absent rather than equality. + assert_eq!( + std::fs::metadata(&dir).unwrap().permissions().mode() & 0o007, + 0, + "the capture directory must not be world-accessible" + ); + assert_eq!( + std::fs::metadata(&path).unwrap().permissions().mode() & 0o077, + 0, + "the capture file must be private to its owner" + ); + } + #[test] fn status_reports_the_bounds_it_is_enforcing() { let _guard = crate::instr::test_serial();