diff --git a/CHANGELOG.md b/CHANGELOG.md index 2ae336ff..db74e321 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -139,6 +139,35 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 validation, since it refuses every inbound offer rather than disabling the limit. Existing configurations parse unchanged, the key being optional. +- The UDP transport's listen socket descriptor can now be handed to an + embedder, for hosts that associate a socket with one interface or network + and steer inbound traffic by that association rather than routing by + destination address. On such a host a peer reachable only over a secondary + network fails in a way FIPS can neither see nor fix: the address is + well-formed, the send succeeds, the peer replies, and the host discards the + reply before it reaches our socket, so the link retries msg1 forever with no + error surfaced anywhere. The correction is a socket option chosen against + host state FIPS has no basis to reason about, so the descriptor goes to + whoever does. Call `Node::enable_app_owned_udp_fd()` after `Node::new` and + before `start()`, and read `AppOwnedUdpSocket { instance, fd }` off the + returned channel once the transport is up, following the existing + `enable_app_owned_tun` contract. One message is sent per UDP transport that + binds, so a multi-listener configuration yields all of them; nothing is sent + when no UDP transport is configured or one fails to bind, so an embedder + tells "no socket" from "here is the socket" by the receive timing out. The + `instance` field is the name the listener was configured under, `None` for a + single unnamed instance, and it is what makes more than one listener usable: + transports are created by iterating a map, so arrival order is luck, and an + embedder whose whole purpose is to bind one socket to one network would + otherwise have to guess which socket it just received. Guessing wrong pins + one lane's socket to the other lane's network, which is the failure the seam + exists to correct. FIPS keeps owning the socket, and + the descriptor carries no promise beyond "this is the transport's socket, + and it is open now". Two limits: the per-peer connected-UDP sockets that + Linux and macOS open after `start()` returns are not covered, and a + transport that adopts a socket handed in by the traversal bootstrap does not + fire the seam. Unix only, since the Windows UDP backend has no descriptor. + ### Changed - Node health is determined at start completion instead of unconditionally diff --git a/src/lib.rs b/src/lib.rs index edb3a6f9..dcd35027 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -100,4 +100,6 @@ pub use proto::fmp::{PromotionResult, cross_connection_winner}; pub use peer::{ActivePeer, ConnectivityState, PeerError}; // Re-export node types +#[cfg(unix)] +pub use node::AppOwnedUdpSocket; pub use node::{Node, NodeError, NodeState, UpdatePeersOutcome}; diff --git a/src/node/lifecycle/mod.rs b/src/node/lifecycle/mod.rs index 1e645126..8c54a351 100644 --- a/src/node/lifecycle/mod.rs +++ b/src/node/lifecycle/mod.rs @@ -1485,6 +1485,21 @@ impl Node { match handle.start().await { Ok(()) => { + // Hand the freshly-bound socket to an embedder that + // armed `enable_app_owned_udp_fd`, labelled with the + // instance it belongs to so two UDP listeners can be + // told apart. Non-UDP handles report `None` for the + // fd by construction, so no transport-type test is + // needed here. + #[cfg(unix)] + if let (Some(tx), Some(fd)) = + (&self.supervisor.udp_fd_tx, handle.raw_fd()) + { + let _ = tx.send(crate::node::AppOwnedUdpSocket { + instance: name.clone(), + fd, + }); + } self.transports.insert(id, handle); Event::SubstrateUp { child } } diff --git a/src/node/lifecycle/supervisor.rs b/src/node/lifecycle/supervisor.rs index 0f3c39cf..af5197da 100644 --- a/src/node/lifecycle/supervisor.rs +++ b/src/node/lifecycle/supervisor.rs @@ -698,6 +698,14 @@ pub(crate) struct Supervisor { /// [`Node::dns_local_addr`](crate::Node::dns_local_addr). pub(in crate::node) dns_local_addr: Option, + /// Sender for each UDP listen socket the transport spawn binds — its raw + /// fd and the instance name it was configured under — armed by + /// [`Node::enable_app_owned_udp_fd`](crate::Node::enable_app_owned_udp_fd) + /// and fired from the transport-spawn arm of `start()` once the socket is + /// bound. `None` unless the embedder armed it. + #[cfg(unix)] + pub(in crate::node) udp_fd_tx: Option>, + /// Node-side driver state for the Nostr overlay peer-rendezvous /// subsystem: the engine handle, its startup timestamp, the one-shot /// startup-sweep latch, and the per-peer bootstrap-transport bookkeeping @@ -743,6 +751,8 @@ impl Supervisor { dns_identity_rx: None, dns_task: None, dns_local_addr: None, + #[cfg(unix)] + udp_fd_tx: None, nostr_rendezvous: crate::nostr::RendezvousDriver::default(), lan_rendezvous: None, #[cfg(unix)] diff --git a/src/node/mod.rs b/src/node/mod.rs index 5d22f067..a58f997d 100644 --- a/src/node/mod.rs +++ b/src/node/mod.rs @@ -245,6 +245,37 @@ pub struct UpdatePeersOutcome { pub unchanged: usize, } +/// One bound UDP listen socket, handed to an embedder that armed +/// [`Node::enable_app_owned_udp_fd`]. +/// +/// A bare descriptor would be enough for the single-listener case and useless +/// for any other: a node configured with several named UDP instances +/// ([`TransportInstances::Named`](crate::config::TransportInstances::Named)) +/// delivers one message per instance, and the whole point of the seam — the +/// embedder associating a socket with one host network — needs to know *which* +/// socket it is holding. Naming it here rather than making the embedder infer +/// it from arrival order is deliberate: transports are created from a +/// `HashMap`, so arrival order carries no meaning, and guessing wrong pins a +/// lane's socket to another lane's network, which is precisely the fault this +/// seam exists to correct. +/// +/// A struct rather than a tuple so the receiving side reads as +/// `socket.instance` / `socket.fd`, and so a future addition (the bound local +/// address, say) does not break every embedder. +#[cfg(unix)] +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct AppOwnedUdpSocket { + /// The configured instance name this listener was built from — the key in + /// a `Named` UDP config, and the same name a peer address qualifies its + /// transport field with (`"udp/aware"`, see + /// [`TransportSpec`](crate::config::TransportSpec)). `None` for a + /// `Single` config, which has no name to give. + pub instance: Option, + /// The bound socket's raw descriptor. Borrowed, not owned: FIPS keeps the + /// socket, and the fd is valid only while the transport is running. + pub fd: std::os::unix::io::RawFd, +} + /// Key for addr_to_link reverse lookup. type AddrKey = (TransportId, TransportAddr); @@ -3013,6 +3044,77 @@ impl Node { (outbound_tx, tun_rx) } + /// Set up an **app-owned UDP socket option**: FIPS keeps the socket, and + /// the embedder gets its raw fd so it can apply a host socket option FIPS + /// has no basis to choose. Call this after [`Node::new`] and **before** + /// [`Self::start`] — the fd does not exist until the transport binds. + /// + /// The UDP transport binds one socket and selects the egress path per + /// destination address, which assumes the host routes by destination + /// alone. Not every host does. Where each socket is instead associated + /// with exactly one network interface (or "network") and inbound traffic + /// is steered by that association, a peer reachable only over a secondary, + /// non-default network is unreachable in a way FIPS can neither see nor + /// correct: the address is well-formed, the send succeeds, the peer + /// receives our handshake and replies, and the host discards the reply + /// before it reaches our socket. Handshake msg1 then retries forever with + /// no error surfaced anywhere. Correcting it is a `setsockopt`-class call + /// against host-specific network state, on a descriptor the transport + /// otherwise keeps entirely private. + /// + /// ```no_run + /// # async fn f(node: &mut fips::Node) -> Result<(), Box> { + /// let rx = node.enable_app_owned_udp_fd(); // after new(), before start() + /// node.start().await?; + /// let socket = rx.recv_timeout(std::time::Duration::from_secs(1))?; + /// # let _ = (socket.instance, socket.fd); Ok(()) + /// # } + /// ``` + /// + /// One message is sent per UDP transport that successfully binds — the + /// usual single-listener configuration therefore yields exactly one, while + /// a config with several named UDP listeners yields one per listener, all + /// of which an embedder pinning sockets to a network needs. Each message + /// carries the instance name its listener was configured under + /// ([`AppOwnedUdpSocket::instance`]), which is the only thing that tells + /// two otherwise-identical descriptors apart: pinning the wrong socket to + /// the wrong network is exactly the fault this seam exists to let an + /// embedder fix, so an fd is never handed over unlabelled. A + /// [`Self::stop`] followed by another [`Self::start`] delivers the new + /// socket's fd on the same channel, since that is a genuinely different + /// descriptor. Nothing at all is sent when no UDP transport is configured + /// or a configured one fails to bind, so a receive that times out is how + /// an embedder tells "no socket" from "here is the socket". Calling this + /// twice replaces the first arming: the last receiver wins. + /// + /// Scope, stated precisely so it is not read as more than it is: + /// + /// - The fd is the transport's wildcard listen socket. On targets that + /// also run the per-peer connected-UDP fast path (Linux and macOS), the + /// additional `connect()`-ed sockets that path opens per established + /// peer, after `start()` has returned, are not covered by this seam. On + /// targets without that path the wildcard socket is the only UDP socket + /// the transport opens. + /// - Transports that adopt a socket supplied from outside (the + /// NAT-traversal bootstrap handoff) do not fire this, since whoever + /// supplied the socket already held its fd and could bind it before + /// handover. + /// - FIPS retains ownership. The fd is a borrow valid while the node is + /// running; using it after the transport stops can touch an unrelated + /// reused descriptor. + /// + /// Unix-only: `RawFd` is a unix concept and the Windows UDP backend has no + /// descriptor. The channel is a [`std::sync::mpsc`] one because the + /// embedder is not necessarily on a tokio runtime; the sending end lives on + /// the supervisor as `udp_fd_tx` and [`Self::start`] fires it from the + /// transport-spawn arm, right after the handle reports a successful start. + #[cfg(unix)] + pub fn enable_app_owned_udp_fd(&mut self) -> std::sync::mpsc::Receiver { + let (udp_fd_tx, udp_fd_rx) = std::sync::mpsc::channel(); + self.supervisor.udp_fd_tx = Some(udp_fd_tx); + udp_fd_rx + } + /// Address the built-in `.fips` DNS responder is listening on, or `None` /// when it is not running (`dns.enabled = false`, the bind failed, or the /// node is stopped). diff --git a/src/node/tests/unit.rs b/src/node/tests/unit.rs index 1d1f5cf9..e35061a2 100644 --- a/src/node/tests/unit.rs +++ b/src/node/tests/unit.rs @@ -2981,6 +2981,240 @@ async fn start_skips_system_tun_when_app_owned() { node.stop().await.unwrap(); } +/// Config for the app-owned-UDP-fd tests: one loopback UDP transport on an +/// ephemeral port and no DNS, mirroring `make_healthy_node`. +#[cfg(unix)] +fn udp_loopback_config() -> crate::Config { + let mut config = crate::Config::new(); + config.transports.udp = crate::config::TransportInstances::Single(crate::config::UdpConfig { + bind_addr: Some("127.0.0.1:0".to_string()), + ..Default::default() + }); + config.dns.enabled = false; + config +} + +/// App-owned UDP fd seam: the embedder gets the descriptor of the socket the +/// transport actually bound, and gets it only once the bind has happened — +/// there is no fd to hand out before `start()`. +#[cfg(unix)] +#[tokio::test] +async fn app_owned_udp_fd_seam_delivers_the_bound_socket() { + let mut node = make_node_with(udp_loopback_config()); + let rx = node.enable_app_owned_udp_fd(); + + assert!( + rx.try_recv().is_err(), + "nothing is delivered at arm time — the socket is not bound until start()", + ); + + node.start().await.unwrap(); + + let socket = rx + .try_recv() + .expect("the seam fires once the UDP socket is bound"); + let live_fd = node + .transports + .values() + .next() + .expect("the loopback UDP transport came up") + .raw_fd(); + assert_eq!( + Some(socket.fd), + live_fd, + "the delivered fd must be the live transport's socket, not some other descriptor", + ); + assert_eq!( + socket.instance, None, + "a `Single` UDP config has no instance name to report", + ); + + assert!( + rx.try_recv().is_err(), + "one UDP transport bound means exactly one delivery", + ); + + node.stop().await.unwrap(); +} + +/// One message per UDP transport that binds: an embedder pinning sockets to a +/// network needs every listener's fd, not just the first, so the seam does not +/// latch after the first send. +#[cfg(unix)] +#[tokio::test] +async fn app_owned_udp_fd_seam_delivers_every_udp_listener_that_binds() { + let mut listeners = std::collections::HashMap::new(); + for name in ["main", "backup"] { + listeners.insert( + name.to_string(), + crate::config::UdpConfig { + bind_addr: Some("127.0.0.1:0".to_string()), + ..Default::default() + }, + ); + } + let mut config = crate::Config::new(); + config.transports.udp = crate::config::TransportInstances::Named(listeners); + config.dns.enabled = false; + + let mut node = make_node_with(config); + let rx = node.enable_app_owned_udp_fd(); + node.start().await.unwrap(); + + let mut delivered: Vec<_> = rx + .try_iter() + .map(|socket| (socket.instance, socket.fd)) + .collect(); + delivered.sort_unstable(); + let mut live: Vec<_> = node + .transports + .values() + .filter_map(|handle| { + handle + .raw_fd() + .map(|fd| (handle.name().map(str::to_string), fd)) + }) + .collect(); + live.sort_unstable(); + assert_eq!( + delivered, live, + "every UDP listener that bound must be handed out, not just the first", + ); + assert_eq!(delivered.len(), 2, "both named listeners bound"); + + // The label is what makes two descriptors usable: an embedder pins each + // socket to a different network, and arrival order — the transports come + // out of a `HashMap` — cannot tell it which is which. + let names: Vec<_> = delivered + .iter() + .map(|(instance, _)| instance.as_deref()) + .collect(); + assert!( + names.contains(&Some("main")) && names.contains(&Some("backup")), + "each fd names the configured instance it belongs to, got {names:?}", + ); + assert_ne!( + delivered[0].1, delivered[1].1, + "two instances are two distinct sockets", + ); + + node.stop().await.unwrap(); +} + +/// No UDP transport means no fd: the channel stays silent rather than +/// delivering a sentinel, so a receive that times out is how an embedder tells +/// "there is no socket" from "here is the socket". +#[cfg(unix)] +#[tokio::test] +async fn app_owned_udp_fd_seam_stays_silent_without_a_udp_transport() { + let mut config = crate::Config::new(); + config.dns.enabled = false; + let mut node = make_node_with(config); + let rx = node.enable_app_owned_udp_fd(); + + // A node with no transports configured fails bring-up with + // `NoOperationalTransports`; asserted so this test cannot silently stop + // exercising the no-UDP path if that outcome ever changes. + let started = node.start().await; + assert!( + started.is_err(), + "a transportless node has no operational transports", + ); + + assert!( + rx.try_recv().is_err(), + "no UDP transport means no fd is ever delivered", + ); +} + +/// A UDP transport that never bound has no fd to hand out. The bind address is +/// deliberately unparseable — a busy port would not do it, since +/// `UdpRawSocket::open` sets `SO_REUSEADDR`/`SO_REUSEPORT` before binding. +#[cfg(unix)] +#[tokio::test] +async fn app_owned_udp_fd_seam_stays_silent_when_the_udp_transport_fails_to_start() { + let mut config = crate::Config::new(); + config.transports.udp = crate::config::TransportInstances::Single(crate::config::UdpConfig { + bind_addr: Some("not-a-socket-addr".to_string()), + ..Default::default() + }); + config.dns.enabled = false; + let mut node = make_node_with(config); + let rx = node.enable_app_owned_udp_fd(); + + // Transport-start failure is warn-and-continue; the node's overall start + // outcome is not what this test pins. + let _ = node.start().await; + + assert!( + rx.try_recv().is_err(), + "a UDP transport that never bound has no fd to hand out", + ); +} + +/// The seam is per-`Node` state with a fresh channel per arming, so an embedder +/// that tears the mesh down and rebuilds the node — a radio off→on cycle — gets +/// the new socket on the new node's channel, with nothing shared between them. +#[cfg(unix)] +#[tokio::test] +async fn app_owned_udp_fd_seam_rearms_on_a_rebuilt_node() { + let mut node_a = make_node_with(udp_loopback_config()); + let rx_a = node_a.enable_app_owned_udp_fd(); + node_a.start().await.unwrap(); + rx_a.try_recv() + .expect("node A's socket reached node A's rx"); + node_a.stop().await.unwrap(); + drop(node_a); + + let mut node_b = make_node_with(udp_loopback_config()); + let rx_b = node_b.enable_app_owned_udp_fd(); + node_b.start().await.unwrap(); + rx_b.try_recv() + .expect("node B's socket reached node B's rx"); + + // Nothing from B reaches A's channel. (A is dropped, so this is + // `Disconnected` rather than `Empty`; the specific variant is not part of + // the contract, only that no fd arrives.) + assert!( + rx_a.try_recv().is_err(), + "the channels are per-node — no shared or global arming state", + ); + + node_b.stop().await.unwrap(); +} + +/// Arming twice on the same node replaces the first arming: the last receiver +/// wins and the earlier one is disconnected. Asserted by firing through the +/// installed sender directly, so the test needs no real bind. +#[cfg(unix)] +#[test] +fn app_owned_udp_fd_seam_second_arm_replaces_the_first() { + let mut node = make_node_with(udp_loopback_config()); + let rx1 = node.enable_app_owned_udp_fd(); + let rx2 = node.enable_app_owned_udp_fd(); + + let sent = crate::node::AppOwnedUdpSocket { + instance: Some("aware".to_string()), + fd: 7, + }; + node.supervisor + .udp_fd_tx + .as_ref() + .expect("the second arming installed a sender") + .send(sent.clone()) + .expect("the surviving receiver is live"); + + assert_eq!( + rx2.try_recv().ok(), + Some(sent), + "the last receiver armed is the one the node feeds", + ); + assert!( + rx1.try_recv().is_err(), + "the replaced receiver gets nothing", + ); +} + /// The embedder-facing DNS contract, end to end. /// /// An embedder that owns the TUN fd (Android `VpnService`) has no system DNS diff --git a/src/transport/mod.rs b/src/transport/mod.rs index b78d32bd..4620f9f6 100644 --- a/src/transport/mod.rs +++ b/src/transport/mod.rs @@ -844,6 +844,26 @@ impl TransportHandle { } } + /// Get the raw file descriptor of the bound socket (UDP only, returns None + /// for other transports and before the transport has started). Unix-only, + /// since `RawFd` is a unix concept and the Windows UDP backend has no + /// descriptor. + #[cfg(unix)] + pub fn raw_fd(&self) -> Option { + match self { + TransportHandle::Udp(t) => t.raw_fd(), + #[cfg(any(target_os = "linux", target_os = "macos"))] + TransportHandle::Ethernet(_) => None, + TransportHandle::Tcp(_) => None, + TransportHandle::Tor(_) => None, + TransportHandle::Nym(_) => None, + #[cfg(target_os = "linux")] + TransportHandle::Ble(_) => None, + #[cfg(test)] + TransportHandle::Loopback(_) => None, + } + } + /// Get the interface name (Ethernet only, returns None for other transports). pub fn interface_name(&self) -> Option<&str> { match self { diff --git a/src/transport/udp/mod.rs b/src/transport/udp/mod.rs index 2e8732d1..6f4ff868 100644 --- a/src/transport/udp/mod.rs +++ b/src/transport/udp/mod.rs @@ -85,6 +85,20 @@ impl UdpTransport { self.local_addr } + /// Raw file descriptor of the bound listen socket, or `None` before + /// `start_async` has bound one. + /// + /// Unix-only: `RawFd` is a unix concept and the Windows backend is built + /// on `tokio::net::UdpSocket` with no descriptor to hand out. The socket + /// stays owned by the transport, so the descriptor is a borrow with no + /// lifetime promise beyond "this is the transport's socket, and it is + /// open right now". + #[cfg(unix)] + pub fn raw_fd(&self) -> Option { + use std::os::unix::io::AsRawFd; + self.socket.as_ref().map(|socket| socket.as_raw_fd()) + } + /// Configured recv buffer size — used when opening per-peer /// `ConnectedPeerSocket`s so they get the same buffer ceiling as /// the wildcard listen socket.