diff --git a/CHANGELOG.md b/CHANGELOG.md index 6154cec5..05de8b83 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -169,6 +169,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 or that no source is readable and session pinning is off. An operator on a kernel with no readable source learned this only from a warning at the first failed tick. +- The gateway counts sessions on a kernel without `/proc/net/nf_conntrack`. + When the file is absent it dumps the conntrack table over netlink, as + `conntrack -L` does, so a mapping carrying traffic is pinned instead of + being reclaimed on its TTL and grace period alone. Kernels built without + `CONFIG_NF_CONNTRACK_PROCFS`, such as Ubuntu's, had session pinning off. - The NAT table is rebuilt in one netlink transaction. A rebuild deleted the `fips_gateway` table in a batch of its own, discarded that batch's result, and only then sent the batch that recreated the table, the chains, the diff --git a/Cargo.lock b/Cargo.lock index 1f41a814..336a8e91 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -650,6 +650,12 @@ version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a6ef517f0926dd24a1582492c791b6a4818a4d94e789a334894aa15b0d12f55c" +[[package]] +name = "convert_case" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6245d59a3e82a7fc217c5828a6692dbc6dfb63a0c8c90495621f7b9d79704a0e" + [[package]] name = "convert_case" version = "0.10.0" @@ -760,7 +766,7 @@ checksum = "d8b9f2e4c67f833b660cdb0a3523065869fb35570177239812ed4c905aeff87b" dependencies = [ "bitflags 2.13.1", "crossterm_winapi", - "derive_more", + "derive_more 2.1.1", "document-features", "mio", "parking_lot", @@ -966,6 +972,19 @@ version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" +[[package]] +name = "derive_more" +version = "0.99.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6edb4b64a43d977b8e99788fe3a04d483834fba1215a7e02caa415b626497f7f" +dependencies = [ + "convert_case 0.4.0", + "proc-macro2", + "quote", + "rustc_version", + "syn 2.0.119", +] + [[package]] name = "derive_more" version = "2.1.1" @@ -981,7 +1000,7 @@ version = "2.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "799a97264921d8623a957f6c3b9011f3b5492f557bbb7a5a19b7fa6d06ba8dcb" dependencies = [ - "convert_case", + "convert_case 0.10.0", "proc-macro2", "quote", "rustc_version", @@ -1159,6 +1178,9 @@ dependencies = [ "libc", "libm", "mdns-sd", + "netlink-packet-core", + "netlink-packet-netfilter", + "netlink-sys", "nostr", "nostr-sdk", "portable-atomic", @@ -1954,6 +1976,19 @@ dependencies = [ "paste", ] +[[package]] +name = "netlink-packet-netfilter" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "27b511a24610c054dbbfea4c3e403fe4df2e811df6bc48fa2583b1b93333f020" +dependencies = [ + "bitflags 2.13.1", + "derive_more 0.99.20", + "libc", + "netlink-packet-core", + "zerocopy", +] + [[package]] name = "netlink-packet-route" version = "0.30.0" diff --git a/Cargo.toml b/Cargo.toml index a052ed99..0efde69b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -61,6 +61,11 @@ libc = "0.2" rtnetlink = "0.21.0" rustables = "0.8.7" procfs = { version = "0.18", default-features = false } +# Conntrack dump for kernels without /proc/net/nf_conntrack. 0.3 is the +# release on netlink-packet-core 0.8, the version rtnetlink 0.21 uses. +netlink-packet-netfilter = "0.3" +netlink-packet-core = "0.8" +netlink-sys = "0.8" # bluer/BlueZ needs glibc — see build.rs `bluer_available` cfg gate. [target.'cfg(all(target_os = "linux", not(target_env = "musl")))'.dependencies] diff --git a/docs/design/fips-gateway.md b/docs/design/fips-gateway.md index a70b0afe..51ce4714 100644 --- a/docs/design/fips-gateway.md +++ b/docs/design/fips-gateway.md @@ -263,8 +263,13 @@ Timing: cached DNS responses. - **Tick interval**: the pool re-evaluates state every 10 s. -Active session counts come from `/proc/net/nf_conntrack`: an entry -counts as a session if its original destination is the virtual IP. +Active session counts come from `/proc/net/nf_conntrack`, or, on a +kernel without that file, from a dump of the IPv6 conntrack table over +`NETLINK_NETFILTER`, the request `conntrack -L` makes. The choice is +made on every tick, and the source is logged once at startup. Either +way an entry counts once toward each distinct IPv6 destination among +its original and reply tuples, so an entry counts as a session of a +virtual IP whose address is its original destination. If the pool is exhausted, new DNS queries return `SERVFAIL`. Existing mappings are never evicted prematurely — the correctness of diff --git a/docs/how-to/troubleshoot-gateway.md b/docs/how-to/troubleshoot-gateway.md index 4fbcb419..6b532ded 100644 --- a/docs/how-to/troubleshoot-gateway.md +++ b/docs/how-to/troubleshoot-gateway.md @@ -167,14 +167,19 @@ virtual IP. At startup it reads conntrack once, the same way each - `Conntrack source: proc; session pinning is on`: sessions are read from `/proc/net/nf_conntrack`. +- `Conntrack source: netlink; session pinning is on`: the proc file + is absent, and sessions are read by dumping the conntrack table over + netlink, as `conntrack -L` does. The dump needs `CAP_NET_ADMIN`. - `No conntrack source is readable; session pinning is off`: no - source could be read, and the line carries the error. Every mapping - then reads zero sessions, so a mapping is reclaimed on its TTL and - grace period alone, even while a client that has not re-queried DNS - still has traffic flowing through it. + source could be read, and the line carries the error from each + source it tried. Every mapping then reads zero sessions, so a + mapping is reclaimed on its TTL and grace period alone, even while a + client that has not re-queried DNS still has traffic flowing through + it. The proc file exists only on a kernel built with -`CONFIG_NF_CONNTRACK_PROCFS`, and only once `nf_conntrack` is loaded: +`CONFIG_NF_CONNTRACK_PROCFS`, and only once `nf_conntrack` is loaded. +Without it the gateway uses the netlink dump: ```sh ls /proc/net/nf_conntrack diff --git a/src/bin/fips-gateway.rs b/src/bin/fips-gateway.rs index 48166685..fb2c6164 100644 --- a/src/bin/fips-gateway.rs +++ b/src/bin/fips-gateway.rs @@ -65,9 +65,9 @@ fn elapsed_us(started: Instant) -> u64 { /// A failed read yields an empty snapshot, so every mapping reads zero /// sessions, which is what the pool did with an unreadable source before. The /// alternative, treating "unknown" as "in use", would pin every mapping forever -/// on a kernel with no conntrack proc file and turn a read error into a pool -/// that never reclaims. The cost is the opposite error: a mapping carrying live -/// traffic can be reclaimed early while the source is unreadable. +/// on a kernel with no readable conntrack source and turn a read error into a +/// pool that never reclaims. The cost is the opposite error: a mapping carrying +/// live traffic can be reclaimed early while the source is unreadable. #[cfg(target_os = "linux")] async fn read_conntrack(log: &mut pool::ConntrackReadLog) -> pool::ConntrackSnapshot { use fips::gateway::pool::ConntrackQuerier; @@ -119,16 +119,27 @@ async fn report_conntrack_source() { .unwrap_or_else(|e| { pool::ConntrackProbe::Missing(pool::ConntrackUnreadable { proc: std::io::Error::other(e.to_string()), + netlink: None, }) }); match probe { pool::ConntrackProbe::Found(pool::ConntrackSource::Proc) => { info!("Conntrack source: proc; session pinning is on") } - pool::ConntrackProbe::Missing(e) => warn!( - proc_error = %e.proc, - "No conntrack source is readable; session pinning is off" - ), + pool::ConntrackProbe::Found(pool::ConntrackSource::Netlink) => { + info!("Conntrack source: netlink; session pinning is on") + } + pool::ConntrackProbe::Missing(e) => match e.netlink { + Some(netlink) => warn!( + proc_error = %e.proc, + netlink_error = %netlink, + "No conntrack source is readable; session pinning is off" + ), + None => warn!( + proc_error = %e.proc, + "No conntrack source is readable; session pinning is off" + ), + }, } } diff --git a/src/gateway/conntrack.rs b/src/gateway/conntrack.rs new file mode 100644 index 00000000..23fca85c --- /dev/null +++ b/src/gateway/conntrack.rs @@ -0,0 +1,330 @@ +//! Conntrack sessions read over netlink. +//! +//! A kernel built without `CONFIG_NF_CONNTRACK_PROCFS` has no +//! `/proc/net/nf_conntrack`, while `conntrack -L` still lists the table: it +//! asks the kernel for a dump over `NETLINK_NETFILTER`. This reader does the +//! same, so such a kernel can still pin mappings that carry traffic. + +use super::pool::{ConntrackQuerier, ConntrackSnapshot}; +use netlink_packet_core::{ + NLM_F_DUMP, NLM_F_REQUEST, NetlinkHeader, NetlinkMessage, NetlinkPayload, +}; +use netlink_packet_netfilter::conntrack::{ConntrackAttribute, ConntrackMessage, IPTuple, Tuple}; +use netlink_packet_netfilter::{ + NetfilterHeader, NetfilterMessage, NetfilterMessageInner, NetfilterProtoFamily, +}; +use netlink_sys::{Socket, SocketAddr, protocols::NETLINK_NETFILTER}; +use std::collections::{HashMap, HashSet}; +use std::io; +use std::net::{IpAddr, Ipv6Addr}; +use std::sync::atomic::{AtomicU32, Ordering}; +use std::time::Duration; + +/// Longest wait for each part of the kernel's reply. +/// +/// The read runs on a blocking thread once per tick, so a kernel that never +/// answers must not hold that thread for longer than a tick. +const READ_TIMEOUT: Duration = Duration::from_secs(2); + +/// Sequence number for the next dump request, so a reply to an earlier +/// request cannot be counted as part of this one. +static NEXT_SEQ: AtomicU32 = AtomicU32::new(1); + +/// Whether a dump has more to come after the buffer just counted. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DumpState { + /// The kernel has more of the dump to send. + More, + /// The kernel sent the end-of-dump message. + Done, +} + +/// Conntrack querier that dumps the table over netlink. +/// +/// Needs `CAP_NET_ADMIN` in the gateway's network namespace, which the +/// gateway already needs for its NAT table. +pub struct NetlinkConntrack; + +impl ConntrackQuerier for NetlinkConntrack { + fn snapshot(&self) -> Result { + let mut socket = Socket::new(NETLINK_NETFILTER)?; + socket.bind_auto()?; + socket.connect(&SocketAddr::new(0, 0))?; + socket2::SockRef::from(&socket).set_read_timeout(Some(READ_TIMEOUT))?; + + let seq = NEXT_SEQ.fetch_add(1, Ordering::Relaxed); + socket.send(&dump_request(seq), 0)?; + + let mut counts = HashMap::new(); + loop { + // Sized by peeking first, so a large batch is not truncated. + let (buf, _) = socket.recv_from_full()?; + if count_dump(&buf, seq, &mut counts)? == DumpState::Done { + break; + } + } + Ok(ConntrackSnapshot::from_counts(counts)) + } +} + +/// A request for a dump of the IPv6 conntrack table. +/// +/// The family in the netfilter header makes the kernel leave out IPv4 +/// entries, which could never name a virtual IP. +fn dump_request(seq: u32) -> Vec { + let mut header = NetlinkHeader::default(); + header.flags = NLM_F_REQUEST | NLM_F_DUMP; + header.sequence_number = seq; + let mut message = NetlinkMessage::new( + header, + NetlinkPayload::from(NetfilterMessage::new( + NetfilterHeader::new(NetfilterProtoFamily::IPv6, 0, 0), + ConntrackMessage::Get(vec![]), + )), + ); + message.finalize(); + let mut buf = vec![0; message.buffer_len()]; + message.serialize(&mut buf); + buf +} + +/// Count the conntrack entries in one received buffer by destination. +/// +/// An entry counts once for each distinct IPv6 destination among its original +/// and reply tuples, the same rule the proc-file parser applies to a line. +/// A message carrying another sequence number is skipped. A dump the kernel +/// flags as interrupted is counted as received: reading the proc file is not +/// atomic across the table either, and failing the read would zero every +/// mapping for the tick. +pub fn count_dump( + buf: &[u8], + seq: u32, + counts: &mut HashMap, +) -> Result { + let mut offset = 0; + while offset < buf.len() { + let message = NetlinkMessage::::deserialize(&buf[offset..]) + .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e.to_string()))?; + // Messages are padded to four bytes. The length is at least a header, + // or the parse above would have failed, so the walk always advances. + let len = message.header.length as usize; + offset += (len + 3) & !3; + if message.header.sequence_number != seq { + continue; + } + match message.payload { + NetlinkPayload::Done(_) => return Ok(DumpState::Done), + NetlinkPayload::Error(e) if e.code.is_some() => return Err(e.to_io()), + NetlinkPayload::InnerMessage(NetfilterMessage { + inner: NetfilterMessageInner::Conntrack(ConntrackMessage::New(attrs)), + .. + }) => count_entry(&attrs, counts), + _ => {} + } + } + Ok(DumpState::More) +} + +/// Add one conntrack entry to the counts, once per distinct IPv6 destination +/// among its original and reply tuples. +fn count_entry(attrs: &[ConntrackAttribute], counts: &mut HashMap) { + let mut seen = HashSet::new(); + for attr in attrs { + let tuples = match attr { + ConntrackAttribute::CtaTupleOrig(t) | ConntrackAttribute::CtaTupleReply(t) => t, + _ => continue, + }; + for tuple in tuples { + let Tuple::Ip(ip) = tuple else { continue }; + for field in ip { + if let IPTuple::DestinationAddress(IpAddr::V6(dst)) = field { + seen.insert(*dst); + } + } + } + } + for dst in seen { + *counts.entry(dst).or_insert(0) += 1; + } +} + +#[cfg(test)] +mod tests { + use super::*; + use netlink_packet_core::{DoneMessage, ErrorMessage}; + use std::num::NonZeroI32; + + const SEQ: u32 = 7; + + fn v6(s: &str) -> Ipv6Addr { + s.parse().unwrap() + } + + /// One tuple naming a source and a destination. + fn tuple(src: Ipv6Addr, dst: Ipv6Addr) -> Vec { + vec![Tuple::Ip(vec![ + IPTuple::SourceAddress(IpAddr::V6(src)), + IPTuple::DestinationAddress(IpAddr::V6(dst)), + ])] + } + + /// A conntrack entry as a dump reply carries it. + fn entry(orig: Vec, reply: Vec) -> NetfilterMessage { + NetfilterMessage::new( + NetfilterHeader::new(NetfilterProtoFamily::IPv6, 0, 0), + ConntrackMessage::New(vec![ + ConntrackAttribute::CtaTupleOrig(orig), + ConntrackAttribute::CtaTupleReply(reply), + ]), + ) + } + + /// Serialise one netlink message with the given sequence number. + fn frame(payload: NetlinkPayload, seq: u32) -> Vec { + let mut header = NetlinkHeader::default(); + header.sequence_number = seq; + header.flags = netlink_packet_core::NLM_F_MULTIPART; + let mut message = NetlinkMessage::new(header, payload); + message.finalize(); + let mut buf = vec![0; message.buffer_len()]; + message.serialize(&mut buf); + buf + } + + fn done() -> NetlinkPayload { + NetlinkPayload::Done(DoneMessage::default()) + } + + /// The flow the gateway sees for a LAN client using a virtual IP: the + /// original tuple is client to virtual IP, and the reply, after DNAT and + /// masquerade, is the mesh address back to the gateway. + fn client_flow(virtual_ip: Ipv6Addr) -> NetfilterMessage { + entry( + tuple(v6("fd02::20"), virtual_ip), + tuple(v6("fd9a::1"), v6("fd9a::2")), + ) + } + + fn count(buf: &[u8]) -> (Result, HashMap) { + let mut counts = HashMap::new(); + let state = count_dump(buf, SEQ, &mut counts); + (state, counts) + } + + #[test] + fn netlink_dump_counts_an_entry_whose_original_destination_is_the_virtual_ip() { + let virtual_ip = v6("fd01::1"); + let buf = frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ); + + let (state, counts) = count(&buf); + + assert_eq!(state.unwrap(), DumpState::More); + assert_eq!(counts.get(&virtual_ip).copied(), Some(1)); + // The reply tuple's destination is counted too, as the proc parser + // counts every dst= on the line. + assert_eq!(counts.get(&v6("fd9a::2")).copied(), Some(1)); + } + + #[test] + fn netlink_dump_counts_an_entry_once_when_both_tuples_name_the_address() { + let addr = v6("fd01::1"); + let hairpin = entry(tuple(addr, addr), tuple(addr, addr)); + let buf = frame(NetlinkPayload::from(hairpin), SEQ); + + let (_, counts) = count(&buf); + + assert_eq!(counts.get(&addr).copied(), Some(1)); + } + + #[test] + fn netlink_dump_counts_each_entry_across_several_messages_in_one_buffer() { + let virtual_ip = v6("fd01::1"); + let other = v6("fd01::2"); + let mut buf = frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ); + buf.extend(frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ)); + buf.extend(frame(NetlinkPayload::from(client_flow(other)), SEQ)); + + let (state, counts) = count(&buf); + + assert_eq!(state.unwrap(), DumpState::More); + assert_eq!(counts.get(&virtual_ip).copied(), Some(2)); + assert_eq!(counts.get(&other).copied(), Some(1)); + } + + #[test] + fn netlink_dump_ignores_an_ipv4_entry() { + let v4 = |s: &str| IpAddr::V4(s.parse().unwrap()); + let ipv4 = NetfilterMessage::new( + NetfilterHeader::new(NetfilterProtoFamily::IPv4, 0, 0), + ConntrackMessage::New(vec![ConntrackAttribute::CtaTupleOrig(vec![Tuple::Ip( + vec![ + IPTuple::SourceAddress(v4("192.0.2.1")), + IPTuple::DestinationAddress(v4("192.0.2.2")), + ], + )])]), + ); + let mut buf = frame(NetlinkPayload::from(ipv4), SEQ); + buf.extend(frame(NetlinkPayload::from(client_flow(v6("fd01::1"))), SEQ)); + + let (state, counts) = count(&buf); + + assert_eq!(state.unwrap(), DumpState::More); + assert_eq!(counts.len(), 2, "only the IPv6 entry's two destinations"); + } + + #[test] + fn netlink_dump_reports_done_on_the_done_message() { + let virtual_ip = v6("fd01::1"); + let mut buf = frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ); + buf.extend(frame(done(), SEQ)); + + let (state, counts) = count(&buf); + + assert_eq!(state.unwrap(), DumpState::Done); + assert_eq!(counts.get(&virtual_ip).copied(), Some(1)); + } + + #[test] + fn netlink_dump_turns_an_eperm_error_message_into_permission_denied() { + let mut error = ErrorMessage::default(); + error.code = NonZeroI32::new(-libc::EPERM); + let buf = frame(NetlinkPayload::Error(error), SEQ); + + let (state, _) = count(&buf); + + assert_eq!( + state.expect_err("an error reply must fail the read").kind(), + io::ErrorKind::PermissionDenied + ); + } + + #[test] + fn netlink_dump_skips_a_message_with_another_sequence_number() { + let virtual_ip = v6("fd01::1"); + let mut buf = frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ + 1); + buf.extend(frame(done(), SEQ + 1)); + buf.extend(frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ)); + + let (state, counts) = count(&buf); + + assert_eq!( + state.unwrap(), + DumpState::More, + "another request's end of dump does not end this one" + ); + assert_eq!(counts.get(&virtual_ip).copied(), Some(1)); + } + + #[test] + fn netlink_dump_rejects_a_buffer_that_does_not_parse() { + let mut buf = frame(NetlinkPayload::from(client_flow(v6("fd01::1"))), SEQ); + buf.truncate(buf.len() - 4); + + let (state, _) = count(&buf); + + assert_eq!( + state.expect_err("a short buffer must fail the read").kind(), + io::ErrorKind::InvalidData + ); + } +} diff --git a/src/gateway/mod.rs b/src/gateway/mod.rs index db3f2dec..df3e4ed7 100644 --- a/src/gateway/mod.rs +++ b/src/gateway/mod.rs @@ -3,6 +3,7 @@ //! Allows unmodified LAN hosts to reach FIPS mesh destinations via //! DNS-allocated virtual IPs and kernel nftables NAT. +pub mod conntrack; pub mod control; pub mod dns; pub mod nat; diff --git a/src/gateway/pool.rs b/src/gateway/pool.rs index 202a099b..bac11373 100644 --- a/src/gateway/pool.rs +++ b/src/gateway/pool.rs @@ -109,7 +109,10 @@ pub struct MappingInfo { pub last_ref_secs: u64, } -/// Path the conntrack table is read from. +/// Path the conntrack table is read from when the kernel provides it. +/// +/// A kernel built without `CONFIG_NF_CONNTRACK_PROCFS` has no such file; +/// `SystemConntrack` then dumps the table over netlink instead. const CONNTRACK_PROC_PATH: &str = "/proc/net/nf_conntrack"; /// Active conntrack sessions counted by destination address. @@ -164,6 +167,8 @@ impl ConntrackQuerier for ProcConntrack { pub enum ConntrackSource { /// `/proc/net/nf_conntrack`. Proc, + /// A conntrack table dump over `NETLINK_NETFILTER`. + Netlink, } impl ConntrackSource { @@ -171,6 +176,7 @@ impl ConntrackSource { pub fn name(self) -> &'static str { match self { Self::Proc => "proc", + Self::Netlink => "netlink", } } } @@ -180,12 +186,25 @@ impl ConntrackSource { pub struct ConntrackUnreadable { /// The error reading `/proc/net/nf_conntrack`. pub proc: std::io::Error, + /// The error from the netlink dump, when the proc file was absent and the + /// dump was tried. + pub netlink: Option, } impl ConntrackUnreadable { /// The error that stands for the whole failed read. + /// + /// When the dump was tried, its error is the one that decided the read, so + /// it sets the kind; the absent proc file is kept in the message. Only a + /// proc error that stopped the read before the dump stands alone. fn into_error(self) -> std::io::Error { - self.proc + match self.netlink { + Some(netlink) => std::io::Error::new( + netlink.kind(), + format!("proc: {}; netlink: {netlink}", self.proc), + ), + None => self.proc, + } } } @@ -193,34 +212,55 @@ impl ConntrackUnreadable { /// answered. /// /// The per-tick read and the startup probe both go through this type, so the -/// probe cannot report a source the tick would not use. The querier is a type -/// parameter so tests can substitute fakes. -pub struct SystemConntrack

{ +/// probe cannot report a source the tick would not use. The queriers are type +/// parameters so tests can substitute fakes. +/// +/// The proc file is read first. Only when it is absent is the table dumped +/// over netlink, and that is decided on every read: the file appears once +/// `nf_conntrack` is loaded in the namespace, so a choice fixed at startup +/// could keep using netlink on a kernel that has the file. +pub struct SystemConntrack

{ proc: P, + netlink: N, } -impl SystemConntrack

{ - /// A reader over the given proc querier. - pub fn new(proc: P) -> Self { - Self { proc } +impl SystemConntrack { + /// A reader over the given proc and netlink queriers. + pub fn new(proc: P, netlink: N) -> Self { + Self { proc, netlink } } /// Read conntrack once and say which source the snapshot came from. + /// + /// A proc error other than an absent file, such as a permission error, is + /// returned without trying netlink. pub fn read(&self) -> Result<(ConntrackSource, ConntrackSnapshot), ConntrackUnreadable> { match self.proc.snapshot() { Ok(snapshot) => Ok((ConntrackSource::Proc, snapshot)), - Err(proc) => Err(ConntrackUnreadable { proc }), + Err(proc) if proc.kind() == std::io::ErrorKind::NotFound => { + match self.netlink.snapshot() { + Ok(snapshot) => Ok((ConntrackSource::Netlink, snapshot)), + Err(netlink) => Err(ConntrackUnreadable { + proc, + netlink: Some(netlink), + }), + } + } + Err(proc) => Err(ConntrackUnreadable { + proc, + netlink: None, + }), } } } impl Default for SystemConntrack { fn default() -> Self { - Self::new(ProcConntrack) + Self::new(ProcConntrack, super::conntrack::NetlinkConntrack) } } -impl ConntrackQuerier for SystemConntrack

{ +impl ConntrackQuerier for SystemConntrack { fn snapshot(&self) -> Result { self.read() .map(|(_, snapshot)| snapshot) @@ -239,7 +279,9 @@ pub enum ConntrackProbe { } /// Read conntrack once, as a tick would, and report which source answered. -pub fn probe_conntrack(reader: &SystemConntrack

) -> ConntrackProbe { +pub fn probe_conntrack( + reader: &SystemConntrack, +) -> ConntrackProbe { match reader.read() { Ok((source, _)) => ConntrackProbe::Found(source), Err(e) => ConntrackProbe::Missing(e), @@ -294,12 +336,12 @@ pub enum ReadReport { /// Remembers the last conntrack read outcome. /// -/// A kernel built without `CONFIG_NF_CONNTRACK_PROCFS` has no -/// `/proc/net/nf_conntrack` at all, so every read fails the same way and a -/// per-tick warning would repeat for the life of the process. Warning on a -/// change of outcome still separates "the source is unreadable" from "there -/// are no sessions", which the pool could not distinguish before, without -/// filling the log. +/// When no source is readable, for example a kernel with no +/// `/proc/net/nf_conntrack` whose netlink dump is refused, every read fails +/// the same way and a per-tick warning would repeat for the life of the +/// process. Warning on a change of outcome still separates "the source is +/// unreadable" from "there are no sessions", which the pool could not +/// distinguish before, without filling the log. #[derive(Debug, Default)] pub struct ConntrackReadLog { last: Option>, @@ -1238,7 +1280,7 @@ mod tests { #[test] fn conntrack_probe_names_the_proc_source_when_the_proc_read_succeeds() { - let reader = SystemConntrack::new(FixedRead(None)); + let reader = SystemConntrack::new(FixedRead(None), NOT_CALLED); match probe_conntrack(&reader) { ConntrackProbe::Found(source) => { @@ -1251,7 +1293,10 @@ mod tests { #[test] fn conntrack_probe_reports_missing_with_the_error_when_the_proc_read_fails() { - let reader = SystemConntrack::new(FixedRead(Some(std::io::ErrorKind::PermissionDenied))); + let reader = SystemConntrack::new( + FixedRead(Some(std::io::ErrorKind::PermissionDenied)), + NOT_CALLED, + ); match probe_conntrack(&reader) { ConntrackProbe::Missing(e) => { @@ -1260,4 +1305,89 @@ mod tests { ConntrackProbe::Found(source) => panic!("expected no source, got {source:?}"), } } + + /// A netlink stand-in for tests where the dump must not be reached. It + /// fails with a kind no test expects, so reaching it shows in the result. + const NOT_CALLED: FixedRead = FixedRead(Some(std::io::ErrorKind::Unsupported)); + + /// A conntrack querier that reports one session to a fixed address. + struct OneSession(Ipv6Addr); + + impl ConntrackQuerier for OneSession { + fn snapshot(&self) -> Result { + Ok(ConntrackSnapshot::from_counts(HashMap::from([(self.0, 1)]))) + } + } + + #[test] + fn system_conntrack_falls_back_to_netlink_when_the_proc_file_is_absent() { + let addr: Ipv6Addr = "fd01::1".parse().unwrap(); + let reader = SystemConntrack::new( + FixedRead(Some(std::io::ErrorKind::NotFound)), + OneSession(addr), + ); + + let (source, snapshot) = reader.read().expect("the netlink dump answered"); + + assert_eq!(source, ConntrackSource::Netlink); + assert_eq!(source.name(), "netlink"); + assert_eq!(snapshot.sessions_for(addr), 1); + } + + #[test] + fn system_conntrack_does_not_fall_back_on_a_proc_error_other_than_not_found() { + let addr: Ipv6Addr = "fd01::1".parse().unwrap(); + let reader = SystemConntrack::new( + FixedRead(Some(std::io::ErrorKind::PermissionDenied)), + OneSession(addr), + ); + + let e = reader + .read() + .expect_err("a denied proc read is not a missing file"); + + assert_eq!(e.proc.kind(), std::io::ErrorKind::PermissionDenied); + assert!(e.netlink.is_none(), "netlink was not tried"); + assert_eq!( + reader.snapshot().unwrap_err().kind(), + std::io::ErrorKind::PermissionDenied + ); + } + + #[test] + fn system_conntrack_prefers_proc_when_it_reads() { + let proc_addr: Ipv6Addr = "fd01::1".parse().unwrap(); + let netlink_addr: Ipv6Addr = "fd01::2".parse().unwrap(); + let reader = SystemConntrack::new(OneSession(proc_addr), OneSession(netlink_addr)); + + let (source, snapshot) = reader.read().expect("the proc file answered"); + + assert_eq!(source, ConntrackSource::Proc); + assert_eq!(snapshot.sessions_for(proc_addr), 1); + assert_eq!(snapshot.sessions_for(netlink_addr), 0); + } + + #[test] + fn conntrack_probe_reports_both_errors_when_neither_source_reads() { + let reader = SystemConntrack::new( + FixedRead(Some(std::io::ErrorKind::NotFound)), + FixedRead(Some(std::io::ErrorKind::PermissionDenied)), + ); + + match probe_conntrack(&reader) { + ConntrackProbe::Missing(e) => { + assert_eq!(e.proc.kind(), std::io::ErrorKind::NotFound); + assert_eq!( + e.netlink.as_ref().map(std::io::Error::kind), + Some(std::io::ErrorKind::PermissionDenied) + ); + } + ConntrackProbe::Found(source) => panic!("expected no source, got {source:?}"), + } + // The per-tick read reports the error that decided it: the dump's. + assert_eq!( + reader.snapshot().unwrap_err().kind(), + std::io::ErrorKind::PermissionDenied + ); + } } diff --git a/testing/static/scripts/gateway-test.sh b/testing/static/scripts/gateway-test.sh index db625cea..dd441b91 100755 --- a/testing/static/scripts/gateway-test.sh +++ b/testing/static/scripts/gateway-test.sh @@ -155,21 +155,25 @@ fi # The gateway names its conntrack source once at startup, before the DNS # resolver starts, so by now the line is in the log. Ask the gateway's own -# namespace which source it should have found. A failed `docker logs` reds the -# check rather than counting as zero lines. +# namespace which source it should have found: the proc file when it exists, +# and otherwise the netlink dump, which the container's NET_ADMIN allows. A +# failed `docker logs` reds the check rather than counting as zero lines. if docker exec "$GATEWAY" test -e /proc/net/nf_conntrack; then EXPECT_SRC=proc else - EXPECT_SRC=none + EXPECT_SRC=netlink fi if GW_START_LOG=$(docker logs "$GATEWAY" 2>&1); then SRC_PROC=$(grep -cF 'Conntrack source: proc; session pinning is on' <<< "$GW_START_LOG" || true) + SRC_NETLINK=$(grep -cF 'Conntrack source: netlink; session pinning is on' <<< "$GW_START_LOG" || true) SRC_NONE=$(grep -cF 'No conntrack source is readable; session pinning is off' <<< "$GW_START_LOG" || true) case "$EXPECT_SRC" in - proc) SRC_OK=$([ "$SRC_PROC" -eq 1 ] && [ "$SRC_NONE" -eq 0 ] && echo 0 || echo 1) ;; - *) SRC_OK=$([ "$SRC_NONE" -eq 1 ] && [ "$SRC_PROC" -eq 0 ] && echo 0 || echo 1) ;; + proc) SRC_HIT=$SRC_PROC ;; + *) SRC_HIT=$SRC_NETLINK ;; esac - check "Conntrack source line at startup (expect $EXPECT_SRC; proc lines $SRC_PROC, none lines $SRC_NONE)" "$SRC_OK" + SRC_ALL=$((SRC_PROC + SRC_NETLINK + SRC_NONE)) + SRC_OK=$([ "$SRC_HIT" -eq 1 ] && [ "$SRC_ALL" -eq 1 ] && echo 0 || echo 1) + check "Conntrack source line at startup (expect $EXPECT_SRC; proc lines $SRC_PROC, netlink lines $SRC_NETLINK, none lines $SRC_NONE)" "$SRC_OK" else check "Conntrack source line at startup (docker logs failed)" 1 fi @@ -299,6 +303,43 @@ else check "nftables DNAT rules" 1 fi +# Phase 5's GET left a TCP conntrack entry to the first virtual IP, which +# stays in the table in TIME_WAIT well past two ticks. The gateway must count +# it. Poll about once a second for 25 tries, which spans two 10s ticks, and +# read the count with a parser that cannot turn a failed query into a number: +# an error response has no `data`, and the parser exits non-zero on it. +if [ -n "$VIRTUAL_IP" ]; then + VIP_SESSIONS=error + for _ in $(seq 1 25); do + VIP_SESSIONS=$(docker exec "$GATEWAY" bash -c \ + 'echo "{\"command\":\"show_mappings\"}" | nc -U -w1 /run/fips/gateway.sock 2>/dev/null' \ + | VIP="$VIRTUAL_IP" python3 -c " +import os, sys, json +r = json.load(sys.stdin) +data = r.get('data') +if not isinstance(data, dict) or not isinstance(data.get('mappings'), list): + sys.exit(1) +hits = [m for m in data['mappings'] if m.get('virtual_ip') == os.environ['VIP']] +if len(hits) != 1 or not isinstance(hits[0].get('sessions'), int): + sys.exit(1) +print(hits[0]['sessions']) +" 2>/dev/null || echo "error") + if [ "$VIP_SESSIONS" != error ] && [ "$VIP_SESSIONS" -ge 1 ]; then + break + fi + sleep 1 + done + if [ "$VIP_SESSIONS" != error ] && [ "$VIP_SESSIONS" -ge 1 ]; then + check "Gateway counts a session to $VIRTUAL_IP (sessions $VIP_SESSIONS, source $EXPECT_SRC)" 0 + else + check "Gateway counts a session to $VIRTUAL_IP (sessions $VIP_SESSIONS, source $EXPECT_SRC)" 1 + echo " Kernel conntrack entries to $VIRTUAL_IP:" + docker exec "$GATEWAY" conntrack -L -f ipv6 -d "$VIRTUAL_IP" 2>&1 | sed 's/^/ /' || true + fi +else + check "Gateway counts a session (skipped — no virtual IP)" 1 +fi + # Phase 7: Inbound port forwarding — UDP and a second simultaneous TCP forward. # # Three forwards exercised: