From 4a1d5836531b392243ca101af89c08af64618227 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 19 Sep 2026 01:44:23 +0000 Subject: [PATCH] fix(gateway): read conntrack over netlink when the proc file is absent A kernel built without CONFIG_NF_CONNTRACK_PROCFS, such as Ubuntu's, has no /proc/net/nf_conntrack. The gateway then counted zero sessions for every mapping, so a mapping still carrying traffic was reclaimed on its TTL and grace period alone. When the proc file is absent, the gateway now dumps the IPv6 conntrack table over NETLINK_NETFILTER, the request conntrack -L makes, and counts each entry once per distinct destination, the same rule the proc parser applies. The choice is made on every read, because the proc file appears only once nf_conntrack is loaded. Any other proc error is still reported without falling back. When both sources fail, the per-tick warning carries the netlink error, and the startup line names both errors. The dump uses netlink-packet-netfilter 0.3, the release built on the netlink-packet-core 0.8 that rtnetlink already uses. The gateway suite now expects the netlink source line when the container has no proc file, and checks that the gateway counts the session left by the HTTP request through the first virtual IP. --- CHANGELOG.md | 5 + Cargo.lock | 39 ++- Cargo.toml | 5 + docs/design/fips-gateway.md | 9 +- docs/how-to/troubleshoot-gateway.md | 15 +- src/bin/fips-gateway.rs | 25 +- src/gateway/conntrack.rs | 330 +++++++++++++++++++++++++ src/gateway/mod.rs | 1 + src/gateway/pool.rs | 172 +++++++++++-- testing/static/scripts/gateway-test.sh | 53 +++- 10 files changed, 611 insertions(+), 43 deletions(-) create mode 100644 src/gateway/conntrack.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 6154cec5..05de8b83 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -169,6 +169,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 or that no source is readable and session pinning is off. An operator on a kernel with no readable source learned this only from a warning at the first failed tick. +- The gateway counts sessions on a kernel without `/proc/net/nf_conntrack`. + When the file is absent it dumps the conntrack table over netlink, as + `conntrack -L` does, so a mapping carrying traffic is pinned instead of + being reclaimed on its TTL and grace period alone. Kernels built without + `CONFIG_NF_CONNTRACK_PROCFS`, such as Ubuntu's, had session pinning off. - The NAT table is rebuilt in one netlink transaction. A rebuild deleted the `fips_gateway` table in a batch of its own, discarded that batch's result, and only then sent the batch that recreated the table, the chains, the diff --git a/Cargo.lock b/Cargo.lock index 1f41a814..336a8e91 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -650,6 +650,12 @@ version = "0.10.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a6ef517f0926dd24a1582492c791b6a4818a4d94e789a334894aa15b0d12f55c" +[[package]] +name = "convert_case" +version = "0.4.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6245d59a3e82a7fc217c5828a6692dbc6dfb63a0c8c90495621f7b9d79704a0e" + [[package]] name = "convert_case" version = "0.10.0" @@ -760,7 +766,7 @@ checksum = "d8b9f2e4c67f833b660cdb0a3523065869fb35570177239812ed4c905aeff87b" dependencies = [ "bitflags 2.13.1", "crossterm_winapi", - "derive_more", + "derive_more 2.1.1", "document-features", "mio", "parking_lot", @@ -966,6 +972,19 @@ version = "0.5.8" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7cd812cc2bc1d69d4764bd80df88b4317eaef9e773c75226407d9bc0876b211c" +[[package]] +name = "derive_more" +version = "0.99.20" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6edb4b64a43d977b8e99788fe3a04d483834fba1215a7e02caa415b626497f7f" +dependencies = [ + "convert_case 0.4.0", + "proc-macro2", + "quote", + "rustc_version", + "syn 2.0.119", +] + [[package]] name = "derive_more" version = "2.1.1" @@ -981,7 +1000,7 @@ version = "2.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "799a97264921d8623a957f6c3b9011f3b5492f557bbb7a5a19b7fa6d06ba8dcb" dependencies = [ - "convert_case", + "convert_case 0.10.0", "proc-macro2", "quote", "rustc_version", @@ -1159,6 +1178,9 @@ dependencies = [ "libc", "libm", "mdns-sd", + "netlink-packet-core", + "netlink-packet-netfilter", + "netlink-sys", "nostr", "nostr-sdk", "portable-atomic", @@ -1954,6 +1976,19 @@ dependencies = [ "paste", ] +[[package]] +name = "netlink-packet-netfilter" +version = "0.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "27b511a24610c054dbbfea4c3e403fe4df2e811df6bc48fa2583b1b93333f020" +dependencies = [ + "bitflags 2.13.1", + "derive_more 0.99.20", + "libc", + "netlink-packet-core", + "zerocopy", +] + [[package]] name = "netlink-packet-route" version = "0.30.0" diff --git a/Cargo.toml b/Cargo.toml index a052ed99..0efde69b 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -61,6 +61,11 @@ libc = "0.2" rtnetlink = "0.21.0" rustables = "0.8.7" procfs = { version = "0.18", default-features = false } +# Conntrack dump for kernels without /proc/net/nf_conntrack. 0.3 is the +# release on netlink-packet-core 0.8, the version rtnetlink 0.21 uses. +netlink-packet-netfilter = "0.3" +netlink-packet-core = "0.8" +netlink-sys = "0.8" # bluer/BlueZ needs glibc — see build.rs `bluer_available` cfg gate. [target.'cfg(all(target_os = "linux", not(target_env = "musl")))'.dependencies] diff --git a/docs/design/fips-gateway.md b/docs/design/fips-gateway.md index a70b0afe..51ce4714 100644 --- a/docs/design/fips-gateway.md +++ b/docs/design/fips-gateway.md @@ -263,8 +263,13 @@ Timing: cached DNS responses. - **Tick interval**: the pool re-evaluates state every 10 s. -Active session counts come from `/proc/net/nf_conntrack`: an entry -counts as a session if its original destination is the virtual IP. +Active session counts come from `/proc/net/nf_conntrack`, or, on a +kernel without that file, from a dump of the IPv6 conntrack table over +`NETLINK_NETFILTER`, the request `conntrack -L` makes. The choice is +made on every tick, and the source is logged once at startup. Either +way an entry counts once toward each distinct IPv6 destination among +its original and reply tuples, so an entry counts as a session of a +virtual IP whose address is its original destination. If the pool is exhausted, new DNS queries return `SERVFAIL`. Existing mappings are never evicted prematurely — the correctness of diff --git a/docs/how-to/troubleshoot-gateway.md b/docs/how-to/troubleshoot-gateway.md index 4fbcb419..6b532ded 100644 --- a/docs/how-to/troubleshoot-gateway.md +++ b/docs/how-to/troubleshoot-gateway.md @@ -167,14 +167,19 @@ virtual IP. At startup it reads conntrack once, the same way each - `Conntrack source: proc; session pinning is on`: sessions are read from `/proc/net/nf_conntrack`. +- `Conntrack source: netlink; session pinning is on`: the proc file + is absent, and sessions are read by dumping the conntrack table over + netlink, as `conntrack -L` does. The dump needs `CAP_NET_ADMIN`. - `No conntrack source is readable; session pinning is off`: no - source could be read, and the line carries the error. Every mapping - then reads zero sessions, so a mapping is reclaimed on its TTL and - grace period alone, even while a client that has not re-queried DNS - still has traffic flowing through it. + source could be read, and the line carries the error from each + source it tried. Every mapping then reads zero sessions, so a + mapping is reclaimed on its TTL and grace period alone, even while a + client that has not re-queried DNS still has traffic flowing through + it. The proc file exists only on a kernel built with -`CONFIG_NF_CONNTRACK_PROCFS`, and only once `nf_conntrack` is loaded: +`CONFIG_NF_CONNTRACK_PROCFS`, and only once `nf_conntrack` is loaded. +Without it the gateway uses the netlink dump: ```sh ls /proc/net/nf_conntrack diff --git a/src/bin/fips-gateway.rs b/src/bin/fips-gateway.rs index 48166685..fb2c6164 100644 --- a/src/bin/fips-gateway.rs +++ b/src/bin/fips-gateway.rs @@ -65,9 +65,9 @@ fn elapsed_us(started: Instant) -> u64 { /// A failed read yields an empty snapshot, so every mapping reads zero /// sessions, which is what the pool did with an unreadable source before. The /// alternative, treating "unknown" as "in use", would pin every mapping forever -/// on a kernel with no conntrack proc file and turn a read error into a pool -/// that never reclaims. The cost is the opposite error: a mapping carrying live -/// traffic can be reclaimed early while the source is unreadable. +/// on a kernel with no readable conntrack source and turn a read error into a +/// pool that never reclaims. The cost is the opposite error: a mapping carrying +/// live traffic can be reclaimed early while the source is unreadable. #[cfg(target_os = "linux")] async fn read_conntrack(log: &mut pool::ConntrackReadLog) -> pool::ConntrackSnapshot { use fips::gateway::pool::ConntrackQuerier; @@ -119,16 +119,27 @@ async fn report_conntrack_source() { .unwrap_or_else(|e| { pool::ConntrackProbe::Missing(pool::ConntrackUnreadable { proc: std::io::Error::other(e.to_string()), + netlink: None, }) }); match probe { pool::ConntrackProbe::Found(pool::ConntrackSource::Proc) => { info!("Conntrack source: proc; session pinning is on") } - pool::ConntrackProbe::Missing(e) => warn!( - proc_error = %e.proc, - "No conntrack source is readable; session pinning is off" - ), + pool::ConntrackProbe::Found(pool::ConntrackSource::Netlink) => { + info!("Conntrack source: netlink; session pinning is on") + } + pool::ConntrackProbe::Missing(e) => match e.netlink { + Some(netlink) => warn!( + proc_error = %e.proc, + netlink_error = %netlink, + "No conntrack source is readable; session pinning is off" + ), + None => warn!( + proc_error = %e.proc, + "No conntrack source is readable; session pinning is off" + ), + }, } } diff --git a/src/gateway/conntrack.rs b/src/gateway/conntrack.rs new file mode 100644 index 00000000..23fca85c --- /dev/null +++ b/src/gateway/conntrack.rs @@ -0,0 +1,330 @@ +//! Conntrack sessions read over netlink. +//! +//! A kernel built without `CONFIG_NF_CONNTRACK_PROCFS` has no +//! `/proc/net/nf_conntrack`, while `conntrack -L` still lists the table: it +//! asks the kernel for a dump over `NETLINK_NETFILTER`. This reader does the +//! same, so such a kernel can still pin mappings that carry traffic. + +use super::pool::{ConntrackQuerier, ConntrackSnapshot}; +use netlink_packet_core::{ + NLM_F_DUMP, NLM_F_REQUEST, NetlinkHeader, NetlinkMessage, NetlinkPayload, +}; +use netlink_packet_netfilter::conntrack::{ConntrackAttribute, ConntrackMessage, IPTuple, Tuple}; +use netlink_packet_netfilter::{ + NetfilterHeader, NetfilterMessage, NetfilterMessageInner, NetfilterProtoFamily, +}; +use netlink_sys::{Socket, SocketAddr, protocols::NETLINK_NETFILTER}; +use std::collections::{HashMap, HashSet}; +use std::io; +use std::net::{IpAddr, Ipv6Addr}; +use std::sync::atomic::{AtomicU32, Ordering}; +use std::time::Duration; + +/// Longest wait for each part of the kernel's reply. +/// +/// The read runs on a blocking thread once per tick, so a kernel that never +/// answers must not hold that thread for longer than a tick. +const READ_TIMEOUT: Duration = Duration::from_secs(2); + +/// Sequence number for the next dump request, so a reply to an earlier +/// request cannot be counted as part of this one. +static NEXT_SEQ: AtomicU32 = AtomicU32::new(1); + +/// Whether a dump has more to come after the buffer just counted. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DumpState { + /// The kernel has more of the dump to send. + More, + /// The kernel sent the end-of-dump message. + Done, +} + +/// Conntrack querier that dumps the table over netlink. +/// +/// Needs `CAP_NET_ADMIN` in the gateway's network namespace, which the +/// gateway already needs for its NAT table. +pub struct NetlinkConntrack; + +impl ConntrackQuerier for NetlinkConntrack { + fn snapshot(&self) -> Result { + let mut socket = Socket::new(NETLINK_NETFILTER)?; + socket.bind_auto()?; + socket.connect(&SocketAddr::new(0, 0))?; + socket2::SockRef::from(&socket).set_read_timeout(Some(READ_TIMEOUT))?; + + let seq = NEXT_SEQ.fetch_add(1, Ordering::Relaxed); + socket.send(&dump_request(seq), 0)?; + + let mut counts = HashMap::new(); + loop { + // Sized by peeking first, so a large batch is not truncated. + let (buf, _) = socket.recv_from_full()?; + if count_dump(&buf, seq, &mut counts)? == DumpState::Done { + break; + } + } + Ok(ConntrackSnapshot::from_counts(counts)) + } +} + +/// A request for a dump of the IPv6 conntrack table. +/// +/// The family in the netfilter header makes the kernel leave out IPv4 +/// entries, which could never name a virtual IP. +fn dump_request(seq: u32) -> Vec { + let mut header = NetlinkHeader::default(); + header.flags = NLM_F_REQUEST | NLM_F_DUMP; + header.sequence_number = seq; + let mut message = NetlinkMessage::new( + header, + NetlinkPayload::from(NetfilterMessage::new( + NetfilterHeader::new(NetfilterProtoFamily::IPv6, 0, 0), + ConntrackMessage::Get(vec![]), + )), + ); + message.finalize(); + let mut buf = vec![0; message.buffer_len()]; + message.serialize(&mut buf); + buf +} + +/// Count the conntrack entries in one received buffer by destination. +/// +/// An entry counts once for each distinct IPv6 destination among its original +/// and reply tuples, the same rule the proc-file parser applies to a line. +/// A message carrying another sequence number is skipped. A dump the kernel +/// flags as interrupted is counted as received: reading the proc file is not +/// atomic across the table either, and failing the read would zero every +/// mapping for the tick. +pub fn count_dump( + buf: &[u8], + seq: u32, + counts: &mut HashMap, +) -> Result { + let mut offset = 0; + while offset < buf.len() { + let message = NetlinkMessage::::deserialize(&buf[offset..]) + .map_err(|e| io::Error::new(io::ErrorKind::InvalidData, e.to_string()))?; + // Messages are padded to four bytes. The length is at least a header, + // or the parse above would have failed, so the walk always advances. + let len = message.header.length as usize; + offset += (len + 3) & !3; + if message.header.sequence_number != seq { + continue; + } + match message.payload { + NetlinkPayload::Done(_) => return Ok(DumpState::Done), + NetlinkPayload::Error(e) if e.code.is_some() => return Err(e.to_io()), + NetlinkPayload::InnerMessage(NetfilterMessage { + inner: NetfilterMessageInner::Conntrack(ConntrackMessage::New(attrs)), + .. + }) => count_entry(&attrs, counts), + _ => {} + } + } + Ok(DumpState::More) +} + +/// Add one conntrack entry to the counts, once per distinct IPv6 destination +/// among its original and reply tuples. +fn count_entry(attrs: &[ConntrackAttribute], counts: &mut HashMap) { + let mut seen = HashSet::new(); + for attr in attrs { + let tuples = match attr { + ConntrackAttribute::CtaTupleOrig(t) | ConntrackAttribute::CtaTupleReply(t) => t, + _ => continue, + }; + for tuple in tuples { + let Tuple::Ip(ip) = tuple else { continue }; + for field in ip { + if let IPTuple::DestinationAddress(IpAddr::V6(dst)) = field { + seen.insert(*dst); + } + } + } + } + for dst in seen { + *counts.entry(dst).or_insert(0) += 1; + } +} + +#[cfg(test)] +mod tests { + use super::*; + use netlink_packet_core::{DoneMessage, ErrorMessage}; + use std::num::NonZeroI32; + + const SEQ: u32 = 7; + + fn v6(s: &str) -> Ipv6Addr { + s.parse().unwrap() + } + + /// One tuple naming a source and a destination. + fn tuple(src: Ipv6Addr, dst: Ipv6Addr) -> Vec { + vec![Tuple::Ip(vec![ + IPTuple::SourceAddress(IpAddr::V6(src)), + IPTuple::DestinationAddress(IpAddr::V6(dst)), + ])] + } + + /// A conntrack entry as a dump reply carries it. + fn entry(orig: Vec, reply: Vec) -> NetfilterMessage { + NetfilterMessage::new( + NetfilterHeader::new(NetfilterProtoFamily::IPv6, 0, 0), + ConntrackMessage::New(vec![ + ConntrackAttribute::CtaTupleOrig(orig), + ConntrackAttribute::CtaTupleReply(reply), + ]), + ) + } + + /// Serialise one netlink message with the given sequence number. + fn frame(payload: NetlinkPayload, seq: u32) -> Vec { + let mut header = NetlinkHeader::default(); + header.sequence_number = seq; + header.flags = netlink_packet_core::NLM_F_MULTIPART; + let mut message = NetlinkMessage::new(header, payload); + message.finalize(); + let mut buf = vec![0; message.buffer_len()]; + message.serialize(&mut buf); + buf + } + + fn done() -> NetlinkPayload { + NetlinkPayload::Done(DoneMessage::default()) + } + + /// The flow the gateway sees for a LAN client using a virtual IP: the + /// original tuple is client to virtual IP, and the reply, after DNAT and + /// masquerade, is the mesh address back to the gateway. + fn client_flow(virtual_ip: Ipv6Addr) -> NetfilterMessage { + entry( + tuple(v6("fd02::20"), virtual_ip), + tuple(v6("fd9a::1"), v6("fd9a::2")), + ) + } + + fn count(buf: &[u8]) -> (Result, HashMap) { + let mut counts = HashMap::new(); + let state = count_dump(buf, SEQ, &mut counts); + (state, counts) + } + + #[test] + fn netlink_dump_counts_an_entry_whose_original_destination_is_the_virtual_ip() { + let virtual_ip = v6("fd01::1"); + let buf = frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ); + + let (state, counts) = count(&buf); + + assert_eq!(state.unwrap(), DumpState::More); + assert_eq!(counts.get(&virtual_ip).copied(), Some(1)); + // The reply tuple's destination is counted too, as the proc parser + // counts every dst= on the line. + assert_eq!(counts.get(&v6("fd9a::2")).copied(), Some(1)); + } + + #[test] + fn netlink_dump_counts_an_entry_once_when_both_tuples_name_the_address() { + let addr = v6("fd01::1"); + let hairpin = entry(tuple(addr, addr), tuple(addr, addr)); + let buf = frame(NetlinkPayload::from(hairpin), SEQ); + + let (_, counts) = count(&buf); + + assert_eq!(counts.get(&addr).copied(), Some(1)); + } + + #[test] + fn netlink_dump_counts_each_entry_across_several_messages_in_one_buffer() { + let virtual_ip = v6("fd01::1"); + let other = v6("fd01::2"); + let mut buf = frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ); + buf.extend(frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ)); + buf.extend(frame(NetlinkPayload::from(client_flow(other)), SEQ)); + + let (state, counts) = count(&buf); + + assert_eq!(state.unwrap(), DumpState::More); + assert_eq!(counts.get(&virtual_ip).copied(), Some(2)); + assert_eq!(counts.get(&other).copied(), Some(1)); + } + + #[test] + fn netlink_dump_ignores_an_ipv4_entry() { + let v4 = |s: &str| IpAddr::V4(s.parse().unwrap()); + let ipv4 = NetfilterMessage::new( + NetfilterHeader::new(NetfilterProtoFamily::IPv4, 0, 0), + ConntrackMessage::New(vec![ConntrackAttribute::CtaTupleOrig(vec![Tuple::Ip( + vec![ + IPTuple::SourceAddress(v4("192.0.2.1")), + IPTuple::DestinationAddress(v4("192.0.2.2")), + ], + )])]), + ); + let mut buf = frame(NetlinkPayload::from(ipv4), SEQ); + buf.extend(frame(NetlinkPayload::from(client_flow(v6("fd01::1"))), SEQ)); + + let (state, counts) = count(&buf); + + assert_eq!(state.unwrap(), DumpState::More); + assert_eq!(counts.len(), 2, "only the IPv6 entry's two destinations"); + } + + #[test] + fn netlink_dump_reports_done_on_the_done_message() { + let virtual_ip = v6("fd01::1"); + let mut buf = frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ); + buf.extend(frame(done(), SEQ)); + + let (state, counts) = count(&buf); + + assert_eq!(state.unwrap(), DumpState::Done); + assert_eq!(counts.get(&virtual_ip).copied(), Some(1)); + } + + #[test] + fn netlink_dump_turns_an_eperm_error_message_into_permission_denied() { + let mut error = ErrorMessage::default(); + error.code = NonZeroI32::new(-libc::EPERM); + let buf = frame(NetlinkPayload::Error(error), SEQ); + + let (state, _) = count(&buf); + + assert_eq!( + state.expect_err("an error reply must fail the read").kind(), + io::ErrorKind::PermissionDenied + ); + } + + #[test] + fn netlink_dump_skips_a_message_with_another_sequence_number() { + let virtual_ip = v6("fd01::1"); + let mut buf = frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ + 1); + buf.extend(frame(done(), SEQ + 1)); + buf.extend(frame(NetlinkPayload::from(client_flow(virtual_ip)), SEQ)); + + let (state, counts) = count(&buf); + + assert_eq!( + state.unwrap(), + DumpState::More, + "another request's end of dump does not end this one" + ); + assert_eq!(counts.get(&virtual_ip).copied(), Some(1)); + } + + #[test] + fn netlink_dump_rejects_a_buffer_that_does_not_parse() { + let mut buf = frame(NetlinkPayload::from(client_flow(v6("fd01::1"))), SEQ); + buf.truncate(buf.len() - 4); + + let (state, _) = count(&buf); + + assert_eq!( + state.expect_err("a short buffer must fail the read").kind(), + io::ErrorKind::InvalidData + ); + } +} diff --git a/src/gateway/mod.rs b/src/gateway/mod.rs index db3f2dec..df3e4ed7 100644 --- a/src/gateway/mod.rs +++ b/src/gateway/mod.rs @@ -3,6 +3,7 @@ //! Allows unmodified LAN hosts to reach FIPS mesh destinations via //! DNS-allocated virtual IPs and kernel nftables NAT. +pub mod conntrack; pub mod control; pub mod dns; pub mod nat; diff --git a/src/gateway/pool.rs b/src/gateway/pool.rs index 202a099b..bac11373 100644 --- a/src/gateway/pool.rs +++ b/src/gateway/pool.rs @@ -109,7 +109,10 @@ pub struct MappingInfo { pub last_ref_secs: u64, } -/// Path the conntrack table is read from. +/// Path the conntrack table is read from when the kernel provides it. +/// +/// A kernel built without `CONFIG_NF_CONNTRACK_PROCFS` has no such file; +/// `SystemConntrack` then dumps the table over netlink instead. const CONNTRACK_PROC_PATH: &str = "/proc/net/nf_conntrack"; /// Active conntrack sessions counted by destination address. @@ -164,6 +167,8 @@ impl ConntrackQuerier for ProcConntrack { pub enum ConntrackSource { /// `/proc/net/nf_conntrack`. Proc, + /// A conntrack table dump over `NETLINK_NETFILTER`. + Netlink, } impl ConntrackSource { @@ -171,6 +176,7 @@ impl ConntrackSource { pub fn name(self) -> &'static str { match self { Self::Proc => "proc", + Self::Netlink => "netlink", } } } @@ -180,12 +186,25 @@ impl ConntrackSource { pub struct ConntrackUnreadable { /// The error reading `/proc/net/nf_conntrack`. pub proc: std::io::Error, + /// The error from the netlink dump, when the proc file was absent and the + /// dump was tried. + pub netlink: Option, } impl ConntrackUnreadable { /// The error that stands for the whole failed read. + /// + /// When the dump was tried, its error is the one that decided the read, so + /// it sets the kind; the absent proc file is kept in the message. Only a + /// proc error that stopped the read before the dump stands alone. fn into_error(self) -> std::io::Error { - self.proc + match self.netlink { + Some(netlink) => std::io::Error::new( + netlink.kind(), + format!("proc: {}; netlink: {netlink}", self.proc), + ), + None => self.proc, + } } } @@ -193,34 +212,55 @@ impl ConntrackUnreadable { /// answered. /// /// The per-tick read and the startup probe both go through this type, so the -/// probe cannot report a source the tick would not use. The querier is a type -/// parameter so tests can substitute fakes. -pub struct SystemConntrack

{ +/// probe cannot report a source the tick would not use. The queriers are type +/// parameters so tests can substitute fakes. +/// +/// The proc file is read first. Only when it is absent is the table dumped +/// over netlink, and that is decided on every read: the file appears once +/// `nf_conntrack` is loaded in the namespace, so a choice fixed at startup +/// could keep using netlink on a kernel that has the file. +pub struct SystemConntrack

{ proc: P, + netlink: N, } -impl SystemConntrack

{ - /// A reader over the given proc querier. - pub fn new(proc: P) -> Self { - Self { proc } +impl SystemConntrack { + /// A reader over the given proc and netlink queriers. + pub fn new(proc: P, netlink: N) -> Self { + Self { proc, netlink } } /// Read conntrack once and say which source the snapshot came from. + /// + /// A proc error other than an absent file, such as a permission error, is + /// returned without trying netlink. pub fn read(&self) -> Result<(ConntrackSource, ConntrackSnapshot), ConntrackUnreadable> { match self.proc.snapshot() { Ok(snapshot) => Ok((ConntrackSource::Proc, snapshot)), - Err(proc) => Err(ConntrackUnreadable { proc }), + Err(proc) if proc.kind() == std::io::ErrorKind::NotFound => { + match self.netlink.snapshot() { + Ok(snapshot) => Ok((ConntrackSource::Netlink, snapshot)), + Err(netlink) => Err(ConntrackUnreadable { + proc, + netlink: Some(netlink), + }), + } + } + Err(proc) => Err(ConntrackUnreadable { + proc, + netlink: None, + }), } } } impl Default for SystemConntrack { fn default() -> Self { - Self::new(ProcConntrack) + Self::new(ProcConntrack, super::conntrack::NetlinkConntrack) } } -impl ConntrackQuerier for SystemConntrack

{ +impl ConntrackQuerier for SystemConntrack { fn snapshot(&self) -> Result { self.read() .map(|(_, snapshot)| snapshot) @@ -239,7 +279,9 @@ pub enum ConntrackProbe { } /// Read conntrack once, as a tick would, and report which source answered. -pub fn probe_conntrack(reader: &SystemConntrack

) -> ConntrackProbe { +pub fn probe_conntrack( + reader: &SystemConntrack, +) -> ConntrackProbe { match reader.read() { Ok((source, _)) => ConntrackProbe::Found(source), Err(e) => ConntrackProbe::Missing(e), @@ -294,12 +336,12 @@ pub enum ReadReport { /// Remembers the last conntrack read outcome. /// -/// A kernel built without `CONFIG_NF_CONNTRACK_PROCFS` has no -/// `/proc/net/nf_conntrack` at all, so every read fails the same way and a -/// per-tick warning would repeat for the life of the process. Warning on a -/// change of outcome still separates "the source is unreadable" from "there -/// are no sessions", which the pool could not distinguish before, without -/// filling the log. +/// When no source is readable, for example a kernel with no +/// `/proc/net/nf_conntrack` whose netlink dump is refused, every read fails +/// the same way and a per-tick warning would repeat for the life of the +/// process. Warning on a change of outcome still separates "the source is +/// unreadable" from "there are no sessions", which the pool could not +/// distinguish before, without filling the log. #[derive(Debug, Default)] pub struct ConntrackReadLog { last: Option>, @@ -1238,7 +1280,7 @@ mod tests { #[test] fn conntrack_probe_names_the_proc_source_when_the_proc_read_succeeds() { - let reader = SystemConntrack::new(FixedRead(None)); + let reader = SystemConntrack::new(FixedRead(None), NOT_CALLED); match probe_conntrack(&reader) { ConntrackProbe::Found(source) => { @@ -1251,7 +1293,10 @@ mod tests { #[test] fn conntrack_probe_reports_missing_with_the_error_when_the_proc_read_fails() { - let reader = SystemConntrack::new(FixedRead(Some(std::io::ErrorKind::PermissionDenied))); + let reader = SystemConntrack::new( + FixedRead(Some(std::io::ErrorKind::PermissionDenied)), + NOT_CALLED, + ); match probe_conntrack(&reader) { ConntrackProbe::Missing(e) => { @@ -1260,4 +1305,89 @@ mod tests { ConntrackProbe::Found(source) => panic!("expected no source, got {source:?}"), } } + + /// A netlink stand-in for tests where the dump must not be reached. It + /// fails with a kind no test expects, so reaching it shows in the result. + const NOT_CALLED: FixedRead = FixedRead(Some(std::io::ErrorKind::Unsupported)); + + /// A conntrack querier that reports one session to a fixed address. + struct OneSession(Ipv6Addr); + + impl ConntrackQuerier for OneSession { + fn snapshot(&self) -> Result { + Ok(ConntrackSnapshot::from_counts(HashMap::from([(self.0, 1)]))) + } + } + + #[test] + fn system_conntrack_falls_back_to_netlink_when_the_proc_file_is_absent() { + let addr: Ipv6Addr = "fd01::1".parse().unwrap(); + let reader = SystemConntrack::new( + FixedRead(Some(std::io::ErrorKind::NotFound)), + OneSession(addr), + ); + + let (source, snapshot) = reader.read().expect("the netlink dump answered"); + + assert_eq!(source, ConntrackSource::Netlink); + assert_eq!(source.name(), "netlink"); + assert_eq!(snapshot.sessions_for(addr), 1); + } + + #[test] + fn system_conntrack_does_not_fall_back_on_a_proc_error_other_than_not_found() { + let addr: Ipv6Addr = "fd01::1".parse().unwrap(); + let reader = SystemConntrack::new( + FixedRead(Some(std::io::ErrorKind::PermissionDenied)), + OneSession(addr), + ); + + let e = reader + .read() + .expect_err("a denied proc read is not a missing file"); + + assert_eq!(e.proc.kind(), std::io::ErrorKind::PermissionDenied); + assert!(e.netlink.is_none(), "netlink was not tried"); + assert_eq!( + reader.snapshot().unwrap_err().kind(), + std::io::ErrorKind::PermissionDenied + ); + } + + #[test] + fn system_conntrack_prefers_proc_when_it_reads() { + let proc_addr: Ipv6Addr = "fd01::1".parse().unwrap(); + let netlink_addr: Ipv6Addr = "fd01::2".parse().unwrap(); + let reader = SystemConntrack::new(OneSession(proc_addr), OneSession(netlink_addr)); + + let (source, snapshot) = reader.read().expect("the proc file answered"); + + assert_eq!(source, ConntrackSource::Proc); + assert_eq!(snapshot.sessions_for(proc_addr), 1); + assert_eq!(snapshot.sessions_for(netlink_addr), 0); + } + + #[test] + fn conntrack_probe_reports_both_errors_when_neither_source_reads() { + let reader = SystemConntrack::new( + FixedRead(Some(std::io::ErrorKind::NotFound)), + FixedRead(Some(std::io::ErrorKind::PermissionDenied)), + ); + + match probe_conntrack(&reader) { + ConntrackProbe::Missing(e) => { + assert_eq!(e.proc.kind(), std::io::ErrorKind::NotFound); + assert_eq!( + e.netlink.as_ref().map(std::io::Error::kind), + Some(std::io::ErrorKind::PermissionDenied) + ); + } + ConntrackProbe::Found(source) => panic!("expected no source, got {source:?}"), + } + // The per-tick read reports the error that decided it: the dump's. + assert_eq!( + reader.snapshot().unwrap_err().kind(), + std::io::ErrorKind::PermissionDenied + ); + } } diff --git a/testing/static/scripts/gateway-test.sh b/testing/static/scripts/gateway-test.sh index db625cea..dd441b91 100755 --- a/testing/static/scripts/gateway-test.sh +++ b/testing/static/scripts/gateway-test.sh @@ -155,21 +155,25 @@ fi # The gateway names its conntrack source once at startup, before the DNS # resolver starts, so by now the line is in the log. Ask the gateway's own -# namespace which source it should have found. A failed `docker logs` reds the -# check rather than counting as zero lines. +# namespace which source it should have found: the proc file when it exists, +# and otherwise the netlink dump, which the container's NET_ADMIN allows. A +# failed `docker logs` reds the check rather than counting as zero lines. if docker exec "$GATEWAY" test -e /proc/net/nf_conntrack; then EXPECT_SRC=proc else - EXPECT_SRC=none + EXPECT_SRC=netlink fi if GW_START_LOG=$(docker logs "$GATEWAY" 2>&1); then SRC_PROC=$(grep -cF 'Conntrack source: proc; session pinning is on' <<< "$GW_START_LOG" || true) + SRC_NETLINK=$(grep -cF 'Conntrack source: netlink; session pinning is on' <<< "$GW_START_LOG" || true) SRC_NONE=$(grep -cF 'No conntrack source is readable; session pinning is off' <<< "$GW_START_LOG" || true) case "$EXPECT_SRC" in - proc) SRC_OK=$([ "$SRC_PROC" -eq 1 ] && [ "$SRC_NONE" -eq 0 ] && echo 0 || echo 1) ;; - *) SRC_OK=$([ "$SRC_NONE" -eq 1 ] && [ "$SRC_PROC" -eq 0 ] && echo 0 || echo 1) ;; + proc) SRC_HIT=$SRC_PROC ;; + *) SRC_HIT=$SRC_NETLINK ;; esac - check "Conntrack source line at startup (expect $EXPECT_SRC; proc lines $SRC_PROC, none lines $SRC_NONE)" "$SRC_OK" + SRC_ALL=$((SRC_PROC + SRC_NETLINK + SRC_NONE)) + SRC_OK=$([ "$SRC_HIT" -eq 1 ] && [ "$SRC_ALL" -eq 1 ] && echo 0 || echo 1) + check "Conntrack source line at startup (expect $EXPECT_SRC; proc lines $SRC_PROC, netlink lines $SRC_NETLINK, none lines $SRC_NONE)" "$SRC_OK" else check "Conntrack source line at startup (docker logs failed)" 1 fi @@ -299,6 +303,43 @@ else check "nftables DNAT rules" 1 fi +# Phase 5's GET left a TCP conntrack entry to the first virtual IP, which +# stays in the table in TIME_WAIT well past two ticks. The gateway must count +# it. Poll about once a second for 25 tries, which spans two 10s ticks, and +# read the count with a parser that cannot turn a failed query into a number: +# an error response has no `data`, and the parser exits non-zero on it. +if [ -n "$VIRTUAL_IP" ]; then + VIP_SESSIONS=error + for _ in $(seq 1 25); do + VIP_SESSIONS=$(docker exec "$GATEWAY" bash -c \ + 'echo "{\"command\":\"show_mappings\"}" | nc -U -w1 /run/fips/gateway.sock 2>/dev/null' \ + | VIP="$VIRTUAL_IP" python3 -c " +import os, sys, json +r = json.load(sys.stdin) +data = r.get('data') +if not isinstance(data, dict) or not isinstance(data.get('mappings'), list): + sys.exit(1) +hits = [m for m in data['mappings'] if m.get('virtual_ip') == os.environ['VIP']] +if len(hits) != 1 or not isinstance(hits[0].get('sessions'), int): + sys.exit(1) +print(hits[0]['sessions']) +" 2>/dev/null || echo "error") + if [ "$VIP_SESSIONS" != error ] && [ "$VIP_SESSIONS" -ge 1 ]; then + break + fi + sleep 1 + done + if [ "$VIP_SESSIONS" != error ] && [ "$VIP_SESSIONS" -ge 1 ]; then + check "Gateway counts a session to $VIRTUAL_IP (sessions $VIP_SESSIONS, source $EXPECT_SRC)" 0 + else + check "Gateway counts a session to $VIRTUAL_IP (sessions $VIP_SESSIONS, source $EXPECT_SRC)" 1 + echo " Kernel conntrack entries to $VIRTUAL_IP:" + docker exec "$GATEWAY" conntrack -L -f ipv6 -d "$VIRTUAL_IP" 2>&1 | sed 's/^/ /' || true + fi +else + check "Gateway counts a session (skipped — no virtual IP)" 1 +fi + # Phase 7: Inbound port forwarding — UDP and a second simultaneous TCP forward. # # Three forwards exercised: