diff --git a/CHANGELOG.md b/CHANGELOG.md index db0e8bea..6154cec5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -164,6 +164,11 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 indistinguishable from an idle one. A kernel built without `CONFIG_NF_CONNTRACK_PROCFS` has no `/proc/net/nf_conntrack` at all and fails identically every tick, so a repeat is logged at debug rather than warn. +- The gateway says at startup whether it can read conntrack sessions. It + reads the table once, as each tick does, and logs either the source it read + or that no source is readable and session pinning is off. An operator on a + kernel with no readable source learned this only from a warning at the first + failed tick. - The NAT table is rebuilt in one netlink transaction. A rebuild deleted the `fips_gateway` table in a batch of its own, discarded that batch's result, and only then sent the batch that recreated the table, the chains, the diff --git a/docs/how-to/troubleshoot-gateway.md b/docs/how-to/troubleshoot-gateway.md index 7c570df5..4fbcb419 100644 --- a/docs/how-to/troubleshoot-gateway.md +++ b/docs/how-to/troubleshoot-gateway.md @@ -159,6 +159,28 @@ failed to bind the socket and continued without it (the warning `Failed to bind gateway control socket — continuing without it` is in the journal in that case). +### Session pinning is off + +The gateway keeps a mapping while conntrack shows sessions to its +virtual IP. At startup it reads conntrack once, the same way each +10 s tick does, and logs which source answered: + +- `Conntrack source: proc; session pinning is on`: sessions are read + from `/proc/net/nf_conntrack`. +- `No conntrack source is readable; session pinning is off`: no + source could be read, and the line carries the error. Every mapping + then reads zero sessions, so a mapping is reclaimed on its TTL and + grace period alone, even while a client that has not re-queried DNS + still has traffic flowing through it. + +The proc file exists only on a kernel built with +`CONFIG_NF_CONNTRACK_PROCFS`, and only once `nf_conntrack` is loaded: + +```sh +ls /proc/net/nf_conntrack +grep NF_CONNTRACK_PROCFS /boot/config-$(uname -r) +``` + ## Outbound-half diagnostics Symptoms in this section all involve a LAN client trying to reach a diff --git a/src/bin/fips-gateway.rs b/src/bin/fips-gateway.rs index 6a6979fb..48166685 100644 --- a/src/bin/fips-gateway.rs +++ b/src/bin/fips-gateway.rs @@ -72,7 +72,7 @@ fn elapsed_us(started: Instant) -> u64 { async fn read_conntrack(log: &mut pool::ConntrackReadLog) -> pool::ConntrackSnapshot { use fips::gateway::pool::ConntrackQuerier; - match tokio::task::spawn_blocking(|| pool::ProcConntrack.snapshot()).await { + match tokio::task::spawn_blocking(|| pool::SystemConntrack::default().snapshot()).await { Ok(Ok(snapshot)) => { log.observe(None); snapshot @@ -107,6 +107,31 @@ fn report_unreadable_conntrack( } } +/// Check once at startup which conntrack source the tick will read, and say so. +/// +/// Without this, an operator on a kernel with no readable source learns that +/// session pinning is off only from a warning at the first failed tick. +#[cfg(target_os = "linux")] +async fn report_conntrack_source() { + let probe = + tokio::task::spawn_blocking(|| pool::probe_conntrack(&pool::SystemConntrack::default())) + .await + .unwrap_or_else(|e| { + pool::ConntrackProbe::Missing(pool::ConntrackUnreadable { + proc: std::io::Error::other(e.to_string()), + }) + }); + match probe { + pool::ConntrackProbe::Found(pool::ConntrackSource::Proc) => { + info!("Conntrack source: proc; session pinning is on") + } + pool::ConntrackProbe::Missing(e) => warn!( + proc_error = %e.proc, + "No conntrack source is readable; session pinning is off" + ), + } +} + #[cfg(target_os = "linux")] #[tokio::main(flavor = "current_thread")] async fn main() { @@ -371,6 +396,10 @@ async fn main() { std::process::exit(1); } + // The NAT table exists by now, so a kernel that provides the proc file + // has loaded nf_conntrack and the probe sees what the first tick will. + report_conntrack_source().await; + // --- Channels --- // Pool events (new/removed mappings) → NAT + net modules diff --git a/src/gateway/pool.rs b/src/gateway/pool.rs index 7776b160..202a099b 100644 --- a/src/gateway/pool.rs +++ b/src/gateway/pool.rs @@ -159,6 +159,93 @@ impl ConntrackQuerier for ProcConntrack { } } +/// Where a conntrack snapshot was read from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ConntrackSource { + /// `/proc/net/nf_conntrack`. + Proc, +} + +impl ConntrackSource { + /// Short name of the source. + pub fn name(self) -> &'static str { + match self { + Self::Proc => "proc", + } + } +} + +/// Why no conntrack source could be read. +#[derive(Debug)] +pub struct ConntrackUnreadable { + /// The error reading `/proc/net/nf_conntrack`. + pub proc: std::io::Error, +} + +impl ConntrackUnreadable { + /// The error that stands for the whole failed read. + fn into_error(self) -> std::io::Error { + self.proc + } +} + +/// The conntrack reader the gateway uses, which also says which source +/// answered. +/// +/// The per-tick read and the startup probe both go through this type, so the +/// probe cannot report a source the tick would not use. The querier is a type +/// parameter so tests can substitute fakes. +pub struct SystemConntrack

{ + proc: P, +} + +impl SystemConntrack

{ + /// A reader over the given proc querier. + pub fn new(proc: P) -> Self { + Self { proc } + } + + /// Read conntrack once and say which source the snapshot came from. + pub fn read(&self) -> Result<(ConntrackSource, ConntrackSnapshot), ConntrackUnreadable> { + match self.proc.snapshot() { + Ok(snapshot) => Ok((ConntrackSource::Proc, snapshot)), + Err(proc) => Err(ConntrackUnreadable { proc }), + } + } +} + +impl Default for SystemConntrack { + fn default() -> Self { + Self::new(ProcConntrack) + } +} + +impl ConntrackQuerier for SystemConntrack

{ + fn snapshot(&self) -> Result { + self.read() + .map(|(_, snapshot)| snapshot) + .map_err(ConntrackUnreadable::into_error) + } +} + +/// Outcome of the startup check for a readable conntrack source. +#[derive(Debug)] +pub enum ConntrackProbe { + /// Sessions can be read, from this source. + Found(ConntrackSource), + /// No source can be read, so every mapping reads zero sessions and session + /// pinning is off. + Missing(ConntrackUnreadable), +} + +/// Read conntrack once, as a tick would, and report which source answered. +pub fn probe_conntrack(reader: &SystemConntrack

) -> ConntrackProbe { + match reader.read() { + Ok((source, _)) => ConntrackProbe::Found(source), + Err(e) => ConntrackProbe::Missing(e), + } +} + /// Count conntrack lines by the destination addresses they name. /// /// Every `dst=` value is parsed as an address and compared as an address. The @@ -1135,4 +1222,42 @@ mod tests { "a different failure is a different outcome and is worth a line" ); } + + /// A conntrack querier that succeeds with an empty snapshot, or fails with + /// a fixed error kind. + struct FixedRead(Option); + + impl ConntrackQuerier for FixedRead { + fn snapshot(&self) -> Result { + match self.0 { + None => Ok(ConntrackSnapshot::default()), + Some(kind) => Err(kind.into()), + } + } + } + + #[test] + fn conntrack_probe_names_the_proc_source_when_the_proc_read_succeeds() { + let reader = SystemConntrack::new(FixedRead(None)); + + match probe_conntrack(&reader) { + ConntrackProbe::Found(source) => { + assert_eq!(source, ConntrackSource::Proc); + assert_eq!(source.name(), "proc"); + } + ConntrackProbe::Missing(e) => panic!("expected the proc source, got {e:?}"), + } + } + + #[test] + fn conntrack_probe_reports_missing_with_the_error_when_the_proc_read_fails() { + let reader = SystemConntrack::new(FixedRead(Some(std::io::ErrorKind::PermissionDenied))); + + match probe_conntrack(&reader) { + ConntrackProbe::Missing(e) => { + assert_eq!(e.proc.kind(), std::io::ErrorKind::PermissionDenied); + } + ConntrackProbe::Found(source) => panic!("expected no source, got {source:?}"), + } + } } diff --git a/testing/static/scripts/gateway-test.sh b/testing/static/scripts/gateway-test.sh index 3153bf71..db625cea 100755 --- a/testing/static/scripts/gateway-test.sh +++ b/testing/static/scripts/gateway-test.sh @@ -153,6 +153,27 @@ if [ "$DNS_READY" != true ]; then echo " WARNING: Gateway DNS did not respond within 30s, continuing anyway" fi +# The gateway names its conntrack source once at startup, before the DNS +# resolver starts, so by now the line is in the log. Ask the gateway's own +# namespace which source it should have found. A failed `docker logs` reds the +# check rather than counting as zero lines. +if docker exec "$GATEWAY" test -e /proc/net/nf_conntrack; then + EXPECT_SRC=proc +else + EXPECT_SRC=none +fi +if GW_START_LOG=$(docker logs "$GATEWAY" 2>&1); then + SRC_PROC=$(grep -cF 'Conntrack source: proc; session pinning is on' <<< "$GW_START_LOG" || true) + SRC_NONE=$(grep -cF 'No conntrack source is readable; session pinning is off' <<< "$GW_START_LOG" || true) + case "$EXPECT_SRC" in + proc) SRC_OK=$([ "$SRC_PROC" -eq 1 ] && [ "$SRC_NONE" -eq 0 ] && echo 0 || echo 1) ;; + *) SRC_OK=$([ "$SRC_NONE" -eq 1 ] && [ "$SRC_PROC" -eq 0 ] && echo 0 || echo 1) ;; + esac + check "Conntrack source line at startup (expect $EXPECT_SRC; proc lines $SRC_PROC, none lines $SRC_NONE)" "$SRC_OK" +else + check "Conntrack source line at startup (docker logs failed)" 1 +fi + # Phase 3: Client network setup — route virtual IP pool via gateway echo "" echo "Phase 3: Client network setup"