From 0b44c4623512b316c55d9356b5b85ce298f6f742 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Tue, 6 Oct 2026 00:03:26 +0000 Subject: [PATCH 1/7] Place the gateway's LAN masquerade ahead of the per-mapping SNAT rules The NAT rebuild listed the LAN-side masquerade last in postrouting, after every per-mapping SNAT. Those SNAT rules match on source address alone and NAT statements are terminal, so an inbound port-forwarded connection from a mesh peer that also held a live .fips mapping was rewritten to the peer's pool address instead of the gateway's LAN address. The LAN target then saw an address that changed as mappings came and went and that only a host routing the pool to the gateway could answer. Emit the LAN masquerade right after the fips0 masquerade, before the mapping rules. Prerouting order and the rule shapes are unchanged. A unit test checks the op order. The gateway suite now checks the order in the kernel's table and adds a phase that probes a port forward while the probing peer holds a live mapping, confirming from conntrack that the reply goes to the gateway's LAN address. The design doc describes the order. --- docs/design/fips-gateway.md | 13 +- src/gateway/nat.rs | 49 +++- testing/static/scripts/gateway-test.sh | 321 ++++++++++++++++++++----- 3 files changed, 316 insertions(+), 67 deletions(-) diff --git a/docs/design/fips-gateway.md b/docs/design/fips-gateway.md index 34d15f70..a123ea74 100644 --- a/docs/design/fips-gateway.md +++ b/docs/design/fips-gateway.md @@ -419,6 +419,13 @@ masquerade in the outbound pipeline; the two have disjoint match clauses (different `iifname`/`oifname` combinations) and coexist without interaction when both directions are active. +The per-mapping SNAT rules match on source address alone, so an +inbound forwarded connection from a mesh peer that also holds a live +mapping matches both its SNAT and the LAN-side masquerade. NAT +statements are terminal, and the rebuild places the LAN-side +masquerade ahead of every per-mapping SNAT, so that peer's connection +reaches the LAN target from the gateway's LAN address like any other. + ### Independence From Outbound The inbound half does not require: @@ -449,9 +456,9 @@ sequence is: 1. Add the table (which succeeds whether or not it exists), delete it, and add it again, so the delete always has a target. 2. Add the `prerouting` and `postrouting` chains; the always-on - `oifname fips0` masquerade; per-mapping DNAT/SNAT rules for - every live pool entry; per-port-forward DNAT rules; the LAN-side - masquerade if any port-forwards exist. + `oifname fips0` masquerade; the LAN-side masquerade if any + port-forwards exist; per-mapping DNAT/SNAT rules for every live + pool entry; per-port-forward DNAT rules. 3. Send all of it as one batch, which the kernel applies as a single transaction. Only the last message before the batch end requests an acknowledgement, the socket's send buffer is sized to the diff --git a/src/gateway/nat.rs b/src/gateway/nat.rs index 72de3a89..517814ca 100644 --- a/src/gateway/nat.rs +++ b/src/gateway/nat.rs @@ -341,6 +341,16 @@ impl NatManager { NatOp::FipsMasquerade, ]; + // When any port forwards are configured, one LAN-side masquerade in + // postrouting gives the LAN target the gateway's LAN address as the + // source, so replies flow back through conntrack. It goes ahead of + // every per-mapping SNAT: those match on source address alone, NAT + // statements are terminal, and a SNAT listed first would take an + // inbound forwarded flow from a peer that holds a live mapping. + if !self.port_forwards.is_empty() { + ops.push(NatOp::LanMasquerade); + } + for mapping in self.mappings.values() { ops.push(NatOp::Dnat(mapping.virtual_ip)); ops.push(NatOp::Snat(mapping.virtual_ip)); @@ -348,15 +358,9 @@ impl NatManager { // Inbound port-forward rules. Each forward is one DNAT rule in // prerouting keyed on (iif fips0, nfproto ipv6, l4proto, th dport). - // When any forwards are configured, emit a single LAN-side masquerade - // in postrouting so the LAN target host sees the gateway's LAN address - // as source and replies flow back through conntrack. for index in 0..self.port_forwards.len() { ops.push(NatOp::PortForward(index)); } - if !self.port_forwards.is_empty() { - ops.push(NatOp::LanMasquerade); - } vec![ops] } @@ -1246,6 +1250,39 @@ mod tests { ); } + #[test] + fn rebuild_places_the_lan_masquerade_before_every_mapping_snat() { + let mut mgr = manager_with_mappings(3); + mgr.port_forwards = vec![PortForward { + proto: Proto::Tcp, + listen_port: 8080, + target: SocketAddrV6::new(Ipv6Addr::LOCALHOST, 80, 0, 0), + }]; + + let ops = mgr.rebuild_batches().remove(0); + + let masquerade = ops + .iter() + .position(|op| matches!(op, NatOp::LanMasquerade)) + .expect("a port forward emits the LAN masquerade"); + let snats: Vec = ops + .iter() + .enumerate() + .filter(|(_, op)| matches!(op, NatOp::Snat(_))) + .map(|(i, _)| i) + .collect(); + assert_eq!(snats.len(), 3, "one SNAT per mapping: {ops:?}"); + for snat in snats { + assert!( + masquerade < snat, + "the LAN masquerade at index {masquerade} follows the SNAT at \ + index {snat}, so an inbound forwarded flow from a peer with a \ + live mapping takes the SNAT and bypasses the masquerade: \ + {ops:?}" + ); + } + } + /// The encoded rebuild of a manager holding `count` mappings. fn encoded_rebuild(count: u16) -> Vec { let mgr = manager_with_mappings(count); diff --git a/testing/static/scripts/gateway-test.sh b/testing/static/scripts/gateway-test.sh index c116821d..1a80a3ba 100755 --- a/testing/static/scripts/gateway-test.sh +++ b/testing/static/scripts/gateway-test.sh @@ -168,6 +168,52 @@ print(found[0]) ' } +# How many per-mapping SNAT rules come before the LAN masquerade, from +# `nft list table inet fips_gateway`. NAT statements are terminal, so a SNAT +# listed first takes an inbound forwarded flow from that mapping's peer and +# the masquerade never runs; the right answer is 0. Fails when the listing +# does not hold exactly one LAN masquerade. +snat_before_masq() { + python3 -c ' +import re, sys +snat = 0 +before = None +masq = 0 +for line in sys.stdin: + if "iifname \"fips0\"" in line and re.search(r"\bmasquerade\b", line): + masq += 1 + before = snat + elif re.search(r"\bsaddr [0-9a-f:]+ .*\bsnat\b", line): + snat += 1 +if masq != 1: + sys.exit(1) +print(before) +' +} + +# The SNAT target of the one rule whose source match is mesh address $1, from +# `nft list table inet fips_gateway`. Fails when no rule or more than one +# matches. +snat_to() { + python3 -c ' +import ipaddress, re, sys +want = ipaddress.ip_address(sys.argv[1]) +found = [] +for line in sys.stdin: + m = re.search(r"\bsaddr ([0-9a-f:]+) .*\bsnat\b.*?\bto \[?([0-9a-f:]+)", line) + if not m: + continue + try: + if ipaddress.ip_address(m.group(1)) == want: + found.append(ipaddress.ip_address(m.group(2))) + except ValueError: + sys.exit(1) +if len(found) != 1: + sys.exit(1) +print(found[0]) +' "$@" +} + # The device of the proxy neighbour entry for address $1, from # `ip -6 neigh show proxy`, whose lines read `ADDR dev DEV proxy`. Fails when # no entry matches or matching entries name different devices. @@ -337,6 +383,30 @@ gw_selftest() { gw_case "masq_iface: no LAN masquerade" 1 "" "$nft_nolan" masq_iface || fails=$((fails + 1)) gw_case "masq_iface: empty input" 1 "" "" masq_iface || fails=$((fails + 1)) + # nft_lan lists the SNAT ahead of the LAN masquerade, the order that lets + # a mapped peer's inbound forward bypass the masquerade. nft_fixed moves + # the masquerade ahead of it, and nft_dup adds a second SNAT for the same + # mesh address. + local masq_line nft_fixed nft_dup + masq_line=$(grep 'iifname "fips0" oifname' <<< "$nft_lan") + nft_fixed=$(awk -v m="$masq_line" '$0 == m {next} /saddr .* snat/ {print m} {print}' <<< "$nft_lan") + nft_dup=$(awk '/saddr .* snat/ {print; sub(/fd01::1/, "fd01::2")} {print}' <<< "$nft_lan") + gw_case "snat_before_masq: SNAT listed first" 0 1 "$nft_lan" snat_before_masq || fails=$((fails + 1)) + gw_case "snat_before_masq: masquerade listed first" 0 0 "$nft_fixed" snat_before_masq || fails=$((fails + 1)) + gw_case "snat_before_masq: no LAN masquerade" 1 "" "$nft_nolan" snat_before_masq || fails=$((fails + 1)) + gw_case "snat_before_masq: empty input" 1 "" "" snat_before_masq || fails=$((fails + 1)) + gw_case "snat_to: mesh address, short form" 0 fd01::1 "$nft_lan" \ + snat_to fd3c:9a51:7e02:4b18::2 || fails=$((fails + 1)) + gw_case "snat_to: mesh address, long form" 0 fd01::1 "$nft_lan" \ + snat_to fd3c:9a51:7e02:4b18:0:0:0:2 || fails=$((fails + 1)) + gw_case "snat_to: masquerade listed first" 0 fd01::1 "$nft_fixed" \ + snat_to fd3c:9a51:7e02:4b18::2 || fails=$((fails + 1)) + gw_case "snat_to: another mesh address" 1 "" "$nft_lan" \ + snat_to fd3c:9a51:7e02:4b18::3 || fails=$((fails + 1)) + gw_case "snat_to: two rules for one mesh address" 1 "" "$nft_dup" \ + snat_to fd3c:9a51:7e02:4b18::2 || fails=$((fails + 1)) + gw_case "snat_to: empty input" 1 "" "" snat_to fd3c:9a51:7e02:4b18::2 || fails=$((fails + 1)) + # Captured lines: one run's entries were on eth0 and a wrong-interface # run's on eth1. The mixed inputs join lines from the two captures, since # no single run holds entries on both devices. @@ -463,6 +533,60 @@ lan_agree() { return 0 } +# Start the LAN-side responders the inbound port forwards reach, on gw-client, +# and wait briefly for them to bind. Used by phases 8b and 8c. +gw_responders_start() { + # Start marker HTTP servers on the LAN-side client. + # :8080 → "inbound-forward-ok" (target of tcp 18080) + # :8081 → "inbound-forward-ok-2" (target of tcp 18082) + # `docker exec -d` is required; `docker exec bash -c 'cmd &'` doesn't + # keep the child alive past the exec session, even with nohup. + docker exec "$CLIENT" sh -c ' + mkdir -p /tmp/inbound /tmp/inbound2 + echo "inbound-forward-ok" > /tmp/inbound/index.html + echo "inbound-forward-ok-2" > /tmp/inbound2/index.html + pkill -f "http.server 8080" 2>/dev/null || true + pkill -f "http.server 8081" 2>/dev/null || true + pkill -f "udp_echo.py" 2>/dev/null || true + ' >/dev/null 2>&1 || true + docker exec -d "$CLIENT" python3 -m http.server 8080 --bind :: --directory /tmp/inbound \ + >/dev/null 2>&1 || true + docker exec -d "$CLIENT" python3 -m http.server 8081 --bind :: --directory /tmp/inbound2 \ + >/dev/null 2>&1 || true + + # Start a UDP echo server on the LAN-side client at [::]:8081/udp. + # This is the target of the udp 18081 forward. Stash the script as a + # named file (`udp_echo.py`) so the cleanup pkill above can find it. + docker exec "$CLIENT" sh -c 'cat > /tmp/udp_echo.py <<'\''PYEOF'\'' +import socket, sys +s = socket.socket(socket.AF_INET6, socket.SOCK_DGRAM) +s.bind(("::", 8081)) +while True: + data, addr = s.recvfrom(2048) + s.sendto(b"udp-forward-ok:" + data, addr) +PYEOF' >/dev/null 2>&1 || true + docker exec -d "$CLIENT" python3 /tmp/udp_echo.py >/dev/null 2>&1 || true + + # Give the servers a moment to bind. + for _ in 1 2 3 4 5; do + TCP_READY=$(docker exec "$CLIENT" ss -6lnt 2>/dev/null | grep -cE ':8080|:8081' || true) + UDP_READY=$(docker exec "$CLIENT" ss -6lnu 2>/dev/null | grep -c ':8081' || true) + if [ "$TCP_READY" -ge 2 ] && [ "$UDP_READY" -ge 1 ]; then + break + fi + sleep 1 + done +} + +# Stop the responders gw_responders_start started. +gw_responders_stop() { + docker exec "$CLIENT" sh -c ' + pkill -f "http.server 8080" 2>/dev/null || true + pkill -f "http.server 8081" 2>/dev/null || true + pkill -f "udp_echo.py" 2>/dev/null || true + ' >/dev/null 2>&1 || true +} + echo "=== FIPS Gateway Integration Test ===" echo "" @@ -725,10 +849,12 @@ fi # udp 18081 → [fd02::20]:8081 (6A — UDP DNAT runtime path) # # Checks the DNAT rules and the LAN-side masquerade that set_port_forwards() -# installs. The traffic through them is Phase 8b's: while the Phase 4 -# mapping to gw-server is live, its SNAT rule matches gw-server's inbound -# flows before the LAN masquerade does, so probes sent here would pass -# without the masquerade. +# installs, and that the masquerade is listed ahead of every per-mapping +# SNAT. NAT statements are terminal, so a SNAT listed first would take an +# inbound forwarded flow from that mapping's peer and the target would see a +# pool address. Phase 6's listing holds Phase 4's live mappings, so it has +# SNAT rules to order against. The traffic through the forwards is Phase +# 8b's, with no mapping to gw-server, and Phase 8c's, with one. echo "" echo "Phase 7: Inbound port-forward rules" @@ -762,6 +888,19 @@ if [ -n "$LAN_IF" ] && MASQ_IF=$(masq_iface <<< "$NFT_RULES"); then else check "LAN masquerade on the LAN interface '$LAN_IF' (no single LAN masquerade rule)" 1 fi +# At least one SNAT must be listed, so an empty or reclaimed table cannot +# pass by having nothing to order. +ORDER_SNAT=$(grep -cE "saddr [0-9a-f:]+ .*snat" <<< "$NFT_RULES" || true) +if SNAT_FIRST=$(snat_before_masq <<< "$NFT_RULES"); then + : +else + SNAT_FIRST=error +fi +if [ "$SNAT_FIRST" = "0" ] && [ "$ORDER_SNAT" -ge 1 ]; then + check "LAN masquerade listed ahead of all $ORDER_SNAT SNAT rules" 0 +else + check "LAN masquerade listed ahead of every SNAT rule (SNAT rules before it: $SNAT_FIRST, SNAT rules listed: $ORDER_SNAT)" 1 +fi # Phase 8: TTL expiration and pool reclamation echo "" @@ -805,9 +944,10 @@ fi # # Mesh peer (gw-server) hits each gw-gateway fips0: rule, which DNATs # into the LAN-side gw-client, and the LAN masquerade rewrites the source to -# the gateway's LAN address. Runs after Phase 8 has reclaimed the mapping to -# gw-server, because a live mapping's SNAT rule matches the same flows first -# and would do the rewrite instead. Runs before Phase 9 kills the daemon. +# the gateway's LAN address. This is the case with no mapping to gw-server: +# it runs after Phase 8 has reclaimed the Phase 4 mapping, and the gate below +# confirms none is left. Phase 8c covers the case with a live mapping. Runs +# before Phase 9 kills the daemon. # # The gate reads both the control socket's mappings, a snapshot refreshed # on the pool tick, and the kernel's table, which is what decides the rule @@ -818,6 +958,11 @@ echo "Phase 8b: Inbound port forwards through the LAN masquerade" SERVER_MESH=$(docker exec "$SERVER" bash -c \ "ip -6 -o addr show fips0 | awk '/inet6 fd/ {print \$4}' | cut -d/ -f1 | head -1" \ 2>/dev/null || echo "") +# The gateway's mesh IPv6 (the fd00::/8 address on fips0), which phases 8b and +# 8c both probe whatever 8b's gate decides. +GW_MESH_IP=$(docker exec "$GATEWAY" bash -c \ + "ip -6 -o addr show fips0 | awk '/inet6 fd/ {print \$4}' | cut -d/ -f1 | head -1" \ + 2>/dev/null || echo "") if [ -n "$SERVER_MESH" ] && SERVER_MAPS=$(docker exec "$GATEWAY" bash -c \ 'echo "{\"command\":\"show_mappings\"}" | nc -U -w1 /run/fips/gateway.sock 2>/dev/null' \ | server_mapped "$SERVER_MESH"); then @@ -838,51 +983,7 @@ else fi if [ "$GATE_OK" = true ]; then - # Start marker HTTP servers on the LAN-side client. - # :8080 → "inbound-forward-ok" (target of tcp 18080) - # :8081 → "inbound-forward-ok-2" (target of tcp 18082) - # `docker exec -d` is required; `docker exec bash -c 'cmd &'` doesn't - # keep the child alive past the exec session, even with nohup. - docker exec "$CLIENT" sh -c ' - mkdir -p /tmp/inbound /tmp/inbound2 - echo "inbound-forward-ok" > /tmp/inbound/index.html - echo "inbound-forward-ok-2" > /tmp/inbound2/index.html - pkill -f "http.server 8080" 2>/dev/null || true - pkill -f "http.server 8081" 2>/dev/null || true - pkill -f "udp_echo.py" 2>/dev/null || true - ' >/dev/null 2>&1 || true - docker exec -d "$CLIENT" python3 -m http.server 8080 --bind :: --directory /tmp/inbound \ - >/dev/null 2>&1 || true - docker exec -d "$CLIENT" python3 -m http.server 8081 --bind :: --directory /tmp/inbound2 \ - >/dev/null 2>&1 || true - - # Start a UDP echo server on the LAN-side client at [::]:8081/udp. - # This is the target of the udp 18081 forward. Stash the script as a - # named file (`udp_echo.py`) so the cleanup pkill above can find it. - docker exec "$CLIENT" sh -c 'cat > /tmp/udp_echo.py <<'\''PYEOF'\'' -import socket, sys -s = socket.socket(socket.AF_INET6, socket.SOCK_DGRAM) -s.bind(("::", 8081)) -while True: - data, addr = s.recvfrom(2048) - s.sendto(b"udp-forward-ok:" + data, addr) -PYEOF' >/dev/null 2>&1 || true - docker exec -d "$CLIENT" python3 /tmp/udp_echo.py >/dev/null 2>&1 || true - - # Give the servers a moment to bind. - for _ in 1 2 3 4 5; do - TCP_READY=$(docker exec "$CLIENT" ss -6lnt 2>/dev/null | grep -cE ':8080|:8081' || true) - UDP_READY=$(docker exec "$CLIENT" ss -6lnu 2>/dev/null | grep -c ':8081' || true) - if [ "$TCP_READY" -ge 2 ] && [ "$UDP_READY" -ge 1 ]; then - break - fi - sleep 1 - done - - # Derive the gateway's mesh IPv6 (fd00::/8 address assigned to fips0). - GW_MESH_IP=$(docker exec "$GATEWAY" bash -c \ - "ip -6 -o addr show fips0 | awk '/inet6 fd/ {print \$4}' | cut -d/ -f1 | head -1" \ - 2>/dev/null || echo "") + gw_responders_start if [ -z "$GW_MESH_IP" ]; then check "Gateway fips0 IPv6 address" 1 @@ -951,12 +1052,8 @@ except Exception as e: fi done - # Stop the LAN-side responders; no later phase uses them. - docker exec "$CLIENT" sh -c ' - pkill -f "http.server 8080" 2>/dev/null || true - pkill -f "http.server 8081" 2>/dev/null || true - pkill -f "udp_echo.py" 2>/dev/null || true - ' >/dev/null 2>&1 || true + # Stop the LAN-side responders; Phase 8c starts its own. + gw_responders_stop else check "Inbound HTTP via TCP forward 18080 (skipped: gate)" 1 check "Inbound HTTP via TCP forward 18082 (skipped: gate)" 1 @@ -966,6 +1063,114 @@ else check "Reply to udp 18081 goes to the gateway's LAN address (skipped: gate)" 1 fi +# Phase 8c: Inbound port forward from a peer with a live mapping +# +# A LAN client resolves gw-server first, so the gateway holds a mapping and a +# SNAT rule for gw-server's mesh address, and gw-server then reaches tcp +# 18080. The LAN masquerade must still take the flow, so the reply goes to +# the gateway's LAN address, not to the mapping's pool address. +# +# Timing. On correct code the probe is masqueraded, so no conntrack entry +# names the virtual IP and nothing pins the new mapping. The pool drains it on +# the first tick more than the TTL (5s) after the dig and frees it on the next +# tick more than the grace (5s) later; with the 10s tick the rule can be gone +# about 15s after the dig. So the responders start and conntrack is flushed +# before the dig, and right after the gate a GET from gw-client to the +# virtual IP leaves an entry that names it, which pins the mapping from the +# next tick on. Only the dig, the gate poll and that GET sit inside the 15s. +# Do not add a step that can take seconds between the dig and the pin. +# +# The SNAT rule is read again after the probe. No DNS query happens in +# between, so a rule present at both ends was present during the probe; +# without that check, a mapping reclaimed early would let the reply check +# pass with no SNAT to compete with. +echo "" +echo "Phase 8c: Inbound port forward from a peer with a live mapping" +P8C_MISSING="" +[ -n "$GW_MESH_IP" ] || P8C_MISSING="gateway mesh address" +[ -n "$SERVER_MESH" ] || P8C_MISSING="${P8C_MISSING:+$P8C_MISSING and }$SERVER mesh address" +P8C_GATE=false +if [ -n "$P8C_MISSING" ]; then + check "Mapping and SNAT rule to $SERVER before the probe (skipped: no $P8C_MISSING)" 1 +else + # Outside the window: the responders, and a flush so Phase 8b's entries + # for tcp 18080, whose reply goes to $GW_DNS, cannot answer for this probe. + gw_responders_start + docker exec "$GATEWAY" conntrack -F 2>/dev/null || true + + # The window opens here. + P8C_T0=$SECONDS + P8C_VIP=$(docker exec "$CLIENT" dig +short AAAA "${NPUB_B}.fips" @${GW_DNS} 2>/dev/null \ + | grep -m1 "^fd01::" || true) + P8C_SNAT="" + if [ -n "$P8C_VIP" ]; then + while :; do + if P8C_SNAT=$(docker exec "$GATEWAY" nft list table inet fips_gateway 2>/dev/null \ + | snat_to "$SERVER_MESH") && same_addr "$P8C_SNAT" "$P8C_VIP"; then + P8C_GATE=true + break + fi + if [ $((SECONDS - P8C_T0)) -ge 5 ]; then + break + fi + sleep 0.5 + done + fi + P8C_GATE_VALUES="dig '$P8C_VIP', SNAT target '$P8C_SNAT', $((SECONDS - P8C_T0))s after the dig" + if [ "$P8C_GATE" = true ]; then + check "Mapping and SNAT rule to $SERVER before the probe ($P8C_GATE_VALUES)" 0 + else + check "Mapping and SNAT rule to $SERVER before the probe ($P8C_GATE_VALUES)" 1 + fi +fi + +if [ "$P8C_GATE" = true ]; then + # Pin the mapping: this GET's conntrack entry names the virtual IP. + P8C_PIN=$(docker exec "$CLIENT" curl -6 -s --max-time 3 "http://[$P8C_VIP]:8000/" 2>&1) || true + if echo "$P8C_PIN" | grep -q "Fuck IPs"; then + check "GET from $CLIENT to $P8C_VIP pins the mapping ($((SECONDS - P8C_T0))s after the dig)" 0 + else + check "GET from $CLIENT to $P8C_VIP pins the mapping ($((SECONDS - P8C_T0))s after the dig, response: '${P8C_PIN:0:80}')" 1 + fi + + P8C_RESPONSE=$(docker exec "$SERVER" curl -6 -s --max-time 5 \ + "http://[${GW_MESH_IP}]:18080/" 2>&1) || true + if echo "$P8C_RESPONSE" | grep -qE '^inbound-forward-ok$'; then + check "Inbound HTTP via TCP forward 18080 with a live mapping to $SERVER" 0 + else + check "Inbound HTTP via TCP forward 18080 with a live mapping (response: '${P8C_RESPONSE:0:80}')" 1 + fi + + # The window closes here. + P8C_AFTER="" + if P8C_AFTER=$(docker exec "$GATEWAY" nft list table inet fips_gateway 2>/dev/null \ + | snat_to "$SERVER_MESH") && same_addr "$P8C_AFTER" "$P8C_VIP"; then + check "SNAT rule to $SERVER still present after the probe ($((SECONDS - P8C_T0))s after the dig)" 0 + else + check "SNAT rule to $SERVER gone after the probe (target '$P8C_AFTER', $((SECONDS - P8C_T0))s after the dig); the reply check below proves nothing on this run" 1 + fi + + # The probe's entry does not depend on the mapping still existing. + P8C_FOUND="" + if P8C_CT=$(docker exec "$GATEWAY" conntrack -L -f ipv6 2>/dev/null) \ + && P8C_FOUND=$(reply_dst tcp 18080 <<< "$P8C_CT") \ + && same_addr "$P8C_FOUND" "$GW_DNS"; then + check "Reply to tcp 18080 with a live mapping goes to $P8C_FOUND, the gateway's LAN address $GW_DNS" 0 + else + check "Reply to tcp 18080 with a live mapping goes to '$P8C_FOUND', expected the gateway's LAN address $GW_DNS" 1 + fi +else + P8C_SKIP="skipped: ${P8C_MISSING:+no $P8C_MISSING}" + [ -n "$P8C_MISSING" ] || P8C_SKIP="skipped: gate" + check "GET from $CLIENT pins the mapping ($P8C_SKIP)" 1 + check "Inbound HTTP via TCP forward 18080 with a live mapping ($P8C_SKIP)" 1 + check "SNAT rule to $SERVER still present after the probe ($P8C_SKIP)" 1 + check "Reply to tcp 18080 with a live mapping goes to the gateway's LAN address ($P8C_SKIP)" 1 +fi +if [ -z "$P8C_MISSING" ]; then + gw_responders_stop +fi + # Phase 9: SERVFAIL when daemon DNS is down echo "" echo "Phase 9: SERVFAIL when daemon DNS is down" From faeac11ca86c19a66ac344875f7b9bb50e38f624 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Mon, 5 Oct 2026 23:38:36 +0000 Subject: [PATCH 2/7] Remove the .fips DNS routing on package remove, not only on purge Removing the .deb while fips-dns was not running left the resolver's .fips routing file in place. prerm stops fips-dns, but systemd runs its teardown only for an active unit, and postrm cleaned up the routing only on purge. After an apt remove the resolver kept sending .fips queries to [::1]:5354, where nothing listens once the daemon is gone, so .fips lookups timed out until the file was deleted by hand. postrm now runs the DNS cleanup on both remove and purge. A purge of an installed package runs remove and then purge; the second pass finds no file and restarts nothing. A purge of a package already removed runs only the purge pass, which still cleans up. Configuration, keys and the fips group are still removed only on purge. The deb-install check now removes the package with the routing planted and fips-dns stopped, then plants the routing again and purges from config-files state, so each postrm branch is exercised on its own. The packaging text test reads the remove|purge branch. --- packaging/debian/postrm | 34 ++++-- src/packaging_tests.rs | 20 +-- testing/deb-install/test.sh | 234 ++++++++++++++++++++++++------------ 3 files changed, 190 insertions(+), 98 deletions(-) diff --git a/packaging/debian/postrm b/packaging/debian/postrm index 8f1afd5d..123e48aa 100755 --- a/packaging/debian/postrm +++ b/packaging/debian/postrm @@ -3,21 +3,16 @@ set -e case "$1" in - purge) - # Remove configuration and identity keys - rm -rf /etc/fips/ - - # Remove tmpfiles.d entry - rm -f /usr/lib/tmpfiles.d/fips.conf - - # Remove runtime directory - rm -rf /run/fips/ - + remove|purge) # Remove the DNS routing fips-dns-setup may have written, in case # fips-dns-teardown did not run (prerm's stop runs it only when - # fips-dns.service was active), and make the resolver drop it. The - # paths match packaging/common/fips-dns-teardown, which dpkg has - # already removed, so it cannot be called from here. + # fips-dns.service was active), and make the resolver drop it. This + # runs on remove as well as purge: a removed package leaves nothing + # listening behind the routing, and purging a package already removed + # runs only postrm purge. On a purge of an installed package the + # purge pass finds nothing left and restarts nothing. The paths match + # packaging/common/fips-dns-teardown, which dpkg has already removed, + # so it cannot be called from here. restart_resolved=0 if [ -f /etc/systemd/dns-delegate.d/fips.dns-delegate ]; then rm -f /etc/systemd/dns-delegate.d/fips.dns-delegate @@ -52,6 +47,19 @@ case "$1" in || echo "fips: warning: could not reload NetworkManager; reload it to drop the .fips route" fi fi + ;; +esac + +case "$1" in + purge) + # Remove configuration and identity keys + rm -rf /etc/fips/ + + # Remove tmpfiles.d entry + rm -f /usr/lib/tmpfiles.d/fips.conf + + # Remove runtime directory + rm -rf /run/fips/ # Remove fips system group if getent group fips >/dev/null 2>&1; then diff --git a/src/packaging_tests.rs b/src/packaging_tests.rs index e68df0a9..18b525c9 100644 --- a/src/packaging_tests.rs +++ b/src/packaging_tests.rs @@ -391,17 +391,19 @@ fn freebsd_newsyslog_entry_signals_the_daemon8_supervisor_started_with_sighup_re ); } -/// Pins the DNS cleanup in `postrm purge` and `uninstall.sh` to the files -/// `fips-dns-setup` writes, so a purge after a `fips-dns` that never ran its -/// teardown does not leave the resolver sending `.fips` to a dead responder. +/// Pins the DNS cleanup in `postrm remove` and `postrm purge` and in +/// `uninstall.sh` to the files `fips-dns-setup` writes, so a remove or purge +/// after a `fips-dns` that never ran its teardown does not leave the resolver +/// sending `.fips` to a dead responder. /// /// This is a text test. Each path must appear on an `rm -f` line, but a /// resolver command passes wherever it appears on a code line, including in a -/// message. What `postrm` actually does is covered by the deb-install purge -/// check. No suite runs `uninstall.sh`: its two resolved paths were run once, -/// by hand in a container, and its dnsmasq and NetworkManager paths by nothing. +/// message. What `postrm` actually does is covered by the deb-install remove +/// and purge checks. No suite runs `uninstall.sh`: its two resolved paths were +/// run once, by hand in a container, and its dnsmasq and NetworkManager paths +/// by nothing. #[test] -fn dns_cleanup_in_postrm_purge_and_uninstall_removes_every_file_fips_dns_setup_writes_and_restarts_its_resolver() +fn dns_cleanup_in_postrm_remove_and_purge_and_uninstall_removes_every_file_fips_dns_setup_writes_and_restarts_its_resolver() { let setup = rc_vars(&repo_file("packaging/common/fips-dns-setup")); let teardown = rc_vars(&repo_file("packaging/common/fips-dns-teardown")); @@ -433,8 +435,8 @@ fn dns_cleanup_in_postrm_purge_and_uninstall_removes_every_file_fips_dns_setup_w let scripts = [ ( - "packaging/debian/postrm purge)", - case_branch(&repo_file("packaging/debian/postrm"), "purge"), + "packaging/debian/postrm remove|purge)", + case_branch(&repo_file("packaging/debian/postrm"), "remove|purge"), ), ( "packaging/systemd/uninstall.sh", diff --git a/testing/deb-install/test.sh b/testing/deb-install/test.sh index 63977bd4..6e73d154 100755 --- a/testing/deb-install/test.sh +++ b/testing/deb-install/test.sh @@ -10,8 +10,10 @@ # through the resolver backend that fips-dns-setup configured. Then # exercises fips-gateway against the same daemon to verify the # gateway/daemon default-pairing. Finally it -# purges the package with the DNS routing file planted and fips-dns -# stopped, and checks the file is removed and systemd-resolved restarted. +# removes the package with the DNS routing file planted and fips-dns +# stopped, and checks the file is removed and systemd-resolved restarted, +# then plants the file again and purges the package from config-files +# state, with the same checks. # # This is the most thorough test surface — it exercises: # - cargo deb packaging (binary stripping, dependency declaration) @@ -19,7 +21,8 @@ # of /etc/fips/fips.yaml # - postinst maintainer scripts (systemd unit enablement, # fips-dns.service running fips-dns-setup) -# - postrm purge (removing the DNS routing fips-dns-setup wrote) +# - postrm remove and postrm purge (removing the DNS routing +# fips-dns-setup wrote) # - The fips, fips-dns, and (optionally) fips-gateway systemd units # - End-to-end .fips resolution as a real user would experience it # @@ -342,19 +345,155 @@ check_gateway_default_listener() { fi } -# Purge the package with the DNS routing file planted and fips-dns stopped, and +# Put the saved DNS routing file back at and restart systemd-resolved, +# then check the state in which removal leaves the file behind: the file in +# place, fips-dns.service not active, and the resolver routing .fips to +# [::1]:5354. Every failure is recorded with