From b8c4a5584fd42a9ce2a11cda832b0580fbdc0d8c Mon Sep 17 00:00:00 2001 From: Arjen <18398758+Origami74@users.noreply.github.com> Date: Tue, 1 Sep 2026 13:17:03 +0100 Subject: [PATCH] test(iface-binding): cover an interface present before the daemon starts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every scenario in the suite created its interface after the daemons were already running — that ordering is the boot race the suite was written for. But it means both nodes could only ever reach Present through binder_loop, so the inline bind in start_async, which is the ordinary case on a booted router, had no end-to-end coverage at all. That is where the churn guard went unseeded and the first detach stopped reaching node health, and no existing case could reach it: they all detach from a binding the loop created, which seeds the guard as a side effect. Case (f) adds a third node whose single required interface exists before its daemon does. The gate is what buys that ordering — the harness needs a running container to have a netns to move a veth into, but the daemon must not start until after the move, so node-c comes up parked on a file and the harness releases it once the interface is in place. Then one detach, on a binding the loop did not create, and the node must degrade. Verified against the defect rather than only against the fix: with the guard seed reverted, cases (a) through (e) all still pass and (f) is the only failure. A regression test that has never been seen to fail is a claim, not a test. It also asserts the reverse edge, so Degraded stays a level rather than a latch on this path too. --- testing/iface-binding/docker-compose.yml | 23 +++++++ testing/iface-binding/generate-configs.sh | 44 ++++++++++++++ testing/iface-binding/test.sh | 74 ++++++++++++++++++++++- 3 files changed, 140 insertions(+), 1 deletion(-) diff --git a/testing/iface-binding/docker-compose.yml b/testing/iface-binding/docker-compose.yml index 0d5382c9..fd87ba6b 100644 --- a/testing/iface-binding/docker-compose.yml +++ b/testing/iface-binding/docker-compose.yml @@ -55,3 +55,26 @@ services: - ../docker/resolv.conf:/etc/resolv.conf:ro - ./generated-configs${FIPS_CI_NAME_SUFFIX:-}/node-b/fips.yaml:/etc/fips/fips.yaml:ro - ./generated-configs${FIPS_CI_NAME_SUFFIX:-}/node-b/fips.key:/etc/fips/fips.key:ro + + # The clean-start case. Its interface exists before its daemon does, which is + # the ordinary state of a booted router and the one ordering node-a and + # node-b cannot produce: their interface is created after they are already + # running, so they can only ever bind through the binder loop. + # + # The gate is what buys that ordering. The harness needs a running container + # to have a netns to move a veth into, but the daemon must not start until + # after the move — so the container comes up, parks on this file, and the + # harness releases it once the interface is in place. + node-c: + <<: *fips-common + container_name: fips-ifb-node-c${FIPS_CI_NAME_SUFFIX:-} + hostname: host-c + entrypoint: ["/bin/sh", "-c"] + command: + - | + while [ ! -e /tmp/fips-go ]; do sleep 0.2; done + exec /usr/local/bin/entrypoint.sh + volumes: + - ../docker/resolv.conf:/etc/resolv.conf:ro + - ./generated-configs${FIPS_CI_NAME_SUFFIX:-}/node-c/fips.yaml:/etc/fips/fips.yaml:ro + - ./generated-configs${FIPS_CI_NAME_SUFFIX:-}/node-c/fips.key:/etc/fips/fips.key:ro diff --git a/testing/iface-binding/generate-configs.sh b/testing/iface-binding/generate-configs.sh index 128b31f8..1aab0a76 100755 --- a/testing/iface-binding/generate-configs.sh +++ b/testing/iface-binding/generate-configs.sh @@ -7,6 +7,10 @@ # harness creates the veth pair afterwards # dock fips-dock0 optional — never exists, on any host, ever # +# Plus a third node whose single required interface exists *before* its daemon +# starts — the one ordering the other two cannot produce, and the one that +# `start_async`'s inline bind takes. See node-c in test.sh case (f). +# # There is deliberately no UDP transport. A node whose only transports are # interface-bound is the case that used to be unrecoverable: every transport # skipped at start, nothing retried, and the node up and deaf. @@ -23,6 +27,7 @@ GENERATED_DIR="$SCRIPT_DIR/generated-configs${FIPS_CI_NAME_SUFFIX:-}" # Deterministic test identities (mirrors the firewall/acl-allowlist style). KEY_A="0102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f20" KEY_B="b102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1fb0" +KEY_C="c102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1fc0" write_file() { local path="$1" @@ -73,8 +78,42 @@ peers: [] EOF } +# node-c: one required interface, present at daemon start. +# +# The other two nodes can only ever reach `Present` through the binder loop, +# because their interface does not exist until the harness makes it. That left +# the inline bind in `start_async` — the ordinary case on a booted router — +# with no coverage at all, which is exactly where the churn guard went unseeded +# and the first detach stopped reaching node health. +write_boot_node_config() { + write_file "$GENERATED_DIR/node-c/fips.yaml" </dev/null 2>&1 || true ip_host "ip link del $HOST_VETH_A" >/dev/null 2>&1 || true + docker exec "$NODE_C" ip link del "$BOOT_IFACE" >/dev/null 2>&1 || true + ip_host "ip link del $HOST_VETH_C" >/dev/null 2>&1 || true if [ "$KEEP_UP" = false ]; then docker compose -f "$COMPOSE_FILE" down --volumes --remove-orphans >/dev/null 2>&1 || true fi @@ -97,6 +104,12 @@ ip_host() { --entrypoint /bin/sh "$IMAGE" -c "$1" } +# Docker's own view of a container, so the gate case can wait for a netns +# without implying the daemon inside it has started. +container_state() { + docker inspect -f '{{.State.Status}}' "$1" 2>/dev/null || true +} + container_pid() { docker inspect -f '{{.State.Pid}}' "$1" } @@ -214,7 +227,8 @@ if [ "$SKIP_BUILD" = false ]; then fi log "Generating fixtures" -LAB_IFACE="$LAB_IFACE" DOCK_IFACE="$DOCK_IFACE" bash "$SCRIPT_DIR/generate-configs.sh" +LAB_IFACE="$LAB_IFACE" DOCK_IFACE="$DOCK_IFACE" BOOT_IFACE="$BOOT_IFACE" \ + bash "$SCRIPT_DIR/generate-configs.sh" log "Starting nodes with $LAB_IFACE absent" docker compose -f "$COMPOSE_FILE" up -d @@ -424,6 +438,64 @@ if ! wait_for_at_least 60 1 peer_count "$NODE_A"; then fi pass "(d) peering re-established over the recreated interface" +# ── (f) an interface present before the daemon starts ──────────────────── +# +# Everything above binds through the binder loop, because the interface does +# not exist until the harness makes it. The ordinary case on a booted router is +# the opposite one: the interface is already there and `start_async` binds it +# inline, before the loop is running. +# +# That path published its presence edge outside the churn guard, so the guard +# believed it had announced nothing and the *first* detach asked for no +# retraction. The node kept reporting Running with its only required interface +# gone, and stayed that way until a second detach happened to repair the guard. +# Nothing in cases (a)-(e) can reach it. +log "(f) starting node-c with its interface already present" + +docker compose -f "$COMPOSE_FILE" up -d node-c + +# The container parks on the gate, so this is the netns and not yet the daemon. +if ! wait_for 30 "running" container_state "$NODE_C"; then + fail "(f) node-c container did not start" +fi + +pid_c="$(container_pid "$NODE_C")" +ip_host "set -e + ip link add $HOST_VETH_C type veth peer name $HOST_VETH_D + ip link set $HOST_VETH_C netns $pid_c name $BOOT_IFACE + ip link set $HOST_VETH_D netns $pid_c name ${BOOT_IFACE}p" >/dev/null +docker exec "$NODE_C" ip link set "$BOOT_IFACE" up +docker exec "$NODE_C" ip link set "${BOOT_IFACE}p" up + +# Release the gate. The daemon now starts with the interface already up. +docker exec "$NODE_C" touch /tmp/fips-go + +if ! wait_for 40 "running" node_state "$NODE_C"; then + fail "(f) node-c did not come up Running with its interface present at start" +fi +[ "$(iface_field "$NODE_C" boot presence)" = "present" ] \ + || fail "(f) node-c did not bind $BOOT_IFACE inline at start" +pass "(f) an interface present at start is bound inline and reports Running" + +# The assertion. One detach, on a binding this loop did not create. +docker exec "$NODE_C" ip link set "$BOOT_IFACE" down + +if ! wait_for 30 "absent" iface_field "$NODE_C" boot presence; then + fail "(f) node-c did not notice $BOOT_IFACE going down" +fi +if ! wait_for 30 "degraded" node_state "$NODE_C"; then + fail "(f) node-c stayed Running after its only required interface went \ +away — the start-time bind never reached node health" +fi +pass "(f) the first detach after a clean start degrades the node" + +# And it is still a level, not a latch, on this path too. +docker exec "$NODE_C" ip link set "$BOOT_IFACE" up +if ! wait_for 30 "running" node_state "$NODE_C"; then + fail "(f) node-c stayed Degraded after its interface returned" +fi +pass "(f) health clears again when the interface returns" + # ── final log hygiene ──────────────────────────────────────────────────── # # Four outages happened above (start, down, delete, and node-b's end of the