mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-06 03:28:24 +00:00
test(iface-binding): cover an interface present before the daemon starts
Every scenario in the suite created its interface after the daemons were already running — that ordering is the boot race the suite was written for. But it means both nodes could only ever reach Present through binder_loop, so the inline bind in start_async, which is the ordinary case on a booted router, had no end-to-end coverage at all. That is where the churn guard went unseeded and the first detach stopped reaching node health, and no existing case could reach it: they all detach from a binding the loop created, which seeds the guard as a side effect. Case (f) adds a third node whose single required interface exists before its daemon does. The gate is what buys that ordering — the harness needs a running container to have a netns to move a veth into, but the daemon must not start until after the move, so node-c comes up parked on a file and the harness releases it once the interface is in place. Then one detach, on a binding the loop did not create, and the node must degrade. Verified against the defect rather than only against the fix: with the guard seed reverted, cases (a) through (e) all still pass and (f) is the only failure. A regression test that has never been seen to fail is a claim, not a test. It also asserts the reverse edge, so Degraded stays a level rather than a latch on this path too.
This commit is contained in:
@@ -55,3 +55,26 @@ services:
|
||||
- ../docker/resolv.conf:/etc/resolv.conf:ro
|
||||
- ./generated-configs${FIPS_CI_NAME_SUFFIX:-}/node-b/fips.yaml:/etc/fips/fips.yaml:ro
|
||||
- ./generated-configs${FIPS_CI_NAME_SUFFIX:-}/node-b/fips.key:/etc/fips/fips.key:ro
|
||||
|
||||
# The clean-start case. Its interface exists before its daemon does, which is
|
||||
# the ordinary state of a booted router and the one ordering node-a and
|
||||
# node-b cannot produce: their interface is created after they are already
|
||||
# running, so they can only ever bind through the binder loop.
|
||||
#
|
||||
# The gate is what buys that ordering. The harness needs a running container
|
||||
# to have a netns to move a veth into, but the daemon must not start until
|
||||
# after the move — so the container comes up, parks on this file, and the
|
||||
# harness releases it once the interface is in place.
|
||||
node-c:
|
||||
<<: *fips-common
|
||||
container_name: fips-ifb-node-c${FIPS_CI_NAME_SUFFIX:-}
|
||||
hostname: host-c
|
||||
entrypoint: ["/bin/sh", "-c"]
|
||||
command:
|
||||
- |
|
||||
while [ ! -e /tmp/fips-go ]; do sleep 0.2; done
|
||||
exec /usr/local/bin/entrypoint.sh
|
||||
volumes:
|
||||
- ../docker/resolv.conf:/etc/resolv.conf:ro
|
||||
- ./generated-configs${FIPS_CI_NAME_SUFFIX:-}/node-c/fips.yaml:/etc/fips/fips.yaml:ro
|
||||
- ./generated-configs${FIPS_CI_NAME_SUFFIX:-}/node-c/fips.key:/etc/fips/fips.key:ro
|
||||
|
||||
@@ -7,6 +7,10 @@
|
||||
# harness creates the veth pair afterwards
|
||||
# dock fips-dock0 optional — never exists, on any host, ever
|
||||
#
|
||||
# Plus a third node whose single required interface exists *before* its daemon
|
||||
# starts — the one ordering the other two cannot produce, and the one that
|
||||
# `start_async`'s inline bind takes. See node-c in test.sh case (f).
|
||||
#
|
||||
# There is deliberately no UDP transport. A node whose only transports are
|
||||
# interface-bound is the case that used to be unrecoverable: every transport
|
||||
# skipped at start, nothing retried, and the node up and deaf.
|
||||
@@ -23,6 +27,7 @@ GENERATED_DIR="$SCRIPT_DIR/generated-configs${FIPS_CI_NAME_SUFFIX:-}"
|
||||
# Deterministic test identities (mirrors the firewall/acl-allowlist style).
|
||||
KEY_A="0102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f20"
|
||||
KEY_B="b102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1fb0"
|
||||
KEY_C="c102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1fc0"
|
||||
|
||||
write_file() {
|
||||
local path="$1"
|
||||
@@ -73,8 +78,42 @@ peers: []
|
||||
EOF
|
||||
}
|
||||
|
||||
# node-c: one required interface, present at daemon start.
|
||||
#
|
||||
# The other two nodes can only ever reach `Present` through the binder loop,
|
||||
# because their interface does not exist until the harness makes it. That left
|
||||
# the inline bind in `start_async` — the ordinary case on a booted router —
|
||||
# with no coverage at all, which is exactly where the churn guard went unseeded
|
||||
# and the first detach stopped reaching node health.
|
||||
write_boot_node_config() {
|
||||
write_file "$GENERATED_DIR/node-c/fips.yaml" <<EOF
|
||||
node:
|
||||
identity:
|
||||
persistent: true
|
||||
|
||||
tun:
|
||||
enabled: false
|
||||
|
||||
dns:
|
||||
enabled: false
|
||||
|
||||
transports:
|
||||
ethernet:
|
||||
boot:
|
||||
interface: "$BOOT_IFACE"
|
||||
listen: true
|
||||
announce: true
|
||||
auto_connect: true
|
||||
accept_connections: true
|
||||
beacon_interval_secs: 2
|
||||
|
||||
peers: []
|
||||
EOF
|
||||
}
|
||||
|
||||
LAB_IFACE="${LAB_IFACE:-ve-lab0}"
|
||||
DOCK_IFACE="${DOCK_IFACE:-fips-dock0}"
|
||||
BOOT_IFACE="${BOOT_IFACE:-ve-boot0}"
|
||||
|
||||
echo "Generating interface-binding fixtures..."
|
||||
rm -rf "$GENERATED_DIR"
|
||||
@@ -89,4 +128,9 @@ write_file "$GENERATED_DIR/node-b/fips.key" <<EOF
|
||||
$KEY_B
|
||||
EOF
|
||||
|
||||
write_boot_node_config
|
||||
write_file "$GENERATED_DIR/node-c/fips.key" <<EOF
|
||||
$KEY_C
|
||||
EOF
|
||||
|
||||
echo "Interface-binding fixtures written to $GENERATED_DIR"
|
||||
|
||||
@@ -33,18 +33,23 @@ COMPOSE_FILE="$SCRIPT_DIR/docker-compose.yml"
|
||||
|
||||
NODE_A="fips-ifb-node-a${FIPS_CI_NAME_SUFFIX:-}"
|
||||
NODE_B="fips-ifb-node-b${FIPS_CI_NAME_SUFFIX:-}"
|
||||
NODE_C="fips-ifb-node-c${FIPS_CI_NAME_SUFFIX:-}"
|
||||
|
||||
# The interface each node binds. Same name on both sides: they are in separate
|
||||
# network namespaces, and using one name keeps the fixtures identical.
|
||||
LAB_IFACE="ve-lab0"
|
||||
# The interface that never exists. `optional: true` in both configs.
|
||||
DOCK_IFACE="fips-dock0"
|
||||
# node-c's interface, present before its daemon starts. See case (f).
|
||||
BOOT_IFACE="ve-boot0"
|
||||
|
||||
# Host-side veth names, scoped per run: these live in the host (or Docker VM)
|
||||
# namespace for the moment between creation and the move into the containers,
|
||||
# where two concurrent runs would otherwise collide on one name.
|
||||
HOST_VETH_A="vhifb${FIPS_CI_NAME_SUFFIX:-0}a"
|
||||
HOST_VETH_B="vhifb${FIPS_CI_NAME_SUFFIX:-0}b"
|
||||
HOST_VETH_C="vhifb${FIPS_CI_NAME_SUFFIX:-0}c"
|
||||
HOST_VETH_D="vhifb${FIPS_CI_NAME_SUFFIX:-0}d"
|
||||
|
||||
SKIP_BUILD=false
|
||||
KEEP_UP=false
|
||||
@@ -78,6 +83,8 @@ cleanup() {
|
||||
# succeeded, on the host if the run died between creation and the move.
|
||||
docker exec "$NODE_A" ip link del "$LAB_IFACE" >/dev/null 2>&1 || true
|
||||
ip_host "ip link del $HOST_VETH_A" >/dev/null 2>&1 || true
|
||||
docker exec "$NODE_C" ip link del "$BOOT_IFACE" >/dev/null 2>&1 || true
|
||||
ip_host "ip link del $HOST_VETH_C" >/dev/null 2>&1 || true
|
||||
if [ "$KEEP_UP" = false ]; then
|
||||
docker compose -f "$COMPOSE_FILE" down --volumes --remove-orphans >/dev/null 2>&1 || true
|
||||
fi
|
||||
@@ -97,6 +104,12 @@ ip_host() {
|
||||
--entrypoint /bin/sh "$IMAGE" -c "$1"
|
||||
}
|
||||
|
||||
# Docker's own view of a container, so the gate case can wait for a netns
|
||||
# without implying the daemon inside it has started.
|
||||
container_state() {
|
||||
docker inspect -f '{{.State.Status}}' "$1" 2>/dev/null || true
|
||||
}
|
||||
|
||||
container_pid() {
|
||||
docker inspect -f '{{.State.Pid}}' "$1"
|
||||
}
|
||||
@@ -214,7 +227,8 @@ if [ "$SKIP_BUILD" = false ]; then
|
||||
fi
|
||||
|
||||
log "Generating fixtures"
|
||||
LAB_IFACE="$LAB_IFACE" DOCK_IFACE="$DOCK_IFACE" bash "$SCRIPT_DIR/generate-configs.sh"
|
||||
LAB_IFACE="$LAB_IFACE" DOCK_IFACE="$DOCK_IFACE" BOOT_IFACE="$BOOT_IFACE" \
|
||||
bash "$SCRIPT_DIR/generate-configs.sh"
|
||||
|
||||
log "Starting nodes with $LAB_IFACE absent"
|
||||
docker compose -f "$COMPOSE_FILE" up -d
|
||||
@@ -424,6 +438,64 @@ if ! wait_for_at_least 60 1 peer_count "$NODE_A"; then
|
||||
fi
|
||||
pass "(d) peering re-established over the recreated interface"
|
||||
|
||||
# ── (f) an interface present before the daemon starts ────────────────────
|
||||
#
|
||||
# Everything above binds through the binder loop, because the interface does
|
||||
# not exist until the harness makes it. The ordinary case on a booted router is
|
||||
# the opposite one: the interface is already there and `start_async` binds it
|
||||
# inline, before the loop is running.
|
||||
#
|
||||
# That path published its presence edge outside the churn guard, so the guard
|
||||
# believed it had announced nothing and the *first* detach asked for no
|
||||
# retraction. The node kept reporting Running with its only required interface
|
||||
# gone, and stayed that way until a second detach happened to repair the guard.
|
||||
# Nothing in cases (a)-(e) can reach it.
|
||||
log "(f) starting node-c with its interface already present"
|
||||
|
||||
docker compose -f "$COMPOSE_FILE" up -d node-c
|
||||
|
||||
# The container parks on the gate, so this is the netns and not yet the daemon.
|
||||
if ! wait_for 30 "running" container_state "$NODE_C"; then
|
||||
fail "(f) node-c container did not start"
|
||||
fi
|
||||
|
||||
pid_c="$(container_pid "$NODE_C")"
|
||||
ip_host "set -e
|
||||
ip link add $HOST_VETH_C type veth peer name $HOST_VETH_D
|
||||
ip link set $HOST_VETH_C netns $pid_c name $BOOT_IFACE
|
||||
ip link set $HOST_VETH_D netns $pid_c name ${BOOT_IFACE}p" >/dev/null
|
||||
docker exec "$NODE_C" ip link set "$BOOT_IFACE" up
|
||||
docker exec "$NODE_C" ip link set "${BOOT_IFACE}p" up
|
||||
|
||||
# Release the gate. The daemon now starts with the interface already up.
|
||||
docker exec "$NODE_C" touch /tmp/fips-go
|
||||
|
||||
if ! wait_for 40 "running" node_state "$NODE_C"; then
|
||||
fail "(f) node-c did not come up Running with its interface present at start"
|
||||
fi
|
||||
[ "$(iface_field "$NODE_C" boot presence)" = "present" ] \
|
||||
|| fail "(f) node-c did not bind $BOOT_IFACE inline at start"
|
||||
pass "(f) an interface present at start is bound inline and reports Running"
|
||||
|
||||
# The assertion. One detach, on a binding this loop did not create.
|
||||
docker exec "$NODE_C" ip link set "$BOOT_IFACE" down
|
||||
|
||||
if ! wait_for 30 "absent" iface_field "$NODE_C" boot presence; then
|
||||
fail "(f) node-c did not notice $BOOT_IFACE going down"
|
||||
fi
|
||||
if ! wait_for 30 "degraded" node_state "$NODE_C"; then
|
||||
fail "(f) node-c stayed Running after its only required interface went \
|
||||
away — the start-time bind never reached node health"
|
||||
fi
|
||||
pass "(f) the first detach after a clean start degrades the node"
|
||||
|
||||
# And it is still a level, not a latch, on this path too.
|
||||
docker exec "$NODE_C" ip link set "$BOOT_IFACE" up
|
||||
if ! wait_for 30 "running" node_state "$NODE_C"; then
|
||||
fail "(f) node-c stayed Degraded after its interface returned"
|
||||
fi
|
||||
pass "(f) health clears again when the interface returns"
|
||||
|
||||
# ── final log hygiene ────────────────────────────────────────────────────
|
||||
#
|
||||
# Four outages happened above (start, down, delete, and node-b's end of the
|
||||
|
||||
Reference in New Issue
Block a user