mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-05 19:18:25 +00:00
fix(testing): bound the deb-install service starts so a dead daemon reds instead of hanging
The deb-install suite started fips-dns.service with no timeout. That unit is Type=oneshot with Requires=fips.service, so when the daemon cannot execute -- a broken package, a bad config, a missing capability -- systemd restarts it every five seconds for ever, the oneshot start job is never dispatched, and systemctl start never returns. The suite then produced no FAIL line, no Results line and no exit status at all. Observed at 21 minutes against a package whose binaries could not load. That is the whole class of fault this suite exists to find, so the suite stopped reporting at exactly the point it was most needed. It also matters beyond one test: a hang here blocks the local run that gates artifact publication. Queue that one unit rather than waiting on it, and wait for it to become active before reading /run/fips/dns-backend. RemainAfterExit=yes makes is-active a correct readiness test for the oneshot, and the wait carries a timeout and dumps the journal on failure, so the verdict lands on an assertion instead of on a stalled call. Keep every other start blocking, because the call returning is what synchronises the checks after it -- fips-gateway.service waits up to thirty seconds for fips0 in ExecStartPre, and a caller that does not wait races it. What those calls lacked was a bound, not the wait, so they get one. Bound the whole suite from ci-local.sh as a backstop, and say so explicitly when it fires: a timeout means no assertion was reached, which is not the same as an assertion failing. Verified by breaking what it guards. Against the released 0.5.0 package, whose binaries require a newer glibc than Debian 12 provides, the suite now reds in 36 seconds with the loader error in the journal dump. Against a working package it passes 38 checks across debian12 and ubuntu26, covering both resolver backends.
This commit is contained in:
+16
-4
@@ -999,13 +999,25 @@ run_dns_resolver() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
# Run deb-install harness (multi-distro real-package install)
|
# Run deb-install harness (multi-distro real-package install)
|
||||||
|
#
|
||||||
|
# Bounded because this worker is what gates artifact publication: a suite that
|
||||||
|
# hangs stops every branch publishing, which is worse than a red. The harness
|
||||||
|
# bounds its own service starts now, so this is the backstop for anything else
|
||||||
|
# that wedges -- a docker daemon that stops answering, a container that never
|
||||||
|
# boots. 40 minutes is well above the observed cold-cache cost of a full
|
||||||
|
# five-distro run and is not a performance budget.
|
||||||
|
DEB_INSTALL_TIMEOUT=${DEB_INSTALL_TIMEOUT:-2400}
|
||||||
run_deb_install() {
|
run_deb_install() {
|
||||||
info "[deb-install] Running multi-distro test (slow — builds .deb + per-distro install)"
|
info "[deb-install] Running multi-distro test (slow — builds .deb + per-distro install)"
|
||||||
if bash testing/deb-install/test.sh 2>&1; then
|
local rc=0
|
||||||
record "deb-install" 0
|
timeout "$DEB_INSTALL_TIMEOUT" bash testing/deb-install/test.sh 2>&1 || rc=$?
|
||||||
else
|
if [[ $rc -eq 124 ]]; then
|
||||||
record "deb-install" 1
|
# Say so explicitly. A bare red here reads as a failed assertion, and
|
||||||
|
# the difference matters: a timeout means no assertion was reached.
|
||||||
|
echo " ERROR: deb-install exceeded ${DEB_INSTALL_TIMEOUT}s and was killed;" >&2
|
||||||
|
echo " no verdict was reached, so this is not an assertion failure." >&2
|
||||||
fi
|
fi
|
||||||
|
record "deb-install" "$rc"
|
||||||
}
|
}
|
||||||
|
|
||||||
# Run Tor SOCKS5 outbound test (live Tor network)
|
# Run Tor SOCKS5 outbound test (live Tor network)
|
||||||
|
|||||||
@@ -36,6 +36,16 @@ DEB_CACHE_DIR="$CACHE_DIR/deb"
|
|||||||
BOOT_TIMEOUT=30
|
BOOT_TIMEOUT=30
|
||||||
SERVICE_TIMEOUT=20
|
SERVICE_TIMEOUT=20
|
||||||
DAEMON_TIMEOUT=15
|
DAEMON_TIMEOUT=15
|
||||||
|
# Bounds on a single `systemctl start`, which is not a wait loop and needs its
|
||||||
|
# own limit. See start_unit() for why an unbounded one can never return.
|
||||||
|
UNIT_START_TIMEOUT=30
|
||||||
|
# fips-gateway.service's ExecStartPre waits up to 30s for fips0 to appear, by
|
||||||
|
# design, so its start legitimately takes longer than any other.
|
||||||
|
GATEWAY_START_TIMEOUT=60
|
||||||
|
# The gateway-enable block restarts fips.service inside the container. Above
|
||||||
|
# systemd's default TimeoutStopSec of 90s, so a wedged stop trips this rather
|
||||||
|
# than this cutting a healthy stop short.
|
||||||
|
CONFIG_RESTART_TIMEOUT=120
|
||||||
|
|
||||||
PASS=0
|
PASS=0
|
||||||
FAIL=0
|
FAIL=0
|
||||||
@@ -97,6 +107,41 @@ wait_for_systemd() {
|
|||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Start a unit without waiting for its start job to finish.
|
||||||
|
#
|
||||||
|
# Start a unit and wait for its start job, under a bound.
|
||||||
|
#
|
||||||
|
# Blocking is the right default and the call returning is what synchronises the
|
||||||
|
# checks after it: `fips-gateway.service` in particular has an ExecStartPre that
|
||||||
|
# waits up to 30s for fips0, so a caller that does not wait races it. What the
|
||||||
|
# old code lacked was the bound, not the wait.
|
||||||
|
start_unit() {
|
||||||
|
local name="$1" unit="$2" limit="${3:-$UNIT_START_TIMEOUT}"
|
||||||
|
timeout "$limit" docker exec "$name" systemctl start "$unit" 2>&1
|
||||||
|
}
|
||||||
|
|
||||||
|
# Queue a unit's start job and return without waiting for it.
|
||||||
|
#
|
||||||
|
# For `fips-dns.service` only, and the reason is specific rather than general.
|
||||||
|
# It is Type=oneshot with Requires=fips.service, so its start job waits on a
|
||||||
|
# dependency that a broken daemon never satisfies: fips.service restarts every
|
||||||
|
# 5s for ever and the oneshot's job is never dispatched. `systemctl start` then
|
||||||
|
# never returns. That is the whole class of fault this suite exists to find, and
|
||||||
|
# the suite answered it by hanging -- no FAIL, no Results line, no exit status,
|
||||||
|
# observed at 21 minutes against a package whose binaries could not load.
|
||||||
|
#
|
||||||
|
# Queueing moves the verdict onto the wait_for_service_active call that follows,
|
||||||
|
# which carries a timeout and dumps the journal when it fails. RemainAfterExit=yes
|
||||||
|
# on that unit makes `is-active` a correct readiness test for a oneshot.
|
||||||
|
#
|
||||||
|
# This bounds these call sites, not every `docker exec` in the file. The backstop
|
||||||
|
# for the rest is the caller's own limit: ci-local.sh bounds the whole suite, and
|
||||||
|
# the GitHub leg carries timeout-minutes.
|
||||||
|
start_unit_queued() {
|
||||||
|
local name="$1" unit="$2"
|
||||||
|
timeout "$UNIT_START_TIMEOUT" docker exec "$name" systemctl start --no-block "$unit" 2>&1
|
||||||
|
}
|
||||||
|
|
||||||
wait_for_service_active() {
|
wait_for_service_active() {
|
||||||
local name="$1" service="$2" timeout="${3:-$SERVICE_TIMEOUT}"
|
local name="$1" service="$2" timeout="${3:-$SERVICE_TIMEOUT}"
|
||||||
for _i in $(seq 1 "$timeout"); do
|
for _i in $(seq 1 "$timeout"); do
|
||||||
@@ -349,8 +394,8 @@ DOCKERFILE
|
|||||||
|
|
||||||
# Start the services as a simulated boot. (On a real system,
|
# Start the services as a simulated boot. (On a real system,
|
||||||
# they'd come up on next reboot.)
|
# they'd come up on next reboot.)
|
||||||
docker exec "$name" systemctl start fips.service 2>&1 || true
|
start_unit "$name" fips.service || true
|
||||||
docker exec "$name" systemctl start fips-dns.service 2>&1 || true
|
start_unit_queued "$name" fips-dns.service || true
|
||||||
|
|
||||||
if wait_for_service_active "$name" fips.service; then
|
if wait_for_service_active "$name" fips.service; then
|
||||||
pass "fips.service active after explicit start"
|
pass "fips.service active after explicit start"
|
||||||
@@ -381,10 +426,21 @@ DOCKERFILE
|
|||||||
return
|
return
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Wait for fips-dns.service. This should have run fips-dns-setup
|
# Wait for fips-dns.service to finish. Its start job is queued rather than
|
||||||
# which configures the resolver backend and writes
|
# waited on above, so this is what makes /run/fips/dns-backend safe to read:
|
||||||
# /run/fips/dns-backend.
|
# fips-dns-setup waits up to 30s for fips0 and restarts systemd-resolved
|
||||||
sleep 2
|
# before it writes that file, so reading it on a timer races the setup.
|
||||||
|
# RemainAfterExit=yes makes is-active correct for this oneshot.
|
||||||
|
if wait_for_service_active "$name" fips-dns.service; then
|
||||||
|
pass "fips-dns.service completed"
|
||||||
|
else
|
||||||
|
fail "fips-dns.service did not complete in ${SERVICE_TIMEOUT}s"
|
||||||
|
echo " --- fips-dns.service journal ---"
|
||||||
|
docker exec "$name" journalctl -u fips-dns.service --no-pager 2>&1 | tail -20
|
||||||
|
cleanup_container "$name"
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
|
||||||
local backend
|
local backend
|
||||||
backend=$(docker exec "$name" cat /run/fips/dns-backend 2>/dev/null || echo "(missing)")
|
backend=$(docker exec "$name" cat /run/fips/dns-backend 2>/dev/null || echo "(missing)")
|
||||||
local ver
|
local ver
|
||||||
@@ -441,7 +497,7 @@ DOCKERFILE
|
|||||||
# default preset) and ipv6 forwarding (gateway checks before
|
# default preset) and ipv6 forwarding (gateway checks before
|
||||||
# the DNS upstream check).
|
# the DNS upstream check).
|
||||||
docker exec "$name" sysctl -w net.ipv6.conf.all.forwarding=1 >/dev/null 2>&1 || true
|
docker exec "$name" sysctl -w net.ipv6.conf.all.forwarding=1 >/dev/null 2>&1 || true
|
||||||
docker exec "$name" bash -c '
|
timeout "$CONFIG_RESTART_TIMEOUT" docker exec "$name" bash -c '
|
||||||
systemctl unmask fips-gateway.service 2>/dev/null
|
systemctl unmask fips-gateway.service 2>/dev/null
|
||||||
# Patch in a minimal gateway config since the shipped fips.yaml
|
# Patch in a minimal gateway config since the shipped fips.yaml
|
||||||
# has gateway disabled by default.
|
# has gateway disabled by default.
|
||||||
@@ -482,7 +538,7 @@ EOF
|
|||||||
fail "non-root fips group member cannot reach control socket after restart"
|
fail "non-root fips group member cannot reach control socket after restart"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
docker exec "$name" systemctl start fips-gateway.service >/dev/null 2>&1 || true
|
start_unit "$name" fips-gateway.service "$GATEWAY_START_TIMEOUT" >/dev/null 2>&1 || true
|
||||||
sleep 3
|
sleep 3
|
||||||
if docker exec "$name" journalctl -u fips-gateway.service --no-pager 2>/dev/null \
|
if docker exec "$name" journalctl -u fips-gateway.service --no-pager 2>/dev/null \
|
||||||
| grep -q "DNS upstream is reachable"; then
|
| grep -q "DNS upstream is reachable"; then
|
||||||
|
|||||||
Reference in New Issue
Block a user