mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-05 19:18:25 +00:00
Stop the dns-resolver harness discarding the cause of its own failures
Build and container-start output went to /dev/null, so a failing scenario printed that it had failed and nothing about why. The ubuntu:26.04 e2e failure has been undiagnosable since July for exactly that reason. Output is now captured and emitted only when the command fails, naming which command it was, so a passing run stays as quiet as before. The systemd readiness wait, which is the check that actually times out, dumps the container state, the failed units and the journal when it gives up. This also fixes a defect the change uncovered. The script runs under pipefail, and systemctl is-system-running exits non-zero when the system is degraded, so the pipeline testing for 'running|degraded' failed even when grep matched. The degraded branch was dead, and since systemd inside a container always settles at degraded, every scenario burned the full 30-second boot timeout and emitted a spurious warning about a container that had booted correctly. The readiness poll now matches on the captured state instead of piping into grep. The original ubuntu:26.04 failure does not reproduce on this host, so that half stays open. What changes is that the next occurrence will say why.
This commit is contained in:
@@ -55,50 +55,115 @@ cleanup_container() {
|
||||
docker rm -f "$name" >/dev/null 2>&1 || true
|
||||
}
|
||||
|
||||
# Emit a captured output file to stderr, delimited and labelled with the
|
||||
# command it came from. Callers use this only on failure: a command that
|
||||
# succeeds leaves no trace, so the suite stays quiet when it is green.
|
||||
dump_output() {
|
||||
local label="$1" file="$2"
|
||||
{
|
||||
echo " --- $label failed; captured output follows ---"
|
||||
if [ -s "$file" ]; then
|
||||
cat "$file"
|
||||
else
|
||||
echo " (no output)"
|
||||
fi
|
||||
echo " --- end captured output ---"
|
||||
} >&2
|
||||
}
|
||||
|
||||
# Run a command with both streams captured. Discard the capture on success;
|
||||
# on failure emit it, so the reason a build or a container start died is not
|
||||
# thrown away. Stdin is inherited, so a caller may pipe into it.
|
||||
run_quiet() {
|
||||
local label="$1"
|
||||
shift
|
||||
local out rc=0
|
||||
out=$(mktemp)
|
||||
"$@" >"$out" 2>&1 || rc=$?
|
||||
[ "$rc" -eq 0 ] || dump_output "$label" "$out"
|
||||
rm -f "$out"
|
||||
return "$rc"
|
||||
}
|
||||
|
||||
# Build an image from an inline Dockerfile.
|
||||
build_image() {
|
||||
local tag="$1"
|
||||
shift
|
||||
echo "$@" | docker build -t "$tag" -f - "$REPO_ROOT" >/dev/null 2>&1
|
||||
echo "$@" | run_quiet "docker build -t $tag" \
|
||||
docker build -t "$tag" -f - "$REPO_ROOT"
|
||||
}
|
||||
|
||||
# Start a systemd container in the background.
|
||||
start_systemd_container() {
|
||||
local name="$1" image="$2"
|
||||
cleanup_container "$name"
|
||||
docker run -d --name "$name" \
|
||||
run_quiet "docker run $name" \
|
||||
docker run -d --name "$name" \
|
||||
--label com.corganlabs.fips-ci=1 \
|
||||
--privileged \
|
||||
--cgroupns=host \
|
||||
-v /sys/fs/cgroup:/sys/fs/cgroup:rw \
|
||||
--tmpfs /run --tmpfs /run/lock \
|
||||
"$image" >/dev/null 2>&1
|
||||
"$image"
|
||||
}
|
||||
|
||||
# Same, but with TUN device for the e2e scenario.
|
||||
start_systemd_container_with_tun() {
|
||||
local name="$1" image="$2"
|
||||
cleanup_container "$name"
|
||||
docker run -d --name "$name" \
|
||||
run_quiet "docker run $name (with tun)" \
|
||||
docker run -d --name "$name" \
|
||||
--label com.corganlabs.fips-ci=1 \
|
||||
--privileged \
|
||||
--cgroupns=host \
|
||||
--device /dev/net/tun \
|
||||
-v /sys/fs/cgroup:/sys/fs/cgroup:rw \
|
||||
--tmpfs /run --tmpfs /run/lock \
|
||||
"$image" >/dev/null 2>&1
|
||||
"$image"
|
||||
}
|
||||
|
||||
# Report why systemd never reached a running state. Every probe is
|
||||
# best-effort and captured with its own stderr: the container may have
|
||||
# exited, or never have been created at all, and the docker error text
|
||||
# saying so is itself the diagnosis.
|
||||
dump_systemd_state() {
|
||||
local name="$1" out
|
||||
out=$(mktemp)
|
||||
{
|
||||
echo "== docker ps -a"
|
||||
docker ps -a --filter "name=^${name}$" 2>&1
|
||||
echo "== systemctl is-system-running"
|
||||
docker exec "$name" systemctl is-system-running 2>&1
|
||||
echo "== systemctl list-units --failed"
|
||||
docker exec "$name" systemctl list-units --failed --no-pager 2>&1
|
||||
echo "== journalctl -b (last 100 lines)"
|
||||
docker exec "$name" journalctl -b --no-pager -n 100 2>&1
|
||||
echo "== docker logs (last 100 lines)"
|
||||
docker logs --tail 100 "$name" 2>&1
|
||||
} >"$out" 2>&1
|
||||
dump_output "systemd boot of $name" "$out"
|
||||
rm -f "$out"
|
||||
}
|
||||
|
||||
# Wait for systemd to reach a bootable state inside the container.
|
||||
wait_for_systemd() {
|
||||
local name="$1"
|
||||
local state
|
||||
for _i in $(seq 1 "$BOOT_TIMEOUT"); do
|
||||
if docker exec "$name" systemctl is-system-running --wait 2>/dev/null | grep -qE 'running|degraded'; then
|
||||
return 0
|
||||
fi
|
||||
# Read the state rather than piping it into grep. systemctl exits
|
||||
# non-zero for "degraded", and pipefail turns that into a failed
|
||||
# pipeline even when grep matched, so the piped form could never
|
||||
# accept a degraded boot: in a container systemd-modules-load
|
||||
# always fails, so every scenario burned the full timeout and
|
||||
# warned about a container that had in fact booted.
|
||||
state=$(docker exec "$name" systemctl is-system-running --wait 2>/dev/null)
|
||||
case "$state" in
|
||||
*running*|*degraded*) return 0 ;;
|
||||
esac
|
||||
sleep 1
|
||||
done
|
||||
echo " WARNING: systemd did not reach running state in ${BOOT_TIMEOUT}s (may still work)"
|
||||
dump_systemd_state "$name"
|
||||
return 0
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user