From b82b9bf34edb4ad53218199d0d6b7b9ba4dffb7c Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 19 Sep 2026 02:27:57 +0000 Subject: [PATCH] Keep a failed interop stress rep's containers until their logs are saved The stress loop documented a per-container docker log for every failed rep, but never wrote one. The driver's EXIT trap had already removed the containers by the time the loop listed them, so the listing came back empty, the harvest loop never ran, and nothing said so. Reps now run with the driver's keep-up switch. The loop takes the expected container set from the generated manifest, saves each node's log, and tears the mesh down itself, also on interrupt. A failed rep whose harvest writes fewer non-empty logs than it has nodes fails the run with exit 3, so missing diagnostics can no longer pass silently. --- testing/interop/README.md | 20 ++++-- testing/interop/interop-stress.sh | 114 ++++++++++++++++++++++++++---- 2 files changed, 114 insertions(+), 20 deletions(-) diff --git a/testing/interop/README.md b/testing/interop/README.md index c87cb319..bd92f467 100644 --- a/testing/interop/README.md +++ b/testing/interop/README.md @@ -177,12 +177,17 @@ FIPS_INTEROP_NETEM="delay 10ms 5ms 25% loss 2%" \ wants netem) but still runs a clean baseline loop. Each rep invokes `interop-test.sh` with netem set and captures its full -output and exit code; a rep passes iff `interop-test.sh` exits 0. Artifacts -land in `testing/interop/.stress-runs//`: +output and exit code; a rep passes iff `interop-test.sh` exits 0. Every rep +runs with `FIPS_INTEROP_KEEP_UP=1`, because whether it failed is known only +after the driver exits: the loop saves a failed rep's per-node logs and then +tears the mesh down itself, including when the loop is interrupted. +Artifacts land in `testing/interop/.stress-runs//`: - `rep-NN/driver.log` — full driver output for every rep. - `rep-NN/docker-.log` — per-container `docker logs` for **failed** - reps only. + reps only. The expected containers come from the generated manifest, and + the loop prints `harvest: k of n per-node logs written` for each failed + rep. A log that could not be read or came back empty counts as missing. - `summary.txt` — the aggregate report. The aggregate report gives reps run, passed/failed counts and an integer @@ -193,9 +198,10 @@ pair kind (mixed vs same), and a verdict: - **both mixed and same pairs** → loss-induced general instability; - **same-version control pair only** → the build is unstable against itself. -`interop-stress.sh` exits **non-zero only** for the interop-regression -signal. A sub-100% pass rate under loss is expected and is not by itself a -failure, so every other outcome exits 0. +`interop-stress.sh` exits **1** for the interop-regression signal and **3** +when a failed rep's per-node logs are incomplete (the run's diagnostics are +missing; a regression keeps exit 1). A sub-100% pass rate under loss is +expected and is not by itself a failure, so every other outcome exits 0. ### Options @@ -203,7 +209,7 @@ failure, so every other outcome exits 0. | ---------------------------- | ------------------------------------------------- | | `FIPS_INTEROP_NETEM` | tc-netem string applied to each container's eth0, e.g. `"delay 10ms 5ms 25% loss 1%"`. Passed through `interop-stress.sh` to `interop-test.sh`. | | `REKEY_AFTER_SECS` | Rekey interval for generated configs (default 35).| -| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. | +| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. The stress loop sets it for every rep and tears down itself. | | `FIPS_INTEROP_KEEP_WORKTREES`| `1` = keep `build-images.sh` worktrees (debug). | | `FIPS_INTEROP_RUNS_DIR` | Root for the three scratch dirs — see [Scratch directory location](#scratch-directory-location). | diff --git a/testing/interop/interop-stress.sh b/testing/interop/interop-stress.sh index 6cf98952..a2629bdf 100755 --- a/testing/interop/interop-stress.sh +++ b/testing/interop/interop-stress.sh @@ -19,8 +19,13 @@ # -> the version under test is unstable even against itself. # # A sub-100% pass rate under loss is EXPECTED and is not, by itself, a -# failure. This script exits non-zero ONLY for the interop-regression -# signal (mixed-only failures). +# failure. This script exits 1 for the interop-regression signal +# (mixed-only failures) and 3 when a failed rep's per-node logs could not +# all be saved, so its diagnostics are incomplete. +# +# Every rep runs with FIPS_INTEROP_KEEP_UP=1, because whether a rep failed +# is known only after the driver exits. The loop saves a failed rep's +# per-node logs and then tears the mesh down itself. # # Reps run SERIALLY — interop-test.sh uses fixed container names and a # fixed Docker network, so two reps must never overlap. @@ -52,7 +57,10 @@ # # Artifacts (per invocation): /.stress-runs// # rep-NN/driver.log full interop-test.sh output for that rep. -# rep-NN/docker-.log per-container `docker logs` (FAILED reps only). +# rep-NN/docker-.log +# per-container `docker logs` (FAILED reps only). +# A failed rep with fewer of these than nodes +# makes the run exit 3. # summary.txt the final aggregate report. set -uo pipefail @@ -80,6 +88,65 @@ else fi RUNS_BASE="$INTEROP_RUNS_BASE/.stress-runs" +# The driver's generated mesh (same paths as interop-test.sh), read for the +# expected container set and used to tear a kept-up rep down. +GEN_DIR="$INTEROP_RUNS_BASE/generated-configs" +COMPOSE_FILE="$GEN_DIR/docker-compose.generated.yml" +NODES_ENV="$GEN_DIR/nodes.env" + +# Save every node's `docker logs` into a failed rep's directory. +# +# The expected containers come from the generated manifest rather than +# from `docker ps`: an empty `docker ps` result is how a harvest that saved +# nothing used to pass without a word. A container counts only when +# `docker logs` succeeded and wrote a non-empty file. Returns 0 only when +# every expected container was saved and there was at least one. +harvest_rep() { + local rep_dir="$1" containers tok ctr n=0 k=0 + local missing=() + # A subshell, so the manifest's variables stay out of the loop. + # shellcheck disable=SC1090 + containers="$( [ -f "$NODES_ENV" ] && . "$NODES_ENV" \ + && printf '%s' "${INTEROP_NODE_CONTAINERS:-}" )" || containers="" + for tok in $containers; do + ctr="${tok#*:}" + n=$((n + 1)) + if docker logs "$ctr" >"$rep_dir/docker-$ctr.log" 2>&1 \ + && [ -s "$rep_dir/docker-$ctr.log" ]; then + k=$((k + 1)) + else + missing+=("$ctr") + fi + done + echo " harvest: $k of $n per-node logs written" + if [ "$n" -eq 0 ]; then + echo " HARVEST FAILED: no manifest / empty container list ($NODES_ENV)" + return 1 + fi + if [ "$k" -lt "$n" ]; then + echo " HARVEST FAILED: $k of $n; missing: ${missing[*]}" + return 1 + fi + return 0 +} + +# Tear down a rep's kept-up mesh. A failure is loud but not fatal: the +# next rep's Phase 0 also brings the mesh down before starting. +teardown_rep() { + local rc left + docker compose -f "$COMPOSE_FILE" down --volumes --remove-orphans \ + >/dev/null 2>&1 + rc=$? + MESH_UP=0 + if [ "$rc" -ne 0 ]; then + echo " WARN: teardown failed ($rc)" + fi + left="$(docker ps -a --filter 'name=fips-interop-' --format '{{.Names}}' 2>/dev/null)" + if [ -n "$left" ]; then + echo " WARN: containers remain after teardown: $(echo "$left" | tr '\n' ' ')" + fi + return "$rc" +} # ── Args ───────────────────────────────────────────────────────────── @@ -171,6 +238,13 @@ RUN_TS="$(date -u +%Y-%m-%dT%H-%M-%SZ)" RUN_DIR="$RUNS_BASE/$RUN_TS" mkdir -p "$RUN_DIR" +# A rep runs kept up, so an interrupted or aborted loop must still take +# its mesh down. The INT and TERM traps exit, which runs the EXIT trap. +MESH_UP=0 +trap '[ "$MESH_UP" -eq 1 ] && teardown_rep' EXIT +trap 'echo ""; echo "Interrupted"; exit 130' INT +trap 'echo ""; echo "Terminated"; exit 143' TERM + echo "==============================================================" echo " FIPS Interop Netem Stress Loop" echo "==============================================================" @@ -186,6 +260,9 @@ echo "" PASS_COUNT=0 FAIL_COUNT=0 FAILED_REPS=() +# Failed reps whose per-node logs were not all saved. +HARVEST_FAILS=0 +HARVEST_FAILED_REPS=() # Per-kind connectivity-failure tallies, summed across all failed reps. MIXED_FAILS=0 SAME_FAILS=0 @@ -199,8 +276,10 @@ for ((rep = 1; rep <= REPS; rep++)); do echo "── $rep_id / $REPS ──────────────────────────────────────────" # Run the driver, capturing full output and exit code. Netem is - # passed through the environment; interop-test.sh applies it. - FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \ + # passed through the environment; interop-test.sh applies it. The + # mesh is kept up so a failed rep can be harvested below. + MESH_UP=1 + FIPS_INTEROP_KEEP_UP=1 FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \ bash "$DRIVER" "${DRIVER_ARGS[@]}" >"$driver_log" 2>&1 rc=$? @@ -212,14 +291,11 @@ for ((rep = 1; rep <= REPS; rep++)); do FAILED_REPS+=("$rep_id") echo " FAIL (exit $rc)" - # Preserve each failed container's full docker logs. The driver - # uses fixed container names fips-interop-; harvest every - # container matching that prefix that still exists. - while read -r ctr; do - [ -n "$ctr" ] || continue - docker logs "$ctr" >"$rep_dir/docker-${ctr}.log" 2>&1 || true - done < <(docker ps -a --filter 'name=fips-interop-' \ - --format '{{.Names}}' 2>/dev/null) + # Preserve each node's full docker logs before the teardown below. + if ! harvest_rep "$rep_dir"; then + HARVEST_FAILS=$((HARVEST_FAILS + 1)) + HARVEST_FAILED_REPS+=("$rep_id") + fi # Tally connectivity failures by pair kind, reusing the # pair-attributed lines interop-test.sh prints. Each line is @@ -230,6 +306,8 @@ for ((rep = 1; rep <= REPS; rep++)); do SAME_FAILS=$((SAME_FAILS + s)) echo " connectivity-failure lines: mixed=$m same=$s" fi + + teardown_rep done echo "" @@ -285,6 +363,10 @@ else VERDICT="NO connectivity-pair failures recorded, but $FAIL_COUNT rep(s) still" VERDICT2="failed — on non-connectivity signatures (global-health log patterns or a missing rekey). Check the per-rep driver logs." fi +# A regression keeps precedence; otherwise missing diagnostics fail the run. +if [ "$EXIT_CODE" -eq 0 ] && [ "$HARVEST_FAILS" -gt 0 ]; then + EXIT_CODE=3 +fi { echo "==============================================================" @@ -302,6 +384,9 @@ fi if [ "${#FAILED_REPS[@]}" -gt 0 ]; then echo "Failed reps: ${FAILED_REPS[*]}" fi + if [ "$HARVEST_FAILS" -gt 0 ]; then + echo "Harvest : $HARVEST_FAILS failed rep(s) with incomplete per-node logs: ${HARVEST_FAILED_REPS[*]}" + fi echo "" echo "-- Connectivity-failure attribution (summed over failed reps) --" echo " mixed-version : $MIXED_FAILS failure(s) over $MIXED_PAIRS mixed pairs x $REPS reps" @@ -315,6 +400,9 @@ fi if [ "$EXIT_CODE" -eq 0 ]; then echo "Exit 0: no interop-regression signal (a sub-100% rate under loss" echo " is expected and is not by itself a failure)." + elif [ "$EXIT_CODE" -eq 3 ]; then + echo "Exit 3: per-node logs missing for a failed rep; the run's" + echo " diagnostics are incomplete." else echo "Exit 1: interop-regression signal present." fi