diff --git a/testing/interop/README.md b/testing/interop/README.md index c87cb319..bd92f467 100644 --- a/testing/interop/README.md +++ b/testing/interop/README.md @@ -177,12 +177,17 @@ FIPS_INTEROP_NETEM="delay 10ms 5ms 25% loss 2%" \ wants netem) but still runs a clean baseline loop. Each rep invokes `interop-test.sh` with netem set and captures its full -output and exit code; a rep passes iff `interop-test.sh` exits 0. Artifacts -land in `testing/interop/.stress-runs//`: +output and exit code; a rep passes iff `interop-test.sh` exits 0. Every rep +runs with `FIPS_INTEROP_KEEP_UP=1`, because whether it failed is known only +after the driver exits: the loop saves a failed rep's per-node logs and then +tears the mesh down itself, including when the loop is interrupted. +Artifacts land in `testing/interop/.stress-runs//`: - `rep-NN/driver.log` — full driver output for every rep. - `rep-NN/docker-.log` — per-container `docker logs` for **failed** - reps only. + reps only. The expected containers come from the generated manifest, and + the loop prints `harvest: k of n per-node logs written` for each failed + rep. A log that could not be read or came back empty counts as missing. - `summary.txt` — the aggregate report. The aggregate report gives reps run, passed/failed counts and an integer @@ -193,9 +198,10 @@ pair kind (mixed vs same), and a verdict: - **both mixed and same pairs** → loss-induced general instability; - **same-version control pair only** → the build is unstable against itself. -`interop-stress.sh` exits **non-zero only** for the interop-regression -signal. A sub-100% pass rate under loss is expected and is not by itself a -failure, so every other outcome exits 0. +`interop-stress.sh` exits **1** for the interop-regression signal and **3** +when a failed rep's per-node logs are incomplete (the run's diagnostics are +missing; a regression keeps exit 1). A sub-100% pass rate under loss is +expected and is not by itself a failure, so every other outcome exits 0. ### Options @@ -203,7 +209,7 @@ failure, so every other outcome exits 0. | ---------------------------- | ------------------------------------------------- | | `FIPS_INTEROP_NETEM` | tc-netem string applied to each container's eth0, e.g. `"delay 10ms 5ms 25% loss 1%"`. Passed through `interop-stress.sh` to `interop-test.sh`. | | `REKEY_AFTER_SECS` | Rekey interval for generated configs (default 35).| -| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. | +| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. The stress loop sets it for every rep and tears down itself. | | `FIPS_INTEROP_KEEP_WORKTREES`| `1` = keep `build-images.sh` worktrees (debug). | | `FIPS_INTEROP_RUNS_DIR` | Root for the three scratch dirs — see [Scratch directory location](#scratch-directory-location). | diff --git a/testing/interop/interop-stress.sh b/testing/interop/interop-stress.sh index 6cf98952..a2629bdf 100755 --- a/testing/interop/interop-stress.sh +++ b/testing/interop/interop-stress.sh @@ -19,8 +19,13 @@ # -> the version under test is unstable even against itself. # # A sub-100% pass rate under loss is EXPECTED and is not, by itself, a -# failure. This script exits non-zero ONLY for the interop-regression -# signal (mixed-only failures). +# failure. This script exits 1 for the interop-regression signal +# (mixed-only failures) and 3 when a failed rep's per-node logs could not +# all be saved, so its diagnostics are incomplete. +# +# Every rep runs with FIPS_INTEROP_KEEP_UP=1, because whether a rep failed +# is known only after the driver exits. The loop saves a failed rep's +# per-node logs and then tears the mesh down itself. # # Reps run SERIALLY — interop-test.sh uses fixed container names and a # fixed Docker network, so two reps must never overlap. @@ -52,7 +57,10 @@ # # Artifacts (per invocation): /.stress-runs// # rep-NN/driver.log full interop-test.sh output for that rep. -# rep-NN/docker-.log per-container `docker logs` (FAILED reps only). +# rep-NN/docker-.log +# per-container `docker logs` (FAILED reps only). +# A failed rep with fewer of these than nodes +# makes the run exit 3. # summary.txt the final aggregate report. set -uo pipefail @@ -80,6 +88,65 @@ else fi RUNS_BASE="$INTEROP_RUNS_BASE/.stress-runs" +# The driver's generated mesh (same paths as interop-test.sh), read for the +# expected container set and used to tear a kept-up rep down. +GEN_DIR="$INTEROP_RUNS_BASE/generated-configs" +COMPOSE_FILE="$GEN_DIR/docker-compose.generated.yml" +NODES_ENV="$GEN_DIR/nodes.env" + +# Save every node's `docker logs` into a failed rep's directory. +# +# The expected containers come from the generated manifest rather than +# from `docker ps`: an empty `docker ps` result is how a harvest that saved +# nothing used to pass without a word. A container counts only when +# `docker logs` succeeded and wrote a non-empty file. Returns 0 only when +# every expected container was saved and there was at least one. +harvest_rep() { + local rep_dir="$1" containers tok ctr n=0 k=0 + local missing=() + # A subshell, so the manifest's variables stay out of the loop. + # shellcheck disable=SC1090 + containers="$( [ -f "$NODES_ENV" ] && . "$NODES_ENV" \ + && printf '%s' "${INTEROP_NODE_CONTAINERS:-}" )" || containers="" + for tok in $containers; do + ctr="${tok#*:}" + n=$((n + 1)) + if docker logs "$ctr" >"$rep_dir/docker-$ctr.log" 2>&1 \ + && [ -s "$rep_dir/docker-$ctr.log" ]; then + k=$((k + 1)) + else + missing+=("$ctr") + fi + done + echo " harvest: $k of $n per-node logs written" + if [ "$n" -eq 0 ]; then + echo " HARVEST FAILED: no manifest / empty container list ($NODES_ENV)" + return 1 + fi + if [ "$k" -lt "$n" ]; then + echo " HARVEST FAILED: $k of $n; missing: ${missing[*]}" + return 1 + fi + return 0 +} + +# Tear down a rep's kept-up mesh. A failure is loud but not fatal: the +# next rep's Phase 0 also brings the mesh down before starting. +teardown_rep() { + local rc left + docker compose -f "$COMPOSE_FILE" down --volumes --remove-orphans \ + >/dev/null 2>&1 + rc=$? + MESH_UP=0 + if [ "$rc" -ne 0 ]; then + echo " WARN: teardown failed ($rc)" + fi + left="$(docker ps -a --filter 'name=fips-interop-' --format '{{.Names}}' 2>/dev/null)" + if [ -n "$left" ]; then + echo " WARN: containers remain after teardown: $(echo "$left" | tr '\n' ' ')" + fi + return "$rc" +} # ── Args ───────────────────────────────────────────────────────────── @@ -171,6 +238,13 @@ RUN_TS="$(date -u +%Y-%m-%dT%H-%M-%SZ)" RUN_DIR="$RUNS_BASE/$RUN_TS" mkdir -p "$RUN_DIR" +# A rep runs kept up, so an interrupted or aborted loop must still take +# its mesh down. The INT and TERM traps exit, which runs the EXIT trap. +MESH_UP=0 +trap '[ "$MESH_UP" -eq 1 ] && teardown_rep' EXIT +trap 'echo ""; echo "Interrupted"; exit 130' INT +trap 'echo ""; echo "Terminated"; exit 143' TERM + echo "==============================================================" echo " FIPS Interop Netem Stress Loop" echo "==============================================================" @@ -186,6 +260,9 @@ echo "" PASS_COUNT=0 FAIL_COUNT=0 FAILED_REPS=() +# Failed reps whose per-node logs were not all saved. +HARVEST_FAILS=0 +HARVEST_FAILED_REPS=() # Per-kind connectivity-failure tallies, summed across all failed reps. MIXED_FAILS=0 SAME_FAILS=0 @@ -199,8 +276,10 @@ for ((rep = 1; rep <= REPS; rep++)); do echo "── $rep_id / $REPS ──────────────────────────────────────────" # Run the driver, capturing full output and exit code. Netem is - # passed through the environment; interop-test.sh applies it. - FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \ + # passed through the environment; interop-test.sh applies it. The + # mesh is kept up so a failed rep can be harvested below. + MESH_UP=1 + FIPS_INTEROP_KEEP_UP=1 FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \ bash "$DRIVER" "${DRIVER_ARGS[@]}" >"$driver_log" 2>&1 rc=$? @@ -212,14 +291,11 @@ for ((rep = 1; rep <= REPS; rep++)); do FAILED_REPS+=("$rep_id") echo " FAIL (exit $rc)" - # Preserve each failed container's full docker logs. The driver - # uses fixed container names fips-interop-; harvest every - # container matching that prefix that still exists. - while read -r ctr; do - [ -n "$ctr" ] || continue - docker logs "$ctr" >"$rep_dir/docker-${ctr}.log" 2>&1 || true - done < <(docker ps -a --filter 'name=fips-interop-' \ - --format '{{.Names}}' 2>/dev/null) + # Preserve each node's full docker logs before the teardown below. + if ! harvest_rep "$rep_dir"; then + HARVEST_FAILS=$((HARVEST_FAILS + 1)) + HARVEST_FAILED_REPS+=("$rep_id") + fi # Tally connectivity failures by pair kind, reusing the # pair-attributed lines interop-test.sh prints. Each line is @@ -230,6 +306,8 @@ for ((rep = 1; rep <= REPS; rep++)); do SAME_FAILS=$((SAME_FAILS + s)) echo " connectivity-failure lines: mixed=$m same=$s" fi + + teardown_rep done echo "" @@ -285,6 +363,10 @@ else VERDICT="NO connectivity-pair failures recorded, but $FAIL_COUNT rep(s) still" VERDICT2="failed — on non-connectivity signatures (global-health log patterns or a missing rekey). Check the per-rep driver logs." fi +# A regression keeps precedence; otherwise missing diagnostics fail the run. +if [ "$EXIT_CODE" -eq 0 ] && [ "$HARVEST_FAILS" -gt 0 ]; then + EXIT_CODE=3 +fi { echo "==============================================================" @@ -302,6 +384,9 @@ fi if [ "${#FAILED_REPS[@]}" -gt 0 ]; then echo "Failed reps: ${FAILED_REPS[*]}" fi + if [ "$HARVEST_FAILS" -gt 0 ]; then + echo "Harvest : $HARVEST_FAILS failed rep(s) with incomplete per-node logs: ${HARVEST_FAILED_REPS[*]}" + fi echo "" echo "-- Connectivity-failure attribution (summed over failed reps) --" echo " mixed-version : $MIXED_FAILS failure(s) over $MIXED_PAIRS mixed pairs x $REPS reps" @@ -315,6 +400,9 @@ fi if [ "$EXIT_CODE" -eq 0 ]; then echo "Exit 0: no interop-regression signal (a sub-100% rate under loss" echo " is expected and is not by itself a failure)." + elif [ "$EXIT_CODE" -eq 3 ]; then + echo "Exit 3: per-node logs missing for a failed rep; the run's" + echo " diagnostics are incomplete." else echo "Exit 1: interop-regression signal present." fi