diff --git a/.github/workflows/aur-publish.yml b/.github/workflows/aur-publish.yml index 50eadd1b..d82f8192 100644 --- a/.github/workflows/aur-publish.yml +++ b/.github/workflows/aur-publish.yml @@ -39,10 +39,16 @@ jobs: - name: Install build and lint tooling run: | set -euo pipefail - pacman -Sy --noconfirm --needed base-devel namcap git curl + pacman -Sy --noconfirm --needed base-devel namcap git curl jq - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6 + - name: Test the publish gate + # Fixture tests for the script the publish job below waits with. They + # run here, on every trigger, so a change to the gate is exercised + # before a release tag depends on it. + run: bash packaging/aur/test-await-package-runs.sh + - name: Resolve package version id: ver env: @@ -136,11 +142,21 @@ jobs: # or a manual dispatch (packaging-only republish with explicit tag + pkgrel). # Branch pushes and pull requests build+lint above but never reach this job. # Gated on aur-build so a package that fails to build/lint is never published. + # It also waits for every package-*.yml run on the tag to succeed before it + # pushes: the AUR must not point at a tag whose release assets are still + # uploading or failed, and once it does, withdrawing the tag breaks the AUR + # package (its b2sum pins the tag's source archive). # ─────────────────────────────────────────────────────────────────────────── aur-publish-fips: name: Publish fips to AUR needs: aur-build runs-on: ubuntu-latest + # Above the gate's own 60-minute budget, so the gate reports a timeout + # rather than the runner killing it. + timeout-minutes: 90 + permissions: + contents: read + actions: read if: >- github.event_name == 'workflow_dispatch' || (github.event_name == 'push' @@ -180,6 +196,26 @@ jobs: with: ref: ${{ steps.tag.outputs.tag }} + # The gate script comes from this workflow's own revision, not the tag: + # a dispatch republishing a tag cut before the gate existed would not + # find it in the tag's tree. The workflows it waits on still come from + # the tag's tree, which is what the tag push triggered. + - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6 + with: + path: gate-src + sparse-checkout: packaging/aur + + - name: Wait for the tag's package workflows + env: + GH_TOKEN: ${{ github.token }} + TAG: ${{ steps.tag.outputs.tag }} + run: | + set -euo pipefail + SHA=$(git rev-parse 'HEAD^{commit}') + echo "Tag $TAG is at $SHA" + SHA="$SHA" WORKFLOW_DIR=.github/workflows \ + bash gate-src/packaging/aur/await-package-runs.sh + - name: Patch PKGBUILD with pkgver, pkgrel, conflicts, and b2sums env: TAG: ${{ steps.tag.outputs.tag }} diff --git a/packaging/aur/PKGBUILD b/packaging/aur/PKGBUILD index ac758a6d..9cdbc749 100644 --- a/packaging/aur/PKGBUILD +++ b/packaging/aur/PKGBUILD @@ -6,7 +6,7 @@ pkgdesc="Distributed, decentralized network routing protocol for mesh nodes" url="https://github.com/jmcorgan/fips" license=('MIT') arch=('x86_64') -depends=('gcc-libs' 'glibc') +depends=('dbus' 'gcc-libs' 'glibc') makedepends=('cargo' 'clang') optdepends=('systemd-resolved: .fips DNS resolution') conflicts=('fips-git' 'fips-git-debug') diff --git a/packaging/aur/README.md b/packaging/aur/README.md index ed5a2991..8017349f 100644 --- a/packaging/aur/README.md +++ b/packaging/aur/README.md @@ -20,6 +20,8 @@ This directory contains Arch Linux packaging files for two AUR packages: | `fips-dns.service` | Symlink to `../debian/fips-dns.service` | | `build-aur.sh` | Local `fips-git` build plus namcap validation (run by `make aur`) | | `patch-pkgbuild.sh` | Rewrites `pkgver`, `pkgrel`, `conflicts`, `options`, and `b2sums` in the PKGBUILD at publish time | +| `await-package-runs.sh` | Holds the AUR publish until every `package-*.yml` run for the release tag has succeeded | +| `test-await-package-runs.sh` | Fixture tests for `await-package-runs.sh`, run by the `aur-build` job | Both PKGBUILDs reference files from `packaging/debian/` (service files) and `packaging/common/` (config files) at build time. These are pulled from the diff --git a/packaging/aur/await-package-runs.sh b/packaging/aur/await-package-runs.sh new file mode 100755 index 00000000..dc2fcecd --- /dev/null +++ b/packaging/aur/await-package-runs.sh @@ -0,0 +1,131 @@ +#!/usr/bin/env bash +# Wait until every package workflow run for a release tag has succeeded. +# +# The AUR publish job runs this before it pushes a new pkgver, so the AUR never +# points at a tag whose release assets are still uploading or failed to build. +# Each package-*.yml workflow uploads its assets from a `release` job inside +# the same run, so a successful tag run means that workflow's assets are up. +# +# The population is discovered from the checked-out tree rather than listed +# here: the package-*.yml files at the tag are the workflows the tag push +# triggered. Finding none is a failure, never a pass. +# +# A tag push and the branch push of the same commit each start a run with the +# same head_sha; only the tag run carries the tag name in head_branch, so runs +# are matched on both. If the tag was re-pushed at the same commit, the newest +# matching run decides. +# +# Required environment variables: +# GITHUB_REPOSITORY - owner/repo +# TAG - release tag (e.g. v0.5.1) +# SHA - the commit the tag points at +# GH_TOKEN - token with actions:read (read by gh; not checked here) +# Optional: +# WORKFLOW_DIR - where to discover package-*.yml (default .github/workflows) +# AWAIT_POLLS - number of polls before giving up (default 60) +# AWAIT_INTERVAL - seconds between polls (default 60) +# DISCOVER_ONLY - if set to 1, print the discovered workflows and exit +# +# Exit status: 0 when every discovered workflow has a successful tag run; +# 1 at once when a tag run concludes anything but success; 1 when the poll +# budget runs out with any workflow unobserved, not finished, or unreadable. + +set -euo pipefail + +: "${GITHUB_REPOSITORY:?GITHUB_REPOSITORY must be set}" +: "${TAG:?TAG must be set}" +: "${SHA:?SHA must be set}" +WORKFLOW_DIR="${WORKFLOW_DIR:-.github/workflows}" +AWAIT_POLLS="${AWAIT_POLLS:-60}" +AWAIT_INTERVAL="${AWAIT_INTERVAL:-60}" + +# How to recover once the cause of a failure is fixed. A workflow_dispatch run +# of a package workflow is not a push run of the tag and never satisfies this +# gate; re-running the failed jobs of the tag's own run keeps it one. +RECOVERY="If a run failed: use \"Re-run failed jobs\" on the package workflow's run for $TAG \ +(a new workflow_dispatch run does not count), then re-run the AUR Publish job for $TAG." + +# Print the basenames of the package workflows found in WORKFLOW_DIR. +discover() { + local f + for f in "$WORKFLOW_DIR"/package-*.yml; do + [ -e "$f" ] && basename "$f" + done + return 0 +} + +# Print " " for the newest push run of TAG at +# SHA for one workflow, or "none" if there is no such run yet. On an API or +# parse failure, print the error text and return nonzero. +latest() { + local wf=$1 body + if ! body=$(gh api "repos/$GITHUB_REPOSITORY/actions/workflows/$wf/runs?head_sha=$SHA&event=push&per_page=100" 2>"$ERRS"); then + tr '\n' ' ' < "$ERRS" + return 1 + fi + printf '%s\n' "$body" | jq -er --arg tag "$TAG" --arg sha "$SHA" ' + [.workflow_runs[] + | select(.head_branch == $tag and .head_sha == $sha and .event == "push")] + | sort_by(.created_at) + | if length == 0 then "none" + else last | "\(.status) \(.conclusion // "none") \(.html_url)" end' 2>&1 +} + +mapfile -t pending < <(discover) +if [ "${#pending[@]}" -eq 0 ]; then + echo "No package-*.yml workflows found in $WORKFLOW_DIR; refusing to treat an empty set as done" >&2 + exit 1 +fi +if [ "${DISCOVER_ONLY:-}" = 1 ]; then + printf '%s\n' "${pending[@]}" + exit 0 +fi + +ERRS=$(mktemp) +trap 'rm -f "$ERRS"' EXIT + +echo "Waiting for $TAG ($SHA) runs of: ${pending[*]}" +declare -A seen +for ((poll = 1; poll <= AWAIT_POLLS; poll++)); do + still=() + for wf in "${pending[@]}"; do + if ! state=$(latest "$wf"); then + seen[$wf]="API error: $state" + still+=("$wf") + continue + fi + seen[$wf]=$state + read -r status conclusion url <<< "$state" + case "$status" in + none) + # The tag run has not been created yet. + still+=("$wf") ;; + completed) + if [ "$conclusion" = success ]; then + echo "$wf: succeeded ($url)" + else + echo "$wf: run for $TAG concluded '$conclusion' ($url); not publishing to the AUR" >&2 + echo "$RECOVERY" >&2 + exit 1 + fi ;; + *) + # queued, in_progress, waiting, requested or pending. + still+=("$wf") ;; + esac + done + pending=("${still[@]}") + if [ "${#pending[@]}" -eq 0 ]; then + echo "Every package workflow run for $TAG has succeeded" + exit 0 + fi + echo "Poll $poll/$AWAIT_POLLS: waiting on ${pending[*]}" + if [ "$poll" -lt "$AWAIT_POLLS" ]; then sleep "$AWAIT_INTERVAL"; fi +done + +echo "Gave up after $AWAIT_POLLS polls; not publishing to the AUR. Still unfinished:" >&2 +for wf in "${pending[@]}"; do + echo " $wf: ${seen[$wf]}" >&2 +done +echo "If a run is still going, re-run the AUR Publish job for $TAG once it has succeeded." >&2 +echo "$RECOVERY" >&2 +exit 1 diff --git a/packaging/aur/test-await-package-runs.sh b/packaging/aur/test-await-package-runs.sh new file mode 100755 index 00000000..42fa2813 --- /dev/null +++ b/packaging/aur/test-await-package-runs.sh @@ -0,0 +1,233 @@ +#!/usr/bin/env bash +# Fixture tests for await-package-runs.sh, the gate that holds the AUR publish +# until every package workflow run for the release tag has succeeded. +# +# Each case writes GitHub API responses into a fixture directory and puts a +# stub `gh` first on PATH. The stub serves ..json for the Nth +# call about a workflow, falls back to .json, and exits 1 when the +# fixture holds a file named `gh-fails`. It counts calls per workflow so a +# case can assert that the gate polled again rather than stopping early. +# +# Needs bash and jq. Exits nonzero if any case fails or if fewer cases ran +# than are defined. +# +# Usage: bash packaging/aur/test-await-package-runs.sh + +set -euo pipefail + +HERE=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +GATE="${GATE:-$HERE/await-package-runs.sh}" +REPO_ROOT=$(cd "$HERE/../.." && pwd) + +TAG=v9.9.9 +SHA=0123456789abcdef0123456789abcdef01234567 +WORKFLOWS=(package-freebsd.yml package-linux.yml package-macos.yml + package-openwrt.yml package-windows.yml) + +WORK=$(mktemp -d) +trap 'rm -rf "$WORK"' EXIT + +# The stub gh. It answers `gh api ` only, which is all the gate uses. +mkdir -p "$WORK/bin" +cat > "$WORK/bin/gh" <<'STUB' +#!/usr/bin/env bash +set -euo pipefail +[ "${1:-}" = "api" ] || { echo "stub gh: unexpected args: $*" >&2; exit 2; } +[ -e "$STUB_FIXTURE/gh-fails" ] && { echo "stub gh: simulated API error" >&2; exit 1; } +wf=$(printf '%s\n' "$2" | sed -E 's|.*/actions/workflows/([^/]+)/runs.*|\1|') +count_file="$STUB_CALLS/$wf" +n=$(( $(cat "$count_file" 2>/dev/null || echo 0) + 1 )) +echo "$n" > "$count_file" +if [ -f "$STUB_FIXTURE/$wf.$n.json" ]; then + cat "$STUB_FIXTURE/$wf.$n.json" +elif [ -f "$STUB_FIXTURE/$wf.json" ]; then + cat "$STUB_FIXTURE/$wf.json" +else + echo '{"total_count":0,"workflow_runs":[]}' +fi +STUB +chmod +x "$WORK/bin/gh" + +# A workflow directory holding the five package workflows as empty files: the +# gate discovers the population by file name only. +mkdir -p "$WORK/workflows" "$WORK/empty-workflows" +for wf in "${WORKFLOWS[@]}"; do : > "$WORK/workflows/$wf"; done + +# Print one run object in the shape of the v0.5.1 API response. +# Args: head_branch status conclusion created_at id +run() { + local conclusion=null + [ "$3" = null ] || conclusion="\"$3\"" + printf '{"id":%s,"name":"Package","head_branch":"%s","head_sha":"%s","event":"push","status":"%s","conclusion":%s,"created_at":"%s","updated_at":"%s","html_url":"https://github.com/example/fips/actions/runs/%s"}' \ + "$5" "$1" "$SHA" "$2" "$conclusion" "$4" "$4" "$5" +} + +# Wrap run objects, given as arguments in the order the API returns them, in +# the list response envelope. +runs() { + local IFS=, + printf '{"total_count":%d,"workflow_runs":[%s]}\n' "$#" "$*" +} + +# Write the healthy v0.5.1 shape for every workflow into a fixture: a tag run +# and a branch run of the same commit, newest first, both successful. +all_green() { + local dir=$1 wf + for wf in "${WORKFLOWS[@]}"; do + runs "$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" \ + "$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$dir/$wf.json" + done +} + +CASES_DEFINED=0 +CASES_RAN=0 +FAILED=0 + +# Run the gate against a fixture and check its exit status and, when a +# pattern is given, that its output states the expected reason (so a red +# caused by a crash in the gate does not pass as the intended red). +# Args: name fixture-dir expect(zero|nonzero) [pattern] [workflow-dir] +# Sets LAST_OUT to the gate's combined output. +check() { + local name=$1 fixture=$2 expect=$3 pattern=${4:-} wfdir=${5:-$WORK/workflows} rc=0 + CASES_RAN=$((CASES_RAN + 1)) + rm -rf "$WORK/calls"; mkdir -p "$WORK/calls" + LAST_OUT=$(PATH="$WORK/bin:$PATH" STUB_FIXTURE="$fixture" STUB_CALLS="$WORK/calls" \ + GITHUB_REPOSITORY=example/fips TAG="$TAG" SHA="$SHA" WORKFLOW_DIR="$wfdir" \ + AWAIT_POLLS=3 AWAIT_INTERVAL=0 bash "$GATE" 2>&1) || rc=$? + if { [ "$expect" = zero ] && [ "$rc" -eq 0 ]; } || + { [ "$expect" = nonzero ] && [ "$rc" -ne 0 ]; }; then + if [ -z "$pattern" ] || printf '%s\n' "$LAST_OUT" | grep -qE -- "$pattern"; then + echo "PASS $name (exit $rc)" + return 0 + fi + echo "FAIL $name: exit $rc as expected, but output lacks /$pattern/" + else + echo "FAIL $name: expected $expect exit, got $rc" + fi + printf '%s\n' "$LAST_OUT" | sed 's/^/ /' + FAILED=$((FAILED + 1)) + return 1 +} + +# Record an extra assertion's failure against the case that just ran. +fail() { + echo "FAIL $1" + FAILED=$((FAILED + 1)) +} + +# Make a fresh fixture directory for a case and print its path. +fixture() { + local dir="$WORK/fx/$1" + mkdir -p "$dir" + echo "$dir" +} + +# --- cases ------------------------------------------------------------------- + +# 1: every package workflow has a successful tag run. +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 1); all_green "$fx" +check "1 all tag runs succeeded" "$fx" zero "has succeeded$" || true + +# 2: one tag run failed; the gate must name that workflow. +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 2); all_green "$fx" +runs "$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" \ + "$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$fx/package-macos.yml.json" +if check "2 one tag run failed" "$fx" nonzero "concluded 'failure'"; then + printf '%s\n' "$LAST_OUT" | grep -q 'package-macos.yml' || + fail "2 one tag run failed: output does not name package-macos.yml" +fi + +# 3: one tag run was cancelled. +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 3); all_green "$fx" +runs "$(run "$TAG" completed cancelled 2026-09-06T20:31:48Z 2)" > "$fx/package-openwrt.yml.json" +check "3 one tag run cancelled" "$fx" nonzero "package-openwrt.yml: run for .* concluded 'cancelled'" || true + +# 4: one workflow has only the branch run of the tag's commit (the v0.5.1 +# shape a SHA-only filter would accept). +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 4); all_green "$fx" +runs "$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$fx/package-freebsd.yml.json" +check "4 only a branch run, no tag run" "$fx" nonzero "package-freebsd.yml: none$" || true + +# 5: one tag run is in progress on the first poll and succeeds on the second. +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 5); all_green "$fx" +runs "$(run "$TAG" in_progress null 2026-09-06T20:31:48Z 2)" > "$fx/package-freebsd.yml.1.json" +if check "5 in progress, then succeeded" "$fx" zero "has succeeded$"; then + calls=$(cat "$WORK/calls/package-freebsd.yml" 2>/dev/null || echo 0) + [ "$calls" -ge 2 ] || + fail "5 in progress, then succeeded: gate queried package-freebsd.yml $calls time(s), expected at least 2" +fi + +# 6: one tag run stays in progress for the whole budget. +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 6); all_green "$fx" +runs "$(run "$TAG" in_progress null 2026-09-06T20:31:48Z 2)" > "$fx/package-freebsd.yml.json" +check "6 in progress for the whole budget" "$fx" nonzero "package-freebsd.yml: in_progress " || true + +# 7: no package workflows discovered. +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 7); all_green "$fx" +check "7 empty workflow population" "$fx" nonzero "No package-\*\.yml workflows found" \ + "$WORK/empty-workflows" || true + +# 8a-8d: the tag was re-pushed at the same commit, so two tag runs exist. The +# newest decides. The API lists newest first; 8b and 8c list oldest first so +# that "take the first match" and "take the newest" disagree. +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 8a); all_green "$fx" +runs "$(run "$TAG" completed success 2026-09-06T21:00:00Z 3)" \ + "$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" > "$fx/package-linux.yml.json" +check "8a older failure, newer success, newest first" "$fx" zero "has succeeded$" || true + +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 8b); all_green "$fx" +runs "$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" \ + "$(run "$TAG" completed failure 2026-09-06T21:00:00Z 3)" > "$fx/package-linux.yml.json" +check "8b older success, newer failure, oldest first" "$fx" nonzero \ + "package-linux.yml: run for .* concluded 'failure'" || true + +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 8c); all_green "$fx" +runs "$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" \ + "$(run "$TAG" completed success 2026-09-06T21:00:00Z 3)" > "$fx/package-linux.yml.json" +check "8c older failure, newer success, oldest first" "$fx" zero "has succeeded$" || true + +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 8d); all_green "$fx" +runs "$(run "$TAG" completed failure 2026-09-06T21:00:00Z 3)" \ + "$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" > "$fx/package-linux.yml.json" +check "8d older success, newer failure, newest first" "$fx" nonzero \ + "package-linux.yml: run for .* concluded 'failure'" || true + +# 9: every API call fails. +CASES_DEFINED=$((CASES_DEFINED + 1)) +fx=$(fixture 9); all_green "$fx"; : > "$fx/gh-fails" +check "9 gh fails on every call" "$fx" nonzero "package-linux.yml: API error" || true + +# 10: discovery against the repository's real workflow directory. +CASES_DEFINED=$((CASES_DEFINED + 1)) +CASES_RAN=$((CASES_RAN + 1)) +rc=0 +found=$(DISCOVER_ONLY=1 WORKFLOW_DIR="$REPO_ROOT/.github/workflows" \ + GITHUB_REPOSITORY=example/fips TAG="$TAG" SHA="$SHA" bash "$GATE" 2>&1) || rc=$? +if [ "$rc" -eq 0 ] && [ -n "$found" ] && + printf '%s\n' "$found" | grep -qx 'package-linux.yml'; then + echo "PASS 10 real workflow discovery: $(printf '%s\n' "$found" | tr '\n' ' ')" +else + echo "FAIL 10 real workflow discovery: exit $rc, found: $found" + FAILED=$((FAILED + 1)) +fi + +# ----------------------------------------------------------------------------- + +echo "cases defined: $CASES_DEFINED, ran: $CASES_RAN, failed: $FAILED" +if [ "$CASES_RAN" -ne "$CASES_DEFINED" ]; then + echo "FAIL: not every defined case ran" + exit 1 +fi +[ "$FAILED" -eq 0 ] diff --git a/testing/interop/README.md b/testing/interop/README.md index c87cb319..a481bd0b 100644 --- a/testing/interop/README.md +++ b/testing/interop/README.md @@ -177,12 +177,17 @@ FIPS_INTEROP_NETEM="delay 10ms 5ms 25% loss 2%" \ wants netem) but still runs a clean baseline loop. Each rep invokes `interop-test.sh` with netem set and captures its full -output and exit code; a rep passes iff `interop-test.sh` exits 0. Artifacts -land in `testing/interop/.stress-runs//`: +output and exit code; a rep passes iff `interop-test.sh` exits 0. Every rep +runs with `FIPS_INTEROP_KEEP_UP=1`, because whether it failed is known only +after the driver exits: the loop saves a failed rep's per-node logs and then +tears the mesh down itself, including when the loop is interrupted. +Artifacts land in `testing/interop/.stress-runs//`: - `rep-NN/driver.log` — full driver output for every rep. - `rep-NN/docker-.log` — per-container `docker logs` for **failed** - reps only. + reps only. The expected containers come from the generated manifest, and + the loop prints `harvest: k of n per-node logs written` for each failed + rep. A log that could not be read or came back empty counts as missing. - `summary.txt` — the aggregate report. The aggregate report gives reps run, passed/failed counts and an integer @@ -193,9 +198,10 @@ pair kind (mixed vs same), and a verdict: - **both mixed and same pairs** → loss-induced general instability; - **same-version control pair only** → the build is unstable against itself. -`interop-stress.sh` exits **non-zero only** for the interop-regression -signal. A sub-100% pass rate under loss is expected and is not by itself a -failure, so every other outcome exits 0. +`interop-stress.sh` exits **1** for the interop-regression signal and **3** +when a failed rep's per-node logs are incomplete (the run's diagnostics are +missing; a regression keeps exit 1). A sub-100% pass rate under loss is +expected and is not by itself a failure, so every other outcome exits 0. ### Options @@ -203,7 +209,8 @@ failure, so every other outcome exits 0. | ---------------------------- | ------------------------------------------------- | | `FIPS_INTEROP_NETEM` | tc-netem string applied to each container's eth0, e.g. `"delay 10ms 5ms 25% loss 1%"`. Passed through `interop-stress.sh` to `interop-test.sh`. | | `REKEY_AFTER_SECS` | Rekey interval for generated configs (default 35).| -| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. | +| `CONTROL_MAX_ATTEMPTS` | Control windows Phase 1b measures before Phase 5b abstains (default 3). | +| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. The stress loop sets it for every rep and tears down itself. | | `FIPS_INTEROP_KEEP_WORKTREES`| `1` = keep `build-images.sh` worktrees (debug). | | `FIPS_INTEROP_RUNS_DIR` | Root for the three scratch dirs — see [Scratch directory location](#scratch-directory-location). | @@ -249,6 +256,16 @@ The driver runs seven phases: | 5 | All pairs still ping after the second rekey. | | 6 | Per-node / per-pair interop log analysis. | +When data-plane streams are on (`--topology`, or `FIPS_INTEROP_STREAMS`), +Phase 1b measures stream loss over a quiet control window and Phase 5b +compares the loss across the rekey window against it. A control window +counts only if no FMP rekey cutover happened during it and the cutover +count could be read on every node. Otherwise Phase 1b waits for the +cutovers to settle and re-measures, up to `CONTROL_MAX_ATTEMPTS` times (default +3). If no attempt is clean, Phase 5b prints `ABSTAIN` for every stream and +returns no verdict, and the summary says so. The stress loop reports in how +many reps Phase 5b abstained and in how many it re-measured. + Phase 6 is the interop-specific part. It reports: - **Global health** — panics, `ERROR` lines, `unknown FMP version` drops, diff --git a/testing/interop/interop-stress.sh b/testing/interop/interop-stress.sh index 6cf98952..933da670 100755 --- a/testing/interop/interop-stress.sh +++ b/testing/interop/interop-stress.sh @@ -19,8 +19,13 @@ # -> the version under test is unstable even against itself. # # A sub-100% pass rate under loss is EXPECTED and is not, by itself, a -# failure. This script exits non-zero ONLY for the interop-regression -# signal (mixed-only failures). +# failure. This script exits 1 for the interop-regression signal +# (mixed-only failures) and 3 when a failed rep's per-node logs could not +# all be saved, so its diagnostics are incomplete. +# +# Every rep runs with FIPS_INTEROP_KEEP_UP=1, because whether a rep failed +# is known only after the driver exits. The loop saves a failed rep's +# per-node logs and then tears the mesh down itself. # # Reps run SERIALLY — interop-test.sh uses fixed container names and a # fixed Docker network, so two reps must never overlap. @@ -52,7 +57,10 @@ # # Artifacts (per invocation): /.stress-runs// # rep-NN/driver.log full interop-test.sh output for that rep. -# rep-NN/docker-.log per-container `docker logs` (FAILED reps only). +# rep-NN/docker-.log +# per-container `docker logs` (FAILED reps only). +# A failed rep with fewer of these than nodes +# makes the run exit 3. # summary.txt the final aggregate report. set -uo pipefail @@ -80,6 +88,65 @@ else fi RUNS_BASE="$INTEROP_RUNS_BASE/.stress-runs" +# The driver's generated mesh (same paths as interop-test.sh), read for the +# expected container set and used to tear a kept-up rep down. +GEN_DIR="$INTEROP_RUNS_BASE/generated-configs" +COMPOSE_FILE="$GEN_DIR/docker-compose.generated.yml" +NODES_ENV="$GEN_DIR/nodes.env" + +# Save every node's `docker logs` into a failed rep's directory. +# +# The expected containers come from the generated manifest rather than +# from `docker ps`: an empty `docker ps` result is how a harvest that saved +# nothing used to pass without a word. A container counts only when +# `docker logs` succeeded and wrote a non-empty file. Returns 0 only when +# every expected container was saved and there was at least one. +harvest_rep() { + local rep_dir="$1" containers tok ctr n=0 k=0 + local missing=() + # A subshell, so the manifest's variables stay out of the loop. + # shellcheck disable=SC1090 + containers="$( [ -f "$NODES_ENV" ] && . "$NODES_ENV" \ + && printf '%s' "${INTEROP_NODE_CONTAINERS:-}" )" || containers="" + for tok in $containers; do + ctr="${tok#*:}" + n=$((n + 1)) + if docker logs "$ctr" >"$rep_dir/docker-$ctr.log" 2>&1 \ + && [ -s "$rep_dir/docker-$ctr.log" ]; then + k=$((k + 1)) + else + missing+=("$ctr") + fi + done + echo " harvest: $k of $n per-node logs written" + if [ "$n" -eq 0 ]; then + echo " HARVEST FAILED: no manifest / empty container list ($NODES_ENV)" + return 1 + fi + if [ "$k" -lt "$n" ]; then + echo " HARVEST FAILED: $k of $n; missing: ${missing[*]}" + return 1 + fi + return 0 +} + +# Tear down a rep's kept-up mesh. A failure is loud but not fatal: the +# next rep's Phase 0 also brings the mesh down before starting. +teardown_rep() { + local rc left + docker compose -f "$COMPOSE_FILE" down --volumes --remove-orphans \ + >/dev/null 2>&1 + rc=$? + MESH_UP=0 + if [ "$rc" -ne 0 ]; then + echo " WARN: teardown failed ($rc)" + fi + left="$(docker ps -a --filter 'name=fips-interop-' --format '{{.Names}}' 2>/dev/null)" + if [ -n "$left" ]; then + echo " WARN: containers remain after teardown: $(echo "$left" | tr '\n' ' ')" + fi + return "$rc" +} # ── Args ───────────────────────────────────────────────────────────── @@ -171,6 +238,13 @@ RUN_TS="$(date -u +%Y-%m-%dT%H-%M-%SZ)" RUN_DIR="$RUNS_BASE/$RUN_TS" mkdir -p "$RUN_DIR" +# A rep runs kept up, so an interrupted or aborted loop must still take +# its mesh down. The INT and TERM traps exit, which runs the EXIT trap. +MESH_UP=0 +trap '[ "$MESH_UP" -eq 1 ] && teardown_rep' EXIT +trap 'echo ""; echo "Interrupted"; exit 130' INT +trap 'echo ""; echo "Terminated"; exit 143' TERM + echo "==============================================================" echo " FIPS Interop Netem Stress Loop" echo "==============================================================" @@ -186,6 +260,14 @@ echo "" PASS_COUNT=0 FAIL_COUNT=0 FAILED_REPS=() +# Failed reps whose per-node logs were not all saved. +HARVEST_FAILS=0 +HARVEST_FAILED_REPS=() +# Phase 5b outcomes over the reps that measured a control window (Phase +# 1b ran): reps that abstained, and reps that re-measured at least once. +STREAM_REPS=0 +ABSTAIN_REPS=0 +REMEASURE_REPS=0 # Per-kind connectivity-failure tallies, summed across all failed reps. MIXED_FAILS=0 SAME_FAILS=0 @@ -199,11 +281,27 @@ for ((rep = 1; rep <= REPS; rep++)); do echo "── $rep_id / $REPS ──────────────────────────────────────────" # Run the driver, capturing full output and exit code. Netem is - # passed through the environment; interop-test.sh applies it. - FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \ + # passed through the environment; interop-test.sh applies it. The + # mesh is kept up so a failed rep can be harvested below. + MESH_UP=1 + FIPS_INTEROP_KEEP_UP=1 FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \ bash "$DRIVER" "${DRIVER_ARGS[@]}" >"$driver_log" 2>&1 rc=$? + # Tallied for every rep whatever its verdict: abstaining is the absence + # of a verdict rather than a failure, so without this count a run whose + # Phase 5b always abstains would pass unnoticed. + if grep -q '^Phase 1b: ' "$driver_log"; then + STREAM_REPS=$((STREAM_REPS + 1)) + if grep -q '^ ABSTAIN Data-plane continuity' "$driver_log"; then + ABSTAIN_REPS=$((ABSTAIN_REPS + 1)) + echo " Phase 5b abstained (no clean control window)" + fi + if grep -q 'control window attempt .*re-measuring' "$driver_log"; then + REMEASURE_REPS=$((REMEASURE_REPS + 1)) + fi + fi + if [ "$rc" -eq 0 ]; then PASS_COUNT=$((PASS_COUNT + 1)) echo " PASS (exit 0)" @@ -212,14 +310,11 @@ for ((rep = 1; rep <= REPS; rep++)); do FAILED_REPS+=("$rep_id") echo " FAIL (exit $rc)" - # Preserve each failed container's full docker logs. The driver - # uses fixed container names fips-interop-; harvest every - # container matching that prefix that still exists. - while read -r ctr; do - [ -n "$ctr" ] || continue - docker logs "$ctr" >"$rep_dir/docker-${ctr}.log" 2>&1 || true - done < <(docker ps -a --filter 'name=fips-interop-' \ - --format '{{.Names}}' 2>/dev/null) + # Preserve each node's full docker logs before the teardown below. + if ! harvest_rep "$rep_dir"; then + HARVEST_FAILS=$((HARVEST_FAILS + 1)) + HARVEST_FAILED_REPS+=("$rep_id") + fi # Tally connectivity failures by pair kind, reusing the # pair-attributed lines interop-test.sh prints. Each line is @@ -230,6 +325,8 @@ for ((rep = 1; rep <= REPS; rep++)); do SAME_FAILS=$((SAME_FAILS + s)) echo " connectivity-failure lines: mixed=$m same=$s" fi + + teardown_rep done echo "" @@ -285,6 +382,10 @@ else VERDICT="NO connectivity-pair failures recorded, but $FAIL_COUNT rep(s) still" VERDICT2="failed — on non-connectivity signatures (global-health log patterns or a missing rekey). Check the per-rep driver logs." fi +# A regression keeps precedence; otherwise missing diagnostics fail the run. +if [ "$EXIT_CODE" -eq 0 ] && [ "$HARVEST_FAILS" -gt 0 ]; then + EXIT_CODE=3 +fi { echo "==============================================================" @@ -302,6 +403,12 @@ fi if [ "${#FAILED_REPS[@]}" -gt 0 ]; then echo "Failed reps: ${FAILED_REPS[*]}" fi + if [ "$HARVEST_FAILS" -gt 0 ]; then + echo "Harvest : $HARVEST_FAILS failed rep(s) with incomplete per-node logs: ${HARVEST_FAILED_REPS[*]}" + fi + if [ "$STREAM_REPS" -gt 0 ]; then + echo "Phase 5b : abstained in $ABSTAIN_REPS of $STREAM_REPS reps; re-measured in $REMEASURE_REPS of $STREAM_REPS reps" + fi echo "" echo "-- Connectivity-failure attribution (summed over failed reps) --" echo " mixed-version : $MIXED_FAILS failure(s) over $MIXED_PAIRS mixed pairs x $REPS reps" @@ -315,6 +422,9 @@ fi if [ "$EXIT_CODE" -eq 0 ]; then echo "Exit 0: no interop-regression signal (a sub-100% rate under loss" echo " is expected and is not by itself a failure)." + elif [ "$EXIT_CODE" -eq 3 ]; then + echo "Exit 3: per-node logs missing for a failed rep; the run's" + echo " diagnostics are incomplete." else echo "Exit 1: interop-regression signal present." fi diff --git a/testing/interop/interop-test.sh b/testing/interop/interop-test.sh index 3d8e41dd..913518fe 100755 --- a/testing/interop/interop-test.sh +++ b/testing/interop/interop-test.sh @@ -53,6 +53,9 @@ # by --topology; empty = streams off. # STREAM_LOSS_MARGIN_PCT rekey-vs-control loss margin (default 5). # CONTROL_STREAM_SECS quiet control-window length (default 12). +# CONTROL_MAX_ATTEMPTS control windows measured before Phase 5b +# abstains because every one saw a rekey +# cutover or could not be checked (default 3). # MESH_SIZE_WARMUP Phase 7 bloom warmup, from mesh start, before # any estimate counts (default 300). # MESH_SIZE_SETTLE Phase 7 unbroken in-band window a node must @@ -190,6 +193,14 @@ LOG_POLL_INTERVAL=2 # as on mixed ones. STREAM_RATE_HZ=20 CONTROL_STREAM_SECS="${CONTROL_STREAM_SECS:-12}" # quiet pre-rekey window +# A control window that saw a rekey cutover, or whose cutover count could +# not be read, is no baseline. Phase 1b re-measures it, after waiting for +# the cutover count to hold still for CONTROL_QUIET_SECS (at most +# CONTROL_QUIET_MAX) so the retry does not land in the same cluster of +# cutovers, and Phase 5b abstains when no attempt comes out clean. +CONTROL_MAX_ATTEMPTS="${CONTROL_MAX_ATTEMPTS:-3}" +CONTROL_QUIET_SECS=5 +CONTROL_QUIET_MAX=30 # 5% is 12 packets of the 240-packet control window and ~155 of the # ~3100-packet rekey window. On a path losing up to 6% (the range the # v0.4.2 netem runs baselined at, under `loss 2%` over several hops) the @@ -526,6 +537,37 @@ count_log_pattern() { return 0 } +# Count completed FMP initiator rekey cutovers across all node logs. +# +# The pattern is a literal argument so the log-string guard can check it +# against the daemon source. Prints the count, or `unreadable:` with +# status 1 when a node's logs cannot be read. +fmp_cutover_count() { + count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped' + return $? +} + +# Wait until the FMP cutover count has held unchanged for +# CONTROL_QUIET_SECS, giving up after CONTROL_QUIET_MAX. An unreadable +# count is treated as a change, so it never counts as quiet. +wait_cutover_quiet() { + local start=$SECONDS still=$SECONDS last cur + last="$(fmp_cutover_count)" || last="unreadable" + while (( SECONDS - still < CONTROL_QUIET_SECS )); do + if (( SECONDS - start >= CONTROL_QUIET_MAX )); then + echo " cutover count still moving after ${CONTROL_QUIET_MAX}s; measuring anyway" + return 1 + fi + sleep 1 + cur="$(fmp_cutover_count)" || cur="unreadable" + if [ "$cur" != "$last" ] || [ "$cur" = "unreadable" ]; then + last="$cur" + still=$SECONDS + fi + done + return 0 +} + # Per-node count of a pattern. count_node_pattern() { local node="$1" pattern="$2" @@ -765,25 +807,52 @@ echo "" # ── Phase 1b: data-plane control window (quiet, pre-rekey) ─────────── # # Measure stream loss over a window with NO rekey cutover, as the control -# baseline for the differential. Validated cutover-free by confirming the -# FMP cutover count did not advance during the window. Then launch the +# baseline for the differential. A window counts only when the FMP cutover +# count was read before and after it and did not advance. A window that +# saw a cutover, or could not be checked, is re-measured up to +# CONTROL_MAX_ATTEMPTS times; if none is clean, Phase 5b abstains rather +# than computing a verdict from a contaminated baseline. Then launch the # rekey-window streams, which run in the background across Phases 2-5 and # are collected/asserted in Phase 5b. -control_contaminated=0 +control_abstain=0 if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then - echo "Phase 1b: Data-plane control stream (${CONTROL_STREAM_SECS}s quiet window)" - pre_cut="$(count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped')" - launch_streams CONTROL "$CONTROL_STREAM_SECS" - collect_streams - post_cut="$(count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped')" - if [ "$post_cut" -ne "$pre_cut" ]; then - control_contaminated=1 - echo " WARN a rekey cutover occurred during the control window — control loss may be contaminated" + echo "Phase 1b: Data-plane control stream (${CONTROL_STREAM_SECS}s quiet window, up to ${CONTROL_MAX_ATTEMPTS} attempts)" + control_ok=0 + for ((attempt = 1; attempt <= CONTROL_MAX_ATTEMPTS; attempt++)); do + pre_rc=0 + post_rc=0 + pre_cut="$(fmp_cutover_count)" || pre_rc=$? + launch_streams CONTROL "$CONTROL_STREAM_SECS" + collect_streams + post_cut="$(fmp_cutover_count)" || post_rc=$? + if [ "$pre_rc" -ne 0 ] || ! [[ "$pre_cut" =~ ^[0-9]+$ ]]; then + why="cutover count unreadable ($pre_cut)" + elif [ "$post_rc" -ne 0 ] || ! [[ "$post_cut" =~ ^[0-9]+$ ]]; then + why="cutover count unreadable ($post_cut)" + elif [ "$post_cut" -ne "$pre_cut" ]; then + why="a rekey cutover occurred (+$((post_cut - pre_cut)))" + else + control_ok=1 + echo " control window accepted on attempt $attempt/$CONTROL_MAX_ATTEMPTS" + break + fi + if [ "$attempt" -lt "$CONTROL_MAX_ATTEMPTS" ]; then + echo " control window attempt $attempt/$CONTROL_MAX_ATTEMPTS: $why; re-measuring" + wait_cutover_quiet || true + else + echo " control window attempt $attempt/$CONTROL_MAX_ATTEMPTS: $why" + fi + done + if [ "$control_ok" -eq 0 ]; then + control_abstain=1 + echo " ABSTAIN control window contaminated on all $CONTROL_MAX_ATTEMPTS attempts; Phase 5b will not return a verdict" fi + ctl_note="" + [ "$control_abstain" -eq 1 ] && ctl_note=" (last attempt, contaminated)" for sp in "${STREAM_PAIRS[@]}"; do read -r sf st <<< "$sp" key="$sf->$st" - echo " control $key: tx=${STREAM_TX[CONTROL:$key]:-0} rx=${STREAM_RX[CONTROL:$key]:-0} loss=$(_loss_pct "${STREAM_TX[CONTROL:$key]:-0}" "${STREAM_RX[CONTROL:$key]:-0}")%" + echo " control $key: tx=${STREAM_TX[CONTROL:$key]:-0} rx=${STREAM_RX[CONTROL:$key]:-0} loss=$(_loss_pct "${STREAM_TX[CONTROL:$key]:-0}" "${STREAM_RX[CONTROL:$key]:-0}")%$ctl_note" done echo " Launching rekey-window streams (${REKEY_STREAM_SECS}s, spanning Phases 2-5)" launch_streams REKEY "$REKEY_STREAM_SECS" @@ -863,6 +932,12 @@ if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then if (k <= c + m) print "PASS"; else print "FAIL" }')" delta="$(awk -v c="$cpct" -v k="$rpct" 'BEGIN{ if(c=="NA"||k=="NA") print "NA"; else printf "%+.1f", k-c }')" + # No clean control window: report the losses for reading, but + # compute no verdict from a contaminated baseline. + if [ "$control_abstain" -eq 1 ]; then + printf ' %-16s %11s %11s %8s %s\n' "$key" "$cpct" "$rpct" "$delta" "ABSTAIN" + continue + fi printf ' %-16s %11s %11s %8s %s\n' "$key" "$cpct" "$rpct" "$delta" "$verdict" if [ "$verdict" = "PASS" ]; then PASSED=$((PASSED + 1)) @@ -871,8 +946,11 @@ if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then INTEROP_FAILURES+=("[stream] $key ($(hop_label "$sf" "$st")): rekey-window loss ${rpct}% vs control ${cpct}% (+${STREAM_LOSS_MARGIN_PCT}% margin) -> $verdict") fi done - [ "${control_contaminated:-0}" -eq 1 ] && echo " NOTE: control window saw a cutover; differential may understate rekey loss." - phase_result "Data-plane continuity across rekey" + if [ "$control_abstain" -eq 1 ]; then + echo " ABSTAIN Data-plane continuity across rekey: no uncontaminated control window in $CONTROL_MAX_ATTEMPTS attempts" + else + phase_result "Data-plane continuity across rekey" + fi echo "" fi @@ -1078,6 +1156,13 @@ echo "==============================================================" # `${#arr[@]}` cannot be combined with `:-`; count into a plain var. INTEROP_FAILURE_COUNT="${#INTEROP_FAILURES[@]}" +# An abstaining Phase 5b adds no check, so say so: otherwise a green run +# reads as having covered data-plane continuity across rekey. +if [ "${control_abstain:-0}" -eq 1 ]; then + echo "" + echo "NOTE: Phase 5b abstained (control window contaminated); its rekey-loss check did not run." +fi + if [ "$TOTAL_FAILED" -eq 0 ] && [ "$INTEROP_FAILURE_COUNT" -eq 0 ]; then echo "" echo "PASS: all versions interoperate cleanly across the mesh (spec '$SPEC_STR')."