Merge master into next: AUR release package dbus dependency and publish gate, interop stress log retention and rekey-contaminated control windows

This commit is contained in:
Johnathan Corgan
2026-09-19 07:57:19 +00:00
8 changed files with 650 additions and 36 deletions
+37 -1
View File
@@ -39,10 +39,16 @@ jobs:
- name: Install build and lint tooling
run: |
set -euo pipefail
pacman -Sy --noconfirm --needed base-devel namcap git curl
pacman -Sy --noconfirm --needed base-devel namcap git curl jq
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6
- name: Test the publish gate
# Fixture tests for the script the publish job below waits with. They
# run here, on every trigger, so a change to the gate is exercised
# before a release tag depends on it.
run: bash packaging/aur/test-await-package-runs.sh
- name: Resolve package version
id: ver
env:
@@ -136,11 +142,21 @@ jobs:
# or a manual dispatch (packaging-only republish with explicit tag + pkgrel).
# Branch pushes and pull requests build+lint above but never reach this job.
# Gated on aur-build so a package that fails to build/lint is never published.
# It also waits for every package-*.yml run on the tag to succeed before it
# pushes: the AUR must not point at a tag whose release assets are still
# uploading or failed, and once it does, withdrawing the tag breaks the AUR
# package (its b2sum pins the tag's source archive).
# ───────────────────────────────────────────────────────────────────────────
aur-publish-fips:
name: Publish fips to AUR
needs: aur-build
runs-on: ubuntu-latest
# Above the gate's own 60-minute budget, so the gate reports a timeout
# rather than the runner killing it.
timeout-minutes: 90
permissions:
contents: read
actions: read
if: >-
github.event_name == 'workflow_dispatch'
|| (github.event_name == 'push'
@@ -180,6 +196,26 @@ jobs:
with:
ref: ${{ steps.tag.outputs.tag }}
# The gate script comes from this workflow's own revision, not the tag:
# a dispatch republishing a tag cut before the gate existed would not
# find it in the tag's tree. The workflows it waits on still come from
# the tag's tree, which is what the tag push triggered.
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6
with:
path: gate-src
sparse-checkout: packaging/aur
- name: Wait for the tag's package workflows
env:
GH_TOKEN: ${{ github.token }}
TAG: ${{ steps.tag.outputs.tag }}
run: |
set -euo pipefail
SHA=$(git rev-parse 'HEAD^{commit}')
echo "Tag $TAG is at $SHA"
SHA="$SHA" WORKFLOW_DIR=.github/workflows \
bash gate-src/packaging/aur/await-package-runs.sh
- name: Patch PKGBUILD with pkgver, pkgrel, conflicts, and b2sums
env:
TAG: ${{ steps.tag.outputs.tag }}
+1 -1
View File
@@ -6,7 +6,7 @@ pkgdesc="Distributed, decentralized network routing protocol for mesh nodes"
url="https://github.com/jmcorgan/fips"
license=('MIT')
arch=('x86_64')
depends=('gcc-libs' 'glibc')
depends=('dbus' 'gcc-libs' 'glibc')
makedepends=('cargo' 'clang')
optdepends=('systemd-resolved: .fips DNS resolution')
conflicts=('fips-git' 'fips-git-debug')
+2
View File
@@ -20,6 +20,8 @@ This directory contains Arch Linux packaging files for two AUR packages:
| `fips-dns.service` | Symlink to `../debian/fips-dns.service` |
| `build-aur.sh` | Local `fips-git` build plus namcap validation (run by `make aur`) |
| `patch-pkgbuild.sh` | Rewrites `pkgver`, `pkgrel`, `conflicts`, `options`, and `b2sums` in the PKGBUILD at publish time |
| `await-package-runs.sh` | Holds the AUR publish until every `package-*.yml` run for the release tag has succeeded |
| `test-await-package-runs.sh` | Fixture tests for `await-package-runs.sh`, run by the `aur-build` job |
Both PKGBUILDs reference files from `packaging/debian/` (service files) and
`packaging/common/` (config files) at build time. These are pulled from the
+131
View File
@@ -0,0 +1,131 @@
#!/usr/bin/env bash
# Wait until every package workflow run for a release tag has succeeded.
#
# The AUR publish job runs this before it pushes a new pkgver, so the AUR never
# points at a tag whose release assets are still uploading or failed to build.
# Each package-*.yml workflow uploads its assets from a `release` job inside
# the same run, so a successful tag run means that workflow's assets are up.
#
# The population is discovered from the checked-out tree rather than listed
# here: the package-*.yml files at the tag are the workflows the tag push
# triggered. Finding none is a failure, never a pass.
#
# A tag push and the branch push of the same commit each start a run with the
# same head_sha; only the tag run carries the tag name in head_branch, so runs
# are matched on both. If the tag was re-pushed at the same commit, the newest
# matching run decides.
#
# Required environment variables:
# GITHUB_REPOSITORY - owner/repo
# TAG - release tag (e.g. v0.5.1)
# SHA - the commit the tag points at
# GH_TOKEN - token with actions:read (read by gh; not checked here)
# Optional:
# WORKFLOW_DIR - where to discover package-*.yml (default .github/workflows)
# AWAIT_POLLS - number of polls before giving up (default 60)
# AWAIT_INTERVAL - seconds between polls (default 60)
# DISCOVER_ONLY - if set to 1, print the discovered workflows and exit
#
# Exit status: 0 when every discovered workflow has a successful tag run;
# 1 at once when a tag run concludes anything but success; 1 when the poll
# budget runs out with any workflow unobserved, not finished, or unreadable.
set -euo pipefail
: "${GITHUB_REPOSITORY:?GITHUB_REPOSITORY must be set}"
: "${TAG:?TAG must be set}"
: "${SHA:?SHA must be set}"
WORKFLOW_DIR="${WORKFLOW_DIR:-.github/workflows}"
AWAIT_POLLS="${AWAIT_POLLS:-60}"
AWAIT_INTERVAL="${AWAIT_INTERVAL:-60}"
# How to recover once the cause of a failure is fixed. A workflow_dispatch run
# of a package workflow is not a push run of the tag and never satisfies this
# gate; re-running the failed jobs of the tag's own run keeps it one.
RECOVERY="If a run failed: use \"Re-run failed jobs\" on the package workflow's run for $TAG \
(a new workflow_dispatch run does not count), then re-run the AUR Publish job for $TAG."
# Print the basenames of the package workflows found in WORKFLOW_DIR.
discover() {
local f
for f in "$WORKFLOW_DIR"/package-*.yml; do
[ -e "$f" ] && basename "$f"
done
return 0
}
# Print "<status> <conclusion> <html_url>" for the newest push run of TAG at
# SHA for one workflow, or "none" if there is no such run yet. On an API or
# parse failure, print the error text and return nonzero.
latest() {
local wf=$1 body
if ! body=$(gh api "repos/$GITHUB_REPOSITORY/actions/workflows/$wf/runs?head_sha=$SHA&event=push&per_page=100" 2>"$ERRS"); then
tr '\n' ' ' < "$ERRS"
return 1
fi
printf '%s\n' "$body" | jq -er --arg tag "$TAG" --arg sha "$SHA" '
[.workflow_runs[]
| select(.head_branch == $tag and .head_sha == $sha and .event == "push")]
| sort_by(.created_at)
| if length == 0 then "none"
else last | "\(.status) \(.conclusion // "none") \(.html_url)" end' 2>&1
}
mapfile -t pending < <(discover)
if [ "${#pending[@]}" -eq 0 ]; then
echo "No package-*.yml workflows found in $WORKFLOW_DIR; refusing to treat an empty set as done" >&2
exit 1
fi
if [ "${DISCOVER_ONLY:-}" = 1 ]; then
printf '%s\n' "${pending[@]}"
exit 0
fi
ERRS=$(mktemp)
trap 'rm -f "$ERRS"' EXIT
echo "Waiting for $TAG ($SHA) runs of: ${pending[*]}"
declare -A seen
for ((poll = 1; poll <= AWAIT_POLLS; poll++)); do
still=()
for wf in "${pending[@]}"; do
if ! state=$(latest "$wf"); then
seen[$wf]="API error: $state"
still+=("$wf")
continue
fi
seen[$wf]=$state
read -r status conclusion url <<< "$state"
case "$status" in
none)
# The tag run has not been created yet.
still+=("$wf") ;;
completed)
if [ "$conclusion" = success ]; then
echo "$wf: succeeded ($url)"
else
echo "$wf: run for $TAG concluded '$conclusion' ($url); not publishing to the AUR" >&2
echo "$RECOVERY" >&2
exit 1
fi ;;
*)
# queued, in_progress, waiting, requested or pending.
still+=("$wf") ;;
esac
done
pending=("${still[@]}")
if [ "${#pending[@]}" -eq 0 ]; then
echo "Every package workflow run for $TAG has succeeded"
exit 0
fi
echo "Poll $poll/$AWAIT_POLLS: waiting on ${pending[*]}"
if [ "$poll" -lt "$AWAIT_POLLS" ]; then sleep "$AWAIT_INTERVAL"; fi
done
echo "Gave up after $AWAIT_POLLS polls; not publishing to the AUR. Still unfinished:" >&2
for wf in "${pending[@]}"; do
echo " $wf: ${seen[$wf]}" >&2
done
echo "If a run is still going, re-run the AUR Publish job for $TAG once it has succeeded." >&2
echo "$RECOVERY" >&2
exit 1
+233
View File
@@ -0,0 +1,233 @@
#!/usr/bin/env bash
# Fixture tests for await-package-runs.sh, the gate that holds the AUR publish
# until every package workflow run for the release tag has succeeded.
#
# Each case writes GitHub API responses into a fixture directory and puts a
# stub `gh` first on PATH. The stub serves <workflow>.<call>.json for the Nth
# call about a workflow, falls back to <workflow>.json, and exits 1 when the
# fixture holds a file named `gh-fails`. It counts calls per workflow so a
# case can assert that the gate polled again rather than stopping early.
#
# Needs bash and jq. Exits nonzero if any case fails or if fewer cases ran
# than are defined.
#
# Usage: bash packaging/aur/test-await-package-runs.sh
set -euo pipefail
HERE=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
GATE="${GATE:-$HERE/await-package-runs.sh}"
REPO_ROOT=$(cd "$HERE/../.." && pwd)
TAG=v9.9.9
SHA=0123456789abcdef0123456789abcdef01234567
WORKFLOWS=(package-freebsd.yml package-linux.yml package-macos.yml
package-openwrt.yml package-windows.yml)
WORK=$(mktemp -d)
trap 'rm -rf "$WORK"' EXIT
# The stub gh. It answers `gh api <url>` only, which is all the gate uses.
mkdir -p "$WORK/bin"
cat > "$WORK/bin/gh" <<'STUB'
#!/usr/bin/env bash
set -euo pipefail
[ "${1:-}" = "api" ] || { echo "stub gh: unexpected args: $*" >&2; exit 2; }
[ -e "$STUB_FIXTURE/gh-fails" ] && { echo "stub gh: simulated API error" >&2; exit 1; }
wf=$(printf '%s\n' "$2" | sed -E 's|.*/actions/workflows/([^/]+)/runs.*|\1|')
count_file="$STUB_CALLS/$wf"
n=$(( $(cat "$count_file" 2>/dev/null || echo 0) + 1 ))
echo "$n" > "$count_file"
if [ -f "$STUB_FIXTURE/$wf.$n.json" ]; then
cat "$STUB_FIXTURE/$wf.$n.json"
elif [ -f "$STUB_FIXTURE/$wf.json" ]; then
cat "$STUB_FIXTURE/$wf.json"
else
echo '{"total_count":0,"workflow_runs":[]}'
fi
STUB
chmod +x "$WORK/bin/gh"
# A workflow directory holding the five package workflows as empty files: the
# gate discovers the population by file name only.
mkdir -p "$WORK/workflows" "$WORK/empty-workflows"
for wf in "${WORKFLOWS[@]}"; do : > "$WORK/workflows/$wf"; done
# Print one run object in the shape of the v0.5.1 API response.
# Args: head_branch status conclusion created_at id
run() {
local conclusion=null
[ "$3" = null ] || conclusion="\"$3\""
printf '{"id":%s,"name":"Package","head_branch":"%s","head_sha":"%s","event":"push","status":"%s","conclusion":%s,"created_at":"%s","updated_at":"%s","html_url":"https://github.com/example/fips/actions/runs/%s"}' \
"$5" "$1" "$SHA" "$2" "$conclusion" "$4" "$4" "$5"
}
# Wrap run objects, given as arguments in the order the API returns them, in
# the list response envelope.
runs() {
local IFS=,
printf '{"total_count":%d,"workflow_runs":[%s]}\n' "$#" "$*"
}
# Write the healthy v0.5.1 shape for every workflow into a fixture: a tag run
# and a branch run of the same commit, newest first, both successful.
all_green() {
local dir=$1 wf
for wf in "${WORKFLOWS[@]}"; do
runs "$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" \
"$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$dir/$wf.json"
done
}
CASES_DEFINED=0
CASES_RAN=0
FAILED=0
# Run the gate against a fixture and check its exit status and, when a
# pattern is given, that its output states the expected reason (so a red
# caused by a crash in the gate does not pass as the intended red).
# Args: name fixture-dir expect(zero|nonzero) [pattern] [workflow-dir]
# Sets LAST_OUT to the gate's combined output.
check() {
local name=$1 fixture=$2 expect=$3 pattern=${4:-} wfdir=${5:-$WORK/workflows} rc=0
CASES_RAN=$((CASES_RAN + 1))
rm -rf "$WORK/calls"; mkdir -p "$WORK/calls"
LAST_OUT=$(PATH="$WORK/bin:$PATH" STUB_FIXTURE="$fixture" STUB_CALLS="$WORK/calls" \
GITHUB_REPOSITORY=example/fips TAG="$TAG" SHA="$SHA" WORKFLOW_DIR="$wfdir" \
AWAIT_POLLS=3 AWAIT_INTERVAL=0 bash "$GATE" 2>&1) || rc=$?
if { [ "$expect" = zero ] && [ "$rc" -eq 0 ]; } ||
{ [ "$expect" = nonzero ] && [ "$rc" -ne 0 ]; }; then
if [ -z "$pattern" ] || printf '%s\n' "$LAST_OUT" | grep -qE -- "$pattern"; then
echo "PASS $name (exit $rc)"
return 0
fi
echo "FAIL $name: exit $rc as expected, but output lacks /$pattern/"
else
echo "FAIL $name: expected $expect exit, got $rc"
fi
printf '%s\n' "$LAST_OUT" | sed 's/^/ /'
FAILED=$((FAILED + 1))
return 1
}
# Record an extra assertion's failure against the case that just ran.
fail() {
echo "FAIL $1"
FAILED=$((FAILED + 1))
}
# Make a fresh fixture directory for a case and print its path.
fixture() {
local dir="$WORK/fx/$1"
mkdir -p "$dir"
echo "$dir"
}
# --- cases -------------------------------------------------------------------
# 1: every package workflow has a successful tag run.
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 1); all_green "$fx"
check "1 all tag runs succeeded" "$fx" zero "has succeeded$" || true
# 2: one tag run failed; the gate must name that workflow.
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 2); all_green "$fx"
runs "$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" \
"$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$fx/package-macos.yml.json"
if check "2 one tag run failed" "$fx" nonzero "concluded 'failure'"; then
printf '%s\n' "$LAST_OUT" | grep -q 'package-macos.yml' ||
fail "2 one tag run failed: output does not name package-macos.yml"
fi
# 3: one tag run was cancelled.
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 3); all_green "$fx"
runs "$(run "$TAG" completed cancelled 2026-09-06T20:31:48Z 2)" > "$fx/package-openwrt.yml.json"
check "3 one tag run cancelled" "$fx" nonzero "package-openwrt.yml: run for .* concluded 'cancelled'" || true
# 4: one workflow has only the branch run of the tag's commit (the v0.5.1
# shape a SHA-only filter would accept).
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 4); all_green "$fx"
runs "$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$fx/package-freebsd.yml.json"
check "4 only a branch run, no tag run" "$fx" nonzero "package-freebsd.yml: none$" || true
# 5: one tag run is in progress on the first poll and succeeds on the second.
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 5); all_green "$fx"
runs "$(run "$TAG" in_progress null 2026-09-06T20:31:48Z 2)" > "$fx/package-freebsd.yml.1.json"
if check "5 in progress, then succeeded" "$fx" zero "has succeeded$"; then
calls=$(cat "$WORK/calls/package-freebsd.yml" 2>/dev/null || echo 0)
[ "$calls" -ge 2 ] ||
fail "5 in progress, then succeeded: gate queried package-freebsd.yml $calls time(s), expected at least 2"
fi
# 6: one tag run stays in progress for the whole budget.
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 6); all_green "$fx"
runs "$(run "$TAG" in_progress null 2026-09-06T20:31:48Z 2)" > "$fx/package-freebsd.yml.json"
check "6 in progress for the whole budget" "$fx" nonzero "package-freebsd.yml: in_progress " || true
# 7: no package workflows discovered.
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 7); all_green "$fx"
check "7 empty workflow population" "$fx" nonzero "No package-\*\.yml workflows found" \
"$WORK/empty-workflows" || true
# 8a-8d: the tag was re-pushed at the same commit, so two tag runs exist. The
# newest decides. The API lists newest first; 8b and 8c list oldest first so
# that "take the first match" and "take the newest" disagree.
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 8a); all_green "$fx"
runs "$(run "$TAG" completed success 2026-09-06T21:00:00Z 3)" \
"$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" > "$fx/package-linux.yml.json"
check "8a older failure, newer success, newest first" "$fx" zero "has succeeded$" || true
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 8b); all_green "$fx"
runs "$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" \
"$(run "$TAG" completed failure 2026-09-06T21:00:00Z 3)" > "$fx/package-linux.yml.json"
check "8b older success, newer failure, oldest first" "$fx" nonzero \
"package-linux.yml: run for .* concluded 'failure'" || true
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 8c); all_green "$fx"
runs "$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" \
"$(run "$TAG" completed success 2026-09-06T21:00:00Z 3)" > "$fx/package-linux.yml.json"
check "8c older failure, newer success, oldest first" "$fx" zero "has succeeded$" || true
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 8d); all_green "$fx"
runs "$(run "$TAG" completed failure 2026-09-06T21:00:00Z 3)" \
"$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" > "$fx/package-linux.yml.json"
check "8d older success, newer failure, newest first" "$fx" nonzero \
"package-linux.yml: run for .* concluded 'failure'" || true
# 9: every API call fails.
CASES_DEFINED=$((CASES_DEFINED + 1))
fx=$(fixture 9); all_green "$fx"; : > "$fx/gh-fails"
check "9 gh fails on every call" "$fx" nonzero "package-linux.yml: API error" || true
# 10: discovery against the repository's real workflow directory.
CASES_DEFINED=$((CASES_DEFINED + 1))
CASES_RAN=$((CASES_RAN + 1))
rc=0
found=$(DISCOVER_ONLY=1 WORKFLOW_DIR="$REPO_ROOT/.github/workflows" \
GITHUB_REPOSITORY=example/fips TAG="$TAG" SHA="$SHA" bash "$GATE" 2>&1) || rc=$?
if [ "$rc" -eq 0 ] && [ -n "$found" ] &&
printf '%s\n' "$found" | grep -qx 'package-linux.yml'; then
echo "PASS 10 real workflow discovery: $(printf '%s\n' "$found" | tr '\n' ' ')"
else
echo "FAIL 10 real workflow discovery: exit $rc, found: $found"
FAILED=$((FAILED + 1))
fi
# -----------------------------------------------------------------------------
echo "cases defined: $CASES_DEFINED, ran: $CASES_RAN, failed: $FAILED"
if [ "$CASES_RAN" -ne "$CASES_DEFINED" ]; then
echo "FAIL: not every defined case ran"
exit 1
fi
[ "$FAILED" -eq 0 ]
+24 -7
View File
@@ -177,12 +177,17 @@ FIPS_INTEROP_NETEM="delay 10ms 5ms 25% loss 2%" \
wants netem) but still runs a clean baseline loop.
Each rep invokes `interop-test.sh` with netem set and captures its full
output and exit code; a rep passes iff `interop-test.sh` exits 0. Artifacts
land in `testing/interop/.stress-runs/<UTC-timestamp>/`:
output and exit code; a rep passes iff `interop-test.sh` exits 0. Every rep
runs with `FIPS_INTEROP_KEEP_UP=1`, because whether it failed is known only
after the driver exits: the loop saves a failed rep's per-node logs and then
tears the mesh down itself, including when the loop is interrupted.
Artifacts land in `testing/interop/.stress-runs/<UTC-timestamp>/`:
- `rep-NN/driver.log` — full driver output for every rep.
- `rep-NN/docker-<container>.log` — per-container `docker logs` for **failed**
reps only.
reps only. The expected containers come from the generated manifest, and
the loop prints `harvest: k of n per-node logs written` for each failed
rep. A log that could not be read or came back empty counts as missing.
- `summary.txt` — the aggregate report.
The aggregate report gives reps run, passed/failed counts and an integer
@@ -193,9 +198,10 @@ pair kind (mixed vs same), and a verdict:
- **both mixed and same pairs** → loss-induced general instability;
- **same-version control pair only** → the build is unstable against itself.
`interop-stress.sh` exits **non-zero only** for the interop-regression
signal. A sub-100% pass rate under loss is expected and is not by itself a
failure, so every other outcome exits 0.
`interop-stress.sh` exits **1** for the interop-regression signal and **3**
when a failed rep's per-node logs are incomplete (the run's diagnostics are
missing; a regression keeps exit 1). A sub-100% pass rate under loss is
expected and is not by itself a failure, so every other outcome exits 0.
### Options
@@ -203,7 +209,8 @@ failure, so every other outcome exits 0.
| ---------------------------- | ------------------------------------------------- |
| `FIPS_INTEROP_NETEM` | tc-netem string applied to each container's eth0, e.g. `"delay 10ms 5ms 25% loss 1%"`. Passed through `interop-stress.sh` to `interop-test.sh`. |
| `REKEY_AFTER_SECS` | Rekey interval for generated configs (default 35).|
| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. |
| `CONTROL_MAX_ATTEMPTS` | Control windows Phase 1b measures before Phase 5b abstains (default 3). |
| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. The stress loop sets it for every rep and tears down itself. |
| `FIPS_INTEROP_KEEP_WORKTREES`| `1` = keep `build-images.sh` worktrees (debug). |
| `FIPS_INTEROP_RUNS_DIR` | Root for the three scratch dirs — see [Scratch directory location](#scratch-directory-location). |
@@ -249,6 +256,16 @@ The driver runs seven phases:
| 5 | All pairs still ping after the second rekey. |
| 6 | Per-node / per-pair interop log analysis. |
When data-plane streams are on (`--topology`, or `FIPS_INTEROP_STREAMS`),
Phase 1b measures stream loss over a quiet control window and Phase 5b
compares the loss across the rekey window against it. A control window
counts only if no FMP rekey cutover happened during it and the cutover
count could be read on every node. Otherwise Phase 1b waits for the
cutovers to settle and re-measures, up to `CONTROL_MAX_ATTEMPTS` times (default
3). If no attempt is clean, Phase 5b prints `ABSTAIN` for every stream and
returns no verdict, and the summary says so. The stress loop reports in how
many reps Phase 5b abstained and in how many it re-measured.
Phase 6 is the interop-specific part. It reports:
- **Global health** — panics, `ERROR` lines, `unknown FMP version` drops,
+123 -13
View File
@@ -19,8 +19,13 @@
# -> the version under test is unstable even against itself.
#
# A sub-100% pass rate under loss is EXPECTED and is not, by itself, a
# failure. This script exits non-zero ONLY for the interop-regression
# signal (mixed-only failures).
# failure. This script exits 1 for the interop-regression signal
# (mixed-only failures) and 3 when a failed rep's per-node logs could not
# all be saved, so its diagnostics are incomplete.
#
# Every rep runs with FIPS_INTEROP_KEEP_UP=1, because whether a rep failed
# is known only after the driver exits. The loop saves a failed rep's
# per-node logs and then tears the mesh down itself.
#
# Reps run SERIALLY — interop-test.sh uses fixed container names and a
# fixed Docker network, so two reps must never overlap.
@@ -52,7 +57,10 @@
#
# Artifacts (per invocation): <runs-base>/.stress-runs/<UTC-ts>/
# rep-NN/driver.log full interop-test.sh output for that rep.
# rep-NN/docker-<node>.log per-container `docker logs` (FAILED reps only).
# rep-NN/docker-<container>.log
# per-container `docker logs` (FAILED reps only).
# A failed rep with fewer of these than nodes
# makes the run exit 3.
# summary.txt the final aggregate report.
set -uo pipefail
@@ -80,6 +88,65 @@ else
fi
RUNS_BASE="$INTEROP_RUNS_BASE/.stress-runs"
# The driver's generated mesh (same paths as interop-test.sh), read for the
# expected container set and used to tear a kept-up rep down.
GEN_DIR="$INTEROP_RUNS_BASE/generated-configs"
COMPOSE_FILE="$GEN_DIR/docker-compose.generated.yml"
NODES_ENV="$GEN_DIR/nodes.env"
# Save every node's `docker logs` into a failed rep's directory.
#
# The expected containers come from the generated manifest rather than
# from `docker ps`: an empty `docker ps` result is how a harvest that saved
# nothing used to pass without a word. A container counts only when
# `docker logs` succeeded and wrote a non-empty file. Returns 0 only when
# every expected container was saved and there was at least one.
harvest_rep() {
local rep_dir="$1" containers tok ctr n=0 k=0
local missing=()
# A subshell, so the manifest's variables stay out of the loop.
# shellcheck disable=SC1090
containers="$( [ -f "$NODES_ENV" ] && . "$NODES_ENV" \
&& printf '%s' "${INTEROP_NODE_CONTAINERS:-}" )" || containers=""
for tok in $containers; do
ctr="${tok#*:}"
n=$((n + 1))
if docker logs "$ctr" >"$rep_dir/docker-$ctr.log" 2>&1 \
&& [ -s "$rep_dir/docker-$ctr.log" ]; then
k=$((k + 1))
else
missing+=("$ctr")
fi
done
echo " harvest: $k of $n per-node logs written"
if [ "$n" -eq 0 ]; then
echo " HARVEST FAILED: no manifest / empty container list ($NODES_ENV)"
return 1
fi
if [ "$k" -lt "$n" ]; then
echo " HARVEST FAILED: $k of $n; missing: ${missing[*]}"
return 1
fi
return 0
}
# Tear down a rep's kept-up mesh. A failure is loud but not fatal: the
# next rep's Phase 0 also brings the mesh down before starting.
teardown_rep() {
local rc left
docker compose -f "$COMPOSE_FILE" down --volumes --remove-orphans \
>/dev/null 2>&1
rc=$?
MESH_UP=0
if [ "$rc" -ne 0 ]; then
echo " WARN: teardown failed ($rc)"
fi
left="$(docker ps -a --filter 'name=fips-interop-' --format '{{.Names}}' 2>/dev/null)"
if [ -n "$left" ]; then
echo " WARN: containers remain after teardown: $(echo "$left" | tr '\n' ' ')"
fi
return "$rc"
}
# ── Args ─────────────────────────────────────────────────────────────
@@ -171,6 +238,13 @@ RUN_TS="$(date -u +%Y-%m-%dT%H-%M-%SZ)"
RUN_DIR="$RUNS_BASE/$RUN_TS"
mkdir -p "$RUN_DIR"
# A rep runs kept up, so an interrupted or aborted loop must still take
# its mesh down. The INT and TERM traps exit, which runs the EXIT trap.
MESH_UP=0
trap '[ "$MESH_UP" -eq 1 ] && teardown_rep' EXIT
trap 'echo ""; echo "Interrupted"; exit 130' INT
trap 'echo ""; echo "Terminated"; exit 143' TERM
echo "=============================================================="
echo " FIPS Interop Netem Stress Loop"
echo "=============================================================="
@@ -186,6 +260,14 @@ echo ""
PASS_COUNT=0
FAIL_COUNT=0
FAILED_REPS=()
# Failed reps whose per-node logs were not all saved.
HARVEST_FAILS=0
HARVEST_FAILED_REPS=()
# Phase 5b outcomes over the reps that measured a control window (Phase
# 1b ran): reps that abstained, and reps that re-measured at least once.
STREAM_REPS=0
ABSTAIN_REPS=0
REMEASURE_REPS=0
# Per-kind connectivity-failure tallies, summed across all failed reps.
MIXED_FAILS=0
SAME_FAILS=0
@@ -199,11 +281,27 @@ for ((rep = 1; rep <= REPS; rep++)); do
echo "── $rep_id / $REPS ──────────────────────────────────────────"
# Run the driver, capturing full output and exit code. Netem is
# passed through the environment; interop-test.sh applies it.
FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \
# passed through the environment; interop-test.sh applies it. The
# mesh is kept up so a failed rep can be harvested below.
MESH_UP=1
FIPS_INTEROP_KEEP_UP=1 FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \
bash "$DRIVER" "${DRIVER_ARGS[@]}" >"$driver_log" 2>&1
rc=$?
# Tallied for every rep whatever its verdict: abstaining is the absence
# of a verdict rather than a failure, so without this count a run whose
# Phase 5b always abstains would pass unnoticed.
if grep -q '^Phase 1b: ' "$driver_log"; then
STREAM_REPS=$((STREAM_REPS + 1))
if grep -q '^ ABSTAIN Data-plane continuity' "$driver_log"; then
ABSTAIN_REPS=$((ABSTAIN_REPS + 1))
echo " Phase 5b abstained (no clean control window)"
fi
if grep -q 'control window attempt .*re-measuring' "$driver_log"; then
REMEASURE_REPS=$((REMEASURE_REPS + 1))
fi
fi
if [ "$rc" -eq 0 ]; then
PASS_COUNT=$((PASS_COUNT + 1))
echo " PASS (exit 0)"
@@ -212,14 +310,11 @@ for ((rep = 1; rep <= REPS; rep++)); do
FAILED_REPS+=("$rep_id")
echo " FAIL (exit $rc)"
# Preserve each failed container's full docker logs. The driver
# uses fixed container names fips-interop-<nodeid>; harvest every
# container matching that prefix that still exists.
while read -r ctr; do
[ -n "$ctr" ] || continue
docker logs "$ctr" >"$rep_dir/docker-${ctr}.log" 2>&1 || true
done < <(docker ps -a --filter 'name=fips-interop-' \
--format '{{.Names}}' 2>/dev/null)
# Preserve each node's full docker logs before the teardown below.
if ! harvest_rep "$rep_dir"; then
HARVEST_FAILS=$((HARVEST_FAILS + 1))
HARVEST_FAILED_REPS+=("$rep_id")
fi
# Tally connectivity failures by pair kind, reusing the
# pair-attributed lines interop-test.sh prints. Each line is
@@ -230,6 +325,8 @@ for ((rep = 1; rep <= REPS; rep++)); do
SAME_FAILS=$((SAME_FAILS + s))
echo " connectivity-failure lines: mixed=$m same=$s"
fi
teardown_rep
done
echo ""
@@ -285,6 +382,10 @@ else
VERDICT="NO connectivity-pair failures recorded, but $FAIL_COUNT rep(s) still"
VERDICT2="failed — on non-connectivity signatures (global-health log patterns or a missing rekey). Check the per-rep driver logs."
fi
# A regression keeps precedence; otherwise missing diagnostics fail the run.
if [ "$EXIT_CODE" -eq 0 ] && [ "$HARVEST_FAILS" -gt 0 ]; then
EXIT_CODE=3
fi
{
echo "=============================================================="
@@ -302,6 +403,12 @@ fi
if [ "${#FAILED_REPS[@]}" -gt 0 ]; then
echo "Failed reps: ${FAILED_REPS[*]}"
fi
if [ "$HARVEST_FAILS" -gt 0 ]; then
echo "Harvest : $HARVEST_FAILS failed rep(s) with incomplete per-node logs: ${HARVEST_FAILED_REPS[*]}"
fi
if [ "$STREAM_REPS" -gt 0 ]; then
echo "Phase 5b : abstained in $ABSTAIN_REPS of $STREAM_REPS reps; re-measured in $REMEASURE_REPS of $STREAM_REPS reps"
fi
echo ""
echo "-- Connectivity-failure attribution (summed over failed reps) --"
echo " mixed-version : $MIXED_FAILS failure(s) over $MIXED_PAIRS mixed pairs x $REPS reps"
@@ -315,6 +422,9 @@ fi
if [ "$EXIT_CODE" -eq 0 ]; then
echo "Exit 0: no interop-regression signal (a sub-100% rate under loss"
echo " is expected and is not by itself a failure)."
elif [ "$EXIT_CODE" -eq 3 ]; then
echo "Exit 3: per-node logs missing for a failed rep; the run's"
echo " diagnostics are incomplete."
else
echo "Exit 1: interop-regression signal present."
fi
+99 -14
View File
@@ -53,6 +53,9 @@
# by --topology; empty = streams off.
# STREAM_LOSS_MARGIN_PCT rekey-vs-control loss margin (default 5).
# CONTROL_STREAM_SECS quiet control-window length (default 12).
# CONTROL_MAX_ATTEMPTS control windows measured before Phase 5b
# abstains because every one saw a rekey
# cutover or could not be checked (default 3).
# MESH_SIZE_WARMUP Phase 7 bloom warmup, from mesh start, before
# any estimate counts (default 300).
# MESH_SIZE_SETTLE Phase 7 unbroken in-band window a node must
@@ -190,6 +193,14 @@ LOG_POLL_INTERVAL=2
# as on mixed ones.
STREAM_RATE_HZ=20
CONTROL_STREAM_SECS="${CONTROL_STREAM_SECS:-12}" # quiet pre-rekey window
# A control window that saw a rekey cutover, or whose cutover count could
# not be read, is no baseline. Phase 1b re-measures it, after waiting for
# the cutover count to hold still for CONTROL_QUIET_SECS (at most
# CONTROL_QUIET_MAX) so the retry does not land in the same cluster of
# cutovers, and Phase 5b abstains when no attempt comes out clean.
CONTROL_MAX_ATTEMPTS="${CONTROL_MAX_ATTEMPTS:-3}"
CONTROL_QUIET_SECS=5
CONTROL_QUIET_MAX=30
# 5% is 12 packets of the 240-packet control window and ~155 of the
# ~3100-packet rekey window. On a path losing up to 6% (the range the
# v0.4.2 netem runs baselined at, under `loss 2%` over several hops) the
@@ -526,6 +537,37 @@ count_log_pattern() {
return 0
}
# Count completed FMP initiator rekey cutovers across all node logs.
#
# The pattern is a literal argument so the log-string guard can check it
# against the daemon source. Prints the count, or `unreadable:<ctr>` with
# status 1 when a node's logs cannot be read.
fmp_cutover_count() {
count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped'
return $?
}
# Wait until the FMP cutover count has held unchanged for
# CONTROL_QUIET_SECS, giving up after CONTROL_QUIET_MAX. An unreadable
# count is treated as a change, so it never counts as quiet.
wait_cutover_quiet() {
local start=$SECONDS still=$SECONDS last cur
last="$(fmp_cutover_count)" || last="unreadable"
while (( SECONDS - still < CONTROL_QUIET_SECS )); do
if (( SECONDS - start >= CONTROL_QUIET_MAX )); then
echo " cutover count still moving after ${CONTROL_QUIET_MAX}s; measuring anyway"
return 1
fi
sleep 1
cur="$(fmp_cutover_count)" || cur="unreadable"
if [ "$cur" != "$last" ] || [ "$cur" = "unreadable" ]; then
last="$cur"
still=$SECONDS
fi
done
return 0
}
# Per-node count of a pattern.
count_node_pattern() {
local node="$1" pattern="$2"
@@ -765,25 +807,52 @@ echo ""
# ── Phase 1b: data-plane control window (quiet, pre-rekey) ───────────
#
# Measure stream loss over a window with NO rekey cutover, as the control
# baseline for the differential. Validated cutover-free by confirming the
# FMP cutover count did not advance during the window. Then launch the
# baseline for the differential. A window counts only when the FMP cutover
# count was read before and after it and did not advance. A window that
# saw a cutover, or could not be checked, is re-measured up to
# CONTROL_MAX_ATTEMPTS times; if none is clean, Phase 5b abstains rather
# than computing a verdict from a contaminated baseline. Then launch the
# rekey-window streams, which run in the background across Phases 2-5 and
# are collected/asserted in Phase 5b.
control_contaminated=0
control_abstain=0
if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then
echo "Phase 1b: Data-plane control stream (${CONTROL_STREAM_SECS}s quiet window)"
pre_cut="$(count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped')"
launch_streams CONTROL "$CONTROL_STREAM_SECS"
collect_streams
post_cut="$(count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped')"
if [ "$post_cut" -ne "$pre_cut" ]; then
control_contaminated=1
echo " WARN a rekey cutover occurred during the control window — control loss may be contaminated"
echo "Phase 1b: Data-plane control stream (${CONTROL_STREAM_SECS}s quiet window, up to ${CONTROL_MAX_ATTEMPTS} attempts)"
control_ok=0
for ((attempt = 1; attempt <= CONTROL_MAX_ATTEMPTS; attempt++)); do
pre_rc=0
post_rc=0
pre_cut="$(fmp_cutover_count)" || pre_rc=$?
launch_streams CONTROL "$CONTROL_STREAM_SECS"
collect_streams
post_cut="$(fmp_cutover_count)" || post_rc=$?
if [ "$pre_rc" -ne 0 ] || ! [[ "$pre_cut" =~ ^[0-9]+$ ]]; then
why="cutover count unreadable ($pre_cut)"
elif [ "$post_rc" -ne 0 ] || ! [[ "$post_cut" =~ ^[0-9]+$ ]]; then
why="cutover count unreadable ($post_cut)"
elif [ "$post_cut" -ne "$pre_cut" ]; then
why="a rekey cutover occurred (+$((post_cut - pre_cut)))"
else
control_ok=1
echo " control window accepted on attempt $attempt/$CONTROL_MAX_ATTEMPTS"
break
fi
if [ "$attempt" -lt "$CONTROL_MAX_ATTEMPTS" ]; then
echo " control window attempt $attempt/$CONTROL_MAX_ATTEMPTS: $why; re-measuring"
wait_cutover_quiet || true
else
echo " control window attempt $attempt/$CONTROL_MAX_ATTEMPTS: $why"
fi
done
if [ "$control_ok" -eq 0 ]; then
control_abstain=1
echo " ABSTAIN control window contaminated on all $CONTROL_MAX_ATTEMPTS attempts; Phase 5b will not return a verdict"
fi
ctl_note=""
[ "$control_abstain" -eq 1 ] && ctl_note=" (last attempt, contaminated)"
for sp in "${STREAM_PAIRS[@]}"; do
read -r sf st <<< "$sp"
key="$sf->$st"
echo " control $key: tx=${STREAM_TX[CONTROL:$key]:-0} rx=${STREAM_RX[CONTROL:$key]:-0} loss=$(_loss_pct "${STREAM_TX[CONTROL:$key]:-0}" "${STREAM_RX[CONTROL:$key]:-0}")%"
echo " control $key: tx=${STREAM_TX[CONTROL:$key]:-0} rx=${STREAM_RX[CONTROL:$key]:-0} loss=$(_loss_pct "${STREAM_TX[CONTROL:$key]:-0}" "${STREAM_RX[CONTROL:$key]:-0}")%$ctl_note"
done
echo " Launching rekey-window streams (${REKEY_STREAM_SECS}s, spanning Phases 2-5)"
launch_streams REKEY "$REKEY_STREAM_SECS"
@@ -863,6 +932,12 @@ if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then
if (k <= c + m) print "PASS"; else print "FAIL"
}')"
delta="$(awk -v c="$cpct" -v k="$rpct" 'BEGIN{ if(c=="NA"||k=="NA") print "NA"; else printf "%+.1f", k-c }')"
# No clean control window: report the losses for reading, but
# compute no verdict from a contaminated baseline.
if [ "$control_abstain" -eq 1 ]; then
printf ' %-16s %11s %11s %8s %s\n' "$key" "$cpct" "$rpct" "$delta" "ABSTAIN"
continue
fi
printf ' %-16s %11s %11s %8s %s\n' "$key" "$cpct" "$rpct" "$delta" "$verdict"
if [ "$verdict" = "PASS" ]; then
PASSED=$((PASSED + 1))
@@ -871,8 +946,11 @@ if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then
INTEROP_FAILURES+=("[stream] $key ($(hop_label "$sf" "$st")): rekey-window loss ${rpct}% vs control ${cpct}% (+${STREAM_LOSS_MARGIN_PCT}% margin) -> $verdict")
fi
done
[ "${control_contaminated:-0}" -eq 1 ] && echo " NOTE: control window saw a cutover; differential may understate rekey loss."
phase_result "Data-plane continuity across rekey"
if [ "$control_abstain" -eq 1 ]; then
echo " ABSTAIN Data-plane continuity across rekey: no uncontaminated control window in $CONTROL_MAX_ATTEMPTS attempts"
else
phase_result "Data-plane continuity across rekey"
fi
echo ""
fi
@@ -1078,6 +1156,13 @@ echo "=============================================================="
# `${#arr[@]}` cannot be combined with `:-`; count into a plain var.
INTEROP_FAILURE_COUNT="${#INTEROP_FAILURES[@]}"
# An abstaining Phase 5b adds no check, so say so: otherwise a green run
# reads as having covered data-plane continuity across rekey.
if [ "${control_abstain:-0}" -eq 1 ]; then
echo ""
echo "NOTE: Phase 5b abstained (control window contaminated); its rekey-loss check did not run."
fi
if [ "$TOTAL_FAILED" -eq 0 ] && [ "$INTEROP_FAILURE_COUNT" -eq 0 ]; then
echo ""
echo "PASS: all versions interoperate cleanly across the mesh (spec '$SPEC_STR')."