mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-05 19:18:25 +00:00
Merge master into next: AUR release package dbus dependency and publish gate, interop stress log retention and rekey-contaminated control windows
This commit is contained in:
@@ -39,10 +39,16 @@ jobs:
|
||||
- name: Install build and lint tooling
|
||||
run: |
|
||||
set -euo pipefail
|
||||
pacman -Sy --noconfirm --needed base-devel namcap git curl
|
||||
pacman -Sy --noconfirm --needed base-devel namcap git curl jq
|
||||
|
||||
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6
|
||||
|
||||
- name: Test the publish gate
|
||||
# Fixture tests for the script the publish job below waits with. They
|
||||
# run here, on every trigger, so a change to the gate is exercised
|
||||
# before a release tag depends on it.
|
||||
run: bash packaging/aur/test-await-package-runs.sh
|
||||
|
||||
- name: Resolve package version
|
||||
id: ver
|
||||
env:
|
||||
@@ -136,11 +142,21 @@ jobs:
|
||||
# or a manual dispatch (packaging-only republish with explicit tag + pkgrel).
|
||||
# Branch pushes and pull requests build+lint above but never reach this job.
|
||||
# Gated on aur-build so a package that fails to build/lint is never published.
|
||||
# It also waits for every package-*.yml run on the tag to succeed before it
|
||||
# pushes: the AUR must not point at a tag whose release assets are still
|
||||
# uploading or failed, and once it does, withdrawing the tag breaks the AUR
|
||||
# package (its b2sum pins the tag's source archive).
|
||||
# ───────────────────────────────────────────────────────────────────────────
|
||||
aur-publish-fips:
|
||||
name: Publish fips to AUR
|
||||
needs: aur-build
|
||||
runs-on: ubuntu-latest
|
||||
# Above the gate's own 60-minute budget, so the gate reports a timeout
|
||||
# rather than the runner killing it.
|
||||
timeout-minutes: 90
|
||||
permissions:
|
||||
contents: read
|
||||
actions: read
|
||||
if: >-
|
||||
github.event_name == 'workflow_dispatch'
|
||||
|| (github.event_name == 'push'
|
||||
@@ -180,6 +196,26 @@ jobs:
|
||||
with:
|
||||
ref: ${{ steps.tag.outputs.tag }}
|
||||
|
||||
# The gate script comes from this workflow's own revision, not the tag:
|
||||
# a dispatch republishing a tag cut before the gate existed would not
|
||||
# find it in the tag's tree. The workflows it waits on still come from
|
||||
# the tag's tree, which is what the tag push triggered.
|
||||
- uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6
|
||||
with:
|
||||
path: gate-src
|
||||
sparse-checkout: packaging/aur
|
||||
|
||||
- name: Wait for the tag's package workflows
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
TAG: ${{ steps.tag.outputs.tag }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
SHA=$(git rev-parse 'HEAD^{commit}')
|
||||
echo "Tag $TAG is at $SHA"
|
||||
SHA="$SHA" WORKFLOW_DIR=.github/workflows \
|
||||
bash gate-src/packaging/aur/await-package-runs.sh
|
||||
|
||||
- name: Patch PKGBUILD with pkgver, pkgrel, conflicts, and b2sums
|
||||
env:
|
||||
TAG: ${{ steps.tag.outputs.tag }}
|
||||
|
||||
@@ -6,7 +6,7 @@ pkgdesc="Distributed, decentralized network routing protocol for mesh nodes"
|
||||
url="https://github.com/jmcorgan/fips"
|
||||
license=('MIT')
|
||||
arch=('x86_64')
|
||||
depends=('gcc-libs' 'glibc')
|
||||
depends=('dbus' 'gcc-libs' 'glibc')
|
||||
makedepends=('cargo' 'clang')
|
||||
optdepends=('systemd-resolved: .fips DNS resolution')
|
||||
conflicts=('fips-git' 'fips-git-debug')
|
||||
|
||||
@@ -20,6 +20,8 @@ This directory contains Arch Linux packaging files for two AUR packages:
|
||||
| `fips-dns.service` | Symlink to `../debian/fips-dns.service` |
|
||||
| `build-aur.sh` | Local `fips-git` build plus namcap validation (run by `make aur`) |
|
||||
| `patch-pkgbuild.sh` | Rewrites `pkgver`, `pkgrel`, `conflicts`, `options`, and `b2sums` in the PKGBUILD at publish time |
|
||||
| `await-package-runs.sh` | Holds the AUR publish until every `package-*.yml` run for the release tag has succeeded |
|
||||
| `test-await-package-runs.sh` | Fixture tests for `await-package-runs.sh`, run by the `aur-build` job |
|
||||
|
||||
Both PKGBUILDs reference files from `packaging/debian/` (service files) and
|
||||
`packaging/common/` (config files) at build time. These are pulled from the
|
||||
|
||||
Executable
+131
@@ -0,0 +1,131 @@
|
||||
#!/usr/bin/env bash
|
||||
# Wait until every package workflow run for a release tag has succeeded.
|
||||
#
|
||||
# The AUR publish job runs this before it pushes a new pkgver, so the AUR never
|
||||
# points at a tag whose release assets are still uploading or failed to build.
|
||||
# Each package-*.yml workflow uploads its assets from a `release` job inside
|
||||
# the same run, so a successful tag run means that workflow's assets are up.
|
||||
#
|
||||
# The population is discovered from the checked-out tree rather than listed
|
||||
# here: the package-*.yml files at the tag are the workflows the tag push
|
||||
# triggered. Finding none is a failure, never a pass.
|
||||
#
|
||||
# A tag push and the branch push of the same commit each start a run with the
|
||||
# same head_sha; only the tag run carries the tag name in head_branch, so runs
|
||||
# are matched on both. If the tag was re-pushed at the same commit, the newest
|
||||
# matching run decides.
|
||||
#
|
||||
# Required environment variables:
|
||||
# GITHUB_REPOSITORY - owner/repo
|
||||
# TAG - release tag (e.g. v0.5.1)
|
||||
# SHA - the commit the tag points at
|
||||
# GH_TOKEN - token with actions:read (read by gh; not checked here)
|
||||
# Optional:
|
||||
# WORKFLOW_DIR - where to discover package-*.yml (default .github/workflows)
|
||||
# AWAIT_POLLS - number of polls before giving up (default 60)
|
||||
# AWAIT_INTERVAL - seconds between polls (default 60)
|
||||
# DISCOVER_ONLY - if set to 1, print the discovered workflows and exit
|
||||
#
|
||||
# Exit status: 0 when every discovered workflow has a successful tag run;
|
||||
# 1 at once when a tag run concludes anything but success; 1 when the poll
|
||||
# budget runs out with any workflow unobserved, not finished, or unreadable.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
: "${GITHUB_REPOSITORY:?GITHUB_REPOSITORY must be set}"
|
||||
: "${TAG:?TAG must be set}"
|
||||
: "${SHA:?SHA must be set}"
|
||||
WORKFLOW_DIR="${WORKFLOW_DIR:-.github/workflows}"
|
||||
AWAIT_POLLS="${AWAIT_POLLS:-60}"
|
||||
AWAIT_INTERVAL="${AWAIT_INTERVAL:-60}"
|
||||
|
||||
# How to recover once the cause of a failure is fixed. A workflow_dispatch run
|
||||
# of a package workflow is not a push run of the tag and never satisfies this
|
||||
# gate; re-running the failed jobs of the tag's own run keeps it one.
|
||||
RECOVERY="If a run failed: use \"Re-run failed jobs\" on the package workflow's run for $TAG \
|
||||
(a new workflow_dispatch run does not count), then re-run the AUR Publish job for $TAG."
|
||||
|
||||
# Print the basenames of the package workflows found in WORKFLOW_DIR.
|
||||
discover() {
|
||||
local f
|
||||
for f in "$WORKFLOW_DIR"/package-*.yml; do
|
||||
[ -e "$f" ] && basename "$f"
|
||||
done
|
||||
return 0
|
||||
}
|
||||
|
||||
# Print "<status> <conclusion> <html_url>" for the newest push run of TAG at
|
||||
# SHA for one workflow, or "none" if there is no such run yet. On an API or
|
||||
# parse failure, print the error text and return nonzero.
|
||||
latest() {
|
||||
local wf=$1 body
|
||||
if ! body=$(gh api "repos/$GITHUB_REPOSITORY/actions/workflows/$wf/runs?head_sha=$SHA&event=push&per_page=100" 2>"$ERRS"); then
|
||||
tr '\n' ' ' < "$ERRS"
|
||||
return 1
|
||||
fi
|
||||
printf '%s\n' "$body" | jq -er --arg tag "$TAG" --arg sha "$SHA" '
|
||||
[.workflow_runs[]
|
||||
| select(.head_branch == $tag and .head_sha == $sha and .event == "push")]
|
||||
| sort_by(.created_at)
|
||||
| if length == 0 then "none"
|
||||
else last | "\(.status) \(.conclusion // "none") \(.html_url)" end' 2>&1
|
||||
}
|
||||
|
||||
mapfile -t pending < <(discover)
|
||||
if [ "${#pending[@]}" -eq 0 ]; then
|
||||
echo "No package-*.yml workflows found in $WORKFLOW_DIR; refusing to treat an empty set as done" >&2
|
||||
exit 1
|
||||
fi
|
||||
if [ "${DISCOVER_ONLY:-}" = 1 ]; then
|
||||
printf '%s\n' "${pending[@]}"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
ERRS=$(mktemp)
|
||||
trap 'rm -f "$ERRS"' EXIT
|
||||
|
||||
echo "Waiting for $TAG ($SHA) runs of: ${pending[*]}"
|
||||
declare -A seen
|
||||
for ((poll = 1; poll <= AWAIT_POLLS; poll++)); do
|
||||
still=()
|
||||
for wf in "${pending[@]}"; do
|
||||
if ! state=$(latest "$wf"); then
|
||||
seen[$wf]="API error: $state"
|
||||
still+=("$wf")
|
||||
continue
|
||||
fi
|
||||
seen[$wf]=$state
|
||||
read -r status conclusion url <<< "$state"
|
||||
case "$status" in
|
||||
none)
|
||||
# The tag run has not been created yet.
|
||||
still+=("$wf") ;;
|
||||
completed)
|
||||
if [ "$conclusion" = success ]; then
|
||||
echo "$wf: succeeded ($url)"
|
||||
else
|
||||
echo "$wf: run for $TAG concluded '$conclusion' ($url); not publishing to the AUR" >&2
|
||||
echo "$RECOVERY" >&2
|
||||
exit 1
|
||||
fi ;;
|
||||
*)
|
||||
# queued, in_progress, waiting, requested or pending.
|
||||
still+=("$wf") ;;
|
||||
esac
|
||||
done
|
||||
pending=("${still[@]}")
|
||||
if [ "${#pending[@]}" -eq 0 ]; then
|
||||
echo "Every package workflow run for $TAG has succeeded"
|
||||
exit 0
|
||||
fi
|
||||
echo "Poll $poll/$AWAIT_POLLS: waiting on ${pending[*]}"
|
||||
if [ "$poll" -lt "$AWAIT_POLLS" ]; then sleep "$AWAIT_INTERVAL"; fi
|
||||
done
|
||||
|
||||
echo "Gave up after $AWAIT_POLLS polls; not publishing to the AUR. Still unfinished:" >&2
|
||||
for wf in "${pending[@]}"; do
|
||||
echo " $wf: ${seen[$wf]}" >&2
|
||||
done
|
||||
echo "If a run is still going, re-run the AUR Publish job for $TAG once it has succeeded." >&2
|
||||
echo "$RECOVERY" >&2
|
||||
exit 1
|
||||
Executable
+233
@@ -0,0 +1,233 @@
|
||||
#!/usr/bin/env bash
|
||||
# Fixture tests for await-package-runs.sh, the gate that holds the AUR publish
|
||||
# until every package workflow run for the release tag has succeeded.
|
||||
#
|
||||
# Each case writes GitHub API responses into a fixture directory and puts a
|
||||
# stub `gh` first on PATH. The stub serves <workflow>.<call>.json for the Nth
|
||||
# call about a workflow, falls back to <workflow>.json, and exits 1 when the
|
||||
# fixture holds a file named `gh-fails`. It counts calls per workflow so a
|
||||
# case can assert that the gate polled again rather than stopping early.
|
||||
#
|
||||
# Needs bash and jq. Exits nonzero if any case fails or if fewer cases ran
|
||||
# than are defined.
|
||||
#
|
||||
# Usage: bash packaging/aur/test-await-package-runs.sh
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
HERE=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
|
||||
GATE="${GATE:-$HERE/await-package-runs.sh}"
|
||||
REPO_ROOT=$(cd "$HERE/../.." && pwd)
|
||||
|
||||
TAG=v9.9.9
|
||||
SHA=0123456789abcdef0123456789abcdef01234567
|
||||
WORKFLOWS=(package-freebsd.yml package-linux.yml package-macos.yml
|
||||
package-openwrt.yml package-windows.yml)
|
||||
|
||||
WORK=$(mktemp -d)
|
||||
trap 'rm -rf "$WORK"' EXIT
|
||||
|
||||
# The stub gh. It answers `gh api <url>` only, which is all the gate uses.
|
||||
mkdir -p "$WORK/bin"
|
||||
cat > "$WORK/bin/gh" <<'STUB'
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
[ "${1:-}" = "api" ] || { echo "stub gh: unexpected args: $*" >&2; exit 2; }
|
||||
[ -e "$STUB_FIXTURE/gh-fails" ] && { echo "stub gh: simulated API error" >&2; exit 1; }
|
||||
wf=$(printf '%s\n' "$2" | sed -E 's|.*/actions/workflows/([^/]+)/runs.*|\1|')
|
||||
count_file="$STUB_CALLS/$wf"
|
||||
n=$(( $(cat "$count_file" 2>/dev/null || echo 0) + 1 ))
|
||||
echo "$n" > "$count_file"
|
||||
if [ -f "$STUB_FIXTURE/$wf.$n.json" ]; then
|
||||
cat "$STUB_FIXTURE/$wf.$n.json"
|
||||
elif [ -f "$STUB_FIXTURE/$wf.json" ]; then
|
||||
cat "$STUB_FIXTURE/$wf.json"
|
||||
else
|
||||
echo '{"total_count":0,"workflow_runs":[]}'
|
||||
fi
|
||||
STUB
|
||||
chmod +x "$WORK/bin/gh"
|
||||
|
||||
# A workflow directory holding the five package workflows as empty files: the
|
||||
# gate discovers the population by file name only.
|
||||
mkdir -p "$WORK/workflows" "$WORK/empty-workflows"
|
||||
for wf in "${WORKFLOWS[@]}"; do : > "$WORK/workflows/$wf"; done
|
||||
|
||||
# Print one run object in the shape of the v0.5.1 API response.
|
||||
# Args: head_branch status conclusion created_at id
|
||||
run() {
|
||||
local conclusion=null
|
||||
[ "$3" = null ] || conclusion="\"$3\""
|
||||
printf '{"id":%s,"name":"Package","head_branch":"%s","head_sha":"%s","event":"push","status":"%s","conclusion":%s,"created_at":"%s","updated_at":"%s","html_url":"https://github.com/example/fips/actions/runs/%s"}' \
|
||||
"$5" "$1" "$SHA" "$2" "$conclusion" "$4" "$4" "$5"
|
||||
}
|
||||
|
||||
# Wrap run objects, given as arguments in the order the API returns them, in
|
||||
# the list response envelope.
|
||||
runs() {
|
||||
local IFS=,
|
||||
printf '{"total_count":%d,"workflow_runs":[%s]}\n' "$#" "$*"
|
||||
}
|
||||
|
||||
# Write the healthy v0.5.1 shape for every workflow into a fixture: a tag run
|
||||
# and a branch run of the same commit, newest first, both successful.
|
||||
all_green() {
|
||||
local dir=$1 wf
|
||||
for wf in "${WORKFLOWS[@]}"; do
|
||||
runs "$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" \
|
||||
"$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$dir/$wf.json"
|
||||
done
|
||||
}
|
||||
|
||||
CASES_DEFINED=0
|
||||
CASES_RAN=0
|
||||
FAILED=0
|
||||
|
||||
# Run the gate against a fixture and check its exit status and, when a
|
||||
# pattern is given, that its output states the expected reason (so a red
|
||||
# caused by a crash in the gate does not pass as the intended red).
|
||||
# Args: name fixture-dir expect(zero|nonzero) [pattern] [workflow-dir]
|
||||
# Sets LAST_OUT to the gate's combined output.
|
||||
check() {
|
||||
local name=$1 fixture=$2 expect=$3 pattern=${4:-} wfdir=${5:-$WORK/workflows} rc=0
|
||||
CASES_RAN=$((CASES_RAN + 1))
|
||||
rm -rf "$WORK/calls"; mkdir -p "$WORK/calls"
|
||||
LAST_OUT=$(PATH="$WORK/bin:$PATH" STUB_FIXTURE="$fixture" STUB_CALLS="$WORK/calls" \
|
||||
GITHUB_REPOSITORY=example/fips TAG="$TAG" SHA="$SHA" WORKFLOW_DIR="$wfdir" \
|
||||
AWAIT_POLLS=3 AWAIT_INTERVAL=0 bash "$GATE" 2>&1) || rc=$?
|
||||
if { [ "$expect" = zero ] && [ "$rc" -eq 0 ]; } ||
|
||||
{ [ "$expect" = nonzero ] && [ "$rc" -ne 0 ]; }; then
|
||||
if [ -z "$pattern" ] || printf '%s\n' "$LAST_OUT" | grep -qE -- "$pattern"; then
|
||||
echo "PASS $name (exit $rc)"
|
||||
return 0
|
||||
fi
|
||||
echo "FAIL $name: exit $rc as expected, but output lacks /$pattern/"
|
||||
else
|
||||
echo "FAIL $name: expected $expect exit, got $rc"
|
||||
fi
|
||||
printf '%s\n' "$LAST_OUT" | sed 's/^/ /'
|
||||
FAILED=$((FAILED + 1))
|
||||
return 1
|
||||
}
|
||||
|
||||
# Record an extra assertion's failure against the case that just ran.
|
||||
fail() {
|
||||
echo "FAIL $1"
|
||||
FAILED=$((FAILED + 1))
|
||||
}
|
||||
|
||||
# Make a fresh fixture directory for a case and print its path.
|
||||
fixture() {
|
||||
local dir="$WORK/fx/$1"
|
||||
mkdir -p "$dir"
|
||||
echo "$dir"
|
||||
}
|
||||
|
||||
# --- cases -------------------------------------------------------------------
|
||||
|
||||
# 1: every package workflow has a successful tag run.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 1); all_green "$fx"
|
||||
check "1 all tag runs succeeded" "$fx" zero "has succeeded$" || true
|
||||
|
||||
# 2: one tag run failed; the gate must name that workflow.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 2); all_green "$fx"
|
||||
runs "$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" \
|
||||
"$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$fx/package-macos.yml.json"
|
||||
if check "2 one tag run failed" "$fx" nonzero "concluded 'failure'"; then
|
||||
printf '%s\n' "$LAST_OUT" | grep -q 'package-macos.yml' ||
|
||||
fail "2 one tag run failed: output does not name package-macos.yml"
|
||||
fi
|
||||
|
||||
# 3: one tag run was cancelled.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 3); all_green "$fx"
|
||||
runs "$(run "$TAG" completed cancelled 2026-09-06T20:31:48Z 2)" > "$fx/package-openwrt.yml.json"
|
||||
check "3 one tag run cancelled" "$fx" nonzero "package-openwrt.yml: run for .* concluded 'cancelled'" || true
|
||||
|
||||
# 4: one workflow has only the branch run of the tag's commit (the v0.5.1
|
||||
# shape a SHA-only filter would accept).
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 4); all_green "$fx"
|
||||
runs "$(run maint completed success 2026-09-06T20:31:10Z 1)" > "$fx/package-freebsd.yml.json"
|
||||
check "4 only a branch run, no tag run" "$fx" nonzero "package-freebsd.yml: none$" || true
|
||||
|
||||
# 5: one tag run is in progress on the first poll and succeeds on the second.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 5); all_green "$fx"
|
||||
runs "$(run "$TAG" in_progress null 2026-09-06T20:31:48Z 2)" > "$fx/package-freebsd.yml.1.json"
|
||||
if check "5 in progress, then succeeded" "$fx" zero "has succeeded$"; then
|
||||
calls=$(cat "$WORK/calls/package-freebsd.yml" 2>/dev/null || echo 0)
|
||||
[ "$calls" -ge 2 ] ||
|
||||
fail "5 in progress, then succeeded: gate queried package-freebsd.yml $calls time(s), expected at least 2"
|
||||
fi
|
||||
|
||||
# 6: one tag run stays in progress for the whole budget.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 6); all_green "$fx"
|
||||
runs "$(run "$TAG" in_progress null 2026-09-06T20:31:48Z 2)" > "$fx/package-freebsd.yml.json"
|
||||
check "6 in progress for the whole budget" "$fx" nonzero "package-freebsd.yml: in_progress " || true
|
||||
|
||||
# 7: no package workflows discovered.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 7); all_green "$fx"
|
||||
check "7 empty workflow population" "$fx" nonzero "No package-\*\.yml workflows found" \
|
||||
"$WORK/empty-workflows" || true
|
||||
|
||||
# 8a-8d: the tag was re-pushed at the same commit, so two tag runs exist. The
|
||||
# newest decides. The API lists newest first; 8b and 8c list oldest first so
|
||||
# that "take the first match" and "take the newest" disagree.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 8a); all_green "$fx"
|
||||
runs "$(run "$TAG" completed success 2026-09-06T21:00:00Z 3)" \
|
||||
"$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" > "$fx/package-linux.yml.json"
|
||||
check "8a older failure, newer success, newest first" "$fx" zero "has succeeded$" || true
|
||||
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 8b); all_green "$fx"
|
||||
runs "$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" \
|
||||
"$(run "$TAG" completed failure 2026-09-06T21:00:00Z 3)" > "$fx/package-linux.yml.json"
|
||||
check "8b older success, newer failure, oldest first" "$fx" nonzero \
|
||||
"package-linux.yml: run for .* concluded 'failure'" || true
|
||||
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 8c); all_green "$fx"
|
||||
runs "$(run "$TAG" completed failure 2026-09-06T20:31:48Z 2)" \
|
||||
"$(run "$TAG" completed success 2026-09-06T21:00:00Z 3)" > "$fx/package-linux.yml.json"
|
||||
check "8c older failure, newer success, oldest first" "$fx" zero "has succeeded$" || true
|
||||
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 8d); all_green "$fx"
|
||||
runs "$(run "$TAG" completed failure 2026-09-06T21:00:00Z 3)" \
|
||||
"$(run "$TAG" completed success 2026-09-06T20:31:48Z 2)" > "$fx/package-linux.yml.json"
|
||||
check "8d older success, newer failure, newest first" "$fx" nonzero \
|
||||
"package-linux.yml: run for .* concluded 'failure'" || true
|
||||
|
||||
# 9: every API call fails.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
fx=$(fixture 9); all_green "$fx"; : > "$fx/gh-fails"
|
||||
check "9 gh fails on every call" "$fx" nonzero "package-linux.yml: API error" || true
|
||||
|
||||
# 10: discovery against the repository's real workflow directory.
|
||||
CASES_DEFINED=$((CASES_DEFINED + 1))
|
||||
CASES_RAN=$((CASES_RAN + 1))
|
||||
rc=0
|
||||
found=$(DISCOVER_ONLY=1 WORKFLOW_DIR="$REPO_ROOT/.github/workflows" \
|
||||
GITHUB_REPOSITORY=example/fips TAG="$TAG" SHA="$SHA" bash "$GATE" 2>&1) || rc=$?
|
||||
if [ "$rc" -eq 0 ] && [ -n "$found" ] &&
|
||||
printf '%s\n' "$found" | grep -qx 'package-linux.yml'; then
|
||||
echo "PASS 10 real workflow discovery: $(printf '%s\n' "$found" | tr '\n' ' ')"
|
||||
else
|
||||
echo "FAIL 10 real workflow discovery: exit $rc, found: $found"
|
||||
FAILED=$((FAILED + 1))
|
||||
fi
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
echo "cases defined: $CASES_DEFINED, ran: $CASES_RAN, failed: $FAILED"
|
||||
if [ "$CASES_RAN" -ne "$CASES_DEFINED" ]; then
|
||||
echo "FAIL: not every defined case ran"
|
||||
exit 1
|
||||
fi
|
||||
[ "$FAILED" -eq 0 ]
|
||||
@@ -177,12 +177,17 @@ FIPS_INTEROP_NETEM="delay 10ms 5ms 25% loss 2%" \
|
||||
wants netem) but still runs a clean baseline loop.
|
||||
|
||||
Each rep invokes `interop-test.sh` with netem set and captures its full
|
||||
output and exit code; a rep passes iff `interop-test.sh` exits 0. Artifacts
|
||||
land in `testing/interop/.stress-runs/<UTC-timestamp>/`:
|
||||
output and exit code; a rep passes iff `interop-test.sh` exits 0. Every rep
|
||||
runs with `FIPS_INTEROP_KEEP_UP=1`, because whether it failed is known only
|
||||
after the driver exits: the loop saves a failed rep's per-node logs and then
|
||||
tears the mesh down itself, including when the loop is interrupted.
|
||||
Artifacts land in `testing/interop/.stress-runs/<UTC-timestamp>/`:
|
||||
|
||||
- `rep-NN/driver.log` — full driver output for every rep.
|
||||
- `rep-NN/docker-<container>.log` — per-container `docker logs` for **failed**
|
||||
reps only.
|
||||
reps only. The expected containers come from the generated manifest, and
|
||||
the loop prints `harvest: k of n per-node logs written` for each failed
|
||||
rep. A log that could not be read or came back empty counts as missing.
|
||||
- `summary.txt` — the aggregate report.
|
||||
|
||||
The aggregate report gives reps run, passed/failed counts and an integer
|
||||
@@ -193,9 +198,10 @@ pair kind (mixed vs same), and a verdict:
|
||||
- **both mixed and same pairs** → loss-induced general instability;
|
||||
- **same-version control pair only** → the build is unstable against itself.
|
||||
|
||||
`interop-stress.sh` exits **non-zero only** for the interop-regression
|
||||
signal. A sub-100% pass rate under loss is expected and is not by itself a
|
||||
failure, so every other outcome exits 0.
|
||||
`interop-stress.sh` exits **1** for the interop-regression signal and **3**
|
||||
when a failed rep's per-node logs are incomplete (the run's diagnostics are
|
||||
missing; a regression keeps exit 1). A sub-100% pass rate under loss is
|
||||
expected and is not by itself a failure, so every other outcome exits 0.
|
||||
|
||||
### Options
|
||||
|
||||
@@ -203,7 +209,8 @@ failure, so every other outcome exits 0.
|
||||
| ---------------------------- | ------------------------------------------------- |
|
||||
| `FIPS_INTEROP_NETEM` | tc-netem string applied to each container's eth0, e.g. `"delay 10ms 5ms 25% loss 1%"`. Passed through `interop-stress.sh` to `interop-test.sh`. |
|
||||
| `REKEY_AFTER_SECS` | Rekey interval for generated configs (default 35).|
|
||||
| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. |
|
||||
| `CONTROL_MAX_ATTEMPTS` | Control windows Phase 1b measures before Phase 5b abstains (default 3). |
|
||||
| `FIPS_INTEROP_KEEP_UP` | `1` = leave containers running after the test. The stress loop sets it for every rep and tears down itself. |
|
||||
| `FIPS_INTEROP_KEEP_WORKTREES`| `1` = keep `build-images.sh` worktrees (debug). |
|
||||
| `FIPS_INTEROP_RUNS_DIR` | Root for the three scratch dirs — see [Scratch directory location](#scratch-directory-location). |
|
||||
|
||||
@@ -249,6 +256,16 @@ The driver runs seven phases:
|
||||
| 5 | All pairs still ping after the second rekey. |
|
||||
| 6 | Per-node / per-pair interop log analysis. |
|
||||
|
||||
When data-plane streams are on (`--topology`, or `FIPS_INTEROP_STREAMS`),
|
||||
Phase 1b measures stream loss over a quiet control window and Phase 5b
|
||||
compares the loss across the rekey window against it. A control window
|
||||
counts only if no FMP rekey cutover happened during it and the cutover
|
||||
count could be read on every node. Otherwise Phase 1b waits for the
|
||||
cutovers to settle and re-measures, up to `CONTROL_MAX_ATTEMPTS` times (default
|
||||
3). If no attempt is clean, Phase 5b prints `ABSTAIN` for every stream and
|
||||
returns no verdict, and the summary says so. The stress loop reports in how
|
||||
many reps Phase 5b abstained and in how many it re-measured.
|
||||
|
||||
Phase 6 is the interop-specific part. It reports:
|
||||
|
||||
- **Global health** — panics, `ERROR` lines, `unknown FMP version` drops,
|
||||
|
||||
@@ -19,8 +19,13 @@
|
||||
# -> the version under test is unstable even against itself.
|
||||
#
|
||||
# A sub-100% pass rate under loss is EXPECTED and is not, by itself, a
|
||||
# failure. This script exits non-zero ONLY for the interop-regression
|
||||
# signal (mixed-only failures).
|
||||
# failure. This script exits 1 for the interop-regression signal
|
||||
# (mixed-only failures) and 3 when a failed rep's per-node logs could not
|
||||
# all be saved, so its diagnostics are incomplete.
|
||||
#
|
||||
# Every rep runs with FIPS_INTEROP_KEEP_UP=1, because whether a rep failed
|
||||
# is known only after the driver exits. The loop saves a failed rep's
|
||||
# per-node logs and then tears the mesh down itself.
|
||||
#
|
||||
# Reps run SERIALLY — interop-test.sh uses fixed container names and a
|
||||
# fixed Docker network, so two reps must never overlap.
|
||||
@@ -52,7 +57,10 @@
|
||||
#
|
||||
# Artifacts (per invocation): <runs-base>/.stress-runs/<UTC-ts>/
|
||||
# rep-NN/driver.log full interop-test.sh output for that rep.
|
||||
# rep-NN/docker-<node>.log per-container `docker logs` (FAILED reps only).
|
||||
# rep-NN/docker-<container>.log
|
||||
# per-container `docker logs` (FAILED reps only).
|
||||
# A failed rep with fewer of these than nodes
|
||||
# makes the run exit 3.
|
||||
# summary.txt the final aggregate report.
|
||||
|
||||
set -uo pipefail
|
||||
@@ -80,6 +88,65 @@ else
|
||||
fi
|
||||
|
||||
RUNS_BASE="$INTEROP_RUNS_BASE/.stress-runs"
|
||||
# The driver's generated mesh (same paths as interop-test.sh), read for the
|
||||
# expected container set and used to tear a kept-up rep down.
|
||||
GEN_DIR="$INTEROP_RUNS_BASE/generated-configs"
|
||||
COMPOSE_FILE="$GEN_DIR/docker-compose.generated.yml"
|
||||
NODES_ENV="$GEN_DIR/nodes.env"
|
||||
|
||||
# Save every node's `docker logs` into a failed rep's directory.
|
||||
#
|
||||
# The expected containers come from the generated manifest rather than
|
||||
# from `docker ps`: an empty `docker ps` result is how a harvest that saved
|
||||
# nothing used to pass without a word. A container counts only when
|
||||
# `docker logs` succeeded and wrote a non-empty file. Returns 0 only when
|
||||
# every expected container was saved and there was at least one.
|
||||
harvest_rep() {
|
||||
local rep_dir="$1" containers tok ctr n=0 k=0
|
||||
local missing=()
|
||||
# A subshell, so the manifest's variables stay out of the loop.
|
||||
# shellcheck disable=SC1090
|
||||
containers="$( [ -f "$NODES_ENV" ] && . "$NODES_ENV" \
|
||||
&& printf '%s' "${INTEROP_NODE_CONTAINERS:-}" )" || containers=""
|
||||
for tok in $containers; do
|
||||
ctr="${tok#*:}"
|
||||
n=$((n + 1))
|
||||
if docker logs "$ctr" >"$rep_dir/docker-$ctr.log" 2>&1 \
|
||||
&& [ -s "$rep_dir/docker-$ctr.log" ]; then
|
||||
k=$((k + 1))
|
||||
else
|
||||
missing+=("$ctr")
|
||||
fi
|
||||
done
|
||||
echo " harvest: $k of $n per-node logs written"
|
||||
if [ "$n" -eq 0 ]; then
|
||||
echo " HARVEST FAILED: no manifest / empty container list ($NODES_ENV)"
|
||||
return 1
|
||||
fi
|
||||
if [ "$k" -lt "$n" ]; then
|
||||
echo " HARVEST FAILED: $k of $n; missing: ${missing[*]}"
|
||||
return 1
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
# Tear down a rep's kept-up mesh. A failure is loud but not fatal: the
|
||||
# next rep's Phase 0 also brings the mesh down before starting.
|
||||
teardown_rep() {
|
||||
local rc left
|
||||
docker compose -f "$COMPOSE_FILE" down --volumes --remove-orphans \
|
||||
>/dev/null 2>&1
|
||||
rc=$?
|
||||
MESH_UP=0
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
echo " WARN: teardown failed ($rc)"
|
||||
fi
|
||||
left="$(docker ps -a --filter 'name=fips-interop-' --format '{{.Names}}' 2>/dev/null)"
|
||||
if [ -n "$left" ]; then
|
||||
echo " WARN: containers remain after teardown: $(echo "$left" | tr '\n' ' ')"
|
||||
fi
|
||||
return "$rc"
|
||||
}
|
||||
|
||||
# ── Args ─────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -171,6 +238,13 @@ RUN_TS="$(date -u +%Y-%m-%dT%H-%M-%SZ)"
|
||||
RUN_DIR="$RUNS_BASE/$RUN_TS"
|
||||
mkdir -p "$RUN_DIR"
|
||||
|
||||
# A rep runs kept up, so an interrupted or aborted loop must still take
|
||||
# its mesh down. The INT and TERM traps exit, which runs the EXIT trap.
|
||||
MESH_UP=0
|
||||
trap '[ "$MESH_UP" -eq 1 ] && teardown_rep' EXIT
|
||||
trap 'echo ""; echo "Interrupted"; exit 130' INT
|
||||
trap 'echo ""; echo "Terminated"; exit 143' TERM
|
||||
|
||||
echo "=============================================================="
|
||||
echo " FIPS Interop Netem Stress Loop"
|
||||
echo "=============================================================="
|
||||
@@ -186,6 +260,14 @@ echo ""
|
||||
PASS_COUNT=0
|
||||
FAIL_COUNT=0
|
||||
FAILED_REPS=()
|
||||
# Failed reps whose per-node logs were not all saved.
|
||||
HARVEST_FAILS=0
|
||||
HARVEST_FAILED_REPS=()
|
||||
# Phase 5b outcomes over the reps that measured a control window (Phase
|
||||
# 1b ran): reps that abstained, and reps that re-measured at least once.
|
||||
STREAM_REPS=0
|
||||
ABSTAIN_REPS=0
|
||||
REMEASURE_REPS=0
|
||||
# Per-kind connectivity-failure tallies, summed across all failed reps.
|
||||
MIXED_FAILS=0
|
||||
SAME_FAILS=0
|
||||
@@ -199,11 +281,27 @@ for ((rep = 1; rep <= REPS; rep++)); do
|
||||
echo "── $rep_id / $REPS ──────────────────────────────────────────"
|
||||
|
||||
# Run the driver, capturing full output and exit code. Netem is
|
||||
# passed through the environment; interop-test.sh applies it.
|
||||
FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \
|
||||
# passed through the environment; interop-test.sh applies it. The
|
||||
# mesh is kept up so a failed rep can be harvested below.
|
||||
MESH_UP=1
|
||||
FIPS_INTEROP_KEEP_UP=1 FIPS_INTEROP_NETEM="${FIPS_INTEROP_NETEM:-}" \
|
||||
bash "$DRIVER" "${DRIVER_ARGS[@]}" >"$driver_log" 2>&1
|
||||
rc=$?
|
||||
|
||||
# Tallied for every rep whatever its verdict: abstaining is the absence
|
||||
# of a verdict rather than a failure, so without this count a run whose
|
||||
# Phase 5b always abstains would pass unnoticed.
|
||||
if grep -q '^Phase 1b: ' "$driver_log"; then
|
||||
STREAM_REPS=$((STREAM_REPS + 1))
|
||||
if grep -q '^ ABSTAIN Data-plane continuity' "$driver_log"; then
|
||||
ABSTAIN_REPS=$((ABSTAIN_REPS + 1))
|
||||
echo " Phase 5b abstained (no clean control window)"
|
||||
fi
|
||||
if grep -q 'control window attempt .*re-measuring' "$driver_log"; then
|
||||
REMEASURE_REPS=$((REMEASURE_REPS + 1))
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ "$rc" -eq 0 ]; then
|
||||
PASS_COUNT=$((PASS_COUNT + 1))
|
||||
echo " PASS (exit 0)"
|
||||
@@ -212,14 +310,11 @@ for ((rep = 1; rep <= REPS; rep++)); do
|
||||
FAILED_REPS+=("$rep_id")
|
||||
echo " FAIL (exit $rc)"
|
||||
|
||||
# Preserve each failed container's full docker logs. The driver
|
||||
# uses fixed container names fips-interop-<nodeid>; harvest every
|
||||
# container matching that prefix that still exists.
|
||||
while read -r ctr; do
|
||||
[ -n "$ctr" ] || continue
|
||||
docker logs "$ctr" >"$rep_dir/docker-${ctr}.log" 2>&1 || true
|
||||
done < <(docker ps -a --filter 'name=fips-interop-' \
|
||||
--format '{{.Names}}' 2>/dev/null)
|
||||
# Preserve each node's full docker logs before the teardown below.
|
||||
if ! harvest_rep "$rep_dir"; then
|
||||
HARVEST_FAILS=$((HARVEST_FAILS + 1))
|
||||
HARVEST_FAILED_REPS+=("$rep_id")
|
||||
fi
|
||||
|
||||
# Tally connectivity failures by pair kind, reusing the
|
||||
# pair-attributed lines interop-test.sh prints. Each line is
|
||||
@@ -230,6 +325,8 @@ for ((rep = 1; rep <= REPS; rep++)); do
|
||||
SAME_FAILS=$((SAME_FAILS + s))
|
||||
echo " connectivity-failure lines: mixed=$m same=$s"
|
||||
fi
|
||||
|
||||
teardown_rep
|
||||
done
|
||||
|
||||
echo ""
|
||||
@@ -285,6 +382,10 @@ else
|
||||
VERDICT="NO connectivity-pair failures recorded, but $FAIL_COUNT rep(s) still"
|
||||
VERDICT2="failed — on non-connectivity signatures (global-health log patterns or a missing rekey). Check the per-rep driver logs."
|
||||
fi
|
||||
# A regression keeps precedence; otherwise missing diagnostics fail the run.
|
||||
if [ "$EXIT_CODE" -eq 0 ] && [ "$HARVEST_FAILS" -gt 0 ]; then
|
||||
EXIT_CODE=3
|
||||
fi
|
||||
|
||||
{
|
||||
echo "=============================================================="
|
||||
@@ -302,6 +403,12 @@ fi
|
||||
if [ "${#FAILED_REPS[@]}" -gt 0 ]; then
|
||||
echo "Failed reps: ${FAILED_REPS[*]}"
|
||||
fi
|
||||
if [ "$HARVEST_FAILS" -gt 0 ]; then
|
||||
echo "Harvest : $HARVEST_FAILS failed rep(s) with incomplete per-node logs: ${HARVEST_FAILED_REPS[*]}"
|
||||
fi
|
||||
if [ "$STREAM_REPS" -gt 0 ]; then
|
||||
echo "Phase 5b : abstained in $ABSTAIN_REPS of $STREAM_REPS reps; re-measured in $REMEASURE_REPS of $STREAM_REPS reps"
|
||||
fi
|
||||
echo ""
|
||||
echo "-- Connectivity-failure attribution (summed over failed reps) --"
|
||||
echo " mixed-version : $MIXED_FAILS failure(s) over $MIXED_PAIRS mixed pairs x $REPS reps"
|
||||
@@ -315,6 +422,9 @@ fi
|
||||
if [ "$EXIT_CODE" -eq 0 ]; then
|
||||
echo "Exit 0: no interop-regression signal (a sub-100% rate under loss"
|
||||
echo " is expected and is not by itself a failure)."
|
||||
elif [ "$EXIT_CODE" -eq 3 ]; then
|
||||
echo "Exit 3: per-node logs missing for a failed rep; the run's"
|
||||
echo " diagnostics are incomplete."
|
||||
else
|
||||
echo "Exit 1: interop-regression signal present."
|
||||
fi
|
||||
|
||||
@@ -53,6 +53,9 @@
|
||||
# by --topology; empty = streams off.
|
||||
# STREAM_LOSS_MARGIN_PCT rekey-vs-control loss margin (default 5).
|
||||
# CONTROL_STREAM_SECS quiet control-window length (default 12).
|
||||
# CONTROL_MAX_ATTEMPTS control windows measured before Phase 5b
|
||||
# abstains because every one saw a rekey
|
||||
# cutover or could not be checked (default 3).
|
||||
# MESH_SIZE_WARMUP Phase 7 bloom warmup, from mesh start, before
|
||||
# any estimate counts (default 300).
|
||||
# MESH_SIZE_SETTLE Phase 7 unbroken in-band window a node must
|
||||
@@ -190,6 +193,14 @@ LOG_POLL_INTERVAL=2
|
||||
# as on mixed ones.
|
||||
STREAM_RATE_HZ=20
|
||||
CONTROL_STREAM_SECS="${CONTROL_STREAM_SECS:-12}" # quiet pre-rekey window
|
||||
# A control window that saw a rekey cutover, or whose cutover count could
|
||||
# not be read, is no baseline. Phase 1b re-measures it, after waiting for
|
||||
# the cutover count to hold still for CONTROL_QUIET_SECS (at most
|
||||
# CONTROL_QUIET_MAX) so the retry does not land in the same cluster of
|
||||
# cutovers, and Phase 5b abstains when no attempt comes out clean.
|
||||
CONTROL_MAX_ATTEMPTS="${CONTROL_MAX_ATTEMPTS:-3}"
|
||||
CONTROL_QUIET_SECS=5
|
||||
CONTROL_QUIET_MAX=30
|
||||
# 5% is 12 packets of the 240-packet control window and ~155 of the
|
||||
# ~3100-packet rekey window. On a path losing up to 6% (the range the
|
||||
# v0.4.2 netem runs baselined at, under `loss 2%` over several hops) the
|
||||
@@ -526,6 +537,37 @@ count_log_pattern() {
|
||||
return 0
|
||||
}
|
||||
|
||||
# Count completed FMP initiator rekey cutovers across all node logs.
|
||||
#
|
||||
# The pattern is a literal argument so the log-string guard can check it
|
||||
# against the daemon source. Prints the count, or `unreadable:<ctr>` with
|
||||
# status 1 when a node's logs cannot be read.
|
||||
fmp_cutover_count() {
|
||||
count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped'
|
||||
return $?
|
||||
}
|
||||
|
||||
# Wait until the FMP cutover count has held unchanged for
|
||||
# CONTROL_QUIET_SECS, giving up after CONTROL_QUIET_MAX. An unreadable
|
||||
# count is treated as a change, so it never counts as quiet.
|
||||
wait_cutover_quiet() {
|
||||
local start=$SECONDS still=$SECONDS last cur
|
||||
last="$(fmp_cutover_count)" || last="unreadable"
|
||||
while (( SECONDS - still < CONTROL_QUIET_SECS )); do
|
||||
if (( SECONDS - start >= CONTROL_QUIET_MAX )); then
|
||||
echo " cutover count still moving after ${CONTROL_QUIET_MAX}s; measuring anyway"
|
||||
return 1
|
||||
fi
|
||||
sleep 1
|
||||
cur="$(fmp_cutover_count)" || cur="unreadable"
|
||||
if [ "$cur" != "$last" ] || [ "$cur" = "unreadable" ]; then
|
||||
last="$cur"
|
||||
still=$SECONDS
|
||||
fi
|
||||
done
|
||||
return 0
|
||||
}
|
||||
|
||||
# Per-node count of a pattern.
|
||||
count_node_pattern() {
|
||||
local node="$1" pattern="$2"
|
||||
@@ -765,25 +807,52 @@ echo ""
|
||||
# ── Phase 1b: data-plane control window (quiet, pre-rekey) ───────────
|
||||
#
|
||||
# Measure stream loss over a window with NO rekey cutover, as the control
|
||||
# baseline for the differential. Validated cutover-free by confirming the
|
||||
# FMP cutover count did not advance during the window. Then launch the
|
||||
# baseline for the differential. A window counts only when the FMP cutover
|
||||
# count was read before and after it and did not advance. A window that
|
||||
# saw a cutover, or could not be checked, is re-measured up to
|
||||
# CONTROL_MAX_ATTEMPTS times; if none is clean, Phase 5b abstains rather
|
||||
# than computing a verdict from a contaminated baseline. Then launch the
|
||||
# rekey-window streams, which run in the background across Phases 2-5 and
|
||||
# are collected/asserted in Phase 5b.
|
||||
control_contaminated=0
|
||||
control_abstain=0
|
||||
if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then
|
||||
echo "Phase 1b: Data-plane control stream (${CONTROL_STREAM_SECS}s quiet window)"
|
||||
pre_cut="$(count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped')"
|
||||
launch_streams CONTROL "$CONTROL_STREAM_SECS"
|
||||
collect_streams
|
||||
post_cut="$(count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped')"
|
||||
if [ "$post_cut" -ne "$pre_cut" ]; then
|
||||
control_contaminated=1
|
||||
echo " WARN a rekey cutover occurred during the control window — control loss may be contaminated"
|
||||
echo "Phase 1b: Data-plane control stream (${CONTROL_STREAM_SECS}s quiet window, up to ${CONTROL_MAX_ATTEMPTS} attempts)"
|
||||
control_ok=0
|
||||
for ((attempt = 1; attempt <= CONTROL_MAX_ATTEMPTS; attempt++)); do
|
||||
pre_rc=0
|
||||
post_rc=0
|
||||
pre_cut="$(fmp_cutover_count)" || pre_rc=$?
|
||||
launch_streams CONTROL "$CONTROL_STREAM_SECS"
|
||||
collect_streams
|
||||
post_cut="$(fmp_cutover_count)" || post_rc=$?
|
||||
if [ "$pre_rc" -ne 0 ] || ! [[ "$pre_cut" =~ ^[0-9]+$ ]]; then
|
||||
why="cutover count unreadable ($pre_cut)"
|
||||
elif [ "$post_rc" -ne 0 ] || ! [[ "$post_cut" =~ ^[0-9]+$ ]]; then
|
||||
why="cutover count unreadable ($post_cut)"
|
||||
elif [ "$post_cut" -ne "$pre_cut" ]; then
|
||||
why="a rekey cutover occurred (+$((post_cut - pre_cut)))"
|
||||
else
|
||||
control_ok=1
|
||||
echo " control window accepted on attempt $attempt/$CONTROL_MAX_ATTEMPTS"
|
||||
break
|
||||
fi
|
||||
if [ "$attempt" -lt "$CONTROL_MAX_ATTEMPTS" ]; then
|
||||
echo " control window attempt $attempt/$CONTROL_MAX_ATTEMPTS: $why; re-measuring"
|
||||
wait_cutover_quiet || true
|
||||
else
|
||||
echo " control window attempt $attempt/$CONTROL_MAX_ATTEMPTS: $why"
|
||||
fi
|
||||
done
|
||||
if [ "$control_ok" -eq 0 ]; then
|
||||
control_abstain=1
|
||||
echo " ABSTAIN control window contaminated on all $CONTROL_MAX_ATTEMPTS attempts; Phase 5b will not return a verdict"
|
||||
fi
|
||||
ctl_note=""
|
||||
[ "$control_abstain" -eq 1 ] && ctl_note=" (last attempt, contaminated)"
|
||||
for sp in "${STREAM_PAIRS[@]}"; do
|
||||
read -r sf st <<< "$sp"
|
||||
key="$sf->$st"
|
||||
echo " control $key: tx=${STREAM_TX[CONTROL:$key]:-0} rx=${STREAM_RX[CONTROL:$key]:-0} loss=$(_loss_pct "${STREAM_TX[CONTROL:$key]:-0}" "${STREAM_RX[CONTROL:$key]:-0}")%"
|
||||
echo " control $key: tx=${STREAM_TX[CONTROL:$key]:-0} rx=${STREAM_RX[CONTROL:$key]:-0} loss=$(_loss_pct "${STREAM_TX[CONTROL:$key]:-0}" "${STREAM_RX[CONTROL:$key]:-0}")%$ctl_note"
|
||||
done
|
||||
echo " Launching rekey-window streams (${REKEY_STREAM_SECS}s, spanning Phases 2-5)"
|
||||
launch_streams REKEY "$REKEY_STREAM_SECS"
|
||||
@@ -863,6 +932,12 @@ if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then
|
||||
if (k <= c + m) print "PASS"; else print "FAIL"
|
||||
}')"
|
||||
delta="$(awk -v c="$cpct" -v k="$rpct" 'BEGIN{ if(c=="NA"||k=="NA") print "NA"; else printf "%+.1f", k-c }')"
|
||||
# No clean control window: report the losses for reading, but
|
||||
# compute no verdict from a contaminated baseline.
|
||||
if [ "$control_abstain" -eq 1 ]; then
|
||||
printf ' %-16s %11s %11s %8s %s\n' "$key" "$cpct" "$rpct" "$delta" "ABSTAIN"
|
||||
continue
|
||||
fi
|
||||
printf ' %-16s %11s %11s %8s %s\n' "$key" "$cpct" "$rpct" "$delta" "$verdict"
|
||||
if [ "$verdict" = "PASS" ]; then
|
||||
PASSED=$((PASSED + 1))
|
||||
@@ -871,8 +946,11 @@ if [ "${#STREAM_PAIRS[@]}" -gt 0 ]; then
|
||||
INTEROP_FAILURES+=("[stream] $key ($(hop_label "$sf" "$st")): rekey-window loss ${rpct}% vs control ${cpct}% (+${STREAM_LOSS_MARGIN_PCT}% margin) -> $verdict")
|
||||
fi
|
||||
done
|
||||
[ "${control_contaminated:-0}" -eq 1 ] && echo " NOTE: control window saw a cutover; differential may understate rekey loss."
|
||||
phase_result "Data-plane continuity across rekey"
|
||||
if [ "$control_abstain" -eq 1 ]; then
|
||||
echo " ABSTAIN Data-plane continuity across rekey: no uncontaminated control window in $CONTROL_MAX_ATTEMPTS attempts"
|
||||
else
|
||||
phase_result "Data-plane continuity across rekey"
|
||||
fi
|
||||
echo ""
|
||||
fi
|
||||
|
||||
@@ -1078,6 +1156,13 @@ echo "=============================================================="
|
||||
# `${#arr[@]}` cannot be combined with `:-`; count into a plain var.
|
||||
INTEROP_FAILURE_COUNT="${#INTEROP_FAILURES[@]}"
|
||||
|
||||
# An abstaining Phase 5b adds no check, so say so: otherwise a green run
|
||||
# reads as having covered data-plane continuity across rekey.
|
||||
if [ "${control_abstain:-0}" -eq 1 ]; then
|
||||
echo ""
|
||||
echo "NOTE: Phase 5b abstained (control window contaminated); its rekey-loss check did not run."
|
||||
fi
|
||||
|
||||
if [ "$TOTAL_FAILED" -eq 0 ] && [ "$INTEROP_FAILURE_COUNT" -eq 0 ]; then
|
||||
echo ""
|
||||
echo "PASS: all versions interoperate cleanly across the mesh (spec '$SPEC_STR')."
|
||||
|
||||
Reference in New Issue
Block a user