mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-05 19:18:25 +00:00
Conflicts resolved: - testing/chaos/sim/netem.py: master re-applies veth netem through _apply_veth, on the restarted node and on each surviving neighbour's end of a recreated pair. _apply_veth now installs state.tc_args(), so both ends of a flapped Ethernet link stay at 100% loss across a node restart. On maint only the restarted end was held. - testing/chaos/sim/runner.py: kept master's version. Master already waits for three agreeing tree reads before the final snapshot, which spans the restored nodes' control-socket start-up, so maint's separate control-socket wait is not added and the final snapshot keeps a single wait. - testing/ci-local.sh: run_chaos keeps master's per-run results directory and adds maint's load-window begin mark after it.
1793 lines
75 KiB
Bash
Executable File
1793 lines
75 KiB
Bash
Executable File
#!/bin/bash
|
|
# Run the CI pipeline locally: CI parity check, build, unit tests,
|
|
# integration tests.
|
|
#
|
|
# Usage: ./ci-local.sh [options]
|
|
#
|
|
# Options:
|
|
# --build-only Only run build + clippy
|
|
# --test-only Only run unit tests (skip build, skip integration)
|
|
# --skip-integration Skip integration tests
|
|
# --skip-chaos Skip chaos scenarios
|
|
# --with-tor Include Tor harnesses (off by default — needs live Tor)
|
|
# --only <suite> Run a single integration suite
|
|
# -j, --jobs <N> Max parallel chaos scenarios (default: 4)
|
|
# --list List available integration suites
|
|
# --check-parity Verify this suite set matches ci.yml's integration
|
|
# matrix (see testing/check-ci-parity.sh), then exit
|
|
# --reap Force-remove all leftover FIPS CI resources
|
|
# (containers/networks/volumes carrying the CI label or a
|
|
# fipsci_ compose project, plus every chaos-simulation
|
|
# host veth interface), then exit. See ci-cleanup.sh.
|
|
# Host interfaces carry no label, so they are matched by
|
|
# name shape: this reaps a bare chaos.sh run's live
|
|
# interfaces too, though not its containers. Don't run
|
|
# it while a bare simulation is up.
|
|
# -h, --help Show this help
|
|
#
|
|
# Integration suites (default coverage):
|
|
# static-mesh, static-chain, gateway,
|
|
# firewall, nat-cone, nat-symmetric,
|
|
# nat-lan, nostr-publish-consume, stun-faults,
|
|
# chaos-churn-mixed-10, chaos-ethernet-mesh,
|
|
# chaos-ethernet-only, chaos-ethernet-churn, chaos-tcp-mesh,
|
|
# chaos-congestion-stress,
|
|
# sidecar, dns-resolver, deb-install, medium-change
|
|
#
|
|
# Opt-in (require --with-tor; depend on live Tor network):
|
|
# tor-socks5, tor-directory
|
|
#
|
|
# Deliberately not run by either runner, with the reason for each. Recorded
|
|
# here so that "not in the suite list" stops being indistinguishable from
|
|
# "forgotten", which is what it was until 2026-07-23:
|
|
# interop/ Manual. Driven by interop-stress.sh, which runs N
|
|
# repetitions serially under netem and takes far longer
|
|
# than a CI slot. No CI-sized entry point exists yet;
|
|
# writing one is the work, not adding a line here.
|
|
# boringtun/ Comparative benchmark against a non-FIPS implementation.
|
|
# Measures rather than asserts, and needs a boringtun
|
|
# build CI does not have.
|
|
# iperf-test.sh Bandwidth measurement, no pass/fail. Driven manually by
|
|
# iperf-compare-refs.sh.
|
|
# ecn-ab-compare.sh Manual A/B comparison, renamed from ecn-ab-test.sh
|
|
# because it asserts nothing. Making it gateable needs a
|
|
# calibration corpus that does not exist; see its header.
|
|
# mesh-lab/ Long-running multi-host lab, not a suite.
|
|
# acl-allowlist/ Retired from CI 2026-07-24 as redundant, not unrunnable.
|
|
# The ACL decision is exhaustively unit-tested per npub over
|
|
# real loaded allow/deny files (src/node/acl.rs mod tests:
|
|
# allow-wins, allowlist-miss, deny-only, deny-all,
|
|
# allow_all-override), and the inbound/outbound handshake-
|
|
# admission path is covered in-process over loopback
|
|
# (src/node/tests/acl.rs). The Docker suite's only unique
|
|
# coverage was real-UDP admission, exercised by every other
|
|
# real-transport suite. Still runnable by hand:
|
|
# bash testing/acl-allowlist/test.sh
|
|
# admission-cap Retired from CI 2026-07-24 as redundant, not unrunnable.
|
|
# The inbound max_peers early-gate in handle_msg1 is unit-
|
|
# tested over a real UDP socket by
|
|
# handle_msg1_silent_drops_at_cap_for_new_peer in
|
|
# src/node/tests/unit.rs: it sends a Msg1 from a fresh
|
|
# identity at saturation and polls the sender socket to
|
|
# assert no Msg2 returns — the same wire-observable
|
|
# discriminator the Docker tcpdump used — plus a sibling
|
|
# test for the existing-peer bypass. Still runnable by hand:
|
|
# bash testing/static/scripts/admission-cap-test.sh
|
|
# rekey, rekey-accept-off, rekey-outbound-only
|
|
# Retired from CI 2026-07-24 as redundant, not unrunnable.
|
|
# The rekey timing/choreography decision (trigger, K-bit
|
|
# cutover, drain, jitter, guards) is covered by the sans-IO
|
|
# poll_rekey tests (src/proto/fsp/tests/core.rs,
|
|
# src/proto/fmp/tests/core.rs); the data-plane continuity
|
|
# across a real cutover by rekey_cutover_preserves_data_plane
|
|
# (src/node/tests/session.rs, a real IK rekey over loopback);
|
|
# the accept-off dual-init regression and the
|
|
# udp.outbound_only rekey loop by the should_admit_msg1 and
|
|
# dual-init chartests (src/node/tests/handshake.rs,
|
|
# establish_chartests.rs). Still runnable by hand:
|
|
# bash testing/static/scripts/rekey-test.sh
|
|
# ecn-ab-on, ecn-ab-off, maelstrom, maelstrom-sparse
|
|
# Chaos scenarios excluded from CHAOS_SUITES. The two
|
|
# ecn-ab ones are halves of the manual comparison above.
|
|
# The two maelstrom ones are 600 s stress runs kept for
|
|
# manual investigation. All four still load-check and
|
|
# carry the default max_errors ceiling if run by hand.
|
|
#
|
|
# Exit codes:
|
|
# 0 — all stages passed
|
|
# 1 — one or more stages failed
|
|
# 130 — interrupted by SIGINT (128 + 2; run was cancelled, not a failure)
|
|
# 143 — terminated by SIGTERM (128 + 15; run was cancelled, not a failure)
|
|
#
|
|
# A preempting CI worker maps 130/143 → "cancelled" (discard, do not record a
|
|
# failing commit), 0 → green, any other non-zero → red.
|
|
#
|
|
# Host load annotation: every failed suite in the summary is followed by the
|
|
# CPU pressure stall (/proc/pressure/cpu) over that suite's run and the peak
|
|
# number of other CI runs' containers, labelled LOAD-DEGRADED at or above
|
|
# CI_LOAD_THRESHOLD percent (testing/lib/load-annotate.py). It is context for
|
|
# reading a red and changes no verdict or exit code. The samples go to
|
|
# $FIPS_CI_LOAD_LOG if set (kept after the run), else to a per-run file that
|
|
# teardown removes.
|
|
#
|
|
# ── CI parity invariant ─────────────────────────────────────────────────────
|
|
# This local default suite set and the GitHub integration matrix
|
|
# (.github/workflows/ci.yml) MUST run the same integration suites, EXCEPT for
|
|
# the deliberate local-only entries below. Adding a suite to one runner
|
|
# without the other means "local green" and "GitHub green" stop being
|
|
# equivalent. testing/check-ci-parity.sh enforces this and fails on drift; it
|
|
# runs as the first stage of every local run, before the build.
|
|
#
|
|
# Deliberate local-only (NOT on the GitHub gate), with reason:
|
|
# tor-socks5 — requires live Tor network; opt-in via --with-tor,
|
|
# unreliable on GitHub-hosted runners.
|
|
# tor-directory — same; live Tor dependency.
|
|
#
|
|
# Deliberate GitHub-only (NOT in this local run), with reason:
|
|
# deb-install ubuntu22 on arm64 — this host is x86_64 and cannot run an
|
|
# arm64 package. The guard compares it by distro only and
|
|
# does not let it stand in for the amd64 ubuntu22 leg.
|
|
#
|
|
# The two runners express the same work in different matrix shapes, and the
|
|
# guard compares through that shape rather than around it: chaos legs are
|
|
# compared per scenario (and per flag), deb-install legs per distro. The one
|
|
# leg still compared at leg granularity is dns-resolver — a single suite on
|
|
# both sides that runs all of its scenarios internally. On GitHub it is a job
|
|
# of its own, because it runs the package's binaries and so waits for the
|
|
# package build.
|
|
# ─────────────────────────────────────────────────────────────────────────────
|
|
set -uo pipefail
|
|
|
|
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
|
|
PROJECT_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)"
|
|
|
|
if [[ ! -f "$PROJECT_ROOT/Cargo.toml" ]]; then
|
|
echo "Error: Cannot find Cargo.toml at $PROJECT_ROOT" >&2
|
|
exit 1
|
|
fi
|
|
|
|
cd "$PROJECT_ROOT" || exit 1
|
|
|
|
# ── Configuration ──────────────────────────────────────────────────────────
|
|
|
|
PARALLEL_JOBS=4
|
|
BUILD_ONLY=false
|
|
TEST_ONLY=false
|
|
SKIP_INTEGRATION=false
|
|
SKIP_CHAOS=false
|
|
WITH_TOR=false
|
|
ONLY_SUITE=""
|
|
|
|
# All integration suites matching ci.yml
|
|
STATIC_SUITES=(static-mesh static-chain)
|
|
# Each entry: "display-name scenario [--flag value ...]"
|
|
CHAOS_SUITES=(
|
|
"churn-mixed-10 churn-mixed --nodes 10 --duration 120"
|
|
"ethernet-mesh ethernet-mesh"
|
|
"ethernet-only ethernet-only"
|
|
"ethernet-churn ethernet-churn"
|
|
"tcp-mesh tcp-mesh"
|
|
"congestion-stress congestion-stress"
|
|
)
|
|
# Scenarios retired 2026-07-23 because their subject was a pure decision the
|
|
# Docker harness could not test reliably, now covered by sans-IO unit tests:
|
|
# - cost-reeval, cost-avoidance, cost-stability, depth-vs-cost,
|
|
# mixed-technology, bottleneck-parent: the evaluate_parent() cost/depth/
|
|
# hysteresis decision (root election, MMP measurement lag and hold-down
|
|
# timing all confounded the assertions) -> src/tree/tests.rs.
|
|
# - congestion-drops (never landed): the SO_RXQ_OVFL detection edge. A
|
|
# fresh daemon reader keeps up with container-speed traffic, so the
|
|
# socket overflow could not be provoked deterministically; the FIPS
|
|
# detection logic is now unit-tested in src/node/tests/unit.rs.
|
|
# congestion-stress stays: it exercises the ECN/MMP congestion signals,
|
|
# which do need the real shaped bottleneck queue.
|
|
#
|
|
# Retired 2026-07-24 for a different reason — redundant, not unreliable:
|
|
# - smoke-10: a no-stressor 10-node tree-convergence sanity check (netem
|
|
# off, no ping). The convergence logic it exercised is covered in-process,
|
|
# faster and deterministically, by the loopback spanning-tree harness
|
|
# (src/node/tests/spanning_tree.rs: ring/star/chain/100-node/disconnected)
|
|
# plus end-to-end datagram delivery (src/node/tests/forwarding.rs). Real-
|
|
# UDP convergence smoke still runs via static-mesh and the other chaos
|
|
# scenarios' baseline assertions, so no Docker coverage is lost.
|
|
#
|
|
# Retired 2026-07-26 for a third reason — the guard's power was never
|
|
# established, and unlike the two blocks above this one DOES lose coverage:
|
|
# - bloom-storm: guarded the regression in 0caef2a (fixed by 4cdf382), where
|
|
# a mid-chain tree update changing neither root nor depth leaked downstream
|
|
# as a sustained bloom announce storm. It was never once run against that
|
|
# regressed binary; its ceiling was inferred from a separate post-mortem
|
|
# harness that no longer exists in the tree. On the only regressed
|
|
# measurement that survives, the tail node's rate scales to ~7 sends per
|
|
# 30s, well under the scenario's ceiling of 40 — so it is not established
|
|
# that the assertion could ever have fired on its own bug class. The
|
|
# ceiling is also uniform per node, calibrated against the flap target
|
|
# (legitimately ~24) while the storm's actual signature is at the tail,
|
|
# which sits at 0 on fixed code. COVERAGE GAP: nothing now exercises
|
|
# downstream containment of a mid-chain ancestor swap. The scenario, its
|
|
# README, the link_swap sim primitive and the mesh-lab dispatch all stay
|
|
# on disk and it remains runnable by hand via
|
|
# testing/chaos/scripts/chaos.sh bloom-storm.
|
|
GATEWAY_SUITES=(gateway)
|
|
OPENWRT_SUITES=(openwrt-scripts)
|
|
SIDECAR_SUITES=(sidecar)
|
|
FIREWALL_SUITES=(firewall)
|
|
IFACE_BINDING_SUITES=(iface-binding)
|
|
NAT_SUITES=(cone symmetric lan)
|
|
NOSTR_RELAY_SUITES=(nostr-publish-consume)
|
|
STUN_FAULTS_SUITES=(stun-faults)
|
|
DNS_RESOLVER_SUITES=(dns-resolver)
|
|
NATIVE_API_SUITES=(native-api)
|
|
MEDIUM_CHANGE_SUITES=(medium-change)
|
|
DEB_INSTALL_SUITES=(deb-install)
|
|
TOR_SUITES=(tor-socks5 tor-directory)
|
|
|
|
# ── Colors ─────────────────────────────────────────────────────────────────
|
|
|
|
RED='\033[0;31m'
|
|
GREEN='\033[0;32m'
|
|
YELLOW='\033[1;33m'
|
|
CYAN='\033[0;36m'
|
|
BOLD='\033[1m'
|
|
RESET='\033[0m'
|
|
|
|
# ── Helpers ────────────────────────────────────────────────────────────────
|
|
|
|
stamp() { date '+%H:%M:%S'; }
|
|
|
|
info() { echo -e "${CYAN}[$(stamp)]${RESET} $*"; }
|
|
pass() { echo -e "${GREEN}[$(stamp)] PASS${RESET} $*"; }
|
|
fail() { echo -e "${RED}[$(stamp)] FAIL${RESET} $*"; }
|
|
stage() { echo -e "\n${BOLD}${YELLOW}═══ $* ═══${RESET}\n"; }
|
|
|
|
list_suites() {
|
|
echo "Available integration suites:"
|
|
echo ""
|
|
echo " Static topologies:"
|
|
for s in "${STATIC_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " Gateway:"
|
|
for s in "${GATEWAY_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " OpenWrt packaging:"
|
|
for s in "${OPENWRT_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " Firewall baseline:"
|
|
for s in "${FIREWALL_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " Dynamic interface binding:"
|
|
for s in "${IFACE_BINDING_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " NAT scenarios:"
|
|
for s in "${NAT_SUITES[@]}"; do echo " nat-$s"; done
|
|
echo ""
|
|
echo " Nostr publish/consume:"
|
|
for s in "${NOSTR_RELAY_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " STUN fault-injection:"
|
|
for s in "${STUN_FAULTS_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " Chaos scenarios:"
|
|
for entry in "${CHAOS_SUITES[@]}"; do
|
|
read -ra parts <<< "$entry"
|
|
echo " chaos-${parts[0]} (${parts[*]:1})"
|
|
done
|
|
echo ""
|
|
echo " Sidecar:"
|
|
for s in "${SIDECAR_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " Native API:"
|
|
for s in "${NATIVE_API_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " Medium change:"
|
|
for s in "${MEDIUM_CHANGE_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " DNS resolver:"
|
|
for s in "${DNS_RESOLVER_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " Deb-install:"
|
|
for s in "${DEB_INSTALL_SUITES[@]}"; do echo " $s"; done
|
|
echo ""
|
|
echo " Tor (opt-in via --with-tor):"
|
|
for s in "${TOR_SUITES[@]}"; do echo " $s"; done
|
|
exit 0
|
|
}
|
|
|
|
usage() {
|
|
sed -n '2,/^$/{ s/^# \?//; p }' "$0"
|
|
exit 0
|
|
}
|
|
|
|
# ── Parse arguments ────────────────────────────────────────────────────────
|
|
|
|
while [[ $# -gt 0 ]]; do
|
|
case "$1" in
|
|
--build-only) BUILD_ONLY=true; shift ;;
|
|
--test-only) TEST_ONLY=true; shift ;;
|
|
--skip-integration) SKIP_INTEGRATION=true; shift ;;
|
|
--skip-chaos) SKIP_CHAOS=true; shift ;;
|
|
--with-tor) WITH_TOR=true; shift ;;
|
|
--only) ONLY_SUITE="$2"; shift 2 ;;
|
|
-j|--jobs) PARALLEL_JOBS="$2"; shift 2 ;;
|
|
--list) list_suites ;;
|
|
--check-parity) exec "$SCRIPT_DIR/check-ci-parity.sh" ;;
|
|
--reap) exec "$SCRIPT_DIR/ci-cleanup.sh" ;;
|
|
-h|--help) usage ;;
|
|
*) echo "Unknown option: $1"; usage ;;
|
|
esac
|
|
done
|
|
|
|
# ── Results tracking ──────────────────────────────────────────────────────
|
|
|
|
declare -A RESULTS
|
|
OVERALL=0
|
|
|
|
record() {
|
|
local name="$1" rc="$2"
|
|
RESULTS["$name"]=$rc
|
|
# A chaos suite's window is written by run_chaos, which runs once per
|
|
# scenario; record() runs twice for a parallel one (child and parent).
|
|
[[ "$name" == chaos-* ]] || load_mark end "$name"
|
|
if [[ $rc -ne 0 ]]; then
|
|
OVERALL=1
|
|
fail "$name"
|
|
else
|
|
pass "$name"
|
|
fi
|
|
}
|
|
|
|
# ── Per-run isolation + signal-safe teardown ───────────────────────────────
|
|
#
|
|
# This script may be preempted (a CI worker sends SIGTERM, waits ~30s, then
|
|
# SIGKILL) so it can restart on a newer tip. To make that safe:
|
|
# * every docker resource is namespaced to THIS run (compose project prefix,
|
|
# per-run image tags, per-run build context) so a restart never collides
|
|
# with a dying run, and neither does a concurrent one;
|
|
# * a trap tears down everything this run created on signal/exit, bounded by
|
|
# `timeout` so a stuck `down` cannot wedge the trap (SIGKILL is the backstop).
|
|
|
|
# Derive a run id: honor the worker's $FIPS_CI_RUN_ID, else <short-sha>-<rand>,
|
|
# else $$-<timestamp>. Sanitize to a valid compose project / image-tag token.
|
|
if [[ -n "${FIPS_CI_RUN_ID:-}" ]]; then
|
|
CI_RUN_ID="$FIPS_CI_RUN_ID"
|
|
else
|
|
_ci_sha="$(git -C "$PROJECT_ROOT" rev-parse --short HEAD 2>/dev/null || true)"
|
|
if [[ -n "$_ci_sha" ]]; then
|
|
CI_RUN_ID="${_ci_sha}-${RANDOM}${RANDOM}"
|
|
else
|
|
CI_RUN_ID="$$-$(date +%s)"
|
|
fi
|
|
fi
|
|
CI_RUN_ID="$(printf '%s' "$CI_RUN_ID" | tr '[:upper:]' '[:lower:]' | tr -c 'a-z0-9_-' '-')"
|
|
# A docker image tag and a compose project name must both START with an
|
|
# alphanumeric, so strip any leading -/_ left by sanitization.
|
|
CI_RUN_ID="$(printf '%s' "$CI_RUN_ID" | sed -E 's/^[^a-z0-9]+//')"
|
|
[[ -z "$CI_RUN_ID" ]] && CI_RUN_ID="r$$"
|
|
|
|
CI_PROJECT_PREFIX="fipsci_${CI_RUN_ID}"
|
|
CI_IMAGE_TEST="fips-test:${CI_RUN_ID}"
|
|
CI_IMAGE_APP="fips-test-app:${CI_RUN_ID}"
|
|
CI_LABEL="com.corganlabs.fips-ci=1"
|
|
# Run-scoped companion to CI_LABEL. The generic label is shared by every run on
|
|
# the host, so teardown filters on this one instead — otherwise this run's reap
|
|
# would force-remove a concurrent run's containers out from under it.
|
|
CI_LABEL_RUN="com.corganlabs.fips-ci.run=${CI_RUN_ID}"
|
|
|
|
# Exported so child suite scripts + their compose/`docker run` invocations
|
|
# inherit the run identity. Compose files read FIPS_TEST_IMAGE/FIPS_TEST_APP_IMAGE
|
|
# (default :latest when unset, preserving manual `docker compose` use).
|
|
export FIPS_CI_RUN_ID="$CI_RUN_ID"
|
|
export FIPS_TEST_IMAGE="$CI_IMAGE_TEST"
|
|
export FIPS_TEST_APP_IMAGE="$CI_IMAGE_APP"
|
|
# The build context is this run's too. testing/docker/ is a single directory in
|
|
# the working tree, so two runs racing on its CONTENTS produce a per-run-tagged
|
|
# image built from the other run's binaries — scoping the tag alone does not
|
|
# close that. Absolute, and it has to be: compose interpolates this into
|
|
# build.context but resolves a relative result against the compose FILE's
|
|
# directory, not the working directory.
|
|
CI_BUILD_CONTEXT="$SCRIPT_DIR/docker-${CI_RUN_ID}"
|
|
export FIPS_BUILD_CONTEXT="$CI_BUILD_CONTEXT"
|
|
# Docker container names are GLOBAL — a compose project name does not scope
|
|
# them — so the suite compose files append this suffix to every explicit
|
|
# container_name, and the suite scripts append it wherever they address a
|
|
# container by name. Empty when unset, so a bare `docker compose up` outside
|
|
# this harness still produces today's plain names.
|
|
export FIPS_CI_NAME_SUFFIX="-${CI_RUN_ID}"
|
|
# run_chaos narrows that export to one scenario, and bash scopes a `local`
|
|
# dynamically: a function called while run_chaos is on the stack — the signal
|
|
# trap, on the `--only chaos-*` path where run_chaos is not a subshell — reads
|
|
# the narrowed value, not this one. Snapshot the run-wide value so the rebuild
|
|
# below always sees it, and make it readonly so no scope can shadow it either.
|
|
CI_RUN_NAME_SUFFIX="$FIPS_CI_NAME_SUFFIX"
|
|
readonly CI_RUN_NAME_SUFFIX
|
|
|
|
# Per-suite compose project name: ${prefix}_<suite-or-compose-basename>. Keeps
|
|
# today's intra-run distinctness (one project per compose file / chaos child)
|
|
# while adding the cross-run prefix that scopes the reap.
|
|
ci_project() { printf '%s_%s' "$CI_PROJECT_PREFIX" "$1"; }
|
|
|
|
# ── Host load samples ─────────────────────────────────────────────────────
|
|
#
|
|
# Stall percent at or above which a failed suite is labelled LOAD-DEGRADED.
|
|
# Measured 2026-09-19 on core-vm (12 CPUs): green chaos sets with no foreign
|
|
# load ran at 4-11% CPU stall, and the one reproduced load-sensitive red
|
|
# (churn-mixed-10 answering 9 of 10) ran at 63%, under 24 CPU hogs plus I/O.
|
|
CI_LOAD_THRESHOLD=15
|
|
if [[ -n "${FIPS_CI_LOAD_LOG:-}" ]]; then
|
|
CI_LOAD_LOG="$FIPS_CI_LOAD_LOG"
|
|
CI_LOAD_LOG_OWNED=0
|
|
else
|
|
CI_LOAD_LOG="${TMPDIR:-/tmp}/fips-ci-load-${CI_RUN_ID}.log"
|
|
CI_LOAD_LOG_OWNED=1
|
|
fi
|
|
CI_LOAD_PID=""
|
|
|
|
# Append one window line (begin/end <suite>, or barrier) to the load log.
|
|
load_mark() { echo "$(date +%s.%N) $*" >> "$CI_LOAD_LOG" 2>/dev/null || true; }
|
|
|
|
# Sample CPU pressure and foreign CI runs every 5 s until killed.
|
|
load_sampler() {
|
|
local psi runs
|
|
load_mark barrier
|
|
while true; do
|
|
psi="$(awk '/^some/ { for (i = 1; i <= NF; i++) if ($i ~ /^total=/) { sub("total=", "", $i); print $i } }' /proc/pressure/cpu 2>/dev/null)"
|
|
runs="$(docker ps --filter "label=$CI_LABEL" --format '{{.Label "com.corganlabs.fips-ci.run"}}' 2>/dev/null \
|
|
| sort -u | grep -cvx -e "$CI_RUN_ID" -e '')"
|
|
echo "$(date +%s.%N) psi ${psi:-none} runs ${runs:-0} load1 $(cut -d' ' -f1 /proc/loadavg)" >> "$CI_LOAD_LOG"
|
|
sleep 5
|
|
done
|
|
}
|
|
|
|
# The name suffix one chaos scenario runs under. Its container names, its
|
|
# generated-config directory and the token in its host veth names all derive
|
|
# from this, so teardown recomputes it to know which interfaces are ours. Reads
|
|
# the snapshot rather than the export for the reason given above.
|
|
ci_chaos_suffix() { printf -- '-%s%s' "$1" "$CI_RUN_NAME_SUFFIX"; }
|
|
|
|
# PIDs of in-flight parallel chaos children (subshells). The trap signals these.
|
|
CI_CHAOS_PIDS=()
|
|
CI_CLEANED=0
|
|
|
|
# Best-effort, BOUNDED teardown of every docker resource THIS run may have
|
|
# created. Idempotent (guarded), so the signal and EXIT paths don't double-run.
|
|
# Takes the run's exit status; defaults to non-zero, which is the conservative
|
|
# reading for the signal path.
|
|
ci_teardown() {
|
|
[[ $CI_CLEANED -eq 1 ]] && return 0
|
|
CI_CLEANED=1
|
|
local run_status="${1:-1}"
|
|
|
|
# 0. The load sampler first, so it stops writing before its log goes.
|
|
if [[ -n "$CI_LOAD_PID" ]]; then
|
|
kill "$CI_LOAD_PID" 2>/dev/null || true
|
|
wait "$CI_LOAD_PID" 2>/dev/null || true
|
|
fi
|
|
[[ "$CI_LOAD_LOG_OWNED" -eq 1 ]] && rm -f "$CI_LOAD_LOG"
|
|
|
|
# 1. Propagate to parallel chaos children and reap them (bounded).
|
|
if [[ ${#CI_CHAOS_PIDS[@]} -gt 0 ]]; then
|
|
kill -TERM "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true
|
|
local _end=$(( SECONDS + 10 )) _p
|
|
for _p in "${CI_CHAOS_PIDS[@]}"; do
|
|
while kill -0 "$_p" 2>/dev/null && (( SECONDS < _end )); do
|
|
sleep 0.3
|
|
done
|
|
done
|
|
kill -KILL "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true
|
|
wait "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true
|
|
fi
|
|
|
|
# 2. Remove all compose projects + direct-run resources + per-run images
|
|
# for this run, plus any host veth interface a chaos scenario was
|
|
# killed part-way through creating. Host interfaces carry no docker
|
|
# label, so hand over the suffixes this run's scenarios used and let
|
|
# the reap derive their names — a blind sweep would take a concurrent
|
|
# run's live interfaces with it. ci-cleanup.sh wraps each docker op in
|
|
# `timeout`; bound the whole sweep too so the trap can never wedge.
|
|
# Its stdout is routine progress and goes nowhere, but stderr carries
|
|
# only a skipped sweep or a bad option, and a sweep that quietly stops
|
|
# reaping is how interfaces would accumulate unnoticed. Let it through.
|
|
local _suffixes=() _entry
|
|
for _entry in "${CHAOS_SUITES[@]}"; do
|
|
_suffixes+=("$(ci_chaos_suffix "${_entry%% *}")")
|
|
done
|
|
# The NAT lab's host veths (vn{a,b}{token}{0,1}) carry the RUN-wide
|
|
# suffix, not any chaos scenario's, so its token appears in no suffix
|
|
# above and the reaper's NAT shape would match nothing at all. This is
|
|
# the half of that fix that fails silently: widening the regex in
|
|
# ci-cleanup.sh without this line looks correct and reaps nothing.
|
|
_suffixes+=("$CI_RUN_NAME_SUFFIX")
|
|
timeout 150 bash "$SCRIPT_DIR/ci-cleanup.sh" \
|
|
--label "$CI_LABEL" \
|
|
--run-id "$CI_RUN_ID" \
|
|
--project-prefix "$CI_PROJECT_PREFIX" \
|
|
--images "$CI_IMAGE_TEST $CI_IMAGE_APP" \
|
|
--veth-suffixes "${_suffixes[*]}" >/dev/null || true
|
|
|
|
# 3. The static, firewall and nat suites' generated configs are per-run (a
|
|
# shared directory would let concurrent runs overwrite each other's node
|
|
# configs), so they are this run's to remove. Only on a green run: after
|
|
# a failure they are the evidence of what the failing nodes were actually
|
|
# configured with. Guarded on a non-empty suffix too, since without one
|
|
# the path is the unscoped working directory a developer uses by hand,
|
|
# which is not ours to delete.
|
|
#
|
|
# Every suite whose configs become per-run owes a line here. acl-allowlist
|
|
# also generates per-run now but is not in this runner's suite list, so
|
|
# this teardown never creates its directory and must not remove one.
|
|
if [[ $run_status -eq 0 && -n "${CI_RUN_NAME_SUFFIX:-}" ]]; then
|
|
rm -rf "$SCRIPT_DIR/static/generated-configs${CI_RUN_NAME_SUFFIX}"
|
|
rm -rf "$SCRIPT_DIR/firewall/generated-configs${CI_RUN_NAME_SUFFIX}"
|
|
rm -rf "$SCRIPT_DIR/nat/generated-configs${CI_RUN_NAME_SUFFIX}"
|
|
fi
|
|
|
|
# 4. This run's build context. Unlike the generated configs it is removed
|
|
# on a red run too: it holds the binaries and the Dockerfiles, both
|
|
# reproducible from the commit, so it is never the evidence of a
|
|
# failure. Guarded on the path having been derived at all, so an early
|
|
# exit cannot turn this into `rm -rf $SCRIPT_DIR/docker-`.
|
|
if [[ -n "${CI_BUILD_CONTEXT:-}" && "$CI_BUILD_CONTEXT" != "$SCRIPT_DIR/docker-" ]]; then
|
|
rm -rf "$CI_BUILD_CONTEXT"
|
|
fi
|
|
}
|
|
|
|
on_signal() {
|
|
local sig="$1"
|
|
# Block re-entry and stop the EXIT trap from overriding the signal code.
|
|
trap '' TERM INT EXIT
|
|
echo "" >&2
|
|
fail "Received SIG$sig — cancelling run, tearing down ${CI_PROJECT_PREFIX}"
|
|
ci_teardown
|
|
# 128 + signal number: distinct from 0 (green) / 1 (stage failed).
|
|
if [[ "$sig" == "TERM" ]]; then exit 143; else exit 130; fi
|
|
}
|
|
|
|
on_exit() {
|
|
local code=$?
|
|
trap '' TERM INT EXIT
|
|
ci_teardown "$code"
|
|
exit "$code"
|
|
}
|
|
|
|
trap 'on_signal TERM' TERM
|
|
trap 'on_signal INT' INT
|
|
trap 'on_exit' EXIT
|
|
|
|
# ── Stage 1: Build ─────────────────────────────────────────────────────────
|
|
|
|
run_build() {
|
|
stage "Stage 1: Build"
|
|
|
|
info "sudo nft -c -f packaging/common/fips.nft (nftables ruleset syntax check)"
|
|
if command -v nft &>/dev/null; then
|
|
if sudo nft -c -f packaging/common/fips.nft 2>&1; then
|
|
record "nft-syntax" 0
|
|
else
|
|
record "nft-syntax" 1
|
|
return 1
|
|
fi
|
|
else
|
|
info "nftables not installed; install with 'apt install nftables' to validate fips.nft"
|
|
record "nft-syntax" 1
|
|
return 1
|
|
fi
|
|
|
|
info "cargo build --release --bins --examples"
|
|
# --bins --examples rather than the bare default: the native datagram API's
|
|
# echo server is a cargo example, and the native-api harness's image needs
|
|
# it built here rather than separately. Naming --bins keeps the daemon and
|
|
# its tools in the build, which --examples alone would drop.
|
|
if cargo build --release --bins --examples 2>&1; then
|
|
record "build" 0
|
|
else
|
|
record "build" 1
|
|
return 1
|
|
fi
|
|
|
|
info "cargo fmt --check"
|
|
if cargo fmt --check 2>&1; then
|
|
record "fmt" 0
|
|
else
|
|
record "fmt" 1
|
|
return 1
|
|
fi
|
|
|
|
info "cargo clippy --all-targets --all-features -- -D warnings"
|
|
if cargo clippy --all-targets --all-features -- -D warnings 2>&1; then
|
|
record "clippy" 0
|
|
else
|
|
record "clippy" 1
|
|
return 1
|
|
fi
|
|
|
|
# An optional feature means two source trees, and --all-features lints only
|
|
# one of them. The default build is what ships, so lint it explicitly:
|
|
# without this stage, code that compiles only with `profiling` enabled would
|
|
# pass CI while breaking every release build.
|
|
info "cargo clippy --all-targets -- -D warnings (default features)"
|
|
if cargo clippy --all-targets -- -D warnings 2>&1; then
|
|
record "clippy-default-features" 0
|
|
else
|
|
record "clippy-default-features" 1
|
|
return 1
|
|
fi
|
|
|
|
# Tick-body profiler: the feature-on tree must build too. Added alongside
|
|
# the default-feature stages rather than replacing them. Mirrored in
|
|
# .github/workflows/ci.yml — check-ci-parity.sh compares integration suites
|
|
# only and will not catch a stage added to one runner and not the other.
|
|
info "cargo build --workspace --features profiling"
|
|
if cargo build --workspace --features profiling 2>&1; then
|
|
record "build-profiling" 0
|
|
else
|
|
record "build-profiling" 1
|
|
return 1
|
|
fi
|
|
|
|
# Guard: the effectively-immutable state lives solely in NodeContext. The
|
|
# Node struct must not re-declare a bundled field (config/identity/
|
|
# startup_epoch/started_at/is_leaf_only/node_profile/max_*) — a shadow field
|
|
# would silently reopen dual-store divergence between the struct and the
|
|
# context. Checks the struct *declaration*, so it is wrap-insensitive.
|
|
info "node-context single-store guard"
|
|
if awk '/^pub struct Node \{/,/^\}/' src/node/mod.rs \
|
|
| grep -qE '^[[:space:]]+(config|identity|startup_epoch|started_at|is_leaf_only|node_profile|max_connections|max_peers|max_links):'; then
|
|
fail "Node struct re-declares a bundled immutable field; it must live solely in NodeContext"
|
|
record "node-context-guard" 1
|
|
return 1
|
|
else
|
|
record "node-context-guard" 0
|
|
fi
|
|
}
|
|
|
|
# ── Stage 2: Unit Tests ───────────────────────────────────────────────────
|
|
|
|
run_tests() {
|
|
stage "Stage 2: Unit Tests"
|
|
|
|
local cmd
|
|
if command -v cargo-nextest &>/dev/null; then
|
|
cmd="cargo nextest run --all"
|
|
info "$cmd"
|
|
if $cmd 2>&1; then
|
|
record "unit-tests" 0
|
|
else
|
|
record "unit-tests" 1
|
|
fi
|
|
else
|
|
cmd="cargo test --all"
|
|
info "$cmd (nextest not found, using cargo test)"
|
|
if $cmd 2>&1; then
|
|
record "unit-tests" 0
|
|
else
|
|
record "unit-tests" 1
|
|
fi
|
|
fi
|
|
|
|
# The `profiling` feature adds a module, a recorder and a writer thread that
|
|
# the default-feature run above never compiles. Run the library tests once
|
|
# more with it on so its own tests execute at all. Mirrored in
|
|
# .github/workflows/ci.yml.
|
|
info "cargo test --lib --features profiling"
|
|
if cargo test --lib --features profiling 2>&1; then
|
|
record "unit-tests-profiling" 0
|
|
else
|
|
record "unit-tests-profiling" 1
|
|
fi
|
|
|
|
# Debug-only helpers (anything behind #[cfg(debug_assertions)]) vanish in a
|
|
# release build, so a test calling one without the same gate breaks a build
|
|
# nothing here ever performs: every run above compiles the test target in
|
|
# debug. Compile it in release too, without running it — the point is that
|
|
# it builds at all. Mirrored in .github/workflows/ci.yml; check-ci-parity.sh
|
|
# compares integration suites only and would not catch a stage added to one
|
|
# runner and not the other.
|
|
info "cargo test --release --lib --no-run"
|
|
if cargo test --release --lib --no-run 2>&1; then
|
|
record "release-test-compile" 0
|
|
else
|
|
record "release-test-compile" 1
|
|
fi
|
|
}
|
|
|
|
# ── Stage 3: Integration Tests ─────────────────────────────────────────────
|
|
|
|
# Copy release binaries into a testing subdirectory
|
|
install_binaries() {
|
|
local dest="$1"
|
|
cp target/release/fips "$dest/fips"
|
|
cp target/release/fipsctl "$dest/fipsctl"
|
|
[[ -f target/release/fipstop ]] && cp target/release/fipstop "$dest/fipstop" || true
|
|
[[ -f target/release/fips-gateway ]] && cp target/release/fips-gateway "$dest/fips-gateway" || true
|
|
# Not optional: the native-api harness runs native-echo as one end of its
|
|
# echo check and native-surface as its surface walk, so a missing one must
|
|
# fail the image build rather than fail a check later with the container
|
|
# exiting on a missing entrypoint.
|
|
cp target/release/examples/native-echo "$dest/native-echo"
|
|
cp target/release/examples/native-surface "$dest/native-surface"
|
|
chmod +x "$dest/fips" "$dest/fipsctl" "$dest/native-echo" "$dest/native-surface"
|
|
[[ -f "$dest/fipstop" ]] && chmod +x "$dest/fipstop" || true
|
|
[[ -f "$dest/fips-gateway" ]] && chmod +x "$dest/fips-gateway" || true
|
|
}
|
|
|
|
# Run a static topology test (mesh, chain)
|
|
run_static() {
|
|
local topology="$1"
|
|
local compose="testing/static/docker-compose.yml"
|
|
local rc=0
|
|
export COMPOSE_PROJECT_NAME="$(ci_project static)"
|
|
|
|
info "[$topology] Generating configs"
|
|
bash testing/static/scripts/generate-configs.sh "$topology" || { record "static-$topology" 1; return; }
|
|
|
|
info "[$topology] Starting containers"
|
|
docker compose -f "$compose" --profile "$topology" up -d || { record "static-$topology" 1; return; }
|
|
|
|
info "[$topology] Running ping test"
|
|
if bash testing/static/scripts/ping-test.sh "$topology"; then
|
|
rc=0
|
|
else
|
|
rc=1
|
|
info "[$topology] Collecting failure logs"
|
|
docker compose -f "$compose" --profile "$topology" logs --no-color 2>&1 | tail -100
|
|
fi
|
|
|
|
docker compose -f "$compose" --profile "$topology" down --volumes --remove-orphans 2>/dev/null
|
|
record "static-$topology" $rc
|
|
}
|
|
|
|
# Lines kept from each node's log when a red scenario's results are printed.
|
|
CHAOS_DUMP_LINES=300
|
|
|
|
# Print a red chaos scenario's results into the run log.
|
|
#
|
|
# A CI worker that runs each job in a worktree it deletes afterwards, under a
|
|
# private /tmp, loses the results directory when the run ends, and the log is
|
|
# the only record that survives. Without this such a red cannot be traced node
|
|
# by node. The runner's own log is not repeated here: it already reached the run
|
|
# log as the scenario's output. Each node log is capped, so the total grows with
|
|
# node count; the largest CI scenario has ten nodes.
|
|
chaos_dump() {
|
|
local name="$1" base="$2" dir f
|
|
dir="$(find "$base" -mindepth 1 -maxdepth 1 -type d 2>/dev/null | sort | tail -n 1)"
|
|
if [[ -z "$dir" ]]; then
|
|
echo "[chaos/$name] no results directory under $base"
|
|
return 0
|
|
fi
|
|
echo "===== chaos-$name results from $dir ====="
|
|
for f in status.txt assertions.txt; do
|
|
[[ -f "$dir/$f" ]] || continue
|
|
echo "----- $f -----"
|
|
cat "$dir/$f"
|
|
done
|
|
if [[ -f "$dir/tree-snapshot-final.json" ]]; then
|
|
echo "----- final tree: node, root, parent -----"
|
|
python3 -c '
|
|
import json, sys
|
|
for node, tree in sorted(json.load(open(sys.argv[1])).items()):
|
|
print(node, tree.get("root"), tree.get("parent"))
|
|
' "$dir/tree-snapshot-final.json" || true
|
|
fi
|
|
for f in "$dir"/fips-node-*.log; do
|
|
[[ -f "$f" ]] || continue
|
|
echo "----- $(basename "$f"), last $CHAOS_DUMP_LINES lines -----"
|
|
tail -n "$CHAOS_DUMP_LINES" "$f"
|
|
done
|
|
echo "===== end of chaos-$name results ====="
|
|
return 0
|
|
}
|
|
|
|
# Run a chaos scenario
|
|
run_chaos() {
|
|
local name="$1"
|
|
shift
|
|
local rc=0
|
|
# Distinct project per scenario (chaos children run in parallel). Scoped to
|
|
# this function so the --only path cannot leak it into a later suite.
|
|
local -x COMPOSE_PROJECT_NAME="$(ci_project "chaos-$name")"
|
|
|
|
# Container names and the generated-config directory are GLOBAL and are not
|
|
# scoped by the compose project, so narrow the run-wide suffix to this
|
|
# scenario. Parallel children then cannot claim each other's names or
|
|
# overwrite each other's compose file.
|
|
local suffix
|
|
suffix="$(ci_chaos_suffix "$name")"
|
|
local -x FIPS_CI_NAME_SUFFIX="$suffix"
|
|
|
|
# The same scoping for the results, so a red can find the directory this
|
|
# run wrote rather than guess among earlier runs' timestamps.
|
|
local results="$SCRIPT_DIR/chaos/sim-results/ci$suffix"
|
|
local -x FIPS_SIM_OUTPUT="$results"
|
|
|
|
load_mark begin "chaos-$name"
|
|
info "[chaos/$name] Running simulation"
|
|
if bash testing/chaos/scripts/chaos.sh "$@" 2>&1; then
|
|
rc=0
|
|
else
|
|
rc=1
|
|
chaos_dump "$name" "$results"
|
|
fi
|
|
|
|
load_mark end "chaos-$name"
|
|
record "chaos-$name" $rc
|
|
|
|
# record() ends in pass()/fail(), which are echoes, so it returns 0 for any
|
|
# rc — and without this return so does run_chaos. On the parallel path the
|
|
# function's status is the child subshell's status, and that is the only
|
|
# thing the waiting parent can see: the child's own RESULTS entry and its
|
|
# OVERALL=1 die with its process, and its FAIL line goes to a logfile only
|
|
# the unreachable branch reads. So a bare record() left every scenario
|
|
# recorded as a pass whatever the simulation reported. On the --only path
|
|
# record() already wrote the true rc into the parent's own RESULTS, and
|
|
# returning it as well is inert: this script does not set -e.
|
|
return $rc
|
|
}
|
|
|
|
# Claim a free /64 for this run's gateway-lan network (Mechanism B).
|
|
#
|
|
# gateway-lan cannot float like fips-net: the LAN clients' resolv.conf pins the
|
|
# gateway's LAN address as their nameserver, which must be a literal known
|
|
# before they start. So instead of a fixed fd02::/64 that two concurrent runs
|
|
# both request (the second failing on "Pool overlaps"), claim-and-advance a
|
|
# candidate /64 and let docker's create be the atomic arbiter. The claimed
|
|
# prefix is threaded — via the exported FIPS_GW_LAN6_PREFIX — into the compose
|
|
# ipv6_address pins, the generated resolv.conf, and gateway-test.sh, so every
|
|
# LAN address moves with the claim. i=0 renders today's fd02::/64.
|
|
#
|
|
# Mirrors sidecar/scripts/test-sidecar.sh:alloc_network. Deliberately does NOT
|
|
# discard stderr: only an address-pool conflict is worth advancing on; any other
|
|
# failure is real and burning through 256 candidates would bury the reason.
|
|
claim_gateway_lan6() {
|
|
local net="$1" i err
|
|
for (( i = 0; i < 256; i++ )); do
|
|
if err=$(docker network create --ipv6 \
|
|
--subnet "fd02:0:0:${i}::/64" \
|
|
--label "$CI_LABEL" --label "$CI_LABEL_RUN" \
|
|
"$net" 2>&1); then
|
|
export FIPS_GW_LAN6_PREFIX="fd02:0:0:${i}"
|
|
info "[gateway] Claimed $net on fd02:0:0:${i}::/64"
|
|
return 0
|
|
fi
|
|
case "$err" in
|
|
*"Pool overlaps"*|*"pool overlaps"*) continue ;;
|
|
*) fail "[gateway] docker network create: $err"; return 1 ;;
|
|
esac
|
|
done
|
|
fail "[gateway] no free /64 in fd02:0:0:0-255::/64 after 256 attempts"
|
|
return 1
|
|
}
|
|
|
|
# Run gateway integration test
|
|
run_gateway() {
|
|
local compose="testing/static/docker-compose.yml"
|
|
local ext="testing/static/docker-compose.gateway-external-net.yml"
|
|
local rc=0
|
|
export COMPOSE_PROJECT_NAME="$(ci_project static)"
|
|
export FIPS_GW_LAN_NET="fips-gateway-lan${FIPS_CI_NAME_SUFFIX:-}"
|
|
|
|
# Claim the LAN /64 first: generate-configs writes resolv.conf from the
|
|
# claimed prefix, and inject-config renders the port-forward targets from
|
|
# it, so both must see FIPS_GW_LAN6_PREFIX before they run.
|
|
info "[gateway] Claiming LAN network"
|
|
if ! claim_gateway_lan6 "$FIPS_GW_LAN_NET"; then
|
|
record "gateway" 1
|
|
return
|
|
fi
|
|
|
|
info "[gateway] Generating configs"
|
|
if bash testing/static/scripts/generate-configs.sh gateway gateway-test \
|
|
&& bash testing/static/scripts/gateway-test.sh inject-config \
|
|
&& { info "[gateway] Starting containers"; \
|
|
docker compose -f "$compose" -f "$ext" --profile gateway up -d; }; then
|
|
info "[gateway] Running gateway test"
|
|
if bash testing/static/scripts/gateway-test.sh; then
|
|
rc=0
|
|
else
|
|
rc=1
|
|
info "[gateway] Collecting failure logs"
|
|
docker compose -f "$compose" -f "$ext" --profile gateway logs --no-color 2>&1 | tail -100
|
|
fi
|
|
else
|
|
rc=1
|
|
fi
|
|
|
|
# compose down does not remove an external network, so drop it explicitly.
|
|
docker compose -f "$compose" -f "$ext" --profile gateway down --volumes --remove-orphans 2>/dev/null
|
|
docker network rm "$FIPS_GW_LAN_NET" >/dev/null 2>&1 || true
|
|
record "gateway" $rc
|
|
}
|
|
|
|
# Run sidecar test
|
|
run_sidecar() {
|
|
local rc=0
|
|
export COMPOSE_PROJECT_NAME="$(ci_project sidecar)"
|
|
|
|
info "[sidecar] Running integration test"
|
|
if bash testing/sidecar/scripts/test-sidecar.sh --skip-build 2>&1; then
|
|
rc=0
|
|
else
|
|
rc=1
|
|
fi
|
|
|
|
record "sidecar" $rc
|
|
}
|
|
|
|
# Run the dynamic interface binding integration test.
|
|
#
|
|
# Creates a veth pair from the host namespace after the daemons are up, so it
|
|
# needs the same privileged ip(8) helper the chaos harness uses. Scoped by
|
|
# COMPOSE_PROJECT_NAME like every other suite; the veth names carry the run
|
|
# suffix themselves (see testing/iface-binding/test.sh).
|
|
run_iface_binding() {
|
|
export COMPOSE_PROJECT_NAME="$(ci_project iface_binding)"
|
|
info "[iface-binding] Running integration test"
|
|
if bash testing/iface-binding/test.sh --skip-build 2>&1; then
|
|
record "iface-binding" 0
|
|
else
|
|
record "iface-binding" 1
|
|
fi
|
|
}
|
|
|
|
# Run firewall baseline integration test
|
|
run_firewall() {
|
|
export COMPOSE_PROJECT_NAME="$(ci_project firewall)"
|
|
info "[firewall] Running integration test"
|
|
if bash testing/firewall/test.sh --skip-build 2>&1; then
|
|
record "firewall" 0
|
|
else
|
|
record "firewall" 1
|
|
fi
|
|
}
|
|
|
|
# ── NAT lab: per-run network claim ─────────────────────────────────────────
|
|
#
|
|
# The lab's two bridges pinned 172.31.254.0/24 (wan) and 172.31.10.0/24
|
|
# (shared-lan), so two concurrent runs asked the daemon for the same address
|
|
# space and the second died with `Pool overlaps`. Claim a free /24 for each
|
|
# instead and let docker's own create be the atomic arbiter, exactly as
|
|
# claim_gateway_lan6 and sidecar's alloc_network do. The run-id-derived offset
|
|
# this task considered earlier stays rejected: it makes a collision unlikely
|
|
# rather than impossible, and a collision is the failure being removed.
|
|
#
|
|
# 10.41.0.0/16 is unclaimed in-tree, sits below this host's docker pool
|
|
# (10.128.0.0/9, per /etc/docker/daemon.json), and is also outside docker's
|
|
# stock 172.17-31 / 192.168 default, so the choice holds either way.
|
|
#
|
|
# The claim lives here rather than in the suite scripts because
|
|
# .github/workflows/ci.yml invokes nat-test.sh, nostr-relay-test.sh and
|
|
# stun-faults-test.sh directly and testing/mesh-lab/run-loop.sh invokes
|
|
# `nat-test.sh lan`; none of those want a claim, and all tear down with the
|
|
# base compose file only.
|
|
#
|
|
# One claim per SUITE invocation, not per run — matching run_gateway. Both
|
|
# labels are stamped so ci-cleanup.sh's label sweep can recover the networks
|
|
# when a run is SIGKILLed, which no inline removal can cover.
|
|
CI_NAT_NET_BASE="10.41"
|
|
# 10.42.0.0/16 for the medium-change lab, on the same reasoning as 10.41 above
|
|
# and adjacent to it so the two stay legible as a pair. Its three bridges are
|
|
# claimed per suite invocation exactly as the nat pair are; before that they
|
|
# were three fixed /24s under 172.31.6x, which two concurrent runs both
|
|
# requested and the second lost with `Pool overlaps`.
|
|
CI_MC_NET_BASE="10.42"
|
|
CI_V4_NET_CANDIDATES=256
|
|
CI_NAT_CLAIMED_PREFIX=""
|
|
CI_CLAIMED_V4_PREFIX=""
|
|
|
|
# Claim one free /24 for `net` by walking candidates under `base` and letting
|
|
# docker's own create be the atomic arbiter. Sets CI_CLAIMED_V4_PREFIX.
|
|
#
|
|
# Deliberately does NOT discard stderr: only an address-pool conflict is worth
|
|
# advancing on. Any other failure is real, and burning through 256 candidates
|
|
# would bury the reason.
|
|
#
|
|
# `tag` is the suite word for the log lines only. Both labels are stamped so
|
|
# ci-cleanup.sh's label sweep can recover the network when a run is SIGKILLed,
|
|
# which no inline removal can cover.
|
|
ci_claim_v4_net() {
|
|
local net="$1" base="$2" tag="$3" i err
|
|
CI_CLAIMED_V4_PREFIX=""
|
|
for (( i = 0; i < CI_V4_NET_CANDIDATES; i++ )); do
|
|
if err=$(docker network create \
|
|
--subnet "${base}.${i}.0/24" \
|
|
--label "$CI_LABEL" --label "$CI_LABEL_RUN" \
|
|
"$net" 2>&1); then
|
|
CI_CLAIMED_V4_PREFIX="${base}.${i}"
|
|
info "[$tag] Claimed $net on ${CI_CLAIMED_V4_PREFIX}.0/24"
|
|
return 0
|
|
fi
|
|
case "$err" in
|
|
*"Pool overlaps"*|*"pool overlaps"*) continue ;;
|
|
*) fail "[$tag] docker network create: $err"; return 1 ;;
|
|
esac
|
|
done
|
|
fail "[$tag] no free /24 in ${base}.0.0/16 after ${CI_V4_NET_CANDIDATES} attempts"
|
|
return 1
|
|
}
|
|
|
|
ci_claim_nat_net() {
|
|
ci_claim_v4_net "$1" "$CI_NAT_NET_BASE" nat || return 1
|
|
CI_NAT_CLAIMED_PREFIX="$CI_CLAIMED_V4_PREFIX"
|
|
}
|
|
|
|
# Claim both lab networks and export the prefixes every lab address derives
|
|
# from. A partial claim is rolled back here, because nothing downstream will
|
|
# run to release it.
|
|
ci_claim_nat_networks() {
|
|
unset NAT_WAN_PREFIX NAT_LAN_PREFIX
|
|
export FIPS_NAT_WAN_NET="fips-nat-wan${FIPS_CI_NAME_SUFFIX:-}"
|
|
export FIPS_NAT_LAN_NET="fips-nat-shared-lan${FIPS_CI_NAME_SUFFIX:-}"
|
|
|
|
ci_claim_nat_net "$FIPS_NAT_WAN_NET" || return 1
|
|
local wan="$CI_NAT_CLAIMED_PREFIX"
|
|
if ! ci_claim_nat_net "$FIPS_NAT_LAN_NET"; then
|
|
docker network rm "$FIPS_NAT_WAN_NET" >/dev/null 2>&1 || true
|
|
return 1
|
|
fi
|
|
|
|
export NAT_WAN_PREFIX="$wan"
|
|
export NAT_LAN_PREFIX="$CI_NAT_CLAIMED_PREFIX"
|
|
}
|
|
|
|
# Release both claimed networks. A complete claim left behind would not merely
|
|
# leak: the next nat-family suite in this run would hit `network with name ...
|
|
# already exists`, which is not a pool overlap, so the allocator correctly
|
|
# refuses to advance and fails. One suite failure would cascade into four.
|
|
#
|
|
# Both halves of run_gateway's shape are needed, and the second alone is a trap.
|
|
# `docker network rm` refuses a network that still has endpoints attached, and
|
|
# the nat suites deliberately leave their containers UP on a failure path so the
|
|
# diagnostics they just dumped can be inspected — which is precisely the path
|
|
# this release exists for. Without the `down` the removal silently no-ops
|
|
# exactly when it matters and reports success. Nothing is lost by tearing down
|
|
# here: the diagnostics are already in the run log, and ci_teardown reaps the
|
|
# containers at the end of the run regardless.
|
|
ci_release_nat_networks() {
|
|
docker compose \
|
|
-f testing/nat/docker-compose.yml \
|
|
-f testing/nat/docker-compose.external-net.yml \
|
|
--profile cone --profile symmetric --profile lan \
|
|
--profile nostr-publish-consume --profile stun-faults \
|
|
down --volumes --remove-orphans >/dev/null 2>&1 || true
|
|
[[ -n "${FIPS_NAT_WAN_NET:-}" ]] && \
|
|
{ docker network rm "$FIPS_NAT_WAN_NET" >/dev/null 2>&1 || true; }
|
|
[[ -n "${FIPS_NAT_LAN_NET:-}" ]] && \
|
|
{ docker network rm "$FIPS_NAT_LAN_NET" >/dev/null 2>&1 || true; }
|
|
return 0
|
|
}
|
|
|
|
# Claim all three medium-change lab networks and export both the names the
|
|
# overlay attaches to and the prefixes every lab address derives from. A
|
|
# partial claim is rolled back here, because nothing downstream will run to
|
|
# release it.
|
|
#
|
|
# Three, not two: the lab's whole design rests on node-b being off-link, so the
|
|
# route to it follows node-a's default route. A mix of claimed and compose-made
|
|
# bridges would still come up, with the wrong topology and no error.
|
|
ci_claim_mc_networks() {
|
|
unset MC_PRIMARY_PREFIX MC_SECONDARY_PREFIX MC_FAR_PREFIX
|
|
export FIPS_MC_PRIMARY_NET="fips-mc-primary${FIPS_CI_NAME_SUFFIX:-}"
|
|
export FIPS_MC_SECONDARY_NET="fips-mc-secondary${FIPS_CI_NAME_SUFFIX:-}"
|
|
export FIPS_MC_FAR_NET="fips-mc-far${FIPS_CI_NAME_SUFFIX:-}"
|
|
|
|
ci_claim_v4_net "$FIPS_MC_PRIMARY_NET" "$CI_MC_NET_BASE" medium-change || return 1
|
|
local primary="$CI_CLAIMED_V4_PREFIX"
|
|
|
|
if ! ci_claim_v4_net "$FIPS_MC_SECONDARY_NET" "$CI_MC_NET_BASE" medium-change; then
|
|
docker network rm "$FIPS_MC_PRIMARY_NET" >/dev/null 2>&1 || true
|
|
return 1
|
|
fi
|
|
local secondary="$CI_CLAIMED_V4_PREFIX"
|
|
|
|
if ! ci_claim_v4_net "$FIPS_MC_FAR_NET" "$CI_MC_NET_BASE" medium-change; then
|
|
docker network rm "$FIPS_MC_SECONDARY_NET" >/dev/null 2>&1 || true
|
|
docker network rm "$FIPS_MC_PRIMARY_NET" >/dev/null 2>&1 || true
|
|
return 1
|
|
fi
|
|
|
|
export MC_PRIMARY_PREFIX="$primary"
|
|
export MC_SECONDARY_PREFIX="$secondary"
|
|
export MC_FAR_PREFIX="$CI_CLAIMED_V4_PREFIX"
|
|
}
|
|
|
|
# Release all three claimed networks. Left behind they would not merely leak:
|
|
# the next medium-change invocation in this run would hit `network with name
|
|
# ... already exists`, which is not a pool overlap, so the allocator correctly
|
|
# refuses to advance and fails.
|
|
#
|
|
# The `down` before the removals is load-bearing, not tidiness. `docker network
|
|
# rm` silently no-ops on a network that still has endpoints attached and
|
|
# reports success, and the suite leaves its containers up on some paths. Without
|
|
# the `down` the removal fails to remove exactly when it matters. The overlay is
|
|
# included so the `down` addresses the same project the `up` created.
|
|
ci_release_mc_networks() {
|
|
docker compose \
|
|
-f testing/medium-change/docker-compose.yml \
|
|
-f testing/medium-change/docker-compose.external-net.yml \
|
|
down --volumes --remove-orphans >/dev/null 2>&1 || true
|
|
local net
|
|
for net in "${FIPS_MC_PRIMARY_NET:-}" "${FIPS_MC_SECONDARY_NET:-}" \
|
|
"${FIPS_MC_FAR_NET:-}"; do
|
|
[[ -n "$net" ]] && { docker network rm "$net" >/dev/null 2>&1 || true; }
|
|
done
|
|
return 0
|
|
}
|
|
|
|
# The overlay pointing compose at the claimed networks. APPENDED to any
|
|
# caller-supplied chain rather than replacing it: mesh-lab/run-loop.sh sets
|
|
# FIPS_NAT_EXTRA_COMPOSE for its trace overlay, and overwriting would silently
|
|
# drop that.
|
|
ci_nat_extra_compose() {
|
|
printf '%s' \
|
|
"${FIPS_NAT_EXTRA_COMPOSE:+${FIPS_NAT_EXTRA_COMPOSE}:}testing/nat/docker-compose.external-net.yml"
|
|
}
|
|
|
|
# Run a NAT scenario (cone, symmetric, lan)
|
|
run_nat() {
|
|
local scenario="$1"
|
|
local rc=0
|
|
export COMPOSE_PROJECT_NAME="$(ci_project nat)"
|
|
|
|
# Early return only BEFORE the claim succeeds, so nothing can leak. Every
|
|
# path after it falls through to the single release below — run_gateway's
|
|
# shape. Not a trap: bash traps are not function-scoped, so a `trap ... EXIT`
|
|
# here would replace the script-level on_exit handler for the rest of the
|
|
# run and disable ci_teardown entirely.
|
|
info "[nat-$scenario] Claiming lab networks"
|
|
if ! ci_claim_nat_networks; then
|
|
record "nat-$scenario" 1
|
|
return
|
|
fi
|
|
|
|
local _extra
|
|
_extra="$(ci_nat_extra_compose)"
|
|
local -x FIPS_NAT_EXTRA_COMPOSE="$_extra"
|
|
|
|
info "[nat-$scenario] Running NAT lab"
|
|
bash testing/nat/scripts/nat-test.sh "$scenario" 2>&1 || rc=1
|
|
|
|
ci_release_nat_networks
|
|
record "nat-$scenario" $rc
|
|
}
|
|
|
|
# Run the Nostr overlay advert publish/consume integration test.
|
|
# Two FIPS daemons + the existing strfry relay; exercises Phase 1
|
|
# (A→B publish/consume), Phase 2 (B→A reverse), and Phase 3 (malformed
|
|
# advert injected directly to the relay; consumer-liveness assertion).
|
|
run_nostr_publish_consume() {
|
|
local rc=0
|
|
export COMPOSE_PROJECT_NAME="$(ci_project nat)"
|
|
|
|
info "[nostr-publish-consume] Claiming lab networks"
|
|
if ! ci_claim_nat_networks; then
|
|
record "nostr-publish-consume" 1
|
|
return
|
|
fi
|
|
|
|
local _extra
|
|
_extra="$(ci_nat_extra_compose)"
|
|
local -x FIPS_NAT_EXTRA_COMPOSE="$_extra"
|
|
|
|
info "[nostr-publish-consume] Running Nostr publish/consume test"
|
|
bash testing/nat/scripts/nostr-relay-test.sh 2>&1 || rc=1
|
|
|
|
ci_release_nat_networks
|
|
record "nostr-publish-consume" $rc
|
|
}
|
|
|
|
# Run the STUN fault-injection integration test.
|
|
# One FIPS daemon + a netns-sharing shim that injects tc/iptables faults
|
|
# against UDP egress to the STUN service. Three phases: drop, delay,
|
|
# kill. Asserts the daemon detects each fault, recovers from delay, and
|
|
# never panics.
|
|
run_stun_faults() {
|
|
local rc=0
|
|
export COMPOSE_PROJECT_NAME="$(ci_project nat)"
|
|
|
|
info "[stun-faults] Claiming lab networks"
|
|
if ! ci_claim_nat_networks; then
|
|
record "stun-faults" 1
|
|
return
|
|
fi
|
|
|
|
local _extra
|
|
_extra="$(ci_nat_extra_compose)"
|
|
local -x FIPS_NAT_EXTRA_COMPOSE="$_extra"
|
|
|
|
info "[stun-faults] Running STUN fault-injection test"
|
|
bash testing/nat/scripts/stun-faults-test.sh 2>&1 || rc=1
|
|
|
|
ci_release_nat_networks
|
|
record "stun-faults" $rc
|
|
}
|
|
|
|
# Run the native datagram API harness.
|
|
#
|
|
# Reads FIPS_TEST_IMAGE, so it exercises this run's binaries rather than a
|
|
# separately built one. Its two-node check creates and removes its own docker
|
|
# network, so nothing here has to.
|
|
run_native_api() {
|
|
info "[native-api] Running native datagram API test"
|
|
if FIPS_TEST_IMAGE="$CI_IMAGE_TEST" bash testing/native-api/test.sh 2>&1; then
|
|
record "native-api" 0
|
|
else
|
|
record "native-api" 1
|
|
fi
|
|
}
|
|
|
|
# Run the transport-medium change suite.
|
|
#
|
|
# Owns its own compose project and its own three bridges, so it neither
|
|
# shares container names with the NAT lab nor has to run after it.
|
|
run_medium_change() {
|
|
# Its own project, like every other suite. Without this the lab inherited
|
|
# whichever COMPOSE_PROJECT_NAME the previous suite exported, so its
|
|
# containers and networks were filed under that suite's project — visible
|
|
# in a failure as `fipsci_<runid>_nat_mc-far`, a medium-change network
|
|
# under the nat project.
|
|
local -x COMPOSE_PROJECT_NAME="$(ci_project medium-change)"
|
|
|
|
# Claim before generate-configs: the suite renders every address in the
|
|
# compose file, the node configs and its own assertions from these
|
|
# prefixes, so all of them must see the claimed values.
|
|
info "[medium-change] Claiming lab networks"
|
|
if ! ci_claim_mc_networks; then
|
|
record "medium-change" 1
|
|
return
|
|
fi
|
|
|
|
info "[medium-change] Running transport-medium change test"
|
|
if MC_EXTRA_COMPOSE="testing/medium-change/docker-compose.external-net.yml" \
|
|
FIPS_TEST_IMAGE="$CI_IMAGE_TEST" \
|
|
bash testing/medium-change/scripts/test.sh 2>&1; then
|
|
record "medium-change" 0
|
|
else
|
|
record "medium-change" 1
|
|
fi
|
|
ci_release_mc_networks
|
|
}
|
|
|
|
# Run dns-resolver harness (multi-distro + e2e scenarios)
|
|
#
|
|
# Its e2e scenarios run the fips binaries from the package build_ci_deb
|
|
# produces, the same package deb-install installs, as the GitHub job does.
|
|
run_dns_resolver() {
|
|
if ! build_ci_deb dns-resolver; then
|
|
record "dns-resolver" "$CI_DEB_RC"
|
|
return
|
|
fi
|
|
info "[dns-resolver] Running multi-distro test against $CI_DEB_PATH (slow — builds per-distro images)"
|
|
if bash testing/dns-resolver/test.sh --deb "$CI_DEB_PATH" 2>&1; then
|
|
record "dns-resolver" 0
|
|
else
|
|
record "dns-resolver" 1
|
|
fi
|
|
}
|
|
|
|
# Run deb-install harness (multi-distro real-package install)
|
|
#
|
|
# Bounded because this worker is what gates artifact publication: a suite that
|
|
# hangs stops every branch publishing, which is worse than a red. The harness
|
|
# bounds its own service starts now, so this is the backstop for anything else
|
|
# that wedges -- a docker daemon that stops answering, a container that never
|
|
# boots. 40 minutes is well above the observed cold-cache cost of a full
|
|
# five-distro run and is not a performance budget.
|
|
#
|
|
# It bounds the container build and the five installs separately rather than
|
|
# both together: the budget was sized for the installs, and a cold build that
|
|
# ate into it would shrink theirs. Neither phase is left unbounded, which is
|
|
# the property this exists for.
|
|
DEB_INSTALL_TIMEOUT=${DEB_INSTALL_TIMEOUT:-2400}
|
|
|
|
# Build the package once per process through the shared container script, for
|
|
# every suite that consumes it: deb-install installs it and dns-resolver runs
|
|
# its binaries. The GitHub workflow builds it once in its own job and hands the
|
|
# artifact to both, so doing the same here keeps the two systems the same work.
|
|
# The first caller builds; later callers get the stored result, a failure
|
|
# included, so a failed build is not retried and reported twice as two builds.
|
|
# Sets CI_DEB_PATH on success and CI_DEB_RC (the value to record) on failure.
|
|
CI_DEB_DONE=0
|
|
CI_DEB_PATH=""
|
|
CI_DEB_RC=0
|
|
build_ci_deb() {
|
|
local suite="$1"
|
|
if [[ $CI_DEB_DONE -eq 1 ]]; then
|
|
if [[ $CI_DEB_RC -ne 0 ]]; then
|
|
echo " ERROR: the package build this suite shares failed earlier in the run (exit $CI_DEB_RC)." >&2
|
|
return 1
|
|
fi
|
|
info "[$suite] Reusing the package built earlier in this run: $CI_DEB_PATH"
|
|
return 0
|
|
fi
|
|
CI_DEB_DONE=1
|
|
info "[$suite] Building the package in the pinned container"
|
|
local build_log deb rc=0
|
|
build_log=$(mktemp "/tmp/ci-deb-install-build.XXXXXX")
|
|
# stdout carries the package path on its last line, so it is captured;
|
|
# stderr stays on the console so build progress is still visible live.
|
|
timeout "$DEB_INSTALL_TIMEOUT" \
|
|
bash packaging/debian/build-deb-container.sh | tee "$build_log" || rc=$?
|
|
if [[ $rc -ne 0 ]]; then
|
|
if [[ $rc -eq 124 ]]; then
|
|
echo " ERROR: the container build exceeded ${DEB_INSTALL_TIMEOUT}s and was killed;" >&2
|
|
echo " no package was produced, so this is not an assertion failure." >&2
|
|
else
|
|
echo " ERROR: the container build failed (exit $rc); no package for $suite." >&2
|
|
fi
|
|
rm -f "$build_log"
|
|
CI_DEB_RC=$rc
|
|
return 1
|
|
fi
|
|
deb=$(tail -n 1 "$build_log")
|
|
rm -f "$build_log"
|
|
if [[ -z "$deb" || ! -f "$deb" ]]; then
|
|
echo " ERROR: the container build reported success but its last line of stdout" >&2
|
|
echo " was not a package path: '${deb}'" >&2
|
|
CI_DEB_RC=1
|
|
return 1
|
|
fi
|
|
CI_DEB_PATH="$deb"
|
|
return 0
|
|
}
|
|
|
|
run_deb_install() {
|
|
# Install the one package build_ci_deb built into every distro. The harness
|
|
# would build its own package if handed none, and that fallback goes through
|
|
# the same script -- but the GitHub job builds explicitly and passes
|
|
# `--deb`, so doing it explicitly here too makes the two systems read as the
|
|
# same work rather than leaving a reader to discover that a fallback happens
|
|
# to match. The build is shared with dns-resolver.
|
|
if ! build_ci_deb deb-install; then
|
|
record "deb-install" "$CI_DEB_RC"
|
|
return
|
|
fi
|
|
local deb="$CI_DEB_PATH" rc=0
|
|
info "[deb-install] Running multi-distro install test against $deb"
|
|
timeout "$DEB_INSTALL_TIMEOUT" bash testing/deb-install/test.sh --deb "$deb" 2>&1 || rc=$?
|
|
if [[ $rc -eq 124 ]]; then
|
|
# Say so explicitly. A bare red here reads as a failed assertion, and
|
|
# the difference matters: a timeout means no assertion was reached.
|
|
echo " ERROR: deb-install exceeded ${DEB_INSTALL_TIMEOUT}s and was killed;" >&2
|
|
echo " no verdict was reached, so this is not an assertion failure." >&2
|
|
fi
|
|
record "deb-install" "$rc"
|
|
}
|
|
|
|
# Run Tor SOCKS5 outbound test (live Tor network)
|
|
run_tor_socks5() {
|
|
export COMPOSE_PROJECT_NAME="$(ci_project tor-socks5)"
|
|
info "[tor-socks5] Running Tor SOCKS5 outbound test (live Tor)"
|
|
if bash testing/tor/socks5-outbound/scripts/tor-test.sh 2>&1; then
|
|
record "tor-socks5" 0
|
|
else
|
|
record "tor-socks5" 1
|
|
fi
|
|
}
|
|
|
|
# Run Tor directory-mode test (live Tor network)
|
|
run_tor_directory() {
|
|
export COMPOSE_PROJECT_NAME="$(ci_project tor-directory)"
|
|
info "[tor-directory] Running Tor directory-mode test (live Tor)"
|
|
if bash testing/tor/directory-mode/scripts/directory-test.sh 2>&1; then
|
|
record "tor-directory" 0
|
|
else
|
|
record "tor-directory" 1
|
|
fi
|
|
}
|
|
|
|
# Determine which suites to run and execute them
|
|
run_integration() {
|
|
stage "Stage 3: Integration Tests"
|
|
|
|
# First, and before the build context: the OpenWrt scenarios need no FIPS
|
|
# binary and no test image, so a packaging regression is reported in
|
|
# seconds rather than after the image build.
|
|
if [[ -z "$ONLY_SUITE" ]]; then
|
|
run_openwrt_scripts
|
|
elif [[ "$ONLY_SUITE" == "openwrt-scripts" ]]; then
|
|
run_openwrt_scripts
|
|
return
|
|
fi
|
|
|
|
# Populate THIS run's build context, then install the binaries into it.
|
|
# Everything but the binaries is copied from the tracked context directory;
|
|
# the binaries are installed fresh, and a previous run's are deliberately
|
|
# not carried over, since inheriting them is the failure this scoping
|
|
# exists to prevent.
|
|
info "Preparing build context $CI_BUILD_CONTEXT"
|
|
rm -rf "$CI_BUILD_CONTEXT"
|
|
mkdir -p "$CI_BUILD_CONTEXT" || { record "docker-build" 1; return; }
|
|
local _f
|
|
for _f in "$SCRIPT_DIR"/docker/*; do
|
|
case "$(basename "$_f")" in
|
|
fips|fipsctl|fipstop|fips-gateway|native-echo|native-surface) continue ;;
|
|
esac
|
|
cp -a "$_f" "$CI_BUILD_CONTEXT/" || { record "docker-build" 1; return; }
|
|
done
|
|
|
|
info "Installing release binaries"
|
|
install_binaries "$CI_BUILD_CONTEXT"
|
|
|
|
# Build unified test image once (used by all harnesses). Tag per-run
|
|
# (fips-test:${run}) so a build killed mid-flight never wedges the next
|
|
# run's rebuild, and concurrent runs never clobber each other's image.
|
|
info "Building $CI_IMAGE_TEST Docker image"
|
|
docker build -t "$CI_IMAGE_TEST" --label "$CI_LABEL" --label "$CI_LABEL_RUN" "$CI_BUILD_CONTEXT" --quiet \
|
|
|| { record "docker-build" 1; return; }
|
|
docker build -t "$CI_IMAGE_APP" --label "$CI_LABEL" --label "$CI_LABEL_RUN" \
|
|
-f "$CI_BUILD_CONTEXT/Dockerfile.app" "$CI_BUILD_CONTEXT" --quiet \
|
|
|| { record "docker-build-app" 1; return; }
|
|
# Deliberately NOT retagged to fips-test:latest. Every consumer reads
|
|
# FIPS_TEST_IMAGE, and a bridge back to the shared mutable name would let a
|
|
# consumer that does not keep working silently — resolving whichever
|
|
# concurrent run wrote the tag last, and recording that run's binaries under
|
|
# this run's commit. Without the bridge a missed consumer fails loudly.
|
|
|
|
# Single suite mode
|
|
if [[ -n "$ONLY_SUITE" ]]; then
|
|
run_suite "$ONLY_SUITE"
|
|
return
|
|
fi
|
|
|
|
# Static topologies (sequential — profiles share container names)
|
|
for topo in "${STATIC_SUITES[@]}"; do
|
|
local topology="${topo#static-}"
|
|
run_static "$topology"
|
|
done
|
|
|
|
# Gateway
|
|
run_gateway
|
|
|
|
# Firewall baseline
|
|
run_firewall
|
|
|
|
# Dynamic interface binding
|
|
run_iface_binding
|
|
|
|
# NAT scenarios (sequential — each owns its compose project)
|
|
for scenario in "${NAT_SUITES[@]}"; do
|
|
run_nat "$scenario"
|
|
done
|
|
|
|
# Nostr publish/consume (sequential — shares the NAT compose project)
|
|
for _suite in "${NOSTR_RELAY_SUITES[@]}"; do
|
|
run_nostr_publish_consume
|
|
done
|
|
|
|
# STUN fault-injection (sequential — shares the NAT compose project)
|
|
for _suite in "${STUN_FAULTS_SUITES[@]}"; do
|
|
run_stun_faults
|
|
done
|
|
|
|
# Transport-medium change (sequential — owns its own compose project)
|
|
for _suite in "${MEDIUM_CHANGE_SUITES[@]}"; do
|
|
run_medium_change
|
|
done
|
|
|
|
# Chaos scenarios (parallel, throttled)
|
|
if [[ "$SKIP_CHAOS" != true ]]; then
|
|
info "Running ${#CHAOS_SUITES[@]} chaos scenarios (max $PARALLEL_JOBS parallel)"
|
|
local pids=()
|
|
local suite_names=()
|
|
local running=0
|
|
|
|
for entry in "${CHAOS_SUITES[@]}"; do
|
|
# Parse: "display-name scenario [flags...]"
|
|
read -ra parts <<< "$entry"
|
|
local name="${parts[0]}"
|
|
local args=("${parts[@]:1}")
|
|
|
|
# No --subnet: the sim claims a free /24 itself (sim/netclaim.py).
|
|
# This used to pass 10.30.${chaos_idx}.0/24, which uniquified the
|
|
# children of ONE run against each other and was byte-identical
|
|
# between runs -- so two concurrent ci-local runs both walked
|
|
# 10.30.0 through 10.30.12 and collided on every one. A claim makes
|
|
# the daemon the arbiter, so an overlap is impossible rather than
|
|
# merely unlikely. The range still sits in 10.30.0.0/16, clear of
|
|
# docker's default pool (172.17-31 / 192.168) and of the
|
|
# fixed-subnet suites in 172.x.
|
|
|
|
# Throttle: wait for a slot
|
|
while [[ $running -ge $PARALLEL_JOBS ]]; do
|
|
wait -n -p done_pid 2>/dev/null || true
|
|
running=$((running - 1))
|
|
done
|
|
|
|
# Run in background, capture output to temp file
|
|
local logfile
|
|
logfile=$(mktemp "/tmp/ci-chaos-${name}.XXXXXX")
|
|
(
|
|
run_chaos "$name" "${args[@]}" >"$logfile" 2>&1
|
|
) &
|
|
pids+=($!)
|
|
CI_CHAOS_PIDS+=("$!")
|
|
suite_names+=("$name:$logfile")
|
|
running=$((running + 1))
|
|
done
|
|
|
|
# Wait for all and collect results
|
|
for i in "${!pids[@]}"; do
|
|
local pid="${pids[$i]}"
|
|
local entry="${suite_names[$i]}"
|
|
local scenario="${entry%%:*}"
|
|
local logfile="${entry#*:}"
|
|
|
|
if wait "$pid" 2>/dev/null; then
|
|
record "chaos-$scenario" 0
|
|
else
|
|
record "chaos-$scenario" 1
|
|
# The whole log, not its tail: the child printed the results
|
|
# directory into it, and this is the only place that reaches
|
|
# the run log. The sed keeps the last frame of each line the
|
|
# progress display redraws with carriage returns.
|
|
echo "--- chaos-$scenario output ---"
|
|
sed 's/.*\r//' "$logfile" 2>/dev/null || true
|
|
echo "---"
|
|
fi
|
|
rm -f "$logfile"
|
|
done
|
|
# All chaos children have been waited on; clear so a later signal does
|
|
# not try to kill already-reaped PIDs.
|
|
CI_CHAOS_PIDS=()
|
|
# The next sequential suite's window starts here, not at the last
|
|
# suite before the chaos block.
|
|
load_mark barrier
|
|
fi
|
|
|
|
# Sidecar
|
|
run_sidecar
|
|
|
|
# Native datagram API (light — one single-node run plus a two-node pair)
|
|
run_native_api
|
|
|
|
# DNS resolver multi-distro suite (heavy — per-distro systemd images)
|
|
run_dns_resolver
|
|
|
|
# Deb-install multi-distro suite (heavy — builds .deb + per-distro install)
|
|
run_deb_install
|
|
|
|
# Tor (opt-in via --with-tor; depends on live Tor network)
|
|
if [[ "$WITH_TOR" == true ]]; then
|
|
run_tor_socks5
|
|
run_tor_directory
|
|
fi
|
|
}
|
|
|
|
# Run a single named suite
|
|
run_suite() {
|
|
local suite="$1"
|
|
case "$suite" in
|
|
static-mesh|static-chain)
|
|
run_static "${suite#static-}" ;;
|
|
gateway)
|
|
run_gateway ;;
|
|
openwrt-scripts)
|
|
run_openwrt_scripts ;;
|
|
firewall)
|
|
run_firewall ;;
|
|
iface-binding)
|
|
run_iface_binding ;;
|
|
nat-cone|nat-symmetric|nat-lan)
|
|
run_nat "${suite#nat-}" ;;
|
|
nostr-publish-consume)
|
|
run_nostr_publish_consume ;;
|
|
stun-faults)
|
|
run_stun_faults ;;
|
|
chaos-*)
|
|
local chaos_name="${suite#chaos-}"
|
|
local found=false
|
|
for entry in "${CHAOS_SUITES[@]}"; do
|
|
read -ra parts <<< "$entry"
|
|
if [[ "${parts[0]}" == "$chaos_name" ]]; then
|
|
run_chaos "$chaos_name" "${parts[@]:1}"
|
|
found=true
|
|
break
|
|
fi
|
|
done
|
|
if [[ "$found" != true ]]; then
|
|
# Fall back to using the name as the scenario directly. Note
|
|
# teardown rebuilds veth suffixes from CHAOS_SUITES only, so a
|
|
# scenario named here but absent from that list is outside the
|
|
# host-interface reap. Harmless while every ethernet-transport
|
|
# scenario is declared there; add one that isn't and its
|
|
# orphaned pairs will need an unscoped `--reap` to clear.
|
|
run_chaos "$chaos_name" "$chaos_name"
|
|
fi
|
|
;;
|
|
sidecar)
|
|
run_sidecar ;;
|
|
dns-resolver)
|
|
run_dns_resolver ;;
|
|
native-api)
|
|
run_native_api ;;
|
|
medium-change)
|
|
run_medium_change ;;
|
|
deb-install)
|
|
run_deb_install ;;
|
|
tor-socks5)
|
|
run_tor_socks5 ;;
|
|
tor-directory)
|
|
run_tor_directory ;;
|
|
*)
|
|
fail "Unknown suite: $suite"
|
|
record "$suite" 1 ;;
|
|
esac
|
|
}
|
|
|
|
# ── Summary ────────────────────────────────────────────────────────────────
|
|
|
|
print_summary() {
|
|
stage "Summary"
|
|
|
|
local passed=0 failed=0 total=0
|
|
for name in $(echo "${!RESULTS[@]}" | tr ' ' '\n' | sort); do
|
|
local rc="${RESULTS[$name]}"
|
|
total=$((total + 1))
|
|
if [[ $rc -eq 0 ]]; then
|
|
passed=$((passed + 1))
|
|
echo -e " ${GREEN}✓${RESET} $name"
|
|
else
|
|
failed=$((failed + 1))
|
|
echo -e " ${RED}✗${RESET} $name"
|
|
python3 "$SCRIPT_DIR/lib/load-annotate.py" "$CI_LOAD_LOG" \
|
|
--threshold "$CI_LOAD_THRESHOLD" "$name" 2>&1 | sed 's/^/ /'
|
|
fi
|
|
done
|
|
|
|
echo ""
|
|
python3 "$SCRIPT_DIR/lib/load-annotate.py" "$CI_LOAD_LOG" --run 2>&1 | sed 's/^/ /'
|
|
echo -e " ${BOLD}Total: $total Passed: $passed Failed: $failed${RESET}"
|
|
echo ""
|
|
|
|
if [[ $OVERALL -eq 0 ]]; then
|
|
echo -e " ${GREEN}${BOLD}ALL PASSED${RESET}"
|
|
else
|
|
echo -e " ${RED}${BOLD}FAILED${RESET}"
|
|
fi
|
|
echo ""
|
|
}
|
|
|
|
# Verify the local default suite set and the GitHub matrix still cover the
|
|
# same work. Runs first: it takes about a second, and a divergence should be
|
|
# reported before a half-hour suite rather than after it.
|
|
# The OpenWrt maintainer scripts and the fips-gateway init script ship to
|
|
# routers and run there under ash, never under bash. This runs them under ash
|
|
# in a busybox container against stubbed init scripts, so an install, an
|
|
# upgrade from either generation of the package, and a removal each assert what
|
|
# the package left enabled and running.
|
|
run_openwrt_scripts() {
|
|
local rc=0
|
|
info "[openwrt-scripts] Running the OpenWrt maintainer-script scenarios"
|
|
bash "$SCRIPT_DIR/openwrt/maintainer-scripts-test.sh" || rc=$?
|
|
record "openwrt-scripts" $rc
|
|
return $rc
|
|
}
|
|
|
|
run_ci_parity() {
|
|
local rc=0
|
|
info "[ci-parity] Comparing the local suite set against the GitHub matrix"
|
|
"$SCRIPT_DIR/check-ci-parity.sh" || rc=$?
|
|
record "ci-parity" $rc
|
|
}
|
|
|
|
# Nothing may resolve the shared mutable test image. This run does not write
|
|
# fips-test:latest, so a consumer naming it fails loudly here and now — but only
|
|
# on a host with no hand-built copy lying around, which is not a property to
|
|
# rely on. This is the static half of that.
|
|
run_image_scoping() {
|
|
local rc=0
|
|
info "[image-scoping] Checking that nothing names the shared test image"
|
|
"$SCRIPT_DIR/check-image-scoping.sh" || rc=$?
|
|
record "image-scoping" $rc
|
|
}
|
|
|
|
# Every third-party action must be referenced by commit SHA. A tag is a mutable
|
|
# pointer, and the jobs holding the AUR deploy key and the release signing key
|
|
# are exactly the ones worth repointing it for. Static, and it costs nothing.
|
|
run_action_pins() {
|
|
local rc=0
|
|
info "[action-pins] Checking that every action is pinned to a commit SHA"
|
|
"$SCRIPT_DIR/check-action-pins.sh" || rc=$?
|
|
record "action-pins" $rc
|
|
}
|
|
|
|
# Every reference a source comment makes must resolve for a reader who has only
|
|
# this repository. A comment naming a private planning artifact is dead text to
|
|
# everyone outside the workspace holding it, and publishes that workspace's
|
|
# shape to everyone else. A hand-run grep closes the population that exists on
|
|
# the day it runs; this closes it for every commit after.
|
|
run_comment_refs() {
|
|
local rc=0
|
|
info "[comment-refs] Checking that every source comment resolves in-repo"
|
|
"$SCRIPT_DIR/check-comment-refs.sh" || rc=$?
|
|
record "comment-refs" $rc
|
|
}
|
|
|
|
# Every daemon log string a test matches on must still be emitted by src/.
|
|
# A stale one does not fail — it stops observing, and an expect-zero assertion
|
|
# built on it then passes for the wrong reason.
|
|
run_log_strings() {
|
|
local rc=0
|
|
info "[log-strings] Checking test log matchers against the strings src/ emits"
|
|
python3 "$SCRIPT_DIR/check-log-strings.py" || rc=$?
|
|
record "log-strings" $rc
|
|
}
|
|
|
|
# A shell function's exit status is its last command's. One ending in a log
|
|
# call returns 0 whatever it did, so a caller testing that status has a dead
|
|
# gate — the run_chaos and build_fips_for_e2e defects, one of which certified
|
|
# unbuilt code as green. This enforces an explicit terminal return wherever a
|
|
# caller consumes the status, which makes the class unreachable rather than
|
|
# repairing instances of it.
|
|
run_trailing_log() {
|
|
local rc=0
|
|
info "[trailing-log] Checking for functions whose status is a log call's"
|
|
python3 "$SCRIPT_DIR/check-trailing-log.py" || rc=$?
|
|
record "trailing-log" $rc
|
|
}
|
|
|
|
# Unit tests for the convergence gate every static suite waits on. Hermetic —
|
|
# synthetic ping functions against the same SECONDS clock, no containers — but
|
|
# it drives real timeouts, so it costs about 45 seconds rather than the "few
|
|
# seconds" its own header used to claim.
|
|
#
|
|
# It is here with the other two rather than in the integration stage because it
|
|
# needs nothing built. Note what that buys: wait_until_connected decides whether
|
|
# every static suite proceeds or gives up, so a regression in it turns those
|
|
# suites' verdicts into noise, and until now nothing ran this at all.
|
|
run_wait_converge() {
|
|
local rc=0
|
|
info "[wait-converge] Running convergence-gate unit tests"
|
|
bash "$SCRIPT_DIR/lib/wait-converge-test.sh" || rc=$?
|
|
record "wait-converge" $rc
|
|
}
|
|
|
|
# ── Main ───────────────────────────────────────────────────────────────────
|
|
|
|
main() {
|
|
local start_time=$SECONDS
|
|
|
|
stage "FIPS Local CI"
|
|
info "Project root: $PROJECT_ROOT"
|
|
|
|
load_sampler &
|
|
CI_LOAD_PID=$!
|
|
|
|
# Above the mode branches deliberately, so --only, --test-only and
|
|
# --build-only are gated too: a divergence invalidates any claim that a
|
|
# local run means what a GitHub run means, whichever subset was asked for.
|
|
run_ci_parity
|
|
run_log_strings
|
|
run_trailing_log
|
|
run_image_scoping
|
|
run_action_pins
|
|
run_comment_refs
|
|
run_wait_converge
|
|
|
|
if [[ "$TEST_ONLY" == true ]]; then
|
|
run_tests
|
|
elif [[ "$BUILD_ONLY" == true ]]; then
|
|
run_build
|
|
else
|
|
run_build
|
|
if [[ "${RESULTS[build]:-1}" -ne 0 ]]; then
|
|
fail "Build failed, skipping remaining stages"
|
|
else
|
|
run_tests
|
|
if [[ "$SKIP_INTEGRATION" != true ]]; then
|
|
run_integration
|
|
fi
|
|
fi
|
|
fi
|
|
|
|
print_summary
|
|
|
|
local elapsed=$(( SECONDS - start_time ))
|
|
local mins=$(( elapsed / 60 ))
|
|
local secs=$(( elapsed % 60 ))
|
|
info "Total time: ${mins}m ${secs}s"
|
|
|
|
exit $OVERALL
|
|
}
|
|
|
|
main
|