diff --git a/testing/acl-allowlist/docker-compose.yml b/testing/acl-allowlist/docker-compose.yml index 14ffc57..9413b6a 100644 --- a/testing/acl-allowlist/docker-compose.yml +++ b/testing/acl-allowlist/docker-compose.yml @@ -1,6 +1,8 @@ networks: acl-net: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" ipam: config: - subnet: 172.31.0.0/24 diff --git a/testing/boringtun/docker-compose.yml b/testing/boringtun/docker-compose.yml index 81c1220..40d51e0 100644 --- a/testing/boringtun/docker-compose.yml +++ b/testing/boringtun/docker-compose.yml @@ -6,6 +6,8 @@ networks: bt-net: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" ipam: config: - subnet: 172.99.0.0/24 diff --git a/testing/chaos/scripts/chaos.sh b/testing/chaos/scripts/chaos.sh index b6ed978..0ffdbb9 100755 --- a/testing/chaos/scripts/chaos.sh +++ b/testing/chaos/scripts/chaos.sh @@ -65,6 +65,7 @@ VERBOSE="" SEED="" DURATION="" NODES="" +SUBNET="" while [ $# -gt 0 ]; do case "$1" in @@ -72,6 +73,7 @@ while [ $# -gt 0 ]; do --seed) SEED="$2"; shift 2 ;; --duration) DURATION="$2"; shift 2 ;; --nodes) NODES="$2"; shift 2 ;; + --subnet) SUBNET="$2"; shift 2 ;; --list) list_scenarios ;; -*) echo "Error: Unknown option '$1'" >&2; usage ;; *) @@ -135,6 +137,7 @@ PYTHON_ARGS=("$SCENARIO_FILE") [ -n "$VERBOSE" ] && PYTHON_ARGS+=("$VERBOSE") [ -n "$SEED" ] && PYTHON_ARGS+=("--seed" "$SEED") [ -n "$DURATION" ] && PYTHON_ARGS+=("--duration" "$DURATION") +[ -n "$SUBNET" ] && PYTHON_ARGS+=("--subnet" "$SUBNET") echo "=== FIPS Stochastic Simulation ===" echo "" @@ -143,6 +146,7 @@ echo " File: $SCENARIO_FILE" [ -n "$SEED" ] && echo " Seed: $SEED (override)" [ -n "$DURATION" ] && echo " Duration: ${DURATION}s (override)" [ -n "$NODES" ] && echo " Nodes: $NODES (override)" +[ -n "$SUBNET" ] && echo " Subnet: $SUBNET (override)" echo "" # If --nodes is specified, create a patched copy of the scenario file diff --git a/testing/chaos/sim/__main__.py b/testing/chaos/sim/__main__.py index b8e2684..ba4f2b8 100644 --- a/testing/chaos/sim/__main__.py +++ b/testing/chaos/sim/__main__.py @@ -25,6 +25,12 @@ def main(): "--duration", type=int, default=None, help="Override scenario duration in seconds", ) + parser.add_argument( + "--subnet", type=str, default=None, + help="Override topology subnet CIDR (e.g. 10.30.0.0/24); node IPs " + "derive from it. Used by CI to give each parallel run a " + "non-overlapping network.", + ) args = parser.parse_args() level = logging.DEBUG if args.verbose else logging.INFO @@ -48,6 +54,8 @@ def main(): print("Error: --duration must be >= 1", file=sys.stderr) sys.exit(1) scenario.duration_secs = args.duration + if args.subnet is not None: + scenario.topology.subnet = args.subnet runner = SimRunner(scenario) result = runner.run() diff --git a/testing/chaos/sim/compose.py b/testing/chaos/sim/compose.py index 1358ee1..5eec26b 100644 --- a/testing/chaos/sim/compose.py +++ b/testing/chaos/sim/compose.py @@ -20,6 +20,8 @@ _COMPOSE_TEMPLATE = Template( networks: fips-net: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" ipam: config: - subnet: {{ subnet }} diff --git a/testing/ci-cleanup.sh b/testing/ci-cleanup.sh new file mode 100755 index 0000000..6f4a3c1 --- /dev/null +++ b/testing/ci-cleanup.sh @@ -0,0 +1,106 @@ +#!/bin/bash +# Reap FIPS CI docker resources: containers, networks, volumes, images. +# +# Force-removes everything created by ci-local.sh that is still around — +# whether a run finished cleanly, was preempted (SIGTERM/SIGKILL), OOM-killed, +# or crashed. Two complementary selectors make this robust no matter how a +# prior run died: +# +# 1. The CI label com.corganlabs.fips-ci=1 (attached to every direct +# `docker run`/network/volume ci-local drives). +# 2. The compose project-name prefix fipsci_ (every compose project ci-local +# starts is named fipsci__, so its containers/networks/ +# volumes all carry com.docker.compose.project=fipsci_... and are named +# with that prefix). +# +# Usage: +# ci-cleanup.sh Reap ALL fips-ci resources (any run) +# ci-cleanup.sh --project-prefix P Restrict the compose-project sweep to +# names starting with P (scopes the reap +# to a single run; the label sweep still +# runs) +# ci-cleanup.sh --label L Override the CI label (default above) +# ci-cleanup.sh --images "a b,c" Also `docker rmi -f` these image tags +# (space- or comma-separated) +# +# Safe to run when there is nothing to reap, and safe to run repeatedly. +# Also reachable as `ci-local.sh --reap`. +set -uo pipefail + +LABEL="com.corganlabs.fips-ci=1" +PROJECT_PREFIX="fipsci_" # broad default: every CI run +IMAGES="" + +while [[ $# -gt 0 ]]; do + case "$1" in + --label) LABEL="$2"; shift 2 ;; + --project-prefix) PROJECT_PREFIX="$2"; shift 2 ;; + --images) IMAGES="$2"; shift 2 ;; + -h|--help) sed -n '2,/^set /{ /^set /d; s/^# \?//; p }' "$0"; exit 0 ;; + *) echo "Unknown option: $1" >&2; exit 2 ;; + esac +done + +if ! command -v docker >/dev/null 2>&1; then + # No docker, nothing to reap. + exit 0 +fi +if ! docker info >/dev/null 2>&1; then + # Daemon unreachable; treat as nothing to reap rather than wedging a caller. + exit 0 +fi + +# Each docker mutation is wrapped in `timeout` so a stuck daemon/resource can +# never wedge a caller (ci-local's signal trap relies on this being bounded). +TMO=30 + +# Distinct compose project names (read off container labels) that start with +# the configured prefix. +ci_projects() { + docker ps -a --format '{{.Label "com.docker.compose.project"}}' 2>/dev/null \ + | grep -E "^${PROJECT_PREFIX}" | sort -u +} + +reap_containers() { + # By CI label. + docker ps -aq --filter "label=${LABEL}" 2>/dev/null \ + | xargs -r timeout "$TMO" docker rm -f >/dev/null 2>&1 || true + # By compose project (carried even when container_name is explicit). + local p + for p in $(ci_projects); do + docker ps -aq --filter "label=com.docker.compose.project=${p}" 2>/dev/null \ + | xargs -r timeout "$TMO" docker rm -f >/dev/null 2>&1 || true + done +} + +reap_networks() { + docker network ls -q --filter "label=${LABEL}" 2>/dev/null \ + | xargs -r timeout "$TMO" docker network rm >/dev/null 2>&1 || true + # Compose networks are named _ → match by name prefix so + # orphaned networks (whose containers are already gone) are still caught. + docker network ls --format '{{.Name}}' 2>/dev/null | grep -E "^${PROJECT_PREFIX}" \ + | xargs -r timeout "$TMO" docker network rm >/dev/null 2>&1 || true +} + +reap_volumes() { + docker volume ls -q --filter "label=${LABEL}" 2>/dev/null \ + | xargs -r timeout "$TMO" docker volume rm >/dev/null 2>&1 || true + docker volume ls --format '{{.Name}}' 2>/dev/null | grep -E "^${PROJECT_PREFIX}" \ + | xargs -r timeout "$TMO" docker volume rm >/dev/null 2>&1 || true +} + +reap_images() { + [[ -z "$IMAGES" ]] && return 0 + local imgs + read -ra imgs <<< "${IMAGES//,/ }" + [[ ${#imgs[@]} -eq 0 ]] && return 0 + timeout "$TMO" docker rmi -f "${imgs[@]}" >/dev/null 2>&1 || true +} + +# Order matters: containers reference networks/volumes, so drop them first. +reap_containers +reap_networks +reap_volumes +reap_images + +exit 0 diff --git a/testing/ci-local.sh b/testing/ci-local.sh index d9335d4..968264b 100755 --- a/testing/ci-local.sh +++ b/testing/ci-local.sh @@ -14,6 +14,9 @@ # --list List available integration suites # --check-parity Verify this suite set matches ci.yml's integration # matrix (see testing/check-ci-parity.sh), then exit +# --reap Force-remove all leftover FIPS CI docker resources +# (containers/networks/volumes carrying the CI label or a +# fipsci_ compose project), then exit. See ci-cleanup.sh. # -h, --help Show this help # # Integration suites (default coverage): @@ -32,8 +35,13 @@ # tor-socks5, tor-directory # # Exit codes: -# 0 — all stages passed -# 1 — one or more stages failed +# 0 — all stages passed +# 1 — one or more stages failed +# 130 — interrupted by SIGINT (128 + 2; run was cancelled, not a failure) +# 143 — terminated by SIGTERM (128 + 15; run was cancelled, not a failure) +# +# A preempting CI worker maps 130/143 → "cancelled" (discard, do not record a +# failing commit), 0 → green, any other non-zero → red. # # ── CI parity invariant ───────────────────────────────────────────────────── # This local default suite set and the GitHub integration matrix @@ -193,6 +201,7 @@ while [[ $# -gt 0 ]]; do -j|--jobs) PARALLEL_JOBS="$2"; shift 2 ;; --list) list_suites ;; --check-parity) exec "$SCRIPT_DIR/check-ci-parity.sh" ;; + --reap) exec "$SCRIPT_DIR/ci-cleanup.sh" ;; -h|--help) usage ;; *) echo "Unknown option: $1"; usage ;; esac @@ -214,6 +223,104 @@ record() { fi } +# ── Per-run isolation + signal-safe teardown ─────────────────────────────── +# +# This script may be preempted (a CI worker sends SIGTERM, waits ~30s, then +# SIGKILL) so it can restart on a newer tip. To make that safe: +# * every docker resource is namespaced to THIS run (compose project prefix +# + per-run image tags) so a restart never collides with a dying run; +# * a trap tears down everything this run created on signal/exit, bounded by +# `timeout` so a stuck `down` cannot wedge the trap (SIGKILL is the backstop). + +# Derive a run id: honor the worker's $FIPS_CI_RUN_ID, else -, +# else $$-. Sanitize to a valid compose project / image-tag token. +if [[ -n "${FIPS_CI_RUN_ID:-}" ]]; then + CI_RUN_ID="$FIPS_CI_RUN_ID" +else + _ci_sha="$(git -C "$PROJECT_ROOT" rev-parse --short HEAD 2>/dev/null || true)" + if [[ -n "$_ci_sha" ]]; then + CI_RUN_ID="${_ci_sha}-${RANDOM}${RANDOM}" + else + CI_RUN_ID="$$-$(date +%s)" + fi +fi +CI_RUN_ID="$(printf '%s' "$CI_RUN_ID" | tr '[:upper:]' '[:lower:]' | tr -c 'a-z0-9_-' '-')" +# A docker image tag and a compose project name must both START with an +# alphanumeric, so strip any leading -/_ left by sanitization. +CI_RUN_ID="$(printf '%s' "$CI_RUN_ID" | sed -E 's/^[^a-z0-9]+//')" +[[ -z "$CI_RUN_ID" ]] && CI_RUN_ID="r$$" + +CI_PROJECT_PREFIX="fipsci_${CI_RUN_ID}" +CI_IMAGE_TEST="fips-test:${CI_RUN_ID}" +CI_IMAGE_APP="fips-test-app:${CI_RUN_ID}" +CI_LABEL="com.corganlabs.fips-ci=1" + +# Exported so child suite scripts + their compose/`docker run` invocations +# inherit the run identity. Compose files read FIPS_TEST_IMAGE/FIPS_TEST_APP_IMAGE +# (default :latest when unset, preserving manual `docker compose` use). +export FIPS_CI_RUN_ID="$CI_RUN_ID" +export FIPS_TEST_IMAGE="$CI_IMAGE_TEST" +export FIPS_TEST_APP_IMAGE="$CI_IMAGE_APP" + +# Per-suite compose project name: ${prefix}_. Keeps +# today's intra-run distinctness (one project per compose file / chaos child) +# while adding the cross-run prefix that scopes the reap. +ci_project() { printf '%s_%s' "$CI_PROJECT_PREFIX" "$1"; } + +# PIDs of in-flight parallel chaos children (subshells). The trap signals these. +CI_CHAOS_PIDS=() +CI_CLEANED=0 + +# Best-effort, BOUNDED teardown of every docker resource THIS run may have +# created. Idempotent (guarded), so the signal and EXIT paths don't double-run. +ci_teardown() { + [[ $CI_CLEANED -eq 1 ]] && return 0 + CI_CLEANED=1 + + # 1. Propagate to parallel chaos children and reap them (bounded). + if [[ ${#CI_CHAOS_PIDS[@]} -gt 0 ]]; then + kill -TERM "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true + local _end=$(( SECONDS + 10 )) _p + for _p in "${CI_CHAOS_PIDS[@]}"; do + while kill -0 "$_p" 2>/dev/null && (( SECONDS < _end )); do + sleep 0.3 + done + done + kill -KILL "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true + wait "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true + fi + + # 2. Remove all compose projects + direct-run resources + per-run images + # for this run. ci-cleanup.sh wraps each docker op in `timeout`; bound + # the whole sweep too so the trap can never wedge. + timeout 150 bash "$SCRIPT_DIR/ci-cleanup.sh" \ + --label "$CI_LABEL" \ + --project-prefix "$CI_PROJECT_PREFIX" \ + --images "$CI_IMAGE_TEST $CI_IMAGE_APP" >/dev/null 2>&1 || true +} + +on_signal() { + local sig="$1" + # Block re-entry and stop the EXIT trap from overriding the signal code. + trap '' TERM INT EXIT + echo "" >&2 + fail "Received SIG$sig — cancelling run, tearing down ${CI_PROJECT_PREFIX}" + ci_teardown + # 128 + signal number: distinct from 0 (green) / 1 (stage failed). + if [[ "$sig" == "TERM" ]]; then exit 143; else exit 130; fi +} + +on_exit() { + local code=$? + trap '' TERM INT EXIT + ci_teardown + exit "$code" +} + +trap 'on_signal TERM' TERM +trap 'on_signal INT' INT +trap 'on_exit' EXIT + # ── Stage 1: Build ───────────────────────────────────────────────────────── run_build() { @@ -317,6 +424,7 @@ run_static() { local topology="$1" local compose="testing/static/docker-compose.yml" local rc=0 + export COMPOSE_PROJECT_NAME="$(ci_project static)" info "[$topology] Generating configs" bash testing/static/scripts/generate-configs.sh "$topology" || { record "static-$topology" 1; return; } @@ -341,6 +449,7 @@ run_static() { run_rekey() { local compose="testing/static/docker-compose.yml" local rc=0 + export COMPOSE_PROJECT_NAME="$(ci_project static)" info "[rekey] Generating configs" bash testing/static/scripts/generate-configs.sh rekey || { record "rekey" 1; return; } @@ -369,6 +478,7 @@ run_rekey() { run_admission_cap() { local compose="testing/static/docker-compose.yml" local rc=0 + export COMPOSE_PROJECT_NAME="$(ci_project static)" info "[admission-cap] Generating configs" bash testing/static/scripts/generate-configs.sh mesh || { record "admission-cap" 1; return; } @@ -395,6 +505,9 @@ run_chaos() { local name="$1" shift local rc=0 + # Distinct project per scenario (chaos children run in parallel). When + # invoked from a background subshell this export is local to that child. + export COMPOSE_PROJECT_NAME="$(ci_project "chaos-$name")" info "[chaos/$name] Running simulation" if bash testing/chaos/scripts/chaos.sh "$@" 2>&1; then @@ -410,6 +523,7 @@ run_chaos() { run_gateway() { local compose="testing/static/docker-compose.yml" local rc=0 + export COMPOSE_PROJECT_NAME="$(ci_project static)" info "[gateway] Generating configs" bash testing/static/scripts/generate-configs.sh gateway gateway-test || { record "gateway" 1; return; } @@ -434,6 +548,7 @@ run_gateway() { # Run sidecar test run_sidecar() { local rc=0 + export COMPOSE_PROJECT_NAME="$(ci_project sidecar)" info "[sidecar] Running integration test" if bash testing/sidecar/scripts/test-sidecar.sh --skip-build 2>&1; then @@ -450,6 +565,7 @@ run_sidecar() { run_rekey_accept_off() { local compose="testing/static/docker-compose.yml" local rc=0 + export COMPOSE_PROJECT_NAME="$(ci_project static)" info "[rekey-accept-off] Generating configs" bash testing/static/scripts/generate-configs.sh rekey-accept-off || \ @@ -484,6 +600,7 @@ run_rekey_accept_off() { run_rekey_outbound_only() { local compose="testing/static/docker-compose.yml" local rc=0 + export COMPOSE_PROJECT_NAME="$(ci_project static)" info "[rekey-outbound-only] Generating configs" bash testing/static/scripts/generate-configs.sh rekey-outbound-only || \ @@ -512,6 +629,7 @@ run_rekey_outbound_only() { # Run ACL allowlist integration test run_acl_allowlist() { + export COMPOSE_PROJECT_NAME="$(ci_project acl)" info "[acl-allowlist] Running integration test" if bash testing/acl-allowlist/test.sh --skip-build 2>&1; then record "acl-allowlist" 0 @@ -522,6 +640,7 @@ run_acl_allowlist() { # Run firewall baseline integration test run_firewall() { + export COMPOSE_PROJECT_NAME="$(ci_project firewall)" info "[firewall] Running integration test" if bash testing/firewall/test.sh --skip-build 2>&1; then record "firewall" 0 @@ -533,6 +652,7 @@ run_firewall() { # Run a NAT scenario (cone, symmetric, lan) run_nat() { local scenario="$1" + export COMPOSE_PROJECT_NAME="$(ci_project nat)" info "[nat-$scenario] Running NAT lab" if bash testing/nat/scripts/nat-test.sh "$scenario" 2>&1; then record "nat-$scenario" 0 @@ -546,6 +666,7 @@ run_nat() { # (A→B publish/consume), Phase 2 (B→A reverse), and Phase 3 (malformed # advert injected directly to the relay; consumer-liveness assertion). run_nostr_publish_consume() { + export COMPOSE_PROJECT_NAME="$(ci_project nat)" info "[nostr-publish-consume] Running Nostr publish/consume test" if bash testing/nat/scripts/nostr-relay-test.sh 2>&1; then record "nostr-publish-consume" 0 @@ -560,6 +681,7 @@ run_nostr_publish_consume() { # kill. Asserts the daemon detects each fault, recovers from delay, and # never panics. run_stun_faults() { + export COMPOSE_PROJECT_NAME="$(ci_project nat)" info "[stun-faults] Running STUN fault-injection test" if bash testing/nat/scripts/stun-faults-test.sh 2>&1; then record "stun-faults" 0 @@ -590,6 +712,7 @@ run_deb_install() { # Run Tor SOCKS5 outbound test (live Tor network) run_tor_socks5() { + export COMPOSE_PROJECT_NAME="$(ci_project tor-socks5)" info "[tor-socks5] Running Tor SOCKS5 outbound test (live Tor)" if bash testing/tor/socks5-outbound/scripts/tor-test.sh 2>&1; then record "tor-socks5" 0 @@ -600,6 +723,7 @@ run_tor_socks5() { # Run Tor directory-mode test (live Tor network) run_tor_directory() { + export COMPOSE_PROJECT_NAME="$(ci_project tor-directory)" info "[tor-directory] Running Tor directory-mode test (live Tor)" if bash testing/tor/directory-mode/scripts/directory-test.sh 2>&1; then record "tor-directory" 0 @@ -616,10 +740,20 @@ run_integration() { info "Installing release binaries" install_binaries testing/docker - # Build unified test image once (used by all harnesses) - info "Building fips-test Docker image" - docker build -t fips-test:latest testing/docker --quiet || { record "docker-build" 1; return; } - docker build -t fips-test-app:latest -f testing/docker/Dockerfile.app testing/docker --quiet || { record "docker-build-app" 1; return; } + # Build unified test image once (used by all harnesses). Tag per-run + # (fips-test:${run}) so a build killed mid-flight never wedges the next + # run's rebuild, and concurrent runs never clobber each other's image. + # Then retag :latest for the compose files / harness scripts that still + # reference fips-test:latest directly; the retag happens only after BOTH + # builds succeed, so :latest never points at a half-built image. + info "Building $CI_IMAGE_TEST Docker image" + docker build -t "$CI_IMAGE_TEST" --label "$CI_LABEL" testing/docker --quiet \ + || { record "docker-build" 1; return; } + docker build -t "$CI_IMAGE_APP" --label "$CI_LABEL" \ + -f testing/docker/Dockerfile.app testing/docker --quiet \ + || { record "docker-build-app" 1; return; } + docker tag "$CI_IMAGE_TEST" fips-test:latest + docker tag "$CI_IMAGE_APP" fips-test-app:latest # Single suite mode if [[ -n "$ONLY_SUITE" ]]; then @@ -673,6 +807,7 @@ run_integration() { local pids=() local suite_names=() local running=0 + local chaos_idx=0 for entry in "${CHAOS_SUITES[@]}"; do # Parse: "display-name scenario [flags...]" @@ -680,6 +815,15 @@ run_integration() { local name="${parts[0]}" local args=("${parts[@]:1}") + # Give each chaos child a unique, non-overlapping /24 in 10.30.x so + # parallel children never collide with each other, and so a chaos + # net can never swallow a fixed-subnet suite (sidecar/static/nat in + # 172.x). 10.30.x sits outside docker's default-address-pool range + # (172.17-31 / 192.168), so auto-assigned nets can't land on it + # either. Node IPs derive from this subnet inside the sim. + args+=("--subnet" "10.30.${chaos_idx}.0/24") + chaos_idx=$((chaos_idx + 1)) + # Throttle: wait for a slot while [[ $running -ge $PARALLEL_JOBS ]]; do wait -n -p done_pid 2>/dev/null || true @@ -693,6 +837,7 @@ run_integration() { run_chaos "$name" "${args[@]}" >"$logfile" 2>&1 ) & pids+=($!) + CI_CHAOS_PIDS+=("$!") suite_names+=("$name:$logfile") running=$((running + 1)) done @@ -715,6 +860,9 @@ run_integration() { fi rm -f "$logfile" done + # All chaos children have been waited on; clear so a later signal does + # not try to kill already-reaped PIDs. + CI_CHAOS_PIDS=() fi # Sidecar diff --git a/testing/deb-install/test.sh b/testing/deb-install/test.sh index a17680c..9b4ce15 100755 --- a/testing/deb-install/test.sh +++ b/testing/deb-install/test.sh @@ -65,6 +65,7 @@ start_systemd_container_with_tun() { local name="$1" image="$2" cleanup_container "$name" docker run -d --name "$name" \ + --label com.corganlabs.fips-ci=1 \ --privileged \ --cgroupns=host \ --device /dev/net/tun \ diff --git a/testing/dns-resolver/test.sh b/testing/dns-resolver/test.sh index 887d56b..218f978 100755 --- a/testing/dns-resolver/test.sh +++ b/testing/dns-resolver/test.sh @@ -67,6 +67,7 @@ start_systemd_container() { local name="$1" image="$2" cleanup_container "$name" docker run -d --name "$name" \ + --label com.corganlabs.fips-ci=1 \ --privileged \ --cgroupns=host \ -v /sys/fs/cgroup:/sys/fs/cgroup:rw \ @@ -79,6 +80,7 @@ start_systemd_container_with_tun() { local name="$1" image="$2" cleanup_container "$name" docker run -d --name "$name" \ + --label com.corganlabs.fips-ci=1 \ --privileged \ --cgroupns=host \ --device /dev/net/tun \ diff --git a/testing/firewall/docker-compose.yml b/testing/firewall/docker-compose.yml index 2b1b1c1..97cb5bb 100644 --- a/testing/firewall/docker-compose.yml +++ b/testing/firewall/docker-compose.yml @@ -1,6 +1,8 @@ networks: fw-net: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" ipam: config: - subnet: 172.32.0.0/24 diff --git a/testing/interop/interop-test.sh b/testing/interop/interop-test.sh index 177532c..b2e0866 100755 --- a/testing/interop/interop-test.sh +++ b/testing/interop/interop-test.sh @@ -162,6 +162,13 @@ REKEY_SETTLE=12 # FSP-cutover settle budget (Phase 6) # strict assertion sweep runs. A genuinely stuck pair still fails — the # poll times out and the recording sweep captures it. POST_REKEY_TIMEOUT=45 +# Progress-aware stall budget for the convergence detector +# (wait_for_full_baseline → wait_until_connected). If no additional pair +# becomes reachable for this long while more than the near-converged +# slack of pairs is still down, the detector gives up early instead of +# burning the whole convergence/post-rekey window; any progress resets +# the clock, so a slow-but-converging mesh under netem keeps polling. +RECONVERGE_STALL=15 LOG_POLL_INTERVAL=2 # Data-plane continuity stream (control-differential). Streams run a @@ -434,25 +441,27 @@ ping_all_pairs() { done } +# Convergence detector probe: one full all-pairs ping sweep in the +# "convergence" context (which ping_all_pairs deliberately does NOT +# record as a failure), setting PASSED/FAILED for wait_until_connected. +_baseline_probe() { + ping_all_pairs quiet 1 "convergence" +} + # Poll until every directed pair pings clean, or until timeout. This is a # convergence DETECTOR — used at establishment (Phase 1) and after each # rekey (Phases 3/5). It pings with the "convergence" context, which # ping_all_pairs deliberately does not record as a failure; the caller # runs a separate strict assertion sweep afterwards. +# +# Delegates to the shared progress-aware wait_until_connected so the +# deadline extends while more pairs are still coming up and gives up fast +# on a genuine stall, instead of the prior fixed-deadline poll that could +# false-time-out under heavy CI contention even while still converging. +# Returns 0 once every pair is reachable, 1 on stall/timeout. wait_for_full_baseline() { local timeout="$1" - local start=$SECONDS - local best_passed=0 best_failed="$NUM_DIRECTED" - while (( SECONDS - start < timeout )); do - ping_all_pairs quiet 1 "convergence" - if [ "$PASSED" -gt "$best_passed" ]; then - best_passed="$PASSED"; best_failed="$FAILED" - fi - [ "$FAILED" -eq 0 ] && return 0 - sleep 1 - done - PASSED="$best_passed"; FAILED="$best_failed" - return 1 + wait_until_connected _baseline_probe "$timeout" "$RECONVERGE_STALL" } phase_result() { @@ -759,8 +768,15 @@ echo "" # ── Phase 4: second rekey cycle ────────────────────────────────────── -echo "Phase 4: Second rekey cycle (waiting ${SECOND_REKEY_WAIT}s)" -sleep "$SECOND_REKEY_WAIT" +echo "Phase 4: Second rekey cycle (waiting up to ${SECOND_REKEY_WAIT}s for the next cutover)" +# Poll for the next FMP cutover beyond what Phases 2/3 already saw, using +# the same pre/post cutover-count delta convention as the control window +# (Phase 1b), instead of a blind sleep. Bounded by SECOND_REKEY_WAIT so a +# stalled rekey falls through to the strict Phase 5/6 assertions. +fmp_cutovers_before="$(count_log_pattern 'Rekey cutover complete \(initiator\), K-bit flipped')" +wait_for_log_pattern_count \ + "Rekey cutover complete \(initiator\), K-bit flipped" \ + "$((fmp_cutovers_before + 1))" "$SECOND_REKEY_WAIT" || true echo "" echo "Phase 5: Post-second-rekey connectivity (reconverge within ${POST_REKEY_TIMEOUT}s)" diff --git a/testing/nat/docker-compose.yml b/testing/nat/docker-compose.yml index b7948e3..72fe7ee 100644 --- a/testing/nat/docker-compose.yml +++ b/testing/nat/docker-compose.yml @@ -1,11 +1,15 @@ networks: wan: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" ipam: config: - subnet: 172.31.254.0/24 shared-lan: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" ipam: config: - subnet: 172.31.10.0/24 diff --git a/testing/nat/scripts/nat-test.sh b/testing/nat/scripts/nat-test.sh index 2a9abba..dd8cd33 100755 --- a/testing/nat/scripts/nat-test.sh +++ b/testing/nat/scripts/nat-test.sh @@ -202,7 +202,7 @@ dump_stun_udp_probe() { local capture_file capture_file="$(mktemp)" - docker run --rm --net=container:fips-nat-stun --cap-add NET_ADMIN --cap-add NET_RAW \ + docker run --rm --label com.corganlabs.fips-ci=1 --net=container:fips-nat-stun --cap-add NET_ADMIN --cap-add NET_RAW \ --entrypoint sh "$helper_image" \ -lc "timeout 8 tcpdump -ni any 'udp and not port 53' -c 80" \ >"$capture_file" 2>&1 & diff --git a/testing/nat/scripts/setup-topology.sh b/testing/nat/scripts/setup-topology.sh index db10759..8463ea9 100755 --- a/testing/nat/scripts/setup-topology.sh +++ b/testing/nat/scripts/setup-topology.sh @@ -57,6 +57,7 @@ run_host_ip() { local image="$1" shift docker run --rm \ + --label com.corganlabs.fips-ci=1 \ --privileged \ --net=host \ --pid=host \ diff --git a/testing/sidecar/docker-compose.yml b/testing/sidecar/docker-compose.yml index f9f86f7..801da5b 100644 --- a/testing/sidecar/docker-compose.yml +++ b/testing/sidecar/docker-compose.yml @@ -2,6 +2,12 @@ networks: fips-net: name: ${FIPS_NETWORK:-fips-sidecar-net} driver: bridge + # Reap label: this harness uses fixed -p sidecar-{a,b,c} project names + # (the test execs containers by those derived names), so its resources are + # NOT covered by ci-local's fipsci_ project prefix. Label them directly so + # ci-cleanup.sh can still reap them after a preemption/SIGKILL. + labels: + - "com.corganlabs.fips-ci=1" ipam: config: - subnet: ${FIPS_SUBNET:-172.20.1.0/24} @@ -10,6 +16,8 @@ services: fips: image: fips-test:latest hostname: fips-sidecar + labels: + - "com.corganlabs.fips-ci=1" cap_add: - NET_ADMIN devices: @@ -33,6 +41,8 @@ services: app: image: fips-test-app:latest + labels: + - "com.corganlabs.fips-ci=1" network_mode: "service:fips" depends_on: - fips diff --git a/testing/static/docker-compose.yml b/testing/static/docker-compose.yml index 1309a54..70dda71 100644 --- a/testing/static/docker-compose.yml +++ b/testing/static/docker-compose.yml @@ -1,11 +1,15 @@ networks: fips-net: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" ipam: config: - subnet: 172.20.0.0/24 gateway-lan: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" enable_ipv6: true ipam: config: @@ -534,7 +538,7 @@ services: ipv4_address: 172.20.0.12 gw-client: - image: fips-test-app:latest + image: ${FIPS_TEST_APP_IMAGE:-fips-test-app:latest} profiles: ["gateway"] container_name: fips-gw-client hostname: gw-client @@ -556,7 +560,7 @@ services: # Same image and gateway-lan attachment as # gw-client; the gateway must allocate a distinct virtual IP for it. gw-client-2: - image: fips-test-app:latest + image: ${FIPS_TEST_APP_IMAGE:-fips-test-app:latest} profiles: ["gateway"] container_name: fips-gw-client-2 hostname: gw-client-2 diff --git a/testing/static/scripts/admission-cap-test.sh b/testing/static/scripts/admission-cap-test.sh index c88448f..d5a553a 100755 --- a/testing/static/scripts/admission-cap-test.sh +++ b/testing/static/scripts/admission-cap-test.sh @@ -149,7 +149,7 @@ HELPER_IMAGE=$(docker inspect -f '{{.Config.Image}}' fips-node-$CAP_NODE 2>/dev/ LOAD_PID=$! # Foreground: tcpdump capture for CAPTURE_SECS -docker run --rm --net=container:fips-node-$CAP_NODE \ +docker run --rm --label com.corganlabs.fips-ci=1 --net=container:fips-node-$CAP_NODE \ --cap-add NET_ADMIN --cap-add NET_RAW \ --entrypoint sh "$HELPER_IMAGE" \ -c "timeout $CAPTURE_SECS tcpdump -nn -i any 'udp port 2121' -l 2>&1 || true" \ diff --git a/testing/static/scripts/rekey-test.sh b/testing/static/scripts/rekey-test.sh index a81b49b..529ea10 100755 --- a/testing/static/scripts/rekey-test.sh +++ b/testing/static/scripts/rekey-test.sh @@ -158,9 +158,20 @@ REKEY_SETTLE=12 # > DRAIN_WINDOW_SECS (10) so post-rekey samples are off # fully converged. Keep this bounded to preserve a meaningful scheduling check # while still allowing for log visibility at the timeout edge. FIRST_REKEY_TIMEOUT=$((REKEY_AFTER_SECS + 15)) -SECOND_REKEY_WAIT=40 # wait for second cycle +SECOND_REKEY_WAIT=40 # upper bound for observing the second rekey cutover LOG_EVENT_POLL_INTERVAL=1 +# Post-rekey data-plane reconvergence is polled, not fixed-slept. +# wait_until_connected drives the same all-pairs ping sweep the strict +# per-phase assert depends on: it fails fast when reachability stops +# improving (no new reachable pair for RECONVERGE_STALL seconds while +# more than the near-converged slack of pairs is still down) and extends +# its deadline up to RECONVERGE_TIMEOUT only while pairs keep coming up, +# so a slow-but-converging post-rekey mesh under CI load passes instead +# of false-failing on a fixed settle window. +RECONVERGE_TIMEOUT=45 +RECONVERGE_STALL=10 + TIMEOUT=5 CONVERGENCE_PING_TIMEOUT=1 # Strict-ping retry policy for the per-phase ping_all asserts. Under 1% @@ -392,20 +403,38 @@ assert_min_count "Rekey cutover complete (initiator), K-bit flipped" 1 \ phase_result "FMP rekey events" echo "" -# Verify connectivity after first rekey (strict — no failures allowed) -echo "Phase 3: Post-rekey connectivity (settling ${REKEY_SETTLE}s)" -sleep "$REKEY_SETTLE" +# Verify connectivity after first rekey (strict — no failures allowed). +# Wait for the data plane to reconverge with a progress-aware deadline +# (the same all-pairs ping sweep the strict assert below depends on) +# instead of a fixed settle: fail fast on a genuinely stuck pair, extend +# only while reachability is still climbing. The strict ping_all is the +# actual assertion, run only after the reconverge poll. +echo "Phase 3: Post-rekey connectivity (reconverging up to ${RECONVERGE_TIMEOUT}s)" +wait_until_connected _baseline_ping "$RECONVERGE_TIMEOUT" "$RECONVERGE_STALL" || true ping_all "" "$TIMEOUT" "$MAX_PING_ATTEMPTS" phase_result "Post-first-rekey (all 20 pairs)" echo "" # ── Phase 4: Wait for second rekey cycle ────────────────────────────── -echo "Phase 4: Second rekey cycle (waiting ${SECOND_REKEY_WAIT}s)" -sleep "$SECOND_REKEY_WAIT" +# Poll for the next FMP rekey cutover instead of blind-sleeping: capture +# the cutover count reached so far, then wait until at least one more +# cutover lands — that increment is the second rekey cycle firing. +# Bounded by SECOND_REKEY_WAIT so a stalled rekey still falls through to +# the strict Phase 5/6 assertions rather than hanging. +echo "Phase 4: Second rekey cycle (waiting up to ${SECOND_REKEY_WAIT}s for the next cutover)" +fmp_cutovers_before=$(count_log_pattern "Rekey cutover complete (initiator), K-bit flipped") +wait_for_log_pattern_count \ + "Rekey cutover complete (initiator), K-bit flipped" \ + "$((fmp_cutovers_before + 1))" "$SECOND_REKEY_WAIT" || true -# Verify connectivity after second rekey (back-to-back) -echo "Phase 5: Post-second-rekey connectivity (settling ${REKEY_SETTLE}s)" -sleep "$REKEY_SETTLE" +# Verify connectivity after second rekey (back-to-back). This is the +# site of the recurring post-second-rekey straggler-pair flake: wait for +# the data plane to reconverge with a progress-aware deadline (the same +# all-pairs ping sweep the strict assert below depends on) rather than a +# fixed settle window — fail fast if reachability stops improving, extend +# only while pairs are still coming up. +echo "Phase 5: Post-second-rekey connectivity (reconverging up to ${RECONVERGE_TIMEOUT}s)" +wait_until_connected _baseline_ping "$RECONVERGE_TIMEOUT" "$RECONVERGE_STALL" || true ping_all "" "$TIMEOUT" "$MAX_PING_ATTEMPTS" phase_result "Post-second-rekey (all 20 pairs)" echo "" diff --git a/testing/tor/directory-mode/docker-compose.yml b/testing/tor/directory-mode/docker-compose.yml index d6e37a2..72886eb 100644 --- a/testing/tor/directory-mode/docker-compose.yml +++ b/testing/tor/directory-mode/docker-compose.yml @@ -11,6 +11,8 @@ networks: dir-test: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" services: fips-a: diff --git a/testing/tor/socks5-outbound/docker-compose.yml b/testing/tor/socks5-outbound/docker-compose.yml index 32c361a..e807af5 100644 --- a/testing/tor/socks5-outbound/docker-compose.yml +++ b/testing/tor/socks5-outbound/docker-compose.yml @@ -10,6 +10,8 @@ networks: tor-test: driver: bridge + labels: + - "com.corganlabs.fips-ci=1" services: tor-daemon: