Merge master into next: retry image builds across transient registry failures, reuse the built test image in chaos legs

This commit is contained in:
Johnathan Corgan
2026-09-19 15:46:55 +00:00
9 changed files with 163 additions and 48 deletions
+8 -2
View File
@@ -763,8 +763,10 @@ jobs:
chmod +x _bin/native-echo _bin/native-surface
cp _bin/native-echo testing/docker/native-echo
cp _bin/native-surface testing/docker/native-surface
docker build -t fips-test:latest testing/docker
docker build -t fips-test-app:latest -f testing/docker/Dockerfile.app testing/docker
# Retried: both builds pull from Docker Hub and the Debian mirrors.
source testing/lib/image-build.sh
retry_build "docker build fips-test" docker build -t fips-test:latest testing/docker
retry_build "docker build fips-test-app" docker build -t fips-test-app:latest -f testing/docker/Dockerfile.app testing/docker
# ── Static topology ────────────────────────────────────────────────────
- name: Generate configs (static)
@@ -830,8 +832,12 @@ jobs:
if: matrix.type == 'chaos'
run: pip3 install --quiet pyyaml jinja2
# With FIPS_TEST_IMAGE set, the chaos runner uses the image the job built
# above instead of building fips-test:latest again (testing/chaos/sim/runner.py).
- name: Run chaos scenario
if: matrix.type == 'chaos'
env:
FIPS_TEST_IMAGE: fips-test:latest
run: bash testing/chaos/scripts/chaos.sh ${{ matrix.scenario }} ${{ matrix.chaos_flags }}
- name: Upload sim results on failure (chaos)
+8 -1
View File
@@ -49,12 +49,19 @@ RUN git config --system --add safe.directory /src
# the compiler the rest of CI uses, and bumping the pin rebuilds the image. The
# builder this replaces installed `stable` and never copied rust-toolchain.toml,
# so it compiled with a different compiler from the release and nothing said so.
#
# CARGO_NET_RETRY raises cargo's own retry count for crate downloads, both for
# cargo install below and for the package build that runs in this image; the
# registry is fetched cold on every CI runner and has failed mid-download.
# CARGO_HTTP_MULTIPLEXING is left at its default: turn it off only if HTTP/2
# framing errors from the registry still occur with the higher retry count.
ARG RUST_TOOLCHAIN
ENV RUSTUP_HOME=/usr/local/rustup \
CARGO_HOME=/usr/local/cargo \
CARGO_NET_RETRY=10 \
PATH=/usr/local/cargo/bin:$PATH
RUN test -n "${RUST_TOOLCHAIN}" \
&& curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \
&& curl --retry 3 --retry-connrefused --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \
| sh -s -- -y --profile minimal --default-toolchain "${RUST_TOOLCHAIN}" \
&& chmod -R a+w "$RUSTUP_HOME" "$CARGO_HOME"
+5 -1
View File
@@ -31,6 +31,8 @@ REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
# shellcheck source=../build-floor.env
. "$REPO_ROOT/packaging/build-floor.env"
# shellcheck source=SCRIPTDIR/../../testing/lib/image-build.sh
. "$REPO_ROOT/testing/lib/image-build.sh"
DEST_DIR="$REPO_ROOT/deploy"
VERSION=""
@@ -115,7 +117,9 @@ fi
if [ "$BUILT_IMAGE" -eq 1 ]; then
echo "=== Building $IMAGE_TAG from $FIPS_BUILD_IMAGE with Rust $RUST_TOOLCHAIN ===" >&2
docker build \
# Retried because the build pulls the floor image and fetches apt packages,
# rustup and crates, any of which can fail for a few seconds at a time.
retry_build "docker build $IMAGE_TAG" docker build \
--build-arg "BASE=$FIPS_BUILD_IMAGE" \
--build-arg "RUST_TOOLCHAIN=$RUST_TOOLCHAIN" \
-t "$IMAGE_TAG" \
+13 -7
View File
@@ -1,10 +1,10 @@
#!/bin/bash
# Test the fips Debian package install path across target distros.
#
# Each scenario builds (or reuses) the .deb in a cached Debian 12
# cargo-deb image, boots a systemd container with TUN access for the
# target distro, installs the .deb via `apt
# install ./fips_*.deb`, waits for fips.service + fips-dns.service
# Each scenario takes the .deb from --deb, or builds (or reuses) it
# through packaging/debian/build-deb-container.sh, boots a systemd
# container with TUN access for the target distro, installs the .deb
# via `apt install ./fips_*.deb`, waits for fips.service + fips-dns.service
# to come up, and verifies that `dig @127.0.0.53 AAAA <npub>.fips`
# returns a non-empty AAAA answer through the resolver backend that
# fips-dns-setup configured. Then exercises fips-gateway against the
@@ -31,6 +31,8 @@ set -uo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
# shellcheck source=SCRIPTDIR/../lib/systemd-container.sh
source "$SCRIPT_DIR/../lib/systemd-container.sh"
# shellcheck source=SCRIPTDIR/../lib/image-build.sh
source "$SCRIPT_DIR/../lib/image-build.sh"
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
CACHE_DIR="$SCRIPT_DIR/.cache"
DEB_CACHE_DIR="$CACHE_DIR/deb"
@@ -77,10 +79,13 @@ cleanup_container() {
docker rm -f "$name" >/dev/null 2>&1 || true
}
# Build an image from an inline Dockerfile.
build_image() {
local tag="$1"
shift
echo "$@" | docker build -t "$tag" -f - "$REPO_ROOT" >/dev/null 2>&1
local dockerfile="$*"
retry_build "docker build -t $tag" build_inline "$tag" "$dockerfile" "$REPO_ROOT" || return
return 0
}
# Start the scenario's systemd container. Not privileged: see
@@ -94,7 +99,8 @@ build_image() {
start_systemd_container_with_tun() {
local name="$1" image="$2"
cleanup_container "$name"
docker run -d --name "$name" \
run_quiet "docker run $name (with tun)" \
docker run -d --name "$name" \
--label com.corganlabs.fips-ci=1 \
"${SYSTEMD_CAPS[@]}" \
--cgroupns=host \
@@ -102,7 +108,7 @@ start_systemd_container_with_tun() {
--sysctl net.ipv6.conf.all.forwarding=1 \
-v /sys/fs/cgroup:/sys/fs/cgroup:rw \
--tmpfs /run --tmpfs /run/lock \
"$image" >/dev/null 2>&1 || return
"$image" || return
check_isolation "$name"
return
}
+5 -32
View File
@@ -31,6 +31,8 @@ set -uo pipefail
SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
# shellcheck source=SCRIPTDIR/../lib/systemd-container.sh
source "$SCRIPT_DIR/../lib/systemd-container.sh"
# shellcheck source=SCRIPTDIR/../lib/image-build.sh
source "$SCRIPT_DIR/../lib/image-build.sh"
REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)"
SETUP_SCRIPT="$REPO_ROOT/packaging/common/fips-dns-setup"
TEARDOWN_SCRIPT="$REPO_ROOT/packaging/common/fips-dns-teardown"
@@ -67,42 +69,13 @@ cleanup_container() {
docker rm -f "$name" >/dev/null 2>&1 || true
}
# Emit a captured output file to stderr, delimited and labelled with the
# command it came from. Callers use this only on failure: a command that
# succeeds leaves no trace, so the suite stays quiet when it is green.
dump_output() {
local label="$1" file="$2"
{
echo " --- $label failed; captured output follows ---"
if [ -s "$file" ]; then
cat "$file"
else
echo " (no output)"
fi
echo " --- end captured output ---"
} >&2
}
# Run a command with both streams captured. Discard the capture on success;
# on failure emit it, so the reason a build or a container start died is not
# thrown away. Stdin is inherited, so a caller may pipe into it.
run_quiet() {
local label="$1"
shift
local out rc=0
out=$(mktemp)
"$@" >"$out" 2>&1 || rc=$?
[ "$rc" -eq 0 ] || dump_output "$label" "$out"
rm -f "$out"
return "$rc"
}
# Build an image from an inline Dockerfile.
build_image() {
local tag="$1"
shift
echo "$@" | run_quiet "docker build -t $tag" \
docker build -t "$tag" -f - "$REPO_ROOT"
local dockerfile="$*"
retry_build "docker build -t $tag" build_inline "$tag" "$dockerfile" "$REPO_ROOT" || return
return 0
}
# Start a systemd container in the background. Not privileged: see
+95
View File
@@ -0,0 +1,95 @@
#!/bin/bash
# Shared helpers for building test images and starting test containers.
#
# Source this file to get dump_output(), run_quiet(), build_inline() and
# retry_build():
# source "$SCRIPT_DIR/../lib/image-build.sh"
# echo "$dockerfile" | run_quiet "docker build -t $tag" \
# docker build -t "$tag" -f - "$REPO_ROOT"
# retry_build "docker build $tag" docker build -t "$tag" "$context"
#
# A build or a container start that fails for a reason outside the project,
# such as a registry timeout, is indistinguishable from one the project caused
# unless its output survives. dump_output and run_quiet keep a command quiet
# when it succeeds and print everything it said when it fails.
#
# Image builds reach Docker Hub, ghcr.io and distribution mirrors, and every one
# of those has timed out or served a truncated file in CI for a few seconds at a
# time. retry_build runs a whole build again after such a failure, so the base
# image is resolved again and each RUN step that fetched packages runs again.
# Emit a captured output file to stderr, delimited and labelled with the
# command it came from. Callers use this only on failure: a command that
# succeeds leaves no trace, so the suite stays quiet when it is green.
dump_output() {
local label="$1" file="$2"
{
echo " --- $label failed; captured output follows ---"
if [ -s "$file" ]; then
cat "$file"
else
echo " (no output)"
fi
echo " --- end captured output ---"
} >&2
}
# Run a command with both streams captured. Discard the capture on success;
# on failure emit it, so the reason a build or a container start died is not
# thrown away. Stdin is inherited, so a caller may pipe into it.
run_quiet() {
local label="$1"
shift
local out rc=0
out=$(mktemp)
"$@" >"$out" 2>&1 || rc=$?
[ "$rc" -eq 0 ] || dump_output "$label" "$out"
rm -f "$out"
return "$rc"
}
# Build TAG from the Dockerfile text DOCKERFILE with context CONTEXT, output
# captured by run_quiet. The text is an argument rather than stdin, so that
# retry_build can run the build again: a pipe can be read only once.
build_inline() {
local tag="$1" dockerfile="$2" context="$3"
echo "$dockerfile" | run_quiet "docker build -t $tag" \
docker build -t "$tag" -f - "$context"
}
# Run an image build, and run it again after a failure, up to three attempts
# with 10 s and then 20 s between them. Use it for builds only: a test or a
# container start that fails is a result, and running it again would hide it.
#
# Every message goes to stderr, because some callers promise that their stdout
# holds only their result. A first-attempt success prints nothing. A success
# after a failure says so, and on GitHub Actions also raises a warning on the
# run summary, so a recovered failure is still counted. Returns the last
# attempt's exit status.
retry_build() {
local label="$1"
shift
local attempts=3 attempt=1 rc=0 wait
while true; do
rc=0
if "$@"; then
if [ "$attempt" -gt 1 ]; then
echo "image-build: $label recovered on attempt $attempt of $attempts" >&2
if [ "${GITHUB_ACTIONS:-}" = "true" ]; then
echo "::warning::image-build: $label recovered on attempt $attempt of $attempts" >&2
fi
fi
return 0
else
rc=$?
fi
if [ "$attempt" -ge "$attempts" ]; then
echo "image-build: $label failed on all $attempts attempts (last exit $rc)" >&2
return "$rc"
fi
wait=$((attempt * 10))
echo "image-build: $label failed on attempt $attempt of $attempts (exit $rc); retrying in ${wait}s" >&2
sleep "$wait"
attempt=$((attempt + 1))
done
}
+15 -3
View File
@@ -10,6 +10,7 @@ GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh"
TOPOLOGY_SCRIPT="$SCRIPT_DIR/setup-topology.sh"
WAIT_LIB="$ROOT_DIR/testing/lib/wait-converge.sh"
RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh"
IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh"
# Must track generate-configs.sh's OUTPUT_DIR and the compose bind-mounts: the
# npubs are read back here after the containers are up, so reading a different
# directory than the one the generator wrote pings an npub no node owns.
@@ -45,6 +46,8 @@ fi
source "$WAIT_LIB"
# shellcheck disable=SC1090
source "$RELAY_LIB"
# shellcheck disable=SC1090
source "$IMAGE_LIB"
RELAY_CONTAINER="fips-nat-relay${FIPS_CI_NAME_SUFFIX:-}"
@@ -500,7 +503,10 @@ run_cone() {
echo "=== NAT lab: cone ==="
cleanup
"$GENERATE_SCRIPT" cone
"${COMPOSE[@]}" --profile cone up -d --build --force-recreate
# Build first, with retries, because the build pulls from registries that
# time out now and then; the start is not retried, since it is the test.
retry_build "compose build (cone)" "${COMPOSE[@]}" --profile cone build
"${COMPOSE[@]}" --profile cone up -d --no-build --force-recreate
"$TOPOLOGY_SCRIPT" cone
wait_for_peers fips-nat-cone-a${FIPS_CI_NAME_SUFFIX:-} 1 45 || {
dump_cone_diagnostics
@@ -548,7 +554,10 @@ run_symmetric() {
echo "=== NAT lab: symmetric fallback ==="
cleanup
NAT_MODE_A=symmetric NAT_MODE_B=symmetric "$GENERATE_SCRIPT" symmetric
NAT_MODE_A=symmetric NAT_MODE_B=symmetric "${COMPOSE[@]}" --profile symmetric up -d --build --force-recreate
# Build first, with retries, because the build pulls from registries that
# time out now and then; the start is not retried, since it is the test.
NAT_MODE_A=symmetric NAT_MODE_B=symmetric retry_build "compose build (symmetric)" "${COMPOSE[@]}" --profile symmetric build
NAT_MODE_A=symmetric NAT_MODE_B=symmetric "${COMPOSE[@]}" --profile symmetric up -d --no-build --force-recreate
"$TOPOLOGY_SCRIPT" symmetric
wait_for_peers fips-nat-symmetric-a${FIPS_CI_NAME_SUFFIX:-} 1 60 || {
dump_symmetric_diagnostics
@@ -598,7 +607,10 @@ run_lan() {
echo "=== NAT lab: lan preference ==="
cleanup
"$GENERATE_SCRIPT" lan
"${COMPOSE[@]}" --profile lan up -d --build --force-recreate
# Build first, with retries, because the build pulls from registries that
# time out now and then; the start is not retried, since it is the test.
retry_build "compose build (lan)" "${COMPOSE[@]}" --profile lan build
"${COMPOSE[@]}" --profile lan up -d --no-build --force-recreate
wait_for_peers fips-nat-lan-a${FIPS_CI_NAME_SUFFIX:-} 1 45 || {
dump_lan_diagnostics
return 1
+7 -1
View File
@@ -22,6 +22,7 @@ BUILD_SCRIPT="$ROOT_DIR/testing/scripts/build.sh"
GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh"
WAIT_LIB="$ROOT_DIR/testing/lib/wait-converge.sh"
RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh"
IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh"
# Must track generate-configs.sh's OUTPUT_DIR and the compose bind-mounts.
CONFIG_DIR="$NAT_DIR/generated-configs${FIPS_CI_NAME_SUFFIX:-}"
@@ -55,6 +56,8 @@ RELAY_CONTAINER="fips-nat-relay${FIPS_CI_NAME_SUFFIX:-}"
source "$WAIT_LIB"
# shellcheck disable=SC1090
source "$RELAY_LIB"
# shellcheck disable=SC1090
source "$IMAGE_LIB"
cleanup() {
"${COMPOSE[@]}" --profile "$PROFILE" down -v --remove-orphans \
@@ -430,7 +433,10 @@ run_test() {
cleanup
"$GENERATE_SCRIPT" "$SCENARIO"
"${COMPOSE[@]}" --profile "$PROFILE" up -d --build --force-recreate
# Build first, with retries, because the build pulls from registries that
# time out now and then; the start is not retried, since it is the test.
retry_build "compose build ($PROFILE)" "${COMPOSE[@]}" --profile "$PROFILE" build
"${COMPOSE[@]}" --profile "$PROFILE" up -d --no-build --force-recreate
# Phase 1 + Phase 2 together: each side publishes its own advert,
# subscribes for the other's, then dials. Bidirectional success
+7 -1
View File
@@ -30,6 +30,7 @@ ROOT_DIR="$(cd "$NAT_DIR/../.." && pwd)"
BUILD_SCRIPT="$ROOT_DIR/testing/scripts/build.sh"
GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh"
RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh"
IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh"
PROFILE="stun-faults"
SCENARIO="$PROFILE"
@@ -73,6 +74,8 @@ cleanup() {
# shellcheck disable=SC1090
source "$RELAY_LIB"
# shellcheck disable=SC1090
source "$IMAGE_LIB"
trap 'echo ""; echo "stun-faults-test interrupted"; cleanup; exit 130' INT TERM
@@ -269,7 +272,10 @@ run_test() {
echo "=== stun-faults-test: setup ==="
cleanup
"$GENERATE_SCRIPT" "$SCENARIO"
"${COMPOSE[@]}" --profile "$PROFILE" up -d --build --force-recreate
# Build first, with retries, because the build pulls from registries that
# time out now and then; the start is not retried, since it is the test.
retry_build "compose build ($PROFILE)" "${COMPOSE[@]}" --profile "$PROFILE" build
"${COMPOSE[@]}" --profile "$PROFILE" up -d --no-build --force-recreate
# Give the daemons time to come up. Both fault-node and fault-peer
# need to start, publish their adverts to the relay, and discover