diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 836a4db3..b42f1f10 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -763,8 +763,10 @@ jobs: chmod +x _bin/native-echo _bin/native-surface cp _bin/native-echo testing/docker/native-echo cp _bin/native-surface testing/docker/native-surface - docker build -t fips-test:latest testing/docker - docker build -t fips-test-app:latest -f testing/docker/Dockerfile.app testing/docker + # Retried: both builds pull from Docker Hub and the Debian mirrors. + source testing/lib/image-build.sh + retry_build "docker build fips-test" docker build -t fips-test:latest testing/docker + retry_build "docker build fips-test-app" docker build -t fips-test-app:latest -f testing/docker/Dockerfile.app testing/docker # ── Static topology ──────────────────────────────────────────────────── - name: Generate configs (static) @@ -830,8 +832,12 @@ jobs: if: matrix.type == 'chaos' run: pip3 install --quiet pyyaml jinja2 + # With FIPS_TEST_IMAGE set, the chaos runner uses the image the job built + # above instead of building fips-test:latest again (testing/chaos/sim/runner.py). - name: Run chaos scenario if: matrix.type == 'chaos' + env: + FIPS_TEST_IMAGE: fips-test:latest run: bash testing/chaos/scripts/chaos.sh ${{ matrix.scenario }} ${{ matrix.chaos_flags }} - name: Upload sim results on failure (chaos) diff --git a/packaging/debian/Dockerfile.build b/packaging/debian/Dockerfile.build index 6e69e644..4df8c81c 100644 --- a/packaging/debian/Dockerfile.build +++ b/packaging/debian/Dockerfile.build @@ -49,12 +49,19 @@ RUN git config --system --add safe.directory /src # the compiler the rest of CI uses, and bumping the pin rebuilds the image. The # builder this replaces installed `stable` and never copied rust-toolchain.toml, # so it compiled with a different compiler from the release and nothing said so. +# +# CARGO_NET_RETRY raises cargo's own retry count for crate downloads, both for +# cargo install below and for the package build that runs in this image; the +# registry is fetched cold on every CI runner and has failed mid-download. +# CARGO_HTTP_MULTIPLEXING is left at its default: turn it off only if HTTP/2 +# framing errors from the registry still occur with the higher retry count. ARG RUST_TOOLCHAIN ENV RUSTUP_HOME=/usr/local/rustup \ CARGO_HOME=/usr/local/cargo \ + CARGO_NET_RETRY=10 \ PATH=/usr/local/cargo/bin:$PATH RUN test -n "${RUST_TOOLCHAIN}" \ - && curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + && curl --retry 3 --retry-connrefused --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ | sh -s -- -y --profile minimal --default-toolchain "${RUST_TOOLCHAIN}" \ && chmod -R a+w "$RUSTUP_HOME" "$CARGO_HOME" diff --git a/packaging/debian/build-deb-container.sh b/packaging/debian/build-deb-container.sh index 29f75f59..ca19ddc9 100755 --- a/packaging/debian/build-deb-container.sh +++ b/packaging/debian/build-deb-container.sh @@ -31,6 +31,8 @@ REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" # shellcheck source=../build-floor.env . "$REPO_ROOT/packaging/build-floor.env" +# shellcheck source=SCRIPTDIR/../../testing/lib/image-build.sh +. "$REPO_ROOT/testing/lib/image-build.sh" DEST_DIR="$REPO_ROOT/deploy" VERSION="" @@ -115,7 +117,9 @@ fi if [ "$BUILT_IMAGE" -eq 1 ]; then echo "=== Building $IMAGE_TAG from $FIPS_BUILD_IMAGE with Rust $RUST_TOOLCHAIN ===" >&2 - docker build \ + # Retried because the build pulls the floor image and fetches apt packages, + # rustup and crates, any of which can fail for a few seconds at a time. + retry_build "docker build $IMAGE_TAG" docker build \ --build-arg "BASE=$FIPS_BUILD_IMAGE" \ --build-arg "RUST_TOOLCHAIN=$RUST_TOOLCHAIN" \ -t "$IMAGE_TAG" \ diff --git a/testing/deb-install/test.sh b/testing/deb-install/test.sh index 1cfa5ed3..31f421a2 100755 --- a/testing/deb-install/test.sh +++ b/testing/deb-install/test.sh @@ -1,10 +1,10 @@ #!/bin/bash # Test the fips Debian package install path across target distros. # -# Each scenario builds (or reuses) the .deb in a cached Debian 12 -# cargo-deb image, boots a systemd container with TUN access for the -# target distro, installs the .deb via `apt -# install ./fips_*.deb`, waits for fips.service + fips-dns.service +# Each scenario takes the .deb from --deb, or builds (or reuses) it +# through packaging/debian/build-deb-container.sh, boots a systemd +# container with TUN access for the target distro, installs the .deb +# via `apt install ./fips_*.deb`, waits for fips.service + fips-dns.service # to come up, and verifies that `dig @127.0.0.53 AAAA .fips` # returns a non-empty AAAA answer through the resolver backend that # fips-dns-setup configured. Then exercises fips-gateway against the @@ -31,6 +31,8 @@ set -uo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" # shellcheck source=SCRIPTDIR/../lib/systemd-container.sh source "$SCRIPT_DIR/../lib/systemd-container.sh" +# shellcheck source=SCRIPTDIR/../lib/image-build.sh +source "$SCRIPT_DIR/../lib/image-build.sh" REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" CACHE_DIR="$SCRIPT_DIR/.cache" DEB_CACHE_DIR="$CACHE_DIR/deb" @@ -77,10 +79,13 @@ cleanup_container() { docker rm -f "$name" >/dev/null 2>&1 || true } +# Build an image from an inline Dockerfile. build_image() { local tag="$1" shift - echo "$@" | docker build -t "$tag" -f - "$REPO_ROOT" >/dev/null 2>&1 + local dockerfile="$*" + retry_build "docker build -t $tag" build_inline "$tag" "$dockerfile" "$REPO_ROOT" || return + return 0 } # Start the scenario's systemd container. Not privileged: see @@ -94,7 +99,8 @@ build_image() { start_systemd_container_with_tun() { local name="$1" image="$2" cleanup_container "$name" - docker run -d --name "$name" \ + run_quiet "docker run $name (with tun)" \ + docker run -d --name "$name" \ --label com.corganlabs.fips-ci=1 \ "${SYSTEMD_CAPS[@]}" \ --cgroupns=host \ @@ -102,7 +108,7 @@ start_systemd_container_with_tun() { --sysctl net.ipv6.conf.all.forwarding=1 \ -v /sys/fs/cgroup:/sys/fs/cgroup:rw \ --tmpfs /run --tmpfs /run/lock \ - "$image" >/dev/null 2>&1 || return + "$image" || return check_isolation "$name" return } diff --git a/testing/dns-resolver/test.sh b/testing/dns-resolver/test.sh index 3dccef8b..038a1190 100755 --- a/testing/dns-resolver/test.sh +++ b/testing/dns-resolver/test.sh @@ -31,6 +31,8 @@ set -uo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" # shellcheck source=SCRIPTDIR/../lib/systemd-container.sh source "$SCRIPT_DIR/../lib/systemd-container.sh" +# shellcheck source=SCRIPTDIR/../lib/image-build.sh +source "$SCRIPT_DIR/../lib/image-build.sh" REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" SETUP_SCRIPT="$REPO_ROOT/packaging/common/fips-dns-setup" TEARDOWN_SCRIPT="$REPO_ROOT/packaging/common/fips-dns-teardown" @@ -67,42 +69,13 @@ cleanup_container() { docker rm -f "$name" >/dev/null 2>&1 || true } -# Emit a captured output file to stderr, delimited and labelled with the -# command it came from. Callers use this only on failure: a command that -# succeeds leaves no trace, so the suite stays quiet when it is green. -dump_output() { - local label="$1" file="$2" - { - echo " --- $label failed; captured output follows ---" - if [ -s "$file" ]; then - cat "$file" - else - echo " (no output)" - fi - echo " --- end captured output ---" - } >&2 -} - -# Run a command with both streams captured. Discard the capture on success; -# on failure emit it, so the reason a build or a container start died is not -# thrown away. Stdin is inherited, so a caller may pipe into it. -run_quiet() { - local label="$1" - shift - local out rc=0 - out=$(mktemp) - "$@" >"$out" 2>&1 || rc=$? - [ "$rc" -eq 0 ] || dump_output "$label" "$out" - rm -f "$out" - return "$rc" -} - # Build an image from an inline Dockerfile. build_image() { local tag="$1" shift - echo "$@" | run_quiet "docker build -t $tag" \ - docker build -t "$tag" -f - "$REPO_ROOT" + local dockerfile="$*" + retry_build "docker build -t $tag" build_inline "$tag" "$dockerfile" "$REPO_ROOT" || return + return 0 } # Start a systemd container in the background. Not privileged: see diff --git a/testing/lib/image-build.sh b/testing/lib/image-build.sh new file mode 100644 index 00000000..f7367a7c --- /dev/null +++ b/testing/lib/image-build.sh @@ -0,0 +1,95 @@ +#!/bin/bash +# Shared helpers for building test images and starting test containers. +# +# Source this file to get dump_output(), run_quiet(), build_inline() and +# retry_build(): +# source "$SCRIPT_DIR/../lib/image-build.sh" +# echo "$dockerfile" | run_quiet "docker build -t $tag" \ +# docker build -t "$tag" -f - "$REPO_ROOT" +# retry_build "docker build $tag" docker build -t "$tag" "$context" +# +# A build or a container start that fails for a reason outside the project, +# such as a registry timeout, is indistinguishable from one the project caused +# unless its output survives. dump_output and run_quiet keep a command quiet +# when it succeeds and print everything it said when it fails. +# +# Image builds reach Docker Hub, ghcr.io and distribution mirrors, and every one +# of those has timed out or served a truncated file in CI for a few seconds at a +# time. retry_build runs a whole build again after such a failure, so the base +# image is resolved again and each RUN step that fetched packages runs again. + +# Emit a captured output file to stderr, delimited and labelled with the +# command it came from. Callers use this only on failure: a command that +# succeeds leaves no trace, so the suite stays quiet when it is green. +dump_output() { + local label="$1" file="$2" + { + echo " --- $label failed; captured output follows ---" + if [ -s "$file" ]; then + cat "$file" + else + echo " (no output)" + fi + echo " --- end captured output ---" + } >&2 +} + +# Run a command with both streams captured. Discard the capture on success; +# on failure emit it, so the reason a build or a container start died is not +# thrown away. Stdin is inherited, so a caller may pipe into it. +run_quiet() { + local label="$1" + shift + local out rc=0 + out=$(mktemp) + "$@" >"$out" 2>&1 || rc=$? + [ "$rc" -eq 0 ] || dump_output "$label" "$out" + rm -f "$out" + return "$rc" +} + +# Build TAG from the Dockerfile text DOCKERFILE with context CONTEXT, output +# captured by run_quiet. The text is an argument rather than stdin, so that +# retry_build can run the build again: a pipe can be read only once. +build_inline() { + local tag="$1" dockerfile="$2" context="$3" + echo "$dockerfile" | run_quiet "docker build -t $tag" \ + docker build -t "$tag" -f - "$context" +} + +# Run an image build, and run it again after a failure, up to three attempts +# with 10 s and then 20 s between them. Use it for builds only: a test or a +# container start that fails is a result, and running it again would hide it. +# +# Every message goes to stderr, because some callers promise that their stdout +# holds only their result. A first-attempt success prints nothing. A success +# after a failure says so, and on GitHub Actions also raises a warning on the +# run summary, so a recovered failure is still counted. Returns the last +# attempt's exit status. +retry_build() { + local label="$1" + shift + local attempts=3 attempt=1 rc=0 wait + while true; do + rc=0 + if "$@"; then + if [ "$attempt" -gt 1 ]; then + echo "image-build: $label recovered on attempt $attempt of $attempts" >&2 + if [ "${GITHUB_ACTIONS:-}" = "true" ]; then + echo "::warning::image-build: $label recovered on attempt $attempt of $attempts" >&2 + fi + fi + return 0 + else + rc=$? + fi + if [ "$attempt" -ge "$attempts" ]; then + echo "image-build: $label failed on all $attempts attempts (last exit $rc)" >&2 + return "$rc" + fi + wait=$((attempt * 10)) + echo "image-build: $label failed on attempt $attempt of $attempts (exit $rc); retrying in ${wait}s" >&2 + sleep "$wait" + attempt=$((attempt + 1)) + done +} diff --git a/testing/nat/scripts/nat-test.sh b/testing/nat/scripts/nat-test.sh index 5e9c040e..76df689f 100755 --- a/testing/nat/scripts/nat-test.sh +++ b/testing/nat/scripts/nat-test.sh @@ -10,6 +10,7 @@ GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh" TOPOLOGY_SCRIPT="$SCRIPT_DIR/setup-topology.sh" WAIT_LIB="$ROOT_DIR/testing/lib/wait-converge.sh" RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh" +IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh" # Must track generate-configs.sh's OUTPUT_DIR and the compose bind-mounts: the # npubs are read back here after the containers are up, so reading a different # directory than the one the generator wrote pings an npub no node owns. @@ -45,6 +46,8 @@ fi source "$WAIT_LIB" # shellcheck disable=SC1090 source "$RELAY_LIB" +# shellcheck disable=SC1090 +source "$IMAGE_LIB" RELAY_CONTAINER="fips-nat-relay${FIPS_CI_NAME_SUFFIX:-}" @@ -500,7 +503,10 @@ run_cone() { echo "=== NAT lab: cone ===" cleanup "$GENERATE_SCRIPT" cone - "${COMPOSE[@]}" --profile cone up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + retry_build "compose build (cone)" "${COMPOSE[@]}" --profile cone build + "${COMPOSE[@]}" --profile cone up -d --no-build --force-recreate "$TOPOLOGY_SCRIPT" cone wait_for_peers fips-nat-cone-a${FIPS_CI_NAME_SUFFIX:-} 1 45 || { dump_cone_diagnostics @@ -548,7 +554,10 @@ run_symmetric() { echo "=== NAT lab: symmetric fallback ===" cleanup NAT_MODE_A=symmetric NAT_MODE_B=symmetric "$GENERATE_SCRIPT" symmetric - NAT_MODE_A=symmetric NAT_MODE_B=symmetric "${COMPOSE[@]}" --profile symmetric up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + NAT_MODE_A=symmetric NAT_MODE_B=symmetric retry_build "compose build (symmetric)" "${COMPOSE[@]}" --profile symmetric build + NAT_MODE_A=symmetric NAT_MODE_B=symmetric "${COMPOSE[@]}" --profile symmetric up -d --no-build --force-recreate "$TOPOLOGY_SCRIPT" symmetric wait_for_peers fips-nat-symmetric-a${FIPS_CI_NAME_SUFFIX:-} 1 60 || { dump_symmetric_diagnostics @@ -598,7 +607,10 @@ run_lan() { echo "=== NAT lab: lan preference ===" cleanup "$GENERATE_SCRIPT" lan - "${COMPOSE[@]}" --profile lan up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + retry_build "compose build (lan)" "${COMPOSE[@]}" --profile lan build + "${COMPOSE[@]}" --profile lan up -d --no-build --force-recreate wait_for_peers fips-nat-lan-a${FIPS_CI_NAME_SUFFIX:-} 1 45 || { dump_lan_diagnostics return 1 diff --git a/testing/nat/scripts/nostr-relay-test.sh b/testing/nat/scripts/nostr-relay-test.sh index df6bdfff..e6b01795 100755 --- a/testing/nat/scripts/nostr-relay-test.sh +++ b/testing/nat/scripts/nostr-relay-test.sh @@ -22,6 +22,7 @@ BUILD_SCRIPT="$ROOT_DIR/testing/scripts/build.sh" GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh" WAIT_LIB="$ROOT_DIR/testing/lib/wait-converge.sh" RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh" +IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh" # Must track generate-configs.sh's OUTPUT_DIR and the compose bind-mounts. CONFIG_DIR="$NAT_DIR/generated-configs${FIPS_CI_NAME_SUFFIX:-}" @@ -55,6 +56,8 @@ RELAY_CONTAINER="fips-nat-relay${FIPS_CI_NAME_SUFFIX:-}" source "$WAIT_LIB" # shellcheck disable=SC1090 source "$RELAY_LIB" +# shellcheck disable=SC1090 +source "$IMAGE_LIB" cleanup() { "${COMPOSE[@]}" --profile "$PROFILE" down -v --remove-orphans \ @@ -430,7 +433,10 @@ run_test() { cleanup "$GENERATE_SCRIPT" "$SCENARIO" - "${COMPOSE[@]}" --profile "$PROFILE" up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + retry_build "compose build ($PROFILE)" "${COMPOSE[@]}" --profile "$PROFILE" build + "${COMPOSE[@]}" --profile "$PROFILE" up -d --no-build --force-recreate # Phase 1 + Phase 2 together: each side publishes its own advert, # subscribes for the other's, then dials. Bidirectional success diff --git a/testing/nat/scripts/stun-faults-test.sh b/testing/nat/scripts/stun-faults-test.sh index 149d5b33..57552681 100755 --- a/testing/nat/scripts/stun-faults-test.sh +++ b/testing/nat/scripts/stun-faults-test.sh @@ -30,6 +30,7 @@ ROOT_DIR="$(cd "$NAT_DIR/../.." && pwd)" BUILD_SCRIPT="$ROOT_DIR/testing/scripts/build.sh" GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh" RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh" +IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh" PROFILE="stun-faults" SCENARIO="$PROFILE" @@ -73,6 +74,8 @@ cleanup() { # shellcheck disable=SC1090 source "$RELAY_LIB" +# shellcheck disable=SC1090 +source "$IMAGE_LIB" trap 'echo ""; echo "stun-faults-test interrupted"; cleanup; exit 130' INT TERM @@ -269,7 +272,10 @@ run_test() { echo "=== stun-faults-test: setup ===" cleanup "$GENERATE_SCRIPT" "$SCENARIO" - "${COMPOSE[@]}" --profile "$PROFILE" up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + retry_build "compose build ($PROFILE)" "${COMPOSE[@]}" --profile "$PROFILE" build + "${COMPOSE[@]}" --profile "$PROFILE" up -d --no-build --force-recreate # Give the daemons time to come up. Both fault-node and fault-peer # need to start, publish their adverts to the relay, and discover