From 217bea308626f3bd77cc553e39f2f2135428ef50 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 19 Sep 2026 13:30:33 +0000 Subject: [PATCH 1/4] Show why a deb-install image build or container start failed The deb-install suite sent the output of its runtime image build and of its container start to /dev/null, so a registry timeout or a container that was never created surfaced only as "runtime build failed" with the cause thrown away. The dns-resolver suite already captured that output and printed it on failure. Move its capture helpers into testing/lib/image-build.sh and use them in both suites, so a failed build or start in deb-install prints what docker said. The dns-resolver change is a move only. Also correct the deb-install header, which still described the package build it no longer does itself. --- testing/deb-install/test.sh | 19 ++++++++++------ testing/dns-resolver/test.sh | 32 ++------------------------- testing/lib/image-build.sh | 42 ++++++++++++++++++++++++++++++++++++ 3 files changed, 56 insertions(+), 37 deletions(-) create mode 100644 testing/lib/image-build.sh diff --git a/testing/deb-install/test.sh b/testing/deb-install/test.sh index 1cfa5ed3..5e47606c 100755 --- a/testing/deb-install/test.sh +++ b/testing/deb-install/test.sh @@ -1,10 +1,10 @@ #!/bin/bash # Test the fips Debian package install path across target distros. # -# Each scenario builds (or reuses) the .deb in a cached Debian 12 -# cargo-deb image, boots a systemd container with TUN access for the -# target distro, installs the .deb via `apt -# install ./fips_*.deb`, waits for fips.service + fips-dns.service +# Each scenario takes the .deb from --deb, or builds (or reuses) it +# through packaging/debian/build-deb-container.sh, boots a systemd +# container with TUN access for the target distro, installs the .deb +# via `apt install ./fips_*.deb`, waits for fips.service + fips-dns.service # to come up, and verifies that `dig @127.0.0.53 AAAA .fips` # returns a non-empty AAAA answer through the resolver backend that # fips-dns-setup configured. Then exercises fips-gateway against the @@ -31,6 +31,8 @@ set -uo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" # shellcheck source=SCRIPTDIR/../lib/systemd-container.sh source "$SCRIPT_DIR/../lib/systemd-container.sh" +# shellcheck source=SCRIPTDIR/../lib/image-build.sh +source "$SCRIPT_DIR/../lib/image-build.sh" REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" CACHE_DIR="$SCRIPT_DIR/.cache" DEB_CACHE_DIR="$CACHE_DIR/deb" @@ -77,10 +79,12 @@ cleanup_container() { docker rm -f "$name" >/dev/null 2>&1 || true } +# Build an image from an inline Dockerfile. build_image() { local tag="$1" shift - echo "$@" | docker build -t "$tag" -f - "$REPO_ROOT" >/dev/null 2>&1 + echo "$@" | run_quiet "docker build -t $tag" \ + docker build -t "$tag" -f - "$REPO_ROOT" } # Start the scenario's systemd container. Not privileged: see @@ -94,7 +98,8 @@ build_image() { start_systemd_container_with_tun() { local name="$1" image="$2" cleanup_container "$name" - docker run -d --name "$name" \ + run_quiet "docker run $name (with tun)" \ + docker run -d --name "$name" \ --label com.corganlabs.fips-ci=1 \ "${SYSTEMD_CAPS[@]}" \ --cgroupns=host \ @@ -102,7 +107,7 @@ start_systemd_container_with_tun() { --sysctl net.ipv6.conf.all.forwarding=1 \ -v /sys/fs/cgroup:/sys/fs/cgroup:rw \ --tmpfs /run --tmpfs /run/lock \ - "$image" >/dev/null 2>&1 || return + "$image" || return check_isolation "$name" return } diff --git a/testing/dns-resolver/test.sh b/testing/dns-resolver/test.sh index 3dccef8b..c9b347e4 100755 --- a/testing/dns-resolver/test.sh +++ b/testing/dns-resolver/test.sh @@ -31,6 +31,8 @@ set -uo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" # shellcheck source=SCRIPTDIR/../lib/systemd-container.sh source "$SCRIPT_DIR/../lib/systemd-container.sh" +# shellcheck source=SCRIPTDIR/../lib/image-build.sh +source "$SCRIPT_DIR/../lib/image-build.sh" REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" SETUP_SCRIPT="$REPO_ROOT/packaging/common/fips-dns-setup" TEARDOWN_SCRIPT="$REPO_ROOT/packaging/common/fips-dns-teardown" @@ -67,36 +69,6 @@ cleanup_container() { docker rm -f "$name" >/dev/null 2>&1 || true } -# Emit a captured output file to stderr, delimited and labelled with the -# command it came from. Callers use this only on failure: a command that -# succeeds leaves no trace, so the suite stays quiet when it is green. -dump_output() { - local label="$1" file="$2" - { - echo " --- $label failed; captured output follows ---" - if [ -s "$file" ]; then - cat "$file" - else - echo " (no output)" - fi - echo " --- end captured output ---" - } >&2 -} - -# Run a command with both streams captured. Discard the capture on success; -# on failure emit it, so the reason a build or a container start died is not -# thrown away. Stdin is inherited, so a caller may pipe into it. -run_quiet() { - local label="$1" - shift - local out rc=0 - out=$(mktemp) - "$@" >"$out" 2>&1 || rc=$? - [ "$rc" -eq 0 ] || dump_output "$label" "$out" - rm -f "$out" - return "$rc" -} - # Build an image from an inline Dockerfile. build_image() { local tag="$1" diff --git a/testing/lib/image-build.sh b/testing/lib/image-build.sh new file mode 100644 index 00000000..0cf38435 --- /dev/null +++ b/testing/lib/image-build.sh @@ -0,0 +1,42 @@ +#!/bin/bash +# Shared helpers for building test images and starting test containers. +# +# Source this file to get dump_output() and run_quiet(): +# source "$SCRIPT_DIR/../lib/image-build.sh" +# echo "$dockerfile" | run_quiet "docker build -t $tag" \ +# docker build -t "$tag" -f - "$REPO_ROOT" +# +# A build or a container start that fails for a reason outside the project, +# such as a registry timeout, is indistinguishable from one the project caused +# unless its output survives. These helpers keep a command quiet when it +# succeeds and print everything it said when it fails. + +# Emit a captured output file to stderr, delimited and labelled with the +# command it came from. Callers use this only on failure: a command that +# succeeds leaves no trace, so the suite stays quiet when it is green. +dump_output() { + local label="$1" file="$2" + { + echo " --- $label failed; captured output follows ---" + if [ -s "$file" ]; then + cat "$file" + else + echo " (no output)" + fi + echo " --- end captured output ---" + } >&2 +} + +# Run a command with both streams captured. Discard the capture on success; +# on failure emit it, so the reason a build or a container start died is not +# thrown away. Stdin is inherited, so a caller may pipe into it. +run_quiet() { + local label="$1" + shift + local out rc=0 + out=$(mktemp) + "$@" >"$out" 2>&1 || rc=$? + [ "$rc" -eq 0 ] || dump_output "$label" "$out" + rm -f "$out" + return "$rc" +} From d8574ce8de21333e7f07eed42efeb0f8a266da79 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 19 Sep 2026 13:32:12 +0000 Subject: [PATCH 2/4] Let the CI chaos legs use the test image the job already built The chaos step set no FIPS_TEST_IMAGE, so the chaos runner built fips-test:latest from testing/docker a second time on each of the five chaos legs, after the job had already built it. Each of those builds was one more registry contact that could time out. Pass the job's image in, as the native-api step already does, and the runner uses it. --- .github/workflows/ci.yml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d53d00ce..e360605d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -638,8 +638,12 @@ jobs: if: matrix.type == 'chaos' run: pip3 install --quiet pyyaml jinja2 + # With FIPS_TEST_IMAGE set, the chaos runner uses the image the job built + # above instead of building fips-test:latest again (testing/chaos/sim/runner.py). - name: Run chaos scenario if: matrix.type == 'chaos' + env: + FIPS_TEST_IMAGE: fips-test:latest run: bash testing/chaos/scripts/chaos.sh ${{ matrix.scenario }} ${{ matrix.chaos_flags }} - name: Upload sim results on failure (chaos) From 0c3c0c79c06d030f727f26110449ded535ebc006 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 19 Sep 2026 13:38:47 +0000 Subject: [PATCH 3/4] Retry test, package and NAT lab image builds across transient registry failures Image builds in CI pull base images from Docker Hub and ghcr.io and fetch packages from distribution mirrors, and each has failed a leg for a few seconds at a time: a registry dial timeout, a 502, an apt file truncated mid-fetch. A single failed build failed the leg, and a rerun of the whole workflow was the only recovery. Add retry_build to testing/lib/image-build.sh: up to three attempts with 10 s and then 20 s between them, whole-build so the base is resolved again and every package-fetching RUN step runs again. It is quiet on a first-attempt success, prints each failed attempt and a recovery to stderr, and on GitHub Actions also raises a warning on the run summary when a build recovered, so recovered failures remain countable. It never wraps a test or a container start. Use it for the shared test images in the integration job, the deb-install and dns-resolver runtime images, and the package builder image. Each deb-install and dns-resolver attempt prints its captured output on failure, so a recovered failure still shows its cause. The NAT, nostr publish-consume and STUN fault suites built their lab images inside `docker compose up --build`: the relay image from ghcr.io and Alpine, the STUN server from the Python image, and the routers from Debian with apt. Build the profile's images first through retry_build, then bring the lab up with --no-build. The start is not retried, since a container that fails to start is a test result; only the build is. --- .github/workflows/ci.yml | 6 ++- packaging/debian/build-deb-container.sh | 6 ++- testing/deb-install/test.sh | 5 ++- testing/dns-resolver/test.sh | 5 ++- testing/lib/image-build.sh | 59 +++++++++++++++++++++++-- testing/nat/scripts/nat-test.sh | 18 ++++++-- testing/nat/scripts/nostr-relay-test.sh | 8 +++- testing/nat/scripts/stun-faults-test.sh | 8 +++- 8 files changed, 100 insertions(+), 15 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e360605d..3f5f1fdc 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -587,8 +587,10 @@ jobs: chmod +x _bin/native-echo _bin/native-surface cp _bin/native-echo testing/docker/native-echo cp _bin/native-surface testing/docker/native-surface - docker build -t fips-test:latest testing/docker - docker build -t fips-test-app:latest -f testing/docker/Dockerfile.app testing/docker + # Retried: both builds pull from Docker Hub and the Debian mirrors. + source testing/lib/image-build.sh + retry_build "docker build fips-test" docker build -t fips-test:latest testing/docker + retry_build "docker build fips-test-app" docker build -t fips-test-app:latest -f testing/docker/Dockerfile.app testing/docker # ── Static topology ──────────────────────────────────────────────────── - name: Generate configs (static) diff --git a/packaging/debian/build-deb-container.sh b/packaging/debian/build-deb-container.sh index 29f75f59..ca19ddc9 100755 --- a/packaging/debian/build-deb-container.sh +++ b/packaging/debian/build-deb-container.sh @@ -31,6 +31,8 @@ REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" # shellcheck source=../build-floor.env . "$REPO_ROOT/packaging/build-floor.env" +# shellcheck source=SCRIPTDIR/../../testing/lib/image-build.sh +. "$REPO_ROOT/testing/lib/image-build.sh" DEST_DIR="$REPO_ROOT/deploy" VERSION="" @@ -115,7 +117,9 @@ fi if [ "$BUILT_IMAGE" -eq 1 ]; then echo "=== Building $IMAGE_TAG from $FIPS_BUILD_IMAGE with Rust $RUST_TOOLCHAIN ===" >&2 - docker build \ + # Retried because the build pulls the floor image and fetches apt packages, + # rustup and crates, any of which can fail for a few seconds at a time. + retry_build "docker build $IMAGE_TAG" docker build \ --build-arg "BASE=$FIPS_BUILD_IMAGE" \ --build-arg "RUST_TOOLCHAIN=$RUST_TOOLCHAIN" \ -t "$IMAGE_TAG" \ diff --git a/testing/deb-install/test.sh b/testing/deb-install/test.sh index 5e47606c..31f421a2 100755 --- a/testing/deb-install/test.sh +++ b/testing/deb-install/test.sh @@ -83,8 +83,9 @@ cleanup_container() { build_image() { local tag="$1" shift - echo "$@" | run_quiet "docker build -t $tag" \ - docker build -t "$tag" -f - "$REPO_ROOT" + local dockerfile="$*" + retry_build "docker build -t $tag" build_inline "$tag" "$dockerfile" "$REPO_ROOT" || return + return 0 } # Start the scenario's systemd container. Not privileged: see diff --git a/testing/dns-resolver/test.sh b/testing/dns-resolver/test.sh index c9b347e4..038a1190 100755 --- a/testing/dns-resolver/test.sh +++ b/testing/dns-resolver/test.sh @@ -73,8 +73,9 @@ cleanup_container() { build_image() { local tag="$1" shift - echo "$@" | run_quiet "docker build -t $tag" \ - docker build -t "$tag" -f - "$REPO_ROOT" + local dockerfile="$*" + retry_build "docker build -t $tag" build_inline "$tag" "$dockerfile" "$REPO_ROOT" || return + return 0 } # Start a systemd container in the background. Not privileged: see diff --git a/testing/lib/image-build.sh b/testing/lib/image-build.sh index 0cf38435..f7367a7c 100644 --- a/testing/lib/image-build.sh +++ b/testing/lib/image-build.sh @@ -1,15 +1,22 @@ #!/bin/bash # Shared helpers for building test images and starting test containers. # -# Source this file to get dump_output() and run_quiet(): +# Source this file to get dump_output(), run_quiet(), build_inline() and +# retry_build(): # source "$SCRIPT_DIR/../lib/image-build.sh" # echo "$dockerfile" | run_quiet "docker build -t $tag" \ # docker build -t "$tag" -f - "$REPO_ROOT" +# retry_build "docker build $tag" docker build -t "$tag" "$context" # # A build or a container start that fails for a reason outside the project, # such as a registry timeout, is indistinguishable from one the project caused -# unless its output survives. These helpers keep a command quiet when it -# succeeds and print everything it said when it fails. +# unless its output survives. dump_output and run_quiet keep a command quiet +# when it succeeds and print everything it said when it fails. +# +# Image builds reach Docker Hub, ghcr.io and distribution mirrors, and every one +# of those has timed out or served a truncated file in CI for a few seconds at a +# time. retry_build runs a whole build again after such a failure, so the base +# image is resolved again and each RUN step that fetched packages runs again. # Emit a captured output file to stderr, delimited and labelled with the # command it came from. Callers use this only on failure: a command that @@ -40,3 +47,49 @@ run_quiet() { rm -f "$out" return "$rc" } + +# Build TAG from the Dockerfile text DOCKERFILE with context CONTEXT, output +# captured by run_quiet. The text is an argument rather than stdin, so that +# retry_build can run the build again: a pipe can be read only once. +build_inline() { + local tag="$1" dockerfile="$2" context="$3" + echo "$dockerfile" | run_quiet "docker build -t $tag" \ + docker build -t "$tag" -f - "$context" +} + +# Run an image build, and run it again after a failure, up to three attempts +# with 10 s and then 20 s between them. Use it for builds only: a test or a +# container start that fails is a result, and running it again would hide it. +# +# Every message goes to stderr, because some callers promise that their stdout +# holds only their result. A first-attempt success prints nothing. A success +# after a failure says so, and on GitHub Actions also raises a warning on the +# run summary, so a recovered failure is still counted. Returns the last +# attempt's exit status. +retry_build() { + local label="$1" + shift + local attempts=3 attempt=1 rc=0 wait + while true; do + rc=0 + if "$@"; then + if [ "$attempt" -gt 1 ]; then + echo "image-build: $label recovered on attempt $attempt of $attempts" >&2 + if [ "${GITHUB_ACTIONS:-}" = "true" ]; then + echo "::warning::image-build: $label recovered on attempt $attempt of $attempts" >&2 + fi + fi + return 0 + else + rc=$? + fi + if [ "$attempt" -ge "$attempts" ]; then + echo "image-build: $label failed on all $attempts attempts (last exit $rc)" >&2 + return "$rc" + fi + wait=$((attempt * 10)) + echo "image-build: $label failed on attempt $attempt of $attempts (exit $rc); retrying in ${wait}s" >&2 + sleep "$wait" + attempt=$((attempt + 1)) + done +} diff --git a/testing/nat/scripts/nat-test.sh b/testing/nat/scripts/nat-test.sh index 5e9c040e..76df689f 100755 --- a/testing/nat/scripts/nat-test.sh +++ b/testing/nat/scripts/nat-test.sh @@ -10,6 +10,7 @@ GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh" TOPOLOGY_SCRIPT="$SCRIPT_DIR/setup-topology.sh" WAIT_LIB="$ROOT_DIR/testing/lib/wait-converge.sh" RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh" +IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh" # Must track generate-configs.sh's OUTPUT_DIR and the compose bind-mounts: the # npubs are read back here after the containers are up, so reading a different # directory than the one the generator wrote pings an npub no node owns. @@ -45,6 +46,8 @@ fi source "$WAIT_LIB" # shellcheck disable=SC1090 source "$RELAY_LIB" +# shellcheck disable=SC1090 +source "$IMAGE_LIB" RELAY_CONTAINER="fips-nat-relay${FIPS_CI_NAME_SUFFIX:-}" @@ -500,7 +503,10 @@ run_cone() { echo "=== NAT lab: cone ===" cleanup "$GENERATE_SCRIPT" cone - "${COMPOSE[@]}" --profile cone up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + retry_build "compose build (cone)" "${COMPOSE[@]}" --profile cone build + "${COMPOSE[@]}" --profile cone up -d --no-build --force-recreate "$TOPOLOGY_SCRIPT" cone wait_for_peers fips-nat-cone-a${FIPS_CI_NAME_SUFFIX:-} 1 45 || { dump_cone_diagnostics @@ -548,7 +554,10 @@ run_symmetric() { echo "=== NAT lab: symmetric fallback ===" cleanup NAT_MODE_A=symmetric NAT_MODE_B=symmetric "$GENERATE_SCRIPT" symmetric - NAT_MODE_A=symmetric NAT_MODE_B=symmetric "${COMPOSE[@]}" --profile symmetric up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + NAT_MODE_A=symmetric NAT_MODE_B=symmetric retry_build "compose build (symmetric)" "${COMPOSE[@]}" --profile symmetric build + NAT_MODE_A=symmetric NAT_MODE_B=symmetric "${COMPOSE[@]}" --profile symmetric up -d --no-build --force-recreate "$TOPOLOGY_SCRIPT" symmetric wait_for_peers fips-nat-symmetric-a${FIPS_CI_NAME_SUFFIX:-} 1 60 || { dump_symmetric_diagnostics @@ -598,7 +607,10 @@ run_lan() { echo "=== NAT lab: lan preference ===" cleanup "$GENERATE_SCRIPT" lan - "${COMPOSE[@]}" --profile lan up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + retry_build "compose build (lan)" "${COMPOSE[@]}" --profile lan build + "${COMPOSE[@]}" --profile lan up -d --no-build --force-recreate wait_for_peers fips-nat-lan-a${FIPS_CI_NAME_SUFFIX:-} 1 45 || { dump_lan_diagnostics return 1 diff --git a/testing/nat/scripts/nostr-relay-test.sh b/testing/nat/scripts/nostr-relay-test.sh index df6bdfff..e6b01795 100755 --- a/testing/nat/scripts/nostr-relay-test.sh +++ b/testing/nat/scripts/nostr-relay-test.sh @@ -22,6 +22,7 @@ BUILD_SCRIPT="$ROOT_DIR/testing/scripts/build.sh" GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh" WAIT_LIB="$ROOT_DIR/testing/lib/wait-converge.sh" RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh" +IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh" # Must track generate-configs.sh's OUTPUT_DIR and the compose bind-mounts. CONFIG_DIR="$NAT_DIR/generated-configs${FIPS_CI_NAME_SUFFIX:-}" @@ -55,6 +56,8 @@ RELAY_CONTAINER="fips-nat-relay${FIPS_CI_NAME_SUFFIX:-}" source "$WAIT_LIB" # shellcheck disable=SC1090 source "$RELAY_LIB" +# shellcheck disable=SC1090 +source "$IMAGE_LIB" cleanup() { "${COMPOSE[@]}" --profile "$PROFILE" down -v --remove-orphans \ @@ -430,7 +433,10 @@ run_test() { cleanup "$GENERATE_SCRIPT" "$SCENARIO" - "${COMPOSE[@]}" --profile "$PROFILE" up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + retry_build "compose build ($PROFILE)" "${COMPOSE[@]}" --profile "$PROFILE" build + "${COMPOSE[@]}" --profile "$PROFILE" up -d --no-build --force-recreate # Phase 1 + Phase 2 together: each side publishes its own advert, # subscribes for the other's, then dials. Bidirectional success diff --git a/testing/nat/scripts/stun-faults-test.sh b/testing/nat/scripts/stun-faults-test.sh index 149d5b33..57552681 100755 --- a/testing/nat/scripts/stun-faults-test.sh +++ b/testing/nat/scripts/stun-faults-test.sh @@ -30,6 +30,7 @@ ROOT_DIR="$(cd "$NAT_DIR/../.." && pwd)" BUILD_SCRIPT="$ROOT_DIR/testing/scripts/build.sh" GENERATE_SCRIPT="$SCRIPT_DIR/generate-configs.sh" RELAY_LIB="$ROOT_DIR/testing/lib/relay-verdict.sh" +IMAGE_LIB="$ROOT_DIR/testing/lib/image-build.sh" PROFILE="stun-faults" SCENARIO="$PROFILE" @@ -73,6 +74,8 @@ cleanup() { # shellcheck disable=SC1090 source "$RELAY_LIB" +# shellcheck disable=SC1090 +source "$IMAGE_LIB" trap 'echo ""; echo "stun-faults-test interrupted"; cleanup; exit 130' INT TERM @@ -269,7 +272,10 @@ run_test() { echo "=== stun-faults-test: setup ===" cleanup "$GENERATE_SCRIPT" "$SCENARIO" - "${COMPOSE[@]}" --profile "$PROFILE" up -d --build --force-recreate + # Build first, with retries, because the build pulls from registries that + # time out now and then; the start is not retried, since it is the test. + retry_build "compose build ($PROFILE)" "${COMPOSE[@]}" --profile "$PROFILE" build + "${COMPOSE[@]}" --profile "$PROFILE" up -d --no-build --force-recreate # Give the daemons time to come up. Both fault-node and fault-peer # need to start, publish their adverts to the relay, and discover From 7b005394dfa7a6e5222dd8674a5f74abf95755c8 Mon Sep 17 00:00:00 2001 From: Johnathan Corgan Date: Sat, 19 Sep 2026 13:56:06 +0000 Subject: [PATCH 4/4] Retry crate and toolchain downloads in the package build image The package build fetches every crate cold on each CI runner, and a crates.io download has failed a build mid-transfer. Set CARGO_NET_RETRY to 10 in the build image, which covers both the cargo-deb install at image build time and the package build that runs in the image. Retry the rustup bootstrap download as well. HTTP/2 multiplexing is left on: turning it off is held back unless registry framing errors recur with the higher retry count. Changing the Dockerfile changes the builder image tag, so every host and the CI cache rebuild the image once. --- packaging/debian/Dockerfile.build | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/packaging/debian/Dockerfile.build b/packaging/debian/Dockerfile.build index 6e69e644..4df8c81c 100644 --- a/packaging/debian/Dockerfile.build +++ b/packaging/debian/Dockerfile.build @@ -49,12 +49,19 @@ RUN git config --system --add safe.directory /src # the compiler the rest of CI uses, and bumping the pin rebuilds the image. The # builder this replaces installed `stable` and never copied rust-toolchain.toml, # so it compiled with a different compiler from the release and nothing said so. +# +# CARGO_NET_RETRY raises cargo's own retry count for crate downloads, both for +# cargo install below and for the package build that runs in this image; the +# registry is fetched cold on every CI runner and has failed mid-download. +# CARGO_HTTP_MULTIPLEXING is left at its default: turn it off only if HTTP/2 +# framing errors from the registry still occur with the higher retry count. ARG RUST_TOOLCHAIN ENV RUSTUP_HOME=/usr/local/rustup \ CARGO_HOME=/usr/local/cargo \ + CARGO_NET_RETRY=10 \ PATH=/usr/local/cargo/bin:$PATH RUN test -n "${RUST_TOOLCHAIN}" \ - && curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ + && curl --retry 3 --retry-connrefused --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs \ | sh -s -- -y --profile minimal --default-toolchain "${RUST_TOOLCHAIN}" \ && chmod -R a+w "$RUSTUP_HOME" "$CARGO_HOME"