diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ea094f14..310d8235 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1019,7 +1019,7 @@ jobs: # ───────────────────────────────────────────────────────────────────────────── # Job 5 – Real-deb install across target distros # -# Boots a privileged systemd container per distro, runs `apt install +# Boots a systemd container per distro (not privileged), runs `apt install # ./fips_*.deb` with the package job 4 built, then asserts end-to-end `.fips` # resolution + the gateway/daemon default-pairing. The most thorough single # test surface — exercises packaging, maintainer scripts, systemd unit diff --git a/testing/README.md b/testing/README.md index 36e8b69e..89388b5b 100644 --- a/testing/README.md +++ b/testing/README.md @@ -127,7 +127,7 @@ through the configured backend. ### [deb-install/](deb-install/) -- Debian Package Install -Installs the built `.deb` in privileged systemd containers for each +Installs the built `.deb` in systemd containers for each target distro and verifies unit enablement, conffile placement and end-to-end `.fips` resolution as a user would meet it. diff --git a/testing/deb-install/test.sh b/testing/deb-install/test.sh index 3eaab9ca..ad299d6a 100755 --- a/testing/deb-install/test.sh +++ b/testing/deb-install/test.sh @@ -1,9 +1,9 @@ #!/bin/bash # Test the fips Debian package install path across target distros. # -# Each scenario builds (or reuses) the .deb in a Debian 12 cargo-deb -# builder image (cached), boots a privileged systemd container with -# TUN access for the target distro, installs the .deb via `apt +# Each scenario builds (or reuses) the .deb in a cached Debian 12 +# cargo-deb image, boots a systemd container with TUN access for the +# target distro, installs the .deb via `apt # install ./fips_*.deb`, waits for fips.service + fips-dns.service # to come up, and verifies that `dig @127.0.0.53 AAAA .fips` # returns a non-empty AAAA answer through the resolver backend that @@ -22,12 +22,15 @@ # No args = run all scenarios. # Named args = run only those (e.g., ./test.sh ubuntu26 debian12) # -# Requirements: Docker with privileged container support, /dev/net/tun -# on the host (standard). +# Requirements: Docker able to grant SYS_ADMIN and NET_ADMIN and an +# unconfined AppArmor profile (the containers are not privileged; see +# testing/lib/systemd-container.sh), /dev/net/tun on the host (standard). set -uo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +# shellcheck source=SCRIPTDIR/../lib/systemd-container.sh +source "$SCRIPT_DIR/../lib/systemd-container.sh" REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" CACHE_DIR="$SCRIPT_DIR/.cache" DEB_CACHE_DIR="$CACHE_DIR/deb" @@ -76,17 +79,28 @@ build_image() { echo "$@" | docker build -t "$tag" -f - "$REPO_ROOT" >/dev/null 2>&1 } +# Start the scenario's systemd container. Not privileged: see +# testing/lib/systemd-container.sh for the flags and why. +# +# IPv6 forwarding is set here rather than inside the container because +# /proc/sys is read-only there. fips-gateway checks it before its DNS upstream +# check, so the gateway block needs it; the cost is that forwarding is on for +# every check in the scenario, including the install and resolver checks that +# run before the gateway block. start_systemd_container_with_tun() { local name="$1" image="$2" cleanup_container "$name" docker run -d --name "$name" \ --label com.corganlabs.fips-ci=1 \ - --privileged \ + "${SYSTEMD_CAPS[@]}" \ --cgroupns=host \ --device /dev/net/tun \ + --sysctl net.ipv6.conf.all.forwarding=1 \ -v /sys/fs/cgroup:/sys/fs/cgroup:rw \ --tmpfs /run --tmpfs /run/lock \ - "$image" >/dev/null 2>&1 + "$image" >/dev/null 2>&1 || return + check_isolation "$name" + return } wait_for_systemd() { @@ -499,8 +513,8 @@ DOCKERFILE # the gateway/daemon default-pairing on a real .deb install (no # custom config). Requires enabling the unit (it's not in the # default preset) and ipv6 forwarding (gateway checks before - # the DNS upstream check). - docker exec "$name" sysctl -w net.ipv6.conf.all.forwarding=1 >/dev/null 2>&1 || true + # the DNS upstream check), which the container is started with; + # see start_systemd_container_with_tun. timeout "$CONFIG_RESTART_TIMEOUT" docker exec "$name" bash -c ' systemctl unmask fips-gateway.service 2>/dev/null # Patch in a minimal gateway config since the shipped fips.yaml diff --git a/testing/dns-resolver/test.sh b/testing/dns-resolver/test.sh index bf4ed80b..978598d5 100755 --- a/testing/dns-resolver/test.sh +++ b/testing/dns-resolver/test.sh @@ -18,12 +18,16 @@ # No args = run all scenarios. # Named args = run only those (e.g., ./test.sh debian12-resolved e2e-debian12) # -# Requirements: Docker with privileged container support. The e2e -# scenario also needs /dev/net/tun on the host (standard). +# Requirements: Docker able to grant SYS_ADMIN and NET_ADMIN and an +# unconfined AppArmor profile (the containers are not privileged; see +# testing/lib/systemd-container.sh). The e2e scenario also needs +# /dev/net/tun on the host (standard). set -uo pipefail SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +# shellcheck source=SCRIPTDIR/../lib/systemd-container.sh +source "$SCRIPT_DIR/../lib/systemd-container.sh" REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" SETUP_SCRIPT="$REPO_ROOT/packaging/common/fips-dns-setup" TEARDOWN_SCRIPT="$REPO_ROOT/packaging/common/fips-dns-teardown" @@ -93,33 +97,45 @@ build_image() { docker build -t "$tag" -f - "$REPO_ROOT" } -# Start a systemd container in the background. +# Start a systemd container in the background. Not privileged: see +# testing/lib/systemd-container.sh for the flags and why. start_systemd_container() { local name="$1" image="$2" cleanup_container "$name" run_quiet "docker run $name" \ docker run -d --name "$name" \ --label com.corganlabs.fips-ci=1 \ - --privileged \ + "${SYSTEMD_CAPS[@]}" \ --cgroupns=host \ -v /sys/fs/cgroup:/sys/fs/cgroup:rw \ --tmpfs /run --tmpfs /run/lock \ - "$image" + "$image" || return + check_isolation "$name" + return } # Same, but with TUN device for the e2e scenario. +# +# IPv6 forwarding is set here rather than inside the container because +# /proc/sys is read-only there. fips-gateway checks it before its DNS upstream +# check, so the gateway parity check needs it; the cost is that forwarding is +# on for every check in the scenario, including the daemon, setup and dig +# checks that run before the gateway. start_systemd_container_with_tun() { local name="$1" image="$2" cleanup_container "$name" run_quiet "docker run $name (with tun)" \ docker run -d --name "$name" \ --label com.corganlabs.fips-ci=1 \ - --privileged \ + "${SYSTEMD_CAPS[@]}" \ --cgroupns=host \ --device /dev/net/tun \ + --sysctl net.ipv6.conf.all.forwarding=1 \ -v /sys/fs/cgroup:/sys/fs/cgroup:rw \ --tmpfs /run --tmpfs /run/lock \ - "$image" + "$image" || return + check_isolation "$name" + return } # Report why systemd never reached a running state. Every probe is @@ -900,9 +916,9 @@ gateway: lan_interface: "eth0" EOF' # fips-gateway checks IPv6 forwarding before the DNS upstream - # reachability check; enable forwarding so we get to the check we + # reachability check; the container is started with forwarding on + # (see start_systemd_container_with_tun) so we get to the check we # actually want to test. - docker exec "$name" sysctl -w net.ipv6.conf.all.forwarding=1 >/dev/null 2>&1 || true docker exec -d "$name" bash -c '/usr/bin/fips-gateway --config /tmp/gateway-test.yaml >/var/log/fips-gateway.log 2>&1 || true' # Wait briefly for the upstream-reachability log line to appear diff --git a/testing/lib/systemd-container.sh b/testing/lib/systemd-container.sh new file mode 100644 index 00000000..05623aa7 --- /dev/null +++ b/testing/lib/systemd-container.sh @@ -0,0 +1,91 @@ +#!/bin/bash +# Shared flags and isolation check for test containers that boot systemd. +# +# Source this file from a harness one level under testing/: +# source "$SCRIPT_DIR/../lib/systemd-container.sh" +# docker run -d ... "${SYSTEMD_CAPS[@]}" ... "$image" +# check_isolation "$name" +# +# check_isolation reports through the harness's own pass() and fail(), so +# those must be defined before it is called. +# +# Why these containers are not started with --privileged: a privileged +# container sees the host's real VT and serial devices (/dev/tty0, /dev/tty1, +# /dev/ttyS0, ...) and a writable /proc/sys and /sys. The systemd inside the +# image treats them as its own, so it starts a getty on the host's console, +# where each side's hangup kills the other's getty until the host's unit hits +# its start limit and the machine is left with no console login; logind holds +# a host VT open; and systemd-sysctl applies the image's sysctl.d to the host's +# kernel parameters. Some images also set up the host's virtual consoles and +# run a udev coldplug against the host's /sys. +# +# Without --privileged, docker gives the container only its default /dev and +# mounts /proc/sys and /sys read-only, so those units skip on their own +# conditions. That removes the console and kernel-parameter reach; it does not +# seal the container off from the host. systemd still mounts the host's fusectl +# and a hugetlbfs, and SYS_ADMIN would let a process inside remount /proc/sys +# read-write, which nothing in these images does. +# +# What each flag is for: +# --cap-add SYS_ADMIN mount namespaces for unit sandboxing (PrivateTmp, +# ProtectHome, ProtectKernelModules in the fips units, +# and the sandboxing resolved and logind carry) +# --cap-add NET_ADMIN TUN and dummy fips0 links, and nftables and neighbour +# proxy entries in the container's own network namespace +# --security-opt apparmor=unconfined +# the default AppArmor profile denies the mounts systemd +# makes for that sandboxing; without it boot ends +# degraded with logind and resolved failed. Needed on +# AppArmor hosts, which include Ubuntu hosts and +# GitHub's ubuntu-latest runners. +# shellcheck disable=SC2034 # read by the sourcing harness, not within this file +SYSTEMD_CAPS=( + --cap-add SYS_ADMIN + --cap-add NET_ADMIN + --security-opt apparmor=unconfined +) + +# Fail the suite if a running container can reach the host console or write +# the host's kernel parameters. +# +# Device nodes and mount modes are fixed when the container is created, so one +# look straight after `docker run` is enough; a remount made later from inside +# is out of its reach. The probe prints a marker first and exits 0, so the +# exec's status says only whether the probe ran: a container that could not be +# inspected is a failure, never a pass. On any failure the container is +# removed at once, so a privileged container does not live on to reach the +# host's console. +# +# Returns 0 when the container is isolated, 1 otherwise. +check_isolation() { + local name="$1" out findings + if ! out=$(docker exec "$name" sh -c ' + echo ISOLATION-PROBE + for d in /dev/tty[0-9]* /dev/ttyS[0-9]* /dev/console; do + [ -e "$d" ] && echo "device $d" + done + [ -w /proc/sys/kernel/core_pattern ] && echo "writable /proc/sys" + [ -w /sys/kernel ] && echo "writable /sys" + exit 0' 2>&1); then + fail "isolation: could not inspect $name: ${out:-no output}" + docker rm -f "$name" >/dev/null 2>&1 || true + return 1 + fi + case $'\n'"$out" in + *$'\n'ISOLATION-PROBE*) ;; + *) + fail "isolation: probe of $name printed no marker: ${out:-no output}" + docker rm -f "$name" >/dev/null 2>&1 || true + return 1 + ;; + esac + findings=${out#*ISOLATION-PROBE} + findings=${findings#$'\n'} + if [ -n "$findings" ]; then + fail "isolation: $name reaches the host: ${findings//$'\n'/, }" + docker rm -f "$name" >/dev/null 2>&1 || true + return 1 + fi + pass "isolation: $name has no host console devices and read-only /proc/sys and /sys" + return 0 +}