Files
fips/testing/nat/docker-compose.yml
Johnathan Corgan 965de26239 testing: make ci-local.sh preemption-safe with per-run docker isolation
A CI worker may preempt an in-flight ci-local.sh run (SIGTERM, then SIGKILL
after a grace period) to restart on a newer commit. For that kill to be safe,
the script must clean up after itself and never let a dying run collide with
its restart. It previously had no signal handling, shared the default compose
project name across runs, and tore down each suite only at the suite end.

- Derive a per-run id (honoring FIPS_CI_RUN_ID, else short-sha+random) and
  namespace every docker resource to it: a fipsci_<run>_<suite> compose project
  per suite and per parallel chaos child, and per-run image tags
  (fips-test:<run>, fips-test-app:<run>) retagged to :latest only after both
  builds succeed so :latest never points at a half-built image.
- Install a bounded, idempotent teardown trap on SIGTERM/SIGINT (+ EXIT): reap
  parallel chaos children, then force-remove this run's docker resources via
  the new ci-cleanup.sh, wrapped in timeout so a stuck down cannot wedge it.
- Exit 143 (SIGTERM) / 130 (SIGINT), distinct from 0 (pass) / 1 (failed), so a
  preempting worker tells a cancelled run from a real failure.
- Add ci-cleanup.sh (also ci-local.sh --reap): force-removes leftover CI
  resources by the com.corganlabs.fips-ci=1 label and the fipsci_ project
  prefix, robust to however a prior run died.
- Label every per-suite docker resource so the label sweep reaps it after a
  SIGKILL regardless of network name: direct docker run/network resources, the
  sidecar compose services, and every per-suite compose network (acl-allowlist,
  boringtun, firewall, nat, static, both tor suites, and the chaos generator
  template). Parametrize the static/sidecar compose image refs so the per-run
  tags are honored.
- Give each parallel chaos child a unique /24 from 10.30.x (a new --subnet
  override on the sim CLI, assigned per-child in ci-local.sh) so parallel
  children never collide on a shared docker subnet, and a chaos net can never
  span a fixed-subnet suite (sidecar/static in 172.20.x). 10.30.x sits outside
  docker's default-address-pool range, so an auto-assigned net cannot land on
  it either; node IPs derive from the subnet, so no scenario config changes.
2026-06-29 14:23:46 +00:00

342 lines
9.9 KiB
YAML

networks:
wan:
driver: bridge
labels:
- "com.corganlabs.fips-ci=1"
ipam:
config:
- subnet: 172.31.254.0/24
shared-lan:
driver: bridge
labels:
- "com.corganlabs.fips-ci=1"
ipam:
config:
- subnet: 172.31.10.0/24
volumes:
relay-data:
x-fips-common: &fips-common
image: fips-test:latest
cap_add:
- NET_ADMIN
devices:
- /dev/net/tun:/dev/net/tun
sysctls:
- net.ipv6.conf.all.disable_ipv6=0
restart: "no"
environment:
- RUST_LOG=info,fips::discovery::nostr=debug,fips::node::lifecycle=debug
services:
relay:
build:
context: ../..
dockerfile: examples/sidecar-nostr-relay/Dockerfile.app
container_name: fips-nat-relay
restart: "no"
volumes:
- relay-data:/usr/src/app/strfry-db
- ./relay/strfry.conf:/usr/src/app/strfry.conf:ro
- ../docker/resolv.conf:/etc/resolv.conf:ro
networks:
wan:
ipv4_address: 172.31.254.30
shared-lan:
ipv4_address: 172.31.10.30
stun:
build:
context: ./stun
container_name: fips-nat-stun
restart: "no"
networks:
wan:
ipv4_address: 172.31.254.40
shared-lan:
ipv4_address: 172.31.10.40
nat-a:
build:
context: ./router
profiles: ["cone", "symmetric"]
container_name: fips-nat-router-a
cap_add:
- NET_ADMIN
sysctls:
- net.ipv4.ip_forward=1
restart: "no"
environment:
- NAT_MODE=${NAT_MODE_A:-cone}
- TCP_FORWARD_PORTS=8443
- LAN_IF=eth1
- WAN_IF=eth0
- LAN_HOST=172.31.1.10
- LAN_SUBNET=172.31.1.0/24
- WAN_SUBNET=172.31.254.0/24
- WAN_GATEWAY=172.31.254.1
networks:
wan:
ipv4_address: 172.31.254.10
nat-b:
build:
context: ./router
profiles: ["cone", "symmetric"]
container_name: fips-nat-router-b
cap_add:
- NET_ADMIN
sysctls:
- net.ipv4.ip_forward=1
restart: "no"
environment:
- NAT_MODE=${NAT_MODE_B:-cone}
- TCP_FORWARD_PORTS=8443
- LAN_IF=eth1
- WAN_IF=eth0
- LAN_HOST=172.31.2.10
- LAN_SUBNET=172.31.2.0/24
- WAN_SUBNET=172.31.254.0/24
- WAN_GATEWAY=172.31.254.1
networks:
wan:
ipv4_address: 172.31.254.11
cone-a:
<<: *fips-common
profiles: ["cone"]
container_name: fips-nat-cone-a
hostname: fips-nat-cone-a
depends_on:
- nat-a
- relay
- stun
entrypoint:
- /usr/local/bin/nat-node-entrypoint.sh
environment:
- RUST_LOG=info,fips::discovery::nostr=debug,fips::node::lifecycle=debug
- DATA_IF=eth0
- ROUTE_SUBNET=172.31.254.0/24
- ROUTE_VIA=172.31.1.254
- RELAY_HOST=172.31.254.30
- RELAY_PORT=7777
- STUN_HOST=172.31.254.40
- STUN_PORT=3478
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./node/entrypoint.sh:/usr/local/bin/nat-node-entrypoint.sh:ro
- ./generated-configs/cone/node-a.yaml:/etc/fips/fips.yaml:ro
network_mode: none
cone-b:
<<: *fips-common
profiles: ["cone"]
container_name: fips-nat-cone-b
hostname: fips-nat-cone-b
depends_on:
- nat-b
- relay
- stun
entrypoint:
- /usr/local/bin/nat-node-entrypoint.sh
environment:
- RUST_LOG=info,fips::discovery::nostr=debug,fips::node::lifecycle=debug
- DATA_IF=eth0
- ROUTE_SUBNET=172.31.254.0/24
- ROUTE_VIA=172.31.2.254
- RELAY_HOST=172.31.254.30
- RELAY_PORT=7777
- STUN_HOST=172.31.254.40
- STUN_PORT=3478
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./node/entrypoint.sh:/usr/local/bin/nat-node-entrypoint.sh:ro
- ./generated-configs/cone/node-b.yaml:/etc/fips/fips.yaml:ro
network_mode: none
symmetric-a:
<<: *fips-common
profiles: ["symmetric"]
container_name: fips-nat-symmetric-a
hostname: fips-nat-symmetric-a
depends_on:
- nat-a
- relay
- stun
entrypoint:
- /usr/local/bin/nat-node-entrypoint.sh
environment:
- RUST_LOG=info,fips::discovery::nostr=debug,fips::node::lifecycle=debug
- DATA_IF=eth0
- ROUTE_SUBNET=172.31.254.0/24
- ROUTE_VIA=172.31.1.254
- RELAY_HOST=172.31.254.30
- RELAY_PORT=7777
- STUN_HOST=172.31.254.40
- STUN_PORT=3478
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./node/entrypoint.sh:/usr/local/bin/nat-node-entrypoint.sh:ro
- ./generated-configs/symmetric/node-a.yaml:/etc/fips/fips.yaml:ro
network_mode: none
symmetric-b:
<<: *fips-common
profiles: ["symmetric"]
container_name: fips-nat-symmetric-b
hostname: fips-nat-symmetric-b
depends_on:
- nat-b
- relay
- stun
entrypoint:
- /usr/local/bin/nat-node-entrypoint.sh
environment:
- RUST_LOG=info,fips::discovery::nostr=debug,fips::node::lifecycle=debug
- DATA_IF=eth0
- ROUTE_SUBNET=172.31.254.0/24
- ROUTE_VIA=172.31.2.254
- RELAY_HOST=172.31.254.30
- RELAY_PORT=7777
- STUN_HOST=172.31.254.40
- STUN_PORT=3478
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./node/entrypoint.sh:/usr/local/bin/nat-node-entrypoint.sh:ro
- ./generated-configs/symmetric/node-b.yaml:/etc/fips/fips.yaml:ro
network_mode: none
lan-a:
<<: *fips-common
profiles: ["lan"]
container_name: fips-nat-lan-a
hostname: fips-nat-lan-a
depends_on:
- relay
- stun
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./generated-configs/lan/node-a.yaml:/etc/fips/fips.yaml:ro
networks:
shared-lan:
ipv4_address: 172.31.10.10
lan-b:
<<: *fips-common
profiles: ["lan"]
container_name: fips-nat-lan-b
hostname: fips-nat-lan-b
depends_on:
- relay
- stun
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./generated-configs/lan/node-b.yaml:/etc/fips/fips.yaml:ro
networks:
shared-lan:
ipv4_address: 172.31.10.11
# ── Nostr publish/consume profile ──────────────────────────────────────
# Two FIPS daemons + the existing strfry relay, exercising the overlay
# advert publish → relay → consumer round-trip end-to-end. Both nodes
# share the same LAN bridge as the relay (no NAT in the way) so the
# focus of the test is the Nostr discovery layer rather than NAT
# traversal mechanics. Phase 3 (malformed advert) is driven by a
# one-shot publish from the test runner via the relay's WebSocket.
nostr-pub-a:
<<: *fips-common
profiles: ["nostr-publish-consume"]
container_name: fips-nat-nostr-pub-a
hostname: fips-nat-nostr-pub-a
depends_on:
- relay
- stun
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./generated-configs/nostr-publish-consume/node-a.yaml:/etc/fips/fips.yaml:ro
networks:
shared-lan:
ipv4_address: 172.31.10.20
nostr-pub-b:
<<: *fips-common
profiles: ["nostr-publish-consume"]
container_name: fips-nat-nostr-pub-b
hostname: fips-nat-nostr-pub-b
depends_on:
- relay
- stun
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./generated-configs/nostr-publish-consume/node-b.yaml:/etc/fips/fips.yaml:ro
networks:
shared-lan:
ipv4_address: 172.31.10.21
# ── STUN fault-injection profile ───────────────────────────────────────
# One FIPS daemon + a netns-sharing shim that injects tc/iptables faults
# against UDP egress to the STUN service. The runner script drives the
# shim via `docker exec` (Approach A) — no scripted timing inside the
# shim itself. Three phases:
# 1. drop — 100% UDP egress drop to STUN; assert daemon notices the
# observation timeout and retries.
# 2. delay — ~5s netem delay; assert daemon recovers and STUN succeeds
# again once the rule is removed.
# 3. kill — `docker stop fips-nat-stun`; assert daemon stays up and
# continues to handle "STUN unreachable" gracefully.
# The shim shares the daemon's network namespace so `tc qdisc add dev
# eth0 ...` operates on the daemon's egress path. The shim therefore
# has its own NET_ADMIN cap; the daemon already has one for TUN.
stun-fault-node:
<<: *fips-common
profiles: ["stun-faults"]
container_name: fips-nat-stun-fault-node
hostname: fips-nat-stun-fault-node
depends_on:
- relay
- stun
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./generated-configs/stun-faults/stun-fault-node.yaml:/etc/fips/fips.yaml:ro
networks:
shared-lan:
ipv4_address: 172.31.10.50
# Fault-free peer that publishes a valid overlay advert, so the
# fault-node's NAT-traversal attempt actually reaches
# observe_traversal_addresses() (the STUN client). Without this peer the
# daemon would abort with "no overlay advert" and never generate the
# STUN egress that the shim's tc/iptables rules are meant to drop.
# Intentionally has NO fault shim sharing its netns; runs cleanly.
stun-fault-peer:
<<: *fips-common
profiles: ["stun-faults"]
container_name: fips-nat-stun-fault-peer
hostname: fips-nat-stun-fault-peer
depends_on:
- relay
- stun
volumes:
- ../docker/resolv.conf:/etc/resolv.conf:ro
- ./generated-configs/stun-faults/stun-fault-peer.yaml:/etc/fips/fips.yaml:ro
networks:
shared-lan:
ipv4_address: 172.31.10.51
stun-fault-shim:
image: fips-test:latest
profiles: ["stun-faults"]
container_name: fips-nat-stun-fault-shim
depends_on:
- stun-fault-node
cap_add:
- NET_ADMIN
- NET_RAW
network_mode: "service:stun-fault-node"
restart: "no"
entrypoint:
- /bin/sh
- -c
- "exec sleep infinity"