diff --git a/.github/workflows/aur-publish.yml b/.github/workflows/aur-publish.yml index 022b5d62..af5c49a7 100644 --- a/.github/workflows/aur-publish.yml +++ b/.github/workflows/aur-publish.yml @@ -138,9 +138,11 @@ jobs: bash namcap-gate.sh PKGBUILD echo "::endgroup::" echo "::group::makepkg build" - # --nocheck: skip the PKGBUILD check() (cargo test --lib); the test - # suite is already covered by ci.yml. This job validates packaging. - makepkg -s --noconfirm --nocheck + # No --nocheck. makepkg runs check() by default, so every AUR user + # runs it and this job must too. ci.yml runs the tests, but not this + # way: frozen and offline against what prepare fetched, with the + # Arch toolchain, in this container. + makepkg -s --noconfirm echo "::endgroup::" echo "::group::namcap built package" bash namcap-gate.sh ./*.pkg.tar.* diff --git a/CHANGELOG.md b/CHANGELOG.md index 59ce56df..4df308ad 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -359,6 +359,32 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 is still the oldest Debian-family distribution, which is no longer the oldest distribution outright. +#### Identity and config + +- An ephemeral node no longer writes `fips.key`. It wrote the private key of + an identity it discards at every restart to that file, overwriting any key + already there, including an operator's key when `persistent: true` had been + forgotten. It now writes only `fips.pub`, so the running npub stays visible, + and holds the private key in memory only. A `fips.key` found at an ephemeral + start is moved to `fips.key.unused` with a warning, so an ephemeral node logs + that warning once, on its first start after the upgrade; if that name is + taken or the rename fails, the file is left in place and a warning says so. + For a stable identity, set `node.identity.persistent: true` and restart: the + first persistent start generates and saves a key, so the npub changes once, + at that restart, and is stable from then on. Starting once in ephemeral mode + and then pinning the key it wrote no longer works. + +#### Gateway + +- The gateway's default DNS listen address is now `[::1]:5365`; it was + `[::1]:5353`, the mDNS port, which the daemon's LAN rendezvous, Avahi and + systemd-resolved can hold. On OpenWrt the init script now points dnsmasq at + whatever port `gateway.dns.listen` sets, and an upgrade rewrites the + previously shipped `listen: "[::1]:5353"` line. On other hosts, a resolver + you configured by hand to forward `.fips` to `[::1]:5353` must now forward + to `[::1]:5365`, or set `gateway.dns.listen: "[::1]:5353"` to keep the old + port. The gateway warns at startup when it is configured on 5353. + #### Packaging (Debian) - An upgrade of the `.deb` now reapplies the firewall ruleset in place. Until @@ -394,6 +420,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 maintainer rather than flagged by an advisory. What it buys is that a fresh checkout can resolve the lockfile without reaching for a yanked version. +#### Documentation (native API) + +- The native API documentation now says that on Linux an empty datagram sent + immediately before a close may read as end of file, and is then not + delivered. Linux carries the flow on `SOCK_SEQPACKET`, where a zero-length + datagram that is the last message before a close cannot be told apart from + the close; macOS and FreeBSD carry it on `SOCK_DGRAM` and are not affected. + The `Received::Datagram` rustdoc, which said an empty datagram is never a + close, now says where the exception applies. + ### Removed - **Source-breaking for consumers of the library crate**: `ActivePeer` no @@ -615,6 +651,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 resends when they show a loss, and resends once after 30 seconds when the reports cannot confirm it. Resends over one peer connection are limited to six a minute, and to one a minute while losses persist. +- A spanning-tree announce lost on the link is now resent. A node counted a + tree announce as delivered once the transport accepted it, so after a + dropped datagram or a link outage shorter than the dead timeout the peer kept + our old tree position until the periodic re-broadcast, up to a minute later, + and a node with only one peer had no periodic re-broadcast at all. Meanwhile + the peer could leave destinations out of discovery or route toward them by + stale coordinates. The node now confirms each tree announce from the link's + receiver reports, as it does for bloom filter announces, resends it when + they show a loss, and resends it once after 30 seconds when they cannot + confirm it, under the same limits. #### Session setup @@ -788,6 +834,13 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 warning says which limit refused it. A name that already has a mapping is answered before either limit is consulted, so names in use keep resolving when the pool is full. The limits are compiled in, not configured. +- `fips-gateway` exits when its DNS listener cannot bind, or stops while the + gateway runs, instead of staying up with `.fips` resolution dead, so systemd + or procd restarts it or reports it failed. This applies to a gateway used + only for port forwards too. An "address in use" error names the service + likely to hold the port. On an OpenWrt access point with the gateway + enabled, the gateway had lost its port to the daemon's own mDNS responder + and `.fips` names stopped resolving with nothing reported. #### Nostr and NAT traversal @@ -944,6 +997,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 directory is owned by another account. Rerun the installer after moving files into the directory. A foreground run from an unelevated prompt can no longer read the files there. +- The Windows service installer now creates empty `peers.allow` and + `peers.deny` files in `C:\ProgramData\fips`. While either was missing there, + the service read that file from `\etc\fips` on the system drive, where any + local user can create files, so a planted list was enforced. An empty file + allows every peer; to clear a list, empty its file rather than deleting it. + The installer stops when either file exists under `\etc\fips` and not in + `C:\ProgramData\fips`, so an upgrader's list is neither enforced from the + old location nor dropped unreviewed: review it, move it into + `C:\ProgramData\fips` or delete it, and run the installer again. Windows + upgraders should rerun `install-service.ps1`. ## [0.5.1] - 2026-09-06 diff --git a/docs/design/fips-gateway.md b/docs/design/fips-gateway.md index 51ce4714..47165abb 100644 --- a/docs/design/fips-gateway.md +++ b/docs/design/fips-gateway.md @@ -130,7 +130,7 @@ There is no `fipsctl gateway` subcommand; clients (including │ │ │ ┌──────────────┐ ┌───────────┐ │ │ │ DNS proxy │ │ Virtual │ │ - │ │ ([::1]:5353) │─▶│ IP pool │ │ + │ │ ([::1]:5365) │─▶│ IP pool │ │ │ │ .fips only │ │ (state │ │ │ └──────┬───────┘ │ machine) │ │ │ │ └─────┬─────┘ │ @@ -182,8 +182,10 @@ involving the DNS proxy or the pool. ### DNS Resolution Flow 1. A LAN client sends a DNS query to the gateway's listener (default - `[::1]:5353`, configurable via `gateway.dns.listen`). The default - is loopback-only on an unprivileged port: the canonical deployment + `[::1]:5365`, configurable via `gateway.dns.listen`). The default + is not 5353, the mDNS port, which the daemon's LAN rendezvous and + other mDNS responders hold. It is loopback-only on an unprivileged + port: the canonical deployment has another resolver on the host (dnsmasq, systemd-resolved, BIND) holding port 53 and forwarding `.fips` queries to the gateway over loopback. Operators on a host without a pre-existing resolver on diff --git a/docs/design/fips-spanning-tree.md b/docs/design/fips-spanning-tree.md index c7b5d117..5e3e9cf1 100644 --- a/docs/design/fips-spanning-tree.md +++ b/docs/design/fips-spanning-tree.md @@ -207,6 +207,19 @@ branch, and each only re-announces to its peers. - **Convergence time**: A tree of depth D reconverges in roughly D × 0.5s to D × 1.0s +The transport accepting a TreeAnnounce is not delivery. Each announce stays +outstanding until the link's receiver reports confirm it, by the same rule +as a FilterAnnounce (see "Update Triggers" in +[fips-bloom-filters.md](fips-bloom-filters.md)). It is resent when the +reports show a loss, and once after 30 s when they cannot confirm it, with +the same budgets and backoff: per declaration sequence per session, one +unchecked resend and three on loss, spaced by a per-peer backoff of 1, 2, +4 ... up to 60 s. A resend carries the current declaration, so a peer that +missed an older one gets the newer position, and one that already holds it +ignores it as not fresher. The periodic re-evaluation under Stability +Mechanisms, which re-broadcasts an unchanged declaration, stays as the +backstop on a node with two or more peers. + ### Transitive Trust (v1) In the v1 protocol, only the sender's outer signature on the TreeAnnounce diff --git a/docs/how-to/deploy-gateway.md b/docs/how-to/deploy-gateway.md index cbd80d7d..e1bb3595 100644 --- a/docs/how-to/deploy-gateway.md +++ b/docs/how-to/deploy-gateway.md @@ -143,7 +143,7 @@ virtual IPs, which is the gateway's hard cap regardless of CIDR width. This minimum config is enough to start the gateway. The `dns.*` block -is optional and defaults to `listen: "[::1]:5353"` and +is optional and defaults to `listen: "[::1]:5365"` and `upstream: "[::1]:5354"`. The full block — including `dns.*`, `pool_grace_period`, `conntrack.*`, and `port_forwards[]` — is documented in @@ -196,7 +196,7 @@ Constraints: ```yaml gateway: dns: - listen: "[::1]:5353" + listen: "[::1]:5365" upstream: "[::1]:5354" ttl: 60 ``` @@ -204,15 +204,16 @@ gateway: Common cases: - **Another resolver on the host (the canonical case):** the default - `listen: "[::1]:5353"` is loopback-only on an unprivileged port, + `listen: "[::1]:5365"` is loopback-only on an unprivileged port, so it never conflicts with dnsmasq, systemd-resolved, or BIND holding 53. Configure the existing resolver to forward `.fips` - queries to `[::1]:5353` and you are done — this is what the - OpenWrt ipk does automatically. + queries to `[::1]:5365` and you are done — this is what the + OpenWrt ipk does automatically. On OpenWrt the init script reads + `gateway.dns.listen` and points dnsmasq at whatever port it sets. - **No other resolver on the host:** set `listen: "[::]:53"` explicitly and LAN clients can query the gateway directly. - **systemd-resolved is on port 53:** the default already side-steps - this — leave the listen address at `[::1]:5353` and configure the + this — leave the listen address at `[::1]:5365` and configure the stub or a small forwarder to delegate `.fips` to the gateway. If you would rather have the gateway on 53 directly, disable the systemd stub listener (`DNSStubListener=no` in diff --git a/docs/how-to/persistent-identity.md b/docs/how-to/persistent-identity.md index 975ee503..015c6772 100644 --- a/docs/how-to/persistent-identity.md +++ b/docs/how-to/persistent-identity.md @@ -35,19 +35,12 @@ nodes, and tests where you actively want a fresh identity per run. The Debian/Ubuntu `.deb` and the Arch `fips` AUR package both ship a default `/etc/fips/fips.yaml` with `node.identity.persistent` left as -the upstream default (false), so the daemon writes a fresh keypair to -`/etc/fips/fips.{key,pub}` on every start until you set -`persistent: true`. To pin the current keypair: +the upstream default (false). In that mode the daemon generates a +fresh keypair on every start, holds the private key only in memory, +and writes only `/etc/fips/fips.pub`. To give the node a stable +identity: -1. Install the package and start the daemon once so it generates - `fips.key` / `fips.pub`: - - ```sh - sudo systemctl start fips - sudo systemctl status fips # confirm it came up - ``` - -2. Edit `/etc/fips/fips.yaml` and set: +1. Install the package, then edit `/etc/fips/fips.yaml` and set: ```yaml node: @@ -55,22 +48,31 @@ the upstream default (false), so the daemon writes a fresh keypair to persistent: true ``` -3. Restart the daemon and verify the identity is reused: +2. Start or restart the daemon. On this first persistent start it + generates a keypair and saves it to `/etc/fips/fips.key`: ```sh sudo systemctl restart fips + sudo systemctl status fips # confirm it came up + ``` + +3. Verify the identity: + + ```sh fipsctl show status | grep -E '"npub"|"node_addr"' cat /etc/fips/fips.pub ``` The npub printed by `fipsctl show status` should match - `/etc/fips/fips.pub` and remain stable across subsequent restarts. + `/etc/fips/fips.pub`. If the daemon was already running + ephemeral, the npub changes once at this restart and is stable + from then on. -The package's `postinst` script does **not** generate the keypair — -the daemon does, on first start. This means the keypair is only -present after the first successful daemon start. If the daemon never -came up cleanly (config error, permission problem), the key files -will be missing. +The package's `postinst` script does **not** generate the keypair. +The first successful daemon start with `persistent: true` does, so +`fips.key` is only present after that start. If the daemon never came +up cleanly (config error, permission problem), the key file will be +missing. ### macOS note diff --git a/docs/how-to/troubleshoot-gateway.md b/docs/how-to/troubleshoot-gateway.md index 6b532ded..719c8868 100644 --- a/docs/how-to/troubleshoot-gateway.md +++ b/docs/how-to/troubleshoot-gateway.md @@ -81,28 +81,62 @@ for the full flag list. ### Port conflict on the DNS listen port -Symptom: gateway fails to start with "address already in use" on -the configured `gateway.dns.listen` address. +Symptom: the gateway exits at startup, before it creates the address +pool or the NAT table, and the log carries an error such as this one, +wrapped here for reading: -The default `[::1]:5353` is loopback-only on an unprivileged port and -should not collide with any standard resolver. If you have overridden -`dns.listen` to bind port 53 (or a LAN-side address) and another DNS -server (systemd-resolved, dnsmasq, BIND) is already bound there, -identify it: - -```sh -sudo ss -tulnp | grep ':53' +```text +cannot bind the gateway DNS listener on [::]:53: Address already in +use (os error 98); another DNS server holds port 53: dnsmasq, +systemd-resolved's stub listener, unbound or BIND; find the holder +with `ss -ulpn 'sport = :53'` or `netstat -ulnp`, or set +gateway.dns.listen to a free port and point the resolver that +forwards .fips at it ``` +Under systemd the unit restarts every five seconds and fails the same +way each time; under procd on OpenWrt the service stops respawning +after five failures within an hour. The gateway also exits, after +removing its NAT table and routes, when the DNS resolver stops while +the gateway is running; that log line reads "Gateway DNS resolver +stopped; exiting so the service manager restarts the gateway". + +The error names the service most likely to hold the port: + +- **53**: another DNS server, such as dnsmasq, systemd-resolved's stub + listener, unbound or BIND. +- **5353**: mDNS. The fips daemon's LAN rendezvous + (`node.rendezvous.lan`), avahi-daemon or systemd-resolved's + MulticastDNS. +- **5354**: the fips daemon's own DNS responder. `gateway.dns.listen` + must not be the daemon's DNS port. +- **5355**: LLMNR, held by systemd-resolved unless `LLMNR=no`. +- **5365**, the default: another fips-gateway already running. +- **Any other port**: another process. + +Find the actual holder, replacing the port with your own: + +```sh +sudo ss -ulpn 'sport = :53' +# OpenWrt ships netstat but not ss: +netstat -ulnp +``` + +The default listen address, `[::1]:5365`, is loopback-only on an +unprivileged port. Releases before 0.5.2 defaulted to `[::1]:5353`, +the mDNS port; a config that still sets it explicitly keeps it, and +the gateway warns at startup. On OpenWrt, an upgrade rewrites the +previously shipped `listen: "[::1]:5353"` line to the new default. Two options: -- **Stay on the loopback default.** Drop the override and let the - gateway use `[::1]:5353`. Configure the existing resolver to - forward `.fips` queries to it (the canonical OpenWrt deployment - works this way out of the box). +- **Move the gateway.** Set `gateway.dns.listen` to a free port and + point the resolver that forwards `.fips` at the same port. With the + loopback default, configure the existing resolver to forward `.fips` + queries to `[::1]:5365` (the canonical OpenWrt deployment works this + way out of the box). - **Relocate the conflicting resolver.** Move it to a different port - (or disable it if not needed) and let the gateway bind 53. + (or disable it if not needed) and let the gateway bind the port. Practical for systemd-resolved (set `DNSStubListener=no` in `/etc/systemd/resolved.conf`); rarely worth it for production resolvers. @@ -211,7 +245,7 @@ not running or not enabled. Check that the daemon config has **Step 2.** Verify the gateway is listening on its DNS port: ```sh -sudo ss -tulnp | grep -E ':(53|5353)\b' +sudo ss -tulnp | grep -E ':(53|5365)\b' ``` If nothing is listening on the configured `dns.listen` address, the diff --git a/docs/how-to/use-the-native-datagram-api.md b/docs/how-to/use-the-native-datagram-api.md index adc87334..355ba093 100644 --- a/docs/how-to/use-the-native-datagram-api.md +++ b/docs/how-to/use-the-native-datagram-api.md @@ -286,10 +286,11 @@ port from one dropped because a client was not reading fast enough. For the response shape, see [../reference/control-socket.md](../reference/control-socket.md#read-only-queries). -Reach for this when datagrams go missing. **Four places lose data with +Reach for this when datagrams go missing. **Five places lose data with nothing reported to your program**: a full per-flow queue, a listener that does not accept fast enough, an outbound datagram sent before a session -exists, and an outbound datagram after the transport MTU has fallen. +exists, an outbound datagram after the transport MTU has fallen, and, on +Linux, an empty datagram sent just before a close. [../reference/native-api.md](../reference/native-api.md#where-data-disappears) describes each and what bounds it. diff --git a/docs/how-to/write-a-native-api-client.md b/docs/how-to/write-a-native-api-client.md index bde73efa..a128874a 100644 --- a/docs/how-to/write-a-native-api-client.md +++ b/docs/how-to/write-a-native-api-client.md @@ -85,13 +85,15 @@ that costs a real payload if you get it wrong. A client that sends an empty datagram, then a message, then closes, leaves both queued, and a reader that trusts `POLLHUP` alone discards the message. -**One case has no answer, and you should design around it rather than solve -it.** A zero-length datagram that is the last message before a close is -indistinguishable from the close: reading it drains the queue, and `FIONREAD` -then reports zero because a zero-length message contributes no bytes. If your -protocol gives a zero-length payload a meaning, do not send it as a zero-length -socket message. Carry a one-byte discriminator, and keep the zero-byte read for -end of file alone. +**One case has no answer on Linux, where the pair is `SOCK_SEQPACKET`, and you +should design around it rather than solve it.** A zero-length datagram that is +the last message before a close is indistinguishable from the close: reading it +drains the queue, and `FIONREAD` then reports zero because a zero-length message +contributes no bytes. If your protocol gives a zero-length payload a meaning, do +not send it as a zero-length socket message. Carry a one-byte discriminator, and +keep the zero-byte read for end of file alone. On macOS and FreeBSD the pair is +`SOCK_DGRAM`: the empty datagram reads as zero bytes, and the close is reported +by the read after it. Both directions of the mistake are real. Reading an empty datagram as a close lets a peer tear down a live flow by sending nothing, and presents as a diff --git a/docs/reference/cli-fips-gateway.md b/docs/reference/cli-fips-gateway.md index 5a764fce..572ecee8 100644 --- a/docs/reference/cli-fips-gateway.md +++ b/docs/reference/cli-fips-gateway.md @@ -73,7 +73,7 @@ Linux host) and | Code | Meaning | | ---- | ------- | | `0` | Clean shutdown after `SIGINT` / `SIGTERM`. | -| `1` | Non-Linux platform, configuration load failure, missing or invalid `gateway:` block, NAT/network setup failure, or control-socket bind failure. The reason is printed to stderr or the log before exit. | +| `1` | Non-Linux platform, configuration load failure, missing or invalid `gateway:` block, the DNS listener could not bind or stopped while running, or NAT/network setup failure. The reason is printed to stderr or the log before exit. A control-socket bind failure is logged as a warning and the gateway continues without the socket. | ## Environment diff --git a/docs/reference/cli-fips.md b/docs/reference/cli-fips.md index 653291a1..b147a0f7 100644 --- a/docs/reference/cli-fips.md +++ b/docs/reference/cli-fips.md @@ -76,16 +76,18 @@ first, then `/usr/local/etc/fips`, so the packaged file wins over a leftover `/etc/fips` copy from an earlier install. Windows likewise probes `\etc\fips` on the current drive, then `C:\ProgramData\fips`. -Adjacent to the highest-priority config file the daemon reads (or -writes, on first start) the identity files: +Adjacent to the highest-priority config file the daemon keeps the +identity files: | File | Mode | Purpose | | ---- | ---- | ------- | -| `fips.key` | `0600` | Bech32 nsec for the persistent identity (Unix; on Windows the file takes its directory's ACL, which `install-service.ps1` restricts to SYSTEM and Administrators). | -| `fips.pub` | `0644` | Bech32 npub corresponding to `fips.key`. | +| `fips.key` | `0600` | Bech32 nsec for the persistent identity, written only in persistent mode (Unix; on Windows the file takes its directory's ACL, which `install-service.ps1` restricts to SYSTEM and Administrators). | +| `fips.pub` | `0644` | Bech32 npub of the running identity, written on every start. In persistent mode it corresponds to `fips.key`. | When `node.identity.persistent` is `false` (the default), a fresh -keypair is written to these files on every start. +keypair is generated on every start and only `fips.pub` is written. +A `fips.key` found there is moved aside to `fips.key.unused` and a +warning is logged. On Windows the service writes its log to `C:\ProgramData\fips\fips.log`, rolled at 10 MiB with four old files kept; a foreground run logs to the diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index fbbf4378..17518aa3 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -108,8 +108,10 @@ Identity resolution follows a three-tier priority: 3. **Ephemeral** — when `persistent: false` (default) and no `nsec`, generates a fresh keypair on each start -Key files (`fips.key` with mode 0600, `fips.pub` with mode 0644) are written adjacent -to the highest-priority config file for operator visibility, even in ephemeral mode. +`fips.pub` (mode 0644) is written adjacent to the highest-priority config file +on every start. `fips.key` (mode 0600) is written only in persistent mode. In +ephemeral mode a `fips.key` found at startup is moved aside to +`fips.key.unused` with a warning. ### General @@ -1073,7 +1075,7 @@ Non-`.fips` queries are answered with `REFUSED`. | Parameter | Type | Default | Description | |-----------|------|---------|-------------| -| `gateway.dns.listen` | string | `"[::1]:5353"` | DNS listen address. The default binds IPv6 loopback on an unprivileged port, matching the canonical deployment where another resolver on the host (dnsmasq, systemd-resolved, BIND) holds port 53 and forwards `.fips` queries to the gateway over loopback. Bind on the LAN-side IP (e.g., `"192.168.1.1:53"`) or wildcard (`"[::]:53"`) only on hosts with no other resolver on 53 and where LAN clients query the gateway directly. See [../how-to/troubleshoot-gateway.md](../how-to/troubleshoot-gateway.md). | +| `gateway.dns.listen` | string | `"[::1]:5365"` | DNS listen address. The default binds IPv6 loopback on an unprivileged port, not 5353, the mDNS port, matching the canonical deployment where another resolver on the host (dnsmasq, systemd-resolved, BIND) holds port 53 and forwards `.fips` queries to the gateway over loopback. Bind on the LAN-side IP (e.g., `"192.168.1.1:53"`) or wildcard (`"[::]:53"`) only on hosts with no other resolver on 53 and where LAN clients query the gateway directly. See [../how-to/troubleshoot-gateway.md](../how-to/troubleshoot-gateway.md). | | `gateway.dns.upstream` | string | `"[::1]:5354"` | Upstream FIPS daemon resolver. **Must match the daemon's `dns.bind_addr` and `dns.port`.** Defaults match the daemon defaults (`::1:5354`). A v4 upstream (`"127.0.0.1:5354"`) cannot reach a daemon bound on `[::1]:5354` — Linux IPv6 sockets bound to explicit `::1` do not accept v4-mapped traffic. If you change the daemon's `dns.bind_addr`, update this field accordingly. | | `gateway.dns.ttl` | u32 | `60` | TTL in seconds on AAAA responses returned to LAN clients. Smaller values let the gateway recycle pool addresses faster; larger values reduce LAN-side query traffic. | @@ -1120,7 +1122,7 @@ gateway: pool: "fd01::/112" lan_interface: "enp3s0" dns: - listen: "[::1]:5353" + listen: "[::1]:5365" upstream: "[::1]:5354" ttl: 60 pool_grace_period: 60 diff --git a/docs/reference/native-api.md b/docs/reference/native-api.md index 61fe1748..cea846d8 100644 --- a/docs/reference/native-api.md +++ b/docs/reference/native-api.md @@ -260,13 +260,15 @@ because it latches while messages are still queued: a read that trusted it would report the close early and discard whatever was waiting. The `EPIPE` that `recv` returns means the daemon went away, never that a peer finished. -**One case is reported as `EPIPE` although it is a datagram.** A zero-length -datagram that is the last message before a close is indistinguishable from the -close, because reading it drains the queue and `FIONREAD` then reports zero: a -zero-length message contributes no bytes. Do not give a zero-length payload a -meaning of its own on this API. Carry a one-byte discriminator instead. -Separating the two needs a payload that is never zero bytes on the wire, which -is a protocol change rather than a receive-path one. +**On Linux, one case is reported as `EPIPE` although it is a datagram.** A +zero-length datagram that is the last message before a close is +indistinguishable from the close, because reading it drains the queue and +`FIONREAD` then reports zero: a zero-length message contributes no bytes. macOS +and FreeBSD carry the flow on `SOCK_DGRAM`, where the datagram is delivered as +`Ok(0)` and the close follows it. Do not give a zero-length payload a meaning +of its own on this API. Carry a one-byte discriminator instead. Separating the +two needs a payload that is never zero bytes on the wire, which is a protocol +change rather than a receive-path one. **`peer_addr()`** and **`local_addr()`** return `FipsAddr`, not `io::Result`, unlike their `TcpStream` counterparts. These are field @@ -438,7 +440,7 @@ There is no acknowledgement, no retransmission, no ordering guarantee and no flow control between the two ends. A program that needs confirmation gets it from the peer, in the payload. -Four places lose data with nothing reported to the client. +Five places lose data with nothing reported to the client. **A full per-flow queue.** Inbound datagrams beyond `pending_per_flow` are dropped with a trace log and no client-visible signal. A program that stops @@ -467,6 +469,14 @@ snapshot taken at setup. The daemon re-checks each outbound datagram against the node's current limit and drops it silently if the transport MTU has since fallen. +**An empty datagram sent just before a close, on Linux.** A program that sends +a zero-length datagram and then drops its `FipsStream` may have the daemon read +that datagram as the close: the daemon frees the flow and the empty datagram +never reaches the peer. In the other direction, an empty datagram from the peer +that the daemon delivers immediately before its own half of the flow closes, +as when the daemon stops, reaches `recv` as `EPIPE` rather than `Ok(0)`. macOS +and FreeBSD are not affected. See `recv` above. + ### The drop causes, and what they mean An inbound datagram can be refused for eight reasons, which render as seven diff --git a/docs/tutorials/deploy-fips-gateway.md b/docs/tutorials/deploy-fips-gateway.md index d505eab2..2328d6df 100644 --- a/docs/tutorials/deploy-fips-gateway.md +++ b/docs/tutorials/deploy-fips-gateway.md @@ -132,7 +132,7 @@ gateway: pool: "fd01::/112" # virtual IP range (up to 65535 addresses) lan_interface: "br-lan" # LAN-facing interface for proxy NDP dns: - listen: "[::1]:5353" # gateway DNS bind (IPv6 loopback only) + # listen: "[::1]:5365" # the default; the init script points dnsmasq at this port upstream: "[::1]:5354" # FIPS daemon DNS resolver (matches daemon default) ttl: 60 # DNS TTL and mapping lifetime (seconds) pool_grace_period: 60 # seconds after last session before reclaiming @@ -147,9 +147,10 @@ Three things to notice: - `lan_interface: "br-lan"` — the OpenWrt LAN bridge. The gateway installs proxy-NDP entries on this interface so LAN clients can ARP-equivalent for pool addresses. -- `dns.listen: "[::1]:5353"` — the gateway's DNS bind, pinned to - IPv6 loopback only. dnsmasq, which owns LAN port 53, forwards - `.fips` queries to it. The init script wires up that forwarding; +- `dns.listen`, commented out — the gateway's DNS bind, left at its + default `[::1]:5365`, IPv6 loopback only. dnsmasq, which owns LAN + port 53, forwards `.fips` queries to it. The init script reads + `gateway.dns.listen` and points dnsmasq at whatever port it sets; you don't bind to a LAN address yourself. For the full reference, see @@ -172,7 +173,7 @@ Behind that single command, the init script `/etc/sysctl.d/fips-gateway.conf`. 2. **Reconfigures dnsmasq via UCI** so `.fips` queries arriving at the LAN's port 53 are forwarded to the gateway's loopback - listener on port 5353 instead of going straight to the daemon's + listener on port 5365 instead of going straight to the daemon's resolver on port 5354. (Dnsmasq still owns 53; the gateway sits in front of the daemon for `.fips` only.) 3. **Adds a global-scope IPv6 prefix** to `br-lan`. Without a @@ -242,7 +243,7 @@ Expectations: > **What just happened end to end.** Your client asked dnsmasq for > `test-us01.fips`. Dnsmasq forwarded the query to the gateway's -> loopback listener on port 5353. The gateway forwarded the query on +> loopback listener on port 5365. The gateway forwarded the query on > to the daemon's resolver on port 5354. The daemon answered with > `test-us01`'s mesh address (`fd97:...`). The gateway allocated a > virtual IP from `fd01::/112`, installed nftables DNAT/SNAT/ diff --git a/packaging/common/fips.yaml b/packaging/common/fips.yaml index e23b4bf1..885bec7e 100644 --- a/packaging/common/fips.yaml +++ b/packaging/common/fips.yaml @@ -133,7 +133,7 @@ transports: # pool: "fd01::/112" # lan_interface: "eth0" # dns: -# listen: "[::1]:5353" +# listen: "[::1]:5365" # # upstream must match the daemon's dns.bind_addr above. The # # default "[::1]:5354" matches the daemon's default. If you set # # the daemon to bind on a wildcard ("::") or specific address, diff --git a/packaging/nixos/README.md b/packaging/nixos/README.md index 87f8cd2d..2f6f52b1 100644 --- a/packaging/nixos/README.md +++ b/packaging/nixos/README.md @@ -70,8 +70,8 @@ operator-editable runtime state: | Path | Purpose | Writable | Seeded from | |---|---|---|---| | `/var/lib/fips/fips.yaml` | Main config | yes | `services.fips.configFile` (first run only) | -| `/var/lib/fips/fips.key` | Node identity (private) | yes | generated by fips on first start | -| `/var/lib/fips/fips.pub` | Node identity (public) | yes | generated by fips on first start | +| `/var/lib/fips/fips.key` | Node identity (private) | yes | generated by fips on first start with `persistent: true` | +| `/var/lib/fips/fips.pub` | Node identity (public) | yes | written by fips on every start | | `/etc/fips/hosts` | Static hostname → npub map | yes | shipped `hosts` (first run only) | | `/etc/fips/peers.allow` | Peer allowlist (ACL) | yes | operator-created | | `/etc/fips/peers.deny` | Peer denylist (ACL) | yes | operator-created | diff --git a/packaging/openwrt-ipk/files/etc/fips/fips.yaml b/packaging/openwrt-ipk/files/etc/fips/fips.yaml index 15682cd3..d7d5b28b 100644 --- a/packaging/openwrt-ipk/files/etc/fips/fips.yaml +++ b/packaging/openwrt-ipk/files/etc/fips/fips.yaml @@ -168,15 +168,15 @@ transports: # No BLE transport: OpenWrt builds target musl, which has no BlueZ backend. -# Outbound LAN gateway. dnsmasq forwards .fips queries to listen=[::1]:5353 -# while it runs (configured by the fips-gateway init script). Requires IPv6 -# forwarding enabled. +# Outbound LAN gateway. While it runs, the fips-gateway init script points +# dnsmasq's .fips forwarding at the port dns.listen sets (default [::1]:5365). +# Requires IPv6 forwarding enabled. gateway: enabled: true pool: "fd01::/112" lan_interface: "br-lan" dns: - listen: "[::1]:5353" + # listen: "[::1]:5365" # the default; the init script points dnsmasq at this port upstream: "[::1]:5354" ttl: 60 pool_grace_period: 60 diff --git a/packaging/openwrt-ipk/files/etc/init.d/fips-gateway b/packaging/openwrt-ipk/files/etc/init.d/fips-gateway index e9e16879..2c12326a 100755 --- a/packaging/openwrt-ipk/files/etc/init.d/fips-gateway +++ b/packaging/openwrt-ipk/files/etc/init.d/fips-gateway @@ -15,8 +15,10 @@ STOP=09 PROG=/usr/bin/fips-gateway CONFIG=/etc/fips/fips.yaml -# Port the gateway DNS listens on (must match dns.listen in fips.yaml). -GW_DNS_PORT=5353 +# Port the gateway DNS listens on when gateway.dns.listen is not set. Must +# match DEFAULT_DNS_LISTEN in the gateway's source; the scenario harness checks +# the two agree. The port actually used comes from gateway_dns_port. +GW_DNS_DEFAULT=5365 # Port the FIPS daemon DNS listens on. DAEMON_DNS_PORT=5354 @@ -40,11 +42,11 @@ start_service() { # Load conntrack module for /proc/net/nf_conntrack. modprobe nf_conntrack 2>/dev/null || true - # Redirect dnsmasq .fips forwarding from daemon (5354) to gateway (5353) - # so LAN clients get virtual IPs instead of raw mesh addresses. - # Done early and synchronously so dnsmasq is ready before the gateway - # starts accepting DNS queries. - dnsmasq_swap_fips_upstream "$GW_DNS_PORT" + # Redirect dnsmasq .fips forwarding from the daemon (5354) to the port the + # gateway listens on, so LAN clients get virtual IPs instead of raw mesh + # addresses. Done early and synchronously so dnsmasq is ready before the + # gateway starts accepting DNS queries. + dnsmasq_swap_fips_upstream "$(gateway_dns_port)" sleep 1 # Add a global-scope IPv6 prefix to br-lan so Android/Chrome clients @@ -87,6 +89,35 @@ gateway_config_enabled() { awk '/^gateway:/{found=1; next} found && /^[^ ]/{found=0} found && /enabled:/{gsub(/.*enabled:[[:space:]]*/, ""); gsub(/["'"'"']/, ""); print; exit}' "$CONFIG" } +# Print the port the gateway's DNS listener will bind: the digits after the +# last ":" of the "listen:" value inside the top-level "gateway:" block of +# fips.yaml, or $GW_DNS_DEFAULT when there is no such line or its value does +# not end in a port. Commented lines are skipped, and "listen_port:" (a port +# forward key) does not match. +# +# Block-style YAML only: a flow-style "dns: {listen: ...}" reads as the +# default. The dnsmasq entry this port feeds is always ::1#, so a +# gateway listening only on 127.0.0.1 is still not reachable through it. +gateway_dns_port() { + local port + port="$(awk ' + /^[A-Za-z_]/ { top = $1 } + top != "gateway:" { next } + /^[[:space:]]*#/ { next } + /^[[:space:]]+listen:/ { + v = $0 + sub(/^[[:space:]]+listen:[[:space:]]*/, "", v) + sub(/[[:space:]]+#.*$/, "", v) + gsub(/["'"'"']/, "", v) + sub(/[[:space:]]+$/, "", v) + n = split(v, part, ":") + if (n > 1 && part[n] ~ /^[0-9]+$/) print part[n] + exit + } + ' "$CONFIG" 2>/dev/null)" + echo "${port:-$GW_DNS_DEFAULT}" +} + # Extract the gateway pool CIDR from fips.yaml. # Looks for "pool:" indented under the top-level "gateway:" block. gateway_pool_cidr() { @@ -178,13 +209,20 @@ gateway_remove_global_prefix() { # $1 = target port number dnsmasq_swap_fips_upstream() { local port="$1" + local server - # Remove both possible entries, then add the correct one. - uci -q del_list dhcp.@dnsmasq[0].server="/fips/127.0.0.1#${DAEMON_DNS_PORT}" 2>/dev/null - uci -q del_list dhcp.@dnsmasq[0].server="/fips/127.0.0.1#${GW_DNS_PORT}" 2>/dev/null - # Also handle IPv6 loopback variants. - uci -q del_list dhcp.@dnsmasq[0].server="/fips/::1#${DAEMON_DNS_PORT}" 2>/dev/null - uci -q del_list dhcp.@dnsmasq[0].server="/fips/::1#${GW_DNS_PORT}" 2>/dev/null + # Remove every loopback .fips forward, then add the one for $port. That + # covers the daemon's port, this gateway's, and a stale entry for any other + # local port, such as the old default 5353 or a changed gateway.dns.listen. + # A .fips forward to another host and servers for other domains are kept. + # uci prints a list on one line separated by spaces. + for server in $(uci -q get 'dhcp.@dnsmasq[0].server' 2>/dev/null); do + case "$server" in + "/fips/::1#"* | "/fips/127.0.0.1#"*) + uci -q del_list dhcp.@dnsmasq[0].server="$server" 2>/dev/null + ;; + esac + done uci add_list dhcp.@dnsmasq[0].server="/fips/::1#${port}" uci commit dhcp diff --git a/packaging/openwrt-ipk/files/etc/uci-defaults/90-fips-setup b/packaging/openwrt-ipk/files/etc/uci-defaults/90-fips-setup index a89c27b5..44965971 100644 --- a/packaging/openwrt-ipk/files/etc/uci-defaults/90-fips-setup +++ b/packaging/openwrt-ipk/files/etc/uci-defaults/90-fips-setup @@ -1,8 +1,14 @@ #!/bin/sh # FIPS first-boot setup — runs once after package installation. # Configures the firewall and kernel modules for FIPS operation. -# This script is executed by /etc/rc.d/S19sysctl on first boot and -# then deleted by the UCI defaults mechanism. +# The UCI defaults mechanism runs it once and deletes it when it ends with +# status 0. +# +# It is executed by the package's postinst, but sourced, not executed, by +# OpenWrt's default_postinst for a package built from the SDK feed and by the +# first-boot uci-defaults run after a sysupgrade. Nothing here may exit early, +# change directory or set shell options, and it must end with status 0: a +# non-zero status leaves the script in place to run again on every boot. # --------------------------------------------------------------------------- # 1. Kernel modules @@ -63,9 +69,13 @@ uci commit firewall # dnsmasq init script builds its config from UCI and loads no directory under # /etc. The daemon's DNS responder binds ::1. The 127.0.0.1 del_list removes # the entry older packages added. While fips-gateway runs, its init script -# points this entry at the gateway's DNS port instead. +# points this entry at the gateway's DNS port instead. The two del_lists after +# the first remove the gateway's entry for its old default port, left behind +# by a gateway that stopped without its init script's stop running. uci -q del_list dhcp.@dnsmasq[0].server="/fips/127.0.0.1#5354" 2>/dev/null || true +uci -q del_list dhcp.@dnsmasq[0].server="/fips/::1#5353" 2>/dev/null || true +uci -q del_list dhcp.@dnsmasq[0].server="/fips/127.0.0.1#5353" 2>/dev/null || true uci -q del_list dhcp.@dnsmasq[0].server="/fips/::1#5354" 2>/dev/null || true uci add_list dhcp.@dnsmasq[0].server="/fips/::1#5354" uci -q del_list dhcp.@dnsmasq[0].rebind_domain="fips" 2>/dev/null || true @@ -86,4 +96,40 @@ grep -qxF 'nf_conntrack' /etc/modules.d/nf-conntrack 2>/dev/null || \ # proxy NDP entries are actually added. sysctl -p /etc/sysctl.d/fips-gateway.conf 2>/dev/null || true +# --------------------------------------------------------------------------- +# 5. Gateway DNS listen port +# --------------------------------------------------------------------------- +# Every release up to 0.5.1 shipped the gateway's DNS listener on 5353, the +# mDNS port, which the daemon's LAN rendezvous can hold. fips.yaml is a +# conffile and fips-ap-setup edits it, so an upgrade keeps the old line. +# Rewrite exactly that shipped line, four-space indent and nothing after the +# closing quote, inside the top-level gateway block, to the line a fresh +# install ships. Any other value is left as configured; the gateway warns at +# startup when it is still on the mDNS port. +# +# Runs last, and cannot fail the script: see the note at the top. +fips_migrate_gateway_dns_listen() { + local cfg=/etc/fips/fips.yaml + local old=' listen: "[::1]:5353"' + local new=' # listen: "[::1]:5365" # the default; the init script points dnsmasq at this port' + local msg='fips: moved gateway.dns.listen off the mDNS port 5353 to the default [::1]:5365' + + if [ -f "$cfg" ] && grep -qxF "$old" "$cfg" 2>/dev/null; then + if awk -v old="$old" -v new="$new" ' + /^[A-Za-z_]/ { top = $1 } + top == "gateway:" && $0 == old { print new; changed = 1; next } + { print } + END { exit changed ? 0 : 1 } + ' "$cfg" > "$cfg.tmp" 2>/dev/null && + chmod 600 "$cfg.tmp" 2>/dev/null && + mv -f "$cfg.tmp" "$cfg" 2>/dev/null; then + logger -t fips "$msg" 2>/dev/null || true + echo "$msg" + else + rm -f "$cfg.tmp" 2>/dev/null || true + fi + fi +} +fips_migrate_gateway_dns_listen + exit 0 diff --git a/packaging/systemd/README.install.md b/packaging/systemd/README.install.md index 2448dbf5..01ebfd0a 100644 --- a/packaging/systemd/README.install.md +++ b/packaging/systemd/README.install.md @@ -17,8 +17,8 @@ sudo ./install.sh | fipstop (TUI) | /usr/local/bin/fipstop | | fips-gateway (LAN bridge) | /usr/local/bin/fips-gateway | | Configuration | /etc/fips/fips.yaml | -| Identity key | /etc/fips/fips.key (auto-generated) | -| Public key | /etc/fips/fips.pub (auto-generated) | +| Identity key | /etc/fips/fips.key (generated on first start with `persistent: true`) | +| Public key | /etc/fips/fips.pub (written on every start) | | Hosts file | /etc/fips/hosts | | Firewall baseline | /etc/fips/fips.nft | | Firewall drop-in directory | /etc/fips/fips.d/ | diff --git a/packaging/windows/build-zip.ps1 b/packaging/windows/build-zip.ps1 index 532cca75..a098fff4 100644 --- a/packaging/windows/build-zip.ps1 +++ b/packaging/windows/build-zip.ps1 @@ -101,9 +101,20 @@ Control Socket: Configuration: The service reads C:\ProgramData\fips\fips.yaml, where - install-service.ps1 puts it, and keeps fips.key, hosts, - peers.allow and peers.deny beside it. Edit fips.yaml there - before starting the service. + install-service.ps1 puts it, and keeps hosts, peers.allow, + peers.deny and, with node.identity.persistent: true, fips.key + beside it. Edit fips.yaml there before starting the service. + + install-service.ps1 creates empty peers.allow and peers.deny + there, which allow every peer until you add entries. To clear + a list, empty the file; do not delete it. While either file + is missing from C:\ProgramData\fips, the service still reads + that file from \etc\fips on the system drive, where earlier + releases kept it and any local user can create it. The + installer stops if it finds a file there with none in + C:\ProgramData\fips: review that file, move it into + C:\ProgramData\fips or delete it, and run install-service.ps1 + again. install-service.ps1 restricts C:\ProgramData\fips to SYSTEM and Administrators before writing into it. Reading or editing diff --git a/packaging/windows/install-service.ps1 b/packaging/windows/install-service.ps1 index 8b4bf214..ee293629 100644 --- a/packaging/windows/install-service.ps1 +++ b/packaging/windows/install-service.ps1 @@ -130,6 +130,31 @@ foreach ($item in @(Get-ChildItem -LiteralPath $ConfigDir -Force)) { Write-Host " Restricted $ConfigDir to SYSTEM and Administrators" +# Releases before this one read peers.allow and peers.deny from \etc\fips on +# the system drive, where any local user can create files, and the service +# still reads a file there when it is missing from the config directory. +# Stop rather than enforce, or silently drop, a list nobody has reviewed. +$legacyAclDir = "$env:SystemDrive\etc\fips" +foreach ($name in @("peers.allow", "peers.deny")) { + $legacy = Join-Path $legacyAclDir $name + $current = "$ConfigDir\$name" + if ((Test-Path -LiteralPath $legacy) -and -not (Test-Path -LiteralPath $current)) { + Write-Error "$legacy exists and $current does not, so the service would enforce the old file. Earlier releases read it, and any local user can write there. Review it, then move it to $current or delete it, and run install-service.ps1 again." + exit 1 + } +} + +# Empty peer ACL files allow every peer. Having them here means the service +# never falls back to the \etc\fips copies. Empty them to clear a list; do not +# delete them. +foreach ($name in @("peers.allow", "peers.deny")) { + $aclFile = "$ConfigDir\$name" + if (-not (Test-Path -LiteralPath $aclFile)) { + New-Item -ItemType File -Path $aclFile | Out-Null + Write-Host " Created empty $aclFile (allows every peer until you add entries)" + } +} + # Copy binaries $Binaries = @("fips.exe", "fipsctl.exe", "fipstop.exe") foreach ($bin in $Binaries) { diff --git a/src/bin/fips-gateway.rs b/src/bin/fips-gateway.rs index fb2c6164..3d2b784d 100644 --- a/src/bin/fips-gateway.rs +++ b/src/bin/fips-gateway.rs @@ -251,8 +251,9 @@ async fn main() { std::process::exit(1); } - // Check DNS upstream reachability (proves the FIPS daemon is running) - { + // Check DNS upstream reachability (proves the FIPS daemon is running). + // The resolver later forwards to the address this probe reached. + let upstream_addr = { let upstream = gw_config.dns.upstream(); info!(upstream = %upstream, "Checking DNS upstream reachability"); @@ -363,6 +364,29 @@ async fn main() { ); std::process::exit(1); } + upstream_addr + }; + + // --- Bind the DNS listener --- + // + // Before the pool, NAT table and routes exist, so a port that is already + // taken ends the gateway with nothing to tear down, and a service manager + // restarting it does not churn nftables. + if gw_config.dns.is_mdns() { + warn!( + "gateway.dns.listen uses port 5353, the mDNS port; an mDNS responder (the fips daemon's LAN rendezvous, avahi) will conflict with it; the default is now [::1]:5365" + ); + } + let dns_socket = match dns::bind_listener(gw_config.dns.listen()).await { + Ok(socket) => socket, + Err(e) => { + error!("{e}"); + std::process::exit(1); + } + }; + match dns_socket.local_addr() { + Ok(addr) => info!(addr = %addr, "Gateway DNS resolver listening"), + Err(_) => info!(addr = %gw_config.dns.listen(), "Gateway DNS resolver listening"), } // --- Initialize components --- @@ -421,27 +445,16 @@ async fn main() { // --- Start DNS resolver task --- - let dns_pool = Arc::clone(&ip_pool); - let dns_event_tx = event_tx.clone(); - let dns_shutdown = shutdown_rx.clone(); - let dns_listen = gw_config.dns.listen().to_string(); - let dns_upstream = gw_config.dns.upstream().to_string(); - let dns_ttl = gw_config.dns.ttl(); - - let dns_task = tokio::spawn(async move { - if let Err(e) = dns::run_dns_resolver( - &dns_listen, - &dns_upstream, - dns_ttl, - dns_pool, - dns_event_tx, - dns_shutdown, - ) - .await - { - error!(error = %e, "DNS resolver error"); - } - }); + // Held in an Option because the main loop may see it complete, and a + // completed JoinHandle panics if it is polled again. + let mut dns_task = Some(tokio::spawn(dns::serve( + dns_socket, + upstream_addr, + gw_config.dns.ttl(), + Arc::clone(&ip_pool), + event_tx.clone(), + shutdown_rx.clone(), + ))); // --- Snapshot channel for control socket --- @@ -531,6 +544,7 @@ async fn main() { info!("fips-gateway running"); + let mut exit_code = 0; loop { tokio::select! { Some(event) = event_rx.recv() => { @@ -559,6 +573,25 @@ async fn main() { } } } + // The resolver ends only on shutdown, which has not been + // signalled while this loop runs, so any completion here means + // .fips resolution has stopped. Exit non-zero so systemd or procd + // restarts the gateway or shows it failed. + result = async { dns_task.as_mut().expect("guarded by the precondition").await }, + if dns_task.is_some() => { + dns_task = None; + let cause = match result { + Ok(Ok(())) => "the resolver returned without an error".to_string(), + Ok(Err(e)) => e.to_string(), + Err(e) => e.to_string(), + }; + error!( + cause = %cause, + "Gateway DNS resolver stopped; exiting so the service manager restarts the gateway" + ); + exit_code = 1; + break; + } _ = tokio::signal::ctrl_c() => { info!("Received SIGINT, shutting down"); break; @@ -582,7 +615,9 @@ async fn main() { task.abort(); let _ = task.await; } - let _ = dns_task.await; + if let Some(task) = dns_task { + let _ = task.await; + } let _ = tick_task.await; // Log final pool status @@ -606,4 +641,7 @@ async fn main() { } info!("fips-gateway shutdown complete"); + if exit_code != 0 { + std::process::exit(exit_code); + } } diff --git a/src/bin/fips.rs b/src/bin/fips.rs index 8b9572bf..52bfaf00 100644 --- a/src/bin/fips.rs +++ b/src/bin/fips.rs @@ -484,7 +484,8 @@ mod service { "Configuration: the service reads {}", dir.join("fips.yaml").display() ); - println!(" keep fips.key, hosts, peers.allow and peers.deny beside it."); + println!(" keep hosts, peers.allow and peers.deny beside it, and fips.key"); + println!(" too when node.identity.persistent is true."); println!( "Logs: the service writes {}", dir.join("fips.log").display() diff --git a/src/bin/fipsctl.rs b/src/bin/fipsctl.rs index 5ee737db..3df15b8e 100644 --- a/src/bin/fipsctl.rs +++ b/src/bin/fipsctl.rs @@ -593,8 +593,12 @@ fn main() { eprintln!("{npub}"); eprintln!("Key files written to: {}/", dir.display()); eprintln!(); - eprintln!("NOTE: Set 'node.identity.persistent: true' in fips.yaml"); - eprintln!(" or these keys will be overwritten on next daemon start."); + eprintln!( + "NOTE: Set 'node.identity.persistent: true' in fips.yaml before the next daemon start." + ); + eprintln!( + " Without it the daemon does not use this key: it moves fips.key aside to fips.key.unused." + ); return; } diff --git a/src/config/gateway.rs b/src/config/gateway.rs index b9a96330..22c383d6 100644 --- a/src/config/gateway.rs +++ b/src/config/gateway.rs @@ -9,7 +9,10 @@ use serde::{Deserialize, Serialize}; /// Default gateway DNS listen address. /// -/// Loopback-only on the unprivileged port 5353. The canonical +/// Loopback-only on the unprivileged port 5365, which IANA leaves +/// unassigned and no common resolver uses. It is not 5353, the mDNS +/// port, which the daemon's LAN rendezvous, avahi-daemon and +/// systemd-resolved can hold. The canonical /// gateway deployment is a host already serving DHCP/DNS to a LAN /// segment (e.g., an OpenWrt AP), where port 53 is taken by the /// existing resolver and `.fips` queries are forwarded to the @@ -21,7 +24,7 @@ use serde::{Deserialize, Serialize}; /// explicit `::1` do not accept v4-mapped traffic. Forwarders that /// reach the gateway over IPv4 loopback (`127.0.0.1`) need to be /// pointed at an explicit IPv4 listen address instead. -const DEFAULT_DNS_LISTEN: &str = "[::1]:5353"; +const DEFAULT_DNS_LISTEN: &str = "[::1]:5365"; /// Default upstream DNS resolver (FIPS daemon). /// @@ -131,7 +134,7 @@ pub struct PortForward { /// Gateway DNS resolver configuration (`gateway.dns.*`). #[derive(Debug, Clone, Default, Serialize, Deserialize)] pub struct GatewayDnsConfig { - /// Listen address and port (default: `[::1]:5353`). + /// Listen address and port (default: `[::1]:5365`). #[serde(default, skip_serializing_if = "Option::is_none")] pub listen: Option, @@ -146,7 +149,7 @@ pub struct GatewayDnsConfig { } impl GatewayDnsConfig { - /// Get the listen address (default: `[::1]:5353`). + /// Get the listen address (default: `[::1]:5365`). pub fn listen(&self) -> &str { self.listen.as_deref().unwrap_or(DEFAULT_DNS_LISTEN) } @@ -160,6 +163,18 @@ impl GatewayDnsConfig { pub fn ttl(&self) -> u32 { self.ttl.unwrap_or(DEFAULT_DNS_TTL) } + + /// The port of a listen address: the digits after its last `:`, or + /// `None` when they do not form a port. Works on a hostname form too. + pub(crate) fn port_of(listen: &str) -> Option { + listen.rsplit_once(':')?.1.parse().ok() + } + + /// Whether the listen address is on the mDNS port, which an mDNS + /// responder can take from the gateway at any time. + pub fn is_mdns(&self) -> bool { + Self::port_of(self.listen()) == Some(5353) + } } /// Conntrack timeout overrides (`gateway.conntrack.*`). @@ -218,7 +233,7 @@ lan_interface: "eth0" assert!(!config.enabled); assert_eq!(config.pool, "fd01::/112"); assert_eq!(config.lan_interface, "eth0"); - assert_eq!(config.dns.listen(), "[::1]:5353"); + assert_eq!(config.dns.listen(), "[::1]:5365"); assert_eq!(config.dns.upstream(), "[::1]:5354"); assert_eq!(config.dns.ttl(), 60); assert_eq!(config.grace_period(), 60); @@ -226,6 +241,43 @@ lan_interface: "eth0" assert_eq!(config.conntrack.udp_timeout(), 30); } + #[test] + fn a_listen_on_5353_is_flagged_as_mdns() { + for listen in [ + "[::1]:5353", + "[::]:5353", + "127.0.0.1:5353", + "localhost:5353", + ] { + let dns = GatewayDnsConfig { + listen: Some(listen.to_string()), + ..Default::default() + }; + assert!(dns.is_mdns(), "{listen} must be flagged as the mDNS port"); + } + } + + #[test] + fn the_default_and_other_ports_are_not_flagged_as_mdns() { + assert!(!GatewayDnsConfig::default().is_mdns()); + for listen in [ + "[::1]:5365", + "[::]:53", + "192.168.1.1:53", + "[::]:5355", + "localhost", + ] { + let dns = GatewayDnsConfig { + listen: Some(listen.to_string()), + ..Default::default() + }; + assert!( + !dns.is_mdns(), + "{listen} must not be flagged as the mDNS port" + ); + } + } + #[test] fn test_gateway_config_custom() { let yaml = r#" diff --git a/src/config/mod.rs b/src/config/mod.rs index 08d0a5f0..e2462582 100644 --- a/src/config/mod.rs +++ b/src/config/mod.rs @@ -596,13 +596,60 @@ pub fn write_pub_file(path: &Path, npub: &str) -> Result<(), ConfigError> { Ok(()) } +/// Move a key file found at an ephemeral start to `.unused` beside it, +/// so it is neither used nor overwritten and stays recoverable. +/// +/// Returns the new path when the file was moved. Does nothing when no file is +/// at `key_path`; a dangling symlink counts as a file and is moved as a link. +/// An existing file at the aside path is never replaced: the key is left in +/// place and a warning says so, as it does when the rename fails. Neither case +/// stops the start. +fn retire_key(key_path: &Path) -> Option { + // symlink_metadata rather than exists: a dangling symlink at the key path + // reports exists() == false but is still a file the operator put there. + key_path.symlink_metadata().ok()?; + + let mut name = key_path.file_name()?.to_os_string(); + name.push(".unused"); + let aside = key_path.with_file_name(name); + + let failure = match aside.symlink_metadata() { + Ok(_) => "a file already exists at the aside path".to_string(), + Err(e) if e.kind() != std::io::ErrorKind::NotFound => e.to_string(), + Err(_) => match std::fs::rename(key_path, &aside) { + Ok(()) => { + tracing::warn!( + path = %key_path.display(), + moved_to = %aside.display(), + config_key = "node.identity.persistent", + "An identity key file was found in ephemeral mode and moved aside, not used; \ + set node.identity.persistent: true and move it back to use it" + ); + return Some(aside); + } + Err(e) => e.to_string(), + }, + }; + + tracing::warn!( + path = %key_path.display(), + aside = %aside.display(), + error = %failure, + config_key = "node.identity.persistent", + "An identity key file was found in ephemeral mode and could not be moved aside; \ + it is not used; set node.identity.persistent: true to use it, or remove it" + ); + None +} + /// Resolve identity from config and key file. /// /// Behavior depends on `node.identity.persistent`: /// /// - **`persistent: false`** (default): generate a fresh ephemeral keypair -/// every start. Key files are written for operator visibility but overwritten -/// on each restart. +/// every start. Only `fips.pub` is written, so the running npub is visible; +/// the private key is never written. A `fips.key` already at the path is +/// moved aside to `fips.key.unused` with a warning, not used or overwritten. /// /// - **`persistent: true`**: use three-tier resolution: /// 1. Explicit nsec in config — highest priority @@ -738,8 +785,8 @@ pub fn resolve_identity( } } } else { - // Ephemeral mode (default): fresh keypair every start, write key files - // for operator visibility + // Ephemeral mode (default): a fresh keypair every start, held only in + // memory. Only the public key file is written. let identity = Identity::generate(); // `keypair()` and `secret_key()` each hand back a whole private key // rather than a handle, so both temporaries are bound and erased. @@ -754,25 +801,8 @@ pub fn resolve_identity( let _ = std::fs::create_dir_all(parent); } - // symlink_metadata rather than exists: a dangling symlink at the key - // path reports exists() == false but is still an existing file the - // write is about to act on. - if key_path.symlink_metadata().is_ok() { - tracing::warn!( - path = %key_path.display(), - config_key = "node.identity.persistent", - "An existing key file at this path is being replaced by a fresh ephemeral \ - identity; set node.identity.persistent: true to keep the existing identity" - ); - } + retire_key(&key_path); - if let Err(e) = write_key_file(&key_path, &nsec) { - tracing::warn!( - path = %key_path.display(), - error = %e, - "Failed to write the ephemeral key file" - ); - } if let Err(e) = write_pub_file(&pub_path, &npub) { tracing::warn!( path = %pub_path.display(), @@ -2103,29 +2133,64 @@ node: assert_eq!(fs::read_to_string(&victim).unwrap(), "victim contents\n"); } + /// The names in `dir`, sorted, so a test can assert on the whole directory. + fn dir_names(dir: &Path) -> Vec { + let mut names: Vec = fs::read_dir(dir) + .unwrap() + .map(|e| e.unwrap().file_name().to_string_lossy().into_owned()) + .collect(); + names.sort(); + names + } + + /// The npub a resolved identity runs as. + fn resolved_npub(resolved: &ResolvedIdentity) -> String { + crate::Identity::from_secret_str(&resolved.nsec) + .unwrap() + .npub() + } + #[test] - fn test_ephemeral_over_existing_key_warns() { + fn ephemeral_start_moves_an_existing_key_aside_intact() { let temp_dir = TempDir::new().unwrap(); let config_path = temp_dir.path().join("fips.yaml"); let key_path = temp_dir.path().join("fips.key"); + let aside_path = temp_dir.path().join("fips.key.unused"); fs::write(&config_path, "node:\n identity: {}\n").unwrap(); let identity = crate::Identity::generate(); let existing = crate::encode_nsec(&identity.keypair().secret_key()); write_key_file(&key_path, &existing).unwrap(); + let planted = fs::read(&key_path).unwrap(); let config = Config::load_file(&config_path).unwrap(); let (resolved, logs) = capture_logs(|| resolve_identity(&config, std::slice::from_ref(&config_path)).unwrap()); assert_ne!(resolved.nsec, existing); + assert!( + key_path.symlink_metadata().is_err(), + "the key file must be moved away from the path a persistent start reads" + ); + assert_eq!( + fs::read(&aside_path).unwrap(), + planted, + "the key set aside must hold the planted bytes exactly" + ); + #[cfg(unix)] + { + use std::os::unix::fs::MetadataExt; + assert_eq!(fs::metadata(&aside_path).unwrap().mode() & 0o777, 0o600); + } let warnings = logs.warnings(); assert!( warnings .iter() .any(|w| w.contains(&key_path.display().to_string()) - && w.contains("node.identity.persistent")), - "expected a warning naming the key path and the config key, got {warnings:?}" + && w.contains(&aside_path.display().to_string()) + && w.contains("node.identity.persistent") + && w.contains("and moved aside, not used")), + "expected the moved-aside warning naming both paths and the config key, got {warnings:?}" ); } @@ -2135,6 +2200,7 @@ node: let temp_dir = TempDir::new().unwrap(); let config_path = temp_dir.path().join("fips.yaml"); let key_path = temp_dir.path().join("fips.key"); + let aside_path = temp_dir.path().join("fips.key.unused"); let target = temp_dir.path().join("absent-target"); fs::write(&config_path, "node:\n identity: {}\n").unwrap(); @@ -2148,8 +2214,21 @@ node: assert!( warnings .iter() - .any(|w| w.contains(&key_path.display().to_string())), - "expected a warning naming the key path, got {warnings:?}" + .any(|w| w.contains(&key_path.display().to_string()) + && w.contains("and moved aside, not used")), + "expected the moved-aside warning naming the key path, got {warnings:?}" + ); + assert!( + key_path.symlink_metadata().is_err(), + "the dangling symlink must be moved away from the key path" + ); + assert!( + aside_path + .symlink_metadata() + .unwrap() + .file_type() + .is_symlink(), + "the symlink itself must be what was moved aside, not a file written through it" ); assert!( !target.exists(), @@ -2157,6 +2236,155 @@ node: ); } + #[test] + fn ephemeral_start_leaves_a_key_in_place_when_the_aside_name_is_taken() { + let temp_dir = TempDir::new().unwrap(); + let config_path = temp_dir.path().join("fips.yaml"); + let key_path = temp_dir.path().join("fips.key"); + let aside_path = temp_dir.path().join("fips.key.unused"); + + fs::write(&config_path, "node:\n identity: {}\n").unwrap(); + let current = crate::encode_nsec(&crate::Identity::generate().keypair().secret_key()); + let earlier = crate::encode_nsec(&crate::Identity::generate().keypair().secret_key()); + write_key_file(&key_path, ¤t).unwrap(); + write_key_file(&aside_path, &earlier).unwrap(); + let key_bytes = fs::read(&key_path).unwrap(); + let aside_bytes = fs::read(&aside_path).unwrap(); + + let config = Config::load_file(&config_path).unwrap(); + let (resolved, logs) = + capture_logs(|| resolve_identity(&config, std::slice::from_ref(&config_path)).unwrap()); + + assert!(matches!(resolved.source, IdentitySource::Ephemeral)); + assert_ne!(resolved.nsec, current); + assert_eq!( + fs::read(&key_path).unwrap(), + key_bytes, + "the key must be left in place, not overwritten" + ); + assert_eq!( + fs::read(&aside_path).unwrap(), + aside_bytes, + "the file already at the aside path must not be replaced" + ); + let warnings = logs.warnings(); + assert!( + warnings + .iter() + .any(|w| w.contains(&key_path.display().to_string()) + && w.contains(&aside_path.display().to_string()) + && w.contains("node.identity.persistent") + && w.contains("could not be moved aside")), + "expected the could-not-move warning naming both paths and the config key, got {warnings:?}" + ); + } + + #[cfg(unix)] + #[test] + fn ephemeral_start_proceeds_when_the_key_cannot_be_moved() { + use std::os::unix::fs::PermissionsExt; + + /// Puts the directory's mode back when dropped, so a failing + /// assertion does not leave a read-only directory behind that + /// `TempDir` cannot remove. + struct RestoreMode<'a>(&'a Path); + impl Drop for RestoreMode<'_> { + fn drop(&mut self) { + let _ = fs::set_permissions(self.0, fs::Permissions::from_mode(0o755)); + } + } + + // Coverage gap: root bypasses directory permissions, so the rename + // succeeds and this branch goes unexercised when the suite runs as + // root. The aside-name-taken test still covers the same warning. + if unsafe { libc::geteuid() } == 0 { + eprintln!("skipped: running as root, which a read-only directory does not stop"); + return; + } + + let temp_dir = TempDir::new().unwrap(); + let config_path = temp_dir.path().join("fips.yaml"); + let key_path = temp_dir.path().join("fips.key"); + + fs::write(&config_path, "node:\n identity: {}\n").unwrap(); + let existing = crate::encode_nsec(&crate::Identity::generate().keypair().secret_key()); + write_key_file(&key_path, &existing).unwrap(); + let planted = fs::read(&key_path).unwrap(); + let config = Config::load_file(&config_path).unwrap(); + + fs::set_permissions(temp_dir.path(), fs::Permissions::from_mode(0o555)).unwrap(); + let _restore = RestoreMode(temp_dir.path()); + + let (resolved, logs) = + capture_logs(|| resolve_identity(&config, std::slice::from_ref(&config_path)).unwrap()); + + assert!(matches!(resolved.source, IdentitySource::Ephemeral)); + assert_ne!(resolved.nsec, existing); + assert_eq!( + fs::read(&key_path).unwrap(), + planted, + "a key that cannot be moved must be left as it was, not overwritten" + ); + let warnings = logs.warnings(); + assert!( + warnings + .iter() + .any(|w| w.contains(&key_path.display().to_string()) + && w.contains("could not be moved aside") + && w.contains("os error")), + "expected a warning naming the key path and the rename error, got {warnings:?}" + ); + } + + #[test] + fn ephemeral_restarts_never_leave_a_private_key_file() { + let temp_dir = TempDir::new().unwrap(); + let config_path = temp_dir.path().join("fips.yaml"); + let pub_path = temp_dir.path().join("fips.pub"); + + fs::write(&config_path, "node:\n identity: {}\n").unwrap(); + let config = Config::load_file(&config_path).unwrap(); + + for start in 1..=3 { + let resolved = resolve_identity(&config, std::slice::from_ref(&config_path)).unwrap(); + assert_eq!( + dir_names(temp_dir.path()), + ["fips.pub", "fips.yaml"], + "after start {start} the directory must hold only the config and the public key" + ); + assert_eq!( + fs::read_to_string(&pub_path).unwrap().trim(), + resolved_npub(&resolved), + "after start {start} fips.pub must name the running identity" + ); + } + } + + #[test] + fn persistent_start_ignores_a_key_set_aside() { + let temp_dir = TempDir::new().unwrap(); + let config_path = temp_dir.path().join("fips.yaml"); + let key_path = temp_dir.path().join("fips.key"); + let aside_path = temp_dir.path().join("fips.key.unused"); + + fs::write(&config_path, "node:\n identity:\n persistent: true\n").unwrap(); + let earlier = crate::encode_nsec(&crate::Identity::generate().keypair().secret_key()); + write_key_file(&aside_path, &earlier).unwrap(); + let aside_bytes = fs::read(&aside_path).unwrap(); + + let config = Config::load_file(&config_path).unwrap(); + let resolved = resolve_identity(&config, std::slice::from_ref(&config_path)).unwrap(); + + assert!(matches!(resolved.source, IdentitySource::Generated(_))); + assert_ne!(resolved.nsec, earlier); + assert_eq!(read_key_file(&key_path).unwrap(), resolved.nsec); + assert_eq!( + fs::read(&aside_path).unwrap(), + aside_bytes, + "a persistent start must leave the key set aside untouched" + ); + } + #[cfg(unix)] #[test] fn test_persistent_permissive_key_warns() { @@ -2288,11 +2516,18 @@ node: let resolved = resolve_identity(&config, std::slice::from_ref(&config_path)).unwrap(); assert!(matches!(resolved.source, IdentitySource::Ephemeral)); - // Key files should still be written for operator visibility + // Only the public key is written: an ephemeral private key lives in + // memory and nowhere else. let key_path = temp_dir.path().join("fips.key"); let pub_path = temp_dir.path().join("fips.pub"); - assert!(key_path.exists()); - assert!(pub_path.exists()); + assert_eq!( + key_path.symlink_metadata().unwrap_err().kind(), + std::io::ErrorKind::NotFound + ); + assert_eq!( + fs::read_to_string(&pub_path).unwrap().trim(), + resolved_npub(&resolved) + ); } #[test] diff --git a/src/gateway/dns.rs b/src/gateway/dns.rs index 4d9a069e..97054a5c 100644 --- a/src/gateway/dns.rs +++ b/src/gateway/dns.rs @@ -17,6 +17,7 @@ use tracing::{debug, info, trace, warn}; use super::pool::{PoolEvent, VirtualIpPool}; use crate::NodeAddr; +use crate::config::GatewayDnsConfig; /// Timeout for upstream DNS queries. const UPSTREAM_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(5); @@ -156,25 +157,108 @@ fn build_aaaa_response(query: &Packet, virtual_ip: Ipv6Addr, ttl: u32) -> Option response.build_bytes_vec_compressed().ok() } +/// The gateway DNS listener could not be bound. +/// +/// The message names the listen address and, when the port is already in +/// use, the service most likely to hold it and how to find the holder. +#[derive(Debug, thiserror::Error)] +#[error("cannot bind the gateway DNS listener on {listen}: {source}{}", in_use_hint(.listen, .source))] +pub struct ListenError { + listen: String, + source: std::io::Error, +} + +impl ListenError { + /// The kind of the underlying bind error. + pub fn kind(&self) -> std::io::ErrorKind { + self.source.kind() + } +} + +/// The suffix `ListenError`'s message carries for an address-in-use error: +/// the likely holder of the port, and how to find the actual one. +fn in_use_hint(listen: &str, source: &std::io::Error) -> String { + if source.kind() != std::io::ErrorKind::AddrInUse { + return String::new(); + } + let (holder, port) = match GatewayDnsConfig::port_of(listen) { + Some(port) => (holder_hint(port), port.to_string()), + None => (holder_hint(0), "".to_string()), + }; + format!( + "; {holder}; find the holder with `ss -ulpn 'sport = :{port}'` or `netstat -ulnp`, \ + or set gateway.dns.listen to a free port and point the resolver that forwards .fips at it" + ) +} + +/// The service most likely to hold a DNS listen port that is already in use. +pub(crate) fn holder_hint(port: u16) -> &'static str { + match port { + 53 => { + "another DNS server holds port 53: dnsmasq, systemd-resolved's stub listener, unbound or BIND" + } + 5353 => { + "port 5353 is mDNS: the fips daemon's LAN rendezvous (node.rendezvous.lan), \ + avahi-daemon or systemd-resolved's MulticastDNS may hold it" + } + 5354 => { + "the fips daemon's own DNS responder listens on 5354 by default; \ + gateway.dns.listen must not be the daemon's DNS port" + } + 5355 => "port 5355 is LLMNR, held by systemd-resolved unless LLMNR=no", + 5365 => "another fips-gateway may already be running", + _ => "another process holds it", + } +} + +/// Bind the gateway DNS listener. +/// +/// Called before the gateway creates anything it would have to tear down, so +/// a port that is already taken stops the gateway before it starts. +pub async fn bind_listener(listen: &str) -> Result { + UdpSocket::bind(listen).await.map_err(|source| ListenError { + listen: listen.to_string(), + source, + }) +} + /// Run the gateway DNS resolver. /// -/// Listens for DNS queries, forwards `.fips` queries to the upstream -/// daemon resolver, allocates virtual IPs, and returns them to clients. +/// Binds `listen_addr`, then serves as [`serve`] does. The gateway binary +/// binds and serves separately so that a bind failure stops it at startup. pub async fn run_dns_resolver( listen_addr: &str, upstream_addr: &str, ttl: u32, pool: std::sync::Arc>, event_tx: tokio::sync::mpsc::Sender, - mut shutdown: watch::Receiver, + shutdown: watch::Receiver, ) -> Result<(), std::io::Error> { - let socket = UdpSocket::bind(listen_addr).await?; + let socket = bind_listener(listen_addr) + .await + .map_err(|e| std::io::Error::new(e.kind(), e))?; info!(addr = %listen_addr, "Gateway DNS resolver listening"); let upstream: SocketAddr = upstream_addr .parse() .map_err(|e| std::io::Error::new(std::io::ErrorKind::InvalidInput, e))?; + serve(socket, upstream, ttl, pool, event_tx, shutdown).await +} + +/// Serve DNS queries on a bound listener until shutdown. +/// +/// Forwards `.fips` queries to the upstream daemon resolver, allocates +/// virtual IPs, and returns them to clients. Returns `Ok` on shutdown and +/// `Err` when receiving from the listener fails. +pub async fn serve( + socket: UdpSocket, + upstream: SocketAddr, + ttl: u32, + pool: std::sync::Arc>, + event_tx: tokio::sync::mpsc::Sender, + mut shutdown: watch::Receiver, +) -> Result<(), std::io::Error> { let mut buf = vec![0u8; MAX_DNS_SIZE]; loop { @@ -807,6 +891,69 @@ mod tests { )); } + #[test] + fn an_in_use_hint_names_the_mdns_responders_for_5353() { + let hint = holder_hint(5353); + assert!(hint.contains("mDNS"), "{hint}"); + assert!(hint.contains("node.rendezvous.lan"), "{hint}"); + assert!(hint.contains("avahi-daemon"), "{hint}"); + } + + #[test] + fn an_in_use_hint_names_llmnr_for_5355() { + let hint = holder_hint(5355); + assert!(hint.contains("LLMNR"), "{hint}"); + assert!(!hint.contains("mDNS"), "{hint}"); + } + + #[test] + fn an_in_use_hint_names_the_daemon_for_5354() { + let hint = holder_hint(5354); + assert!(hint.contains("fips daemon's own DNS responder"), "{hint}"); + } + + #[test] + fn an_in_use_hint_names_a_dns_server_for_53() { + let hint = holder_hint(53); + assert!(hint.contains("another DNS server"), "{hint}"); + assert!(hint.contains("dnsmasq"), "{hint}"); + } + + #[test] + fn an_in_use_hint_names_another_gateway_for_the_default_port() { + let hint = holder_hint(5365); + assert!(hint.contains("another fips-gateway"), "{hint}"); + } + + #[test] + fn an_in_use_hint_names_another_process_for_an_unknown_port() { + assert_eq!(holder_hint(40000), "another process holds it"); + } + + #[tokio::test] + async fn binding_a_held_port_fails_with_addr_in_use_and_names_the_port_ss_and_netstat() { + let holder = UdpSocket::bind("[::1]:0").await.unwrap(); + let port = holder.local_addr().unwrap().port(); + let listen = format!("[::1]:{port}"); + + let err = bind_listener(&listen) + .await + .expect_err("binding a held port must fail"); + assert_eq!(err.kind(), std::io::ErrorKind::AddrInUse); + let message = err.to_string(); + assert!(message.contains(&listen), "{message}"); + assert!(message.contains(&format!("sport = :{port}")), "{message}"); + assert!(message.contains("ss -ulpn"), "{message}"); + assert!(message.contains("netstat -ulnp"), "{message}"); + assert!(message.contains(holder_hint(port)), "{message}"); + } + + #[tokio::test] + async fn binding_a_free_port_returns_a_bound_socket() { + let socket = bind_listener("[::1]:0").await.expect("bind a free port"); + assert_ne!(socket.local_addr().unwrap().port(), 0); + } + #[test] fn test_extract_fips_name() { // Build a simple AAAA query for test.fips diff --git a/src/native/client/mod.rs b/src/native/client/mod.rs index 1fb85681..ee4c8ade 100644 --- a/src/native/client/mod.rs +++ b/src/native/client/mod.rs @@ -488,10 +488,12 @@ impl FipsStream { /// [`bytes_queued`](super::seqpacket::bytes_queued) reports nothing behind /// it. The daemon half applies the identical rule, on the same pair. /// - /// One case survives both checks: a zero-length datagram that is the last - /// message before the close is indistinguishable from the close, because - /// reading it drains the queue and a zero-length message contributes no - /// bytes. Do not give a zero-length payload a meaning of its own. + /// One case survives both checks on Linux, where the pair is + /// `SOCK_SEQPACKET`: a zero-length datagram that is the last message before + /// the close is indistinguishable from the close, because reading it drains + /// the queue and a zero-length message contributes no bytes, so it is + /// reported as `EPIPE`. On macOS and FreeBSD it is delivered as `Ok(0)` and + /// the close follows. Do not give a zero-length payload a meaning of its own. /// /// A datagram longer than `buf` is truncated and the remainder discarded, /// which is `SOCK_SEQPACKET` behaviour. Size `buf` at diff --git a/src/native/seqpacket.rs b/src/native/seqpacket.rs index 7e17e2cb..d2537a27 100644 --- a/src/native/seqpacket.rs +++ b/src/native/seqpacket.rs @@ -46,7 +46,11 @@ use tokio::io::unix::AsyncFd; #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Received { /// A datagram of this many bytes. Zero is a legitimate value: a client may - /// send an empty datagram, and that is not the same as closing. + /// send an empty datagram, and while it keeps its half open that is not the + /// same as closing. On Linux, where the pair is `SOCK_SEQPACKET`, an empty + /// datagram that is the last thing a client sends before it closes reads as + /// [`Received::Eof`] instead; `recv_once` says why that one case cannot be + /// told apart. Datagram(usize), /// The peer closed its half of the pair. Eof, @@ -308,14 +312,18 @@ impl Seqpacket { /// message behind it is never taken. /// /// This matters because reading an empty datagram as a close would let a client -/// tear down its own flow by sending nothing, and the defect would present as a -/// spurious disconnect. +/// tear down its own flow by sending nothing while it is still connected, and the +/// defect would present as a spurious disconnect. /// -/// **One case survives and cannot be fixed here.** A zero-length datagram that -/// is the last message before a close is indistinguishable from the close: -/// reading it drains the queue, and every observation then matches a bare end of -/// file. Separating those needs a payload that is never zero bytes on the wire, -/// which is a protocol change rather than a receive-path one. +/// **One case survives on `SOCK_SEQPACKET`, which is Linux, and cannot be fixed +/// here.** A zero-length datagram that is the last message before a close is +/// indistinguishable from the close: reading it drains the queue, and every +/// observation then matches a bare end of file. The flow ends early and the empty +/// datagram is never delivered. Separating those needs a payload that is never +/// zero bytes on the wire, which is a protocol change rather than a receive-path +/// one. On `SOCK_DGRAM`, which macOS and FreeBSD use, the empty datagram is +/// delivered and the close is reported by the read after it; the tests below +/// pin both behaviours. /// /// **`ECONNRESET` is treated as end of file too**, for the platform whose /// datagram sockets report a close that way rather than through `POLLHUP`. Both @@ -671,6 +679,45 @@ mod tests { assert_eq!(recv_bounded(&daemon, &mut buf).await, Received::Eof); } + #[tokio::test] + async fn a_trailing_empty_datagram_before_a_close_reads_as_the_close_only_on_seqpacket() { + // The case recv_once documents as unfixable, pinned per socket type so + // the documentation cannot drift from what the kernel does. On + // SOCK_SEQPACKET the empty datagram is the last thing queued when the + // client closes, so reading it drains the queue with POLLHUP latched + // and it is indistinguishable from the close. On SOCK_DGRAM it is + // delivered and the close is reported by the read after it. The branch + // is on SOCK_TYPE rather than on the OS, because the socket type is the + // property the limitation depends on. + let (daemon, theirs) = pair().unwrap(); + let daemon = Seqpacket::new(daemon).unwrap(); + let theirs = client(theirs); + + // SAFETY: the descriptor is open and owned by `theirs`. + let sent = unsafe { libc::send(theirs.as_raw_fd(), std::ptr::null(), 0, 0) }; + assert_eq!(sent, 0, "{}", io::Error::last_os_error()); + drop(theirs); + + let mut buf = [0u8; 64]; + if SOCK_TYPE == libc::SOCK_SEQPACKET { + assert_eq!( + recv_bounded(&daemon, &mut buf).await, + Received::Eof, + "a trailing empty datagram on SOCK_SEQPACKET was delivered rather than \ + read as the close; the documented limitation no longer holds on this \ + kernel, so re-measure it and correct the docs that describe it" + ); + } else { + assert_eq!( + recv_bounded(&daemon, &mut buf).await, + Received::Datagram(0), + "a trailing empty datagram on SOCK_DGRAM was not delivered; the docs \ + say only SOCK_SEQPACKET platforms lose it" + ); + assert_eq!(recv_bounded(&daemon, &mut buf).await, Received::Eof); + } + } + #[test] fn both_halves_of_a_pair_are_close_on_exec() { // Asserted rather than assumed because the platforms disagree on how it diff --git a/src/node/bloom.rs b/src/node/bloom.rs index e85ea4c9..48c08415 100644 --- a/src/node/bloom.rs +++ b/src/node/bloom.rs @@ -6,7 +6,7 @@ use crate::NodeAddr; use crate::proto::bloom::BloomFilter; use crate::proto::bloom::FilterAnnounce; -use crate::proto::bloom::{LinkEvidence, RrCounters}; +use crate::proto::mmp::delivery::{LinkEvidence, RrCounters}; use super::reject::BloomReject; use super::{Node, NodeError}; @@ -288,8 +288,9 @@ impl Node { /// Read what `peer_addr`'s link shows about delivery of our frames: the /// session's identity and next send counter, and the last ReceiverReport - /// accepted on it. `None` when the peer has no session. - fn link_evidence(&self, peer_addr: &NodeAddr) -> Option { + /// accepted on it. `None` when the peer has no session. Both the filter + /// and the tree announce resends read it. + pub(super) fn link_evidence(&self, peer_addr: &NodeAddr) -> Option { let peer = self.peers.get(peer_addr)?; let session = peer.noise_session()?; let mut epoch = [0u8; 8]; diff --git a/src/node/tests/bloom.rs b/src/node/tests/bloom.rs index 978b57cd..70104193 100644 --- a/src/node/tests/bloom.rs +++ b/src/node/tests/bloom.rs @@ -1013,17 +1013,24 @@ fn zero_intervals(node: &mut Node) { } } +/// One MMP exchange between nodes `a` and `b`: `a` reports, everyone +/// processes, `b` reports, everyone processes. No other node is asked to +/// report. +async fn mmp_between(nodes: &mut [TestNode], a: usize, b: usize) { + zero_intervals(&mut nodes[a].node); + zero_intervals(&mut nodes[b].node); + nodes[a].node.check_mmp_reports().await; + process_available_packets(nodes).await; + zero_intervals(&mut nodes[a].node); + zero_intervals(&mut nodes[b].node); + nodes[b].node.check_mmp_reports().await; + process_available_packets(nodes).await; +} + /// One MMP exchange between M and P: M reports, P processes, P reports, M /// processes. C is never asked to report. async fn mmp_round(nodes: &mut [TestNode]) { - zero_intervals(&mut nodes[M].node); - zero_intervals(&mut nodes[P].node); - nodes[M].node.check_mmp_reports().await; - process_available_packets(nodes).await; - zero_intervals(&mut nodes[M].node); - zero_intervals(&mut nodes[P].node); - nodes[P].node.check_mmp_reports().await; - process_available_packets(nodes).await; + mmp_between(nodes, M, P).await; } /// Process packets on every node until a pass handles none, at most 50 passes. @@ -1050,15 +1057,22 @@ async fn drop_queued(tn: &mut TestNode) -> usize { dropped } -/// ReceiverReports M has seen from P, including stale and duplicate ones. -fn reports_seen(fx: &FlipFixture) -> u64 { - fx.nodes[M] +/// ReceiverReports node `a` has seen from node `b`, including stale and +/// duplicate ones. +fn seen_by(nodes: &[TestNode], a: usize, b: usize) -> u64 { + let remote = *nodes[b].node.node_addr(); + nodes[a] .node - .get_peer(&fx.p) + .get_peer(&remote) .and_then(|peer| peer.mmp()) .map_or(0, |mmp| mmp.metrics.reports_seen()) } +/// ReceiverReports M has seen from P, including stale and duplicate ones. +fn reports_seen(fx: &FlipFixture) -> u64 { + seen_by(&fx.nodes, M, P) +} + /// Whether the filter P stores for M contains the marker. fn holds_marker(fx: &FlipFixture) -> bool { fx.nodes[P] @@ -1068,40 +1082,52 @@ fn holds_marker(fx: &FlipFixture) -> bool { .is_some_and(|filter| filter.contains(&marker())) } -/// Parent-switch counts at M and P before a run of MMP rounds. +/// Parent-switch counts at the two ends of a link before a run of MMP +/// rounds. /// /// A first RTT sample can re-evaluate the parent, and a switch marks every /// peer, which would pass a resend test for a reason unrelated to the resend. struct SwitchGuard { - m: u64, - p: u64, + a: u64, + b: u64, +} + +/// Snapshot the parent-switch counters of nodes `a` and `b`. +fn guard_of(nodes: &[TestNode], a: usize, b: usize) -> SwitchGuard { + SwitchGuard { + a: nodes[a].node.metrics().tree.parent_switches.get(), + b: nodes[b].node.metrics().tree.parent_switches.get(), + } +} + +/// Assert neither node `a` nor node `b` switched parent since `guard`, and +/// `a`'s parent is still `parent`. +fn assert_steady(nodes: &[TestNode], a: usize, b: usize, guard: &SwitchGuard, parent: &NodeAddr) { + assert_eq!( + nodes[a].node.metrics().tree.parent_switches.get(), + guard.a, + "setup: node {a} must not switch parent during the MMP rounds" + ); + assert_eq!( + nodes[b].node.metrics().tree.parent_switches.get(), + guard.b, + "setup: node {b} must not switch parent during the MMP rounds" + ); + assert_eq!( + nodes[a].node.tree_state().my_declaration().parent_id(), + parent, + "setup: node {a}'s parent must not change" + ); } /// Snapshot M's and P's parent-switch counters. fn switch_guard(fx: &FlipFixture) -> SwitchGuard { - SwitchGuard { - m: fx.nodes[M].node.metrics().tree.parent_switches.get(), - p: fx.nodes[P].node.metrics().tree.parent_switches.get(), - } + guard_of(&fx.nodes, M, P) } /// Assert neither M nor P switched parent since `guard`, and M's parent is P. fn assert_unswitched(fx: &FlipFixture, guard: &SwitchGuard) { - assert_eq!( - fx.nodes[M].node.metrics().tree.parent_switches.get(), - guard.m, - "setup: M must not switch parent during the MMP rounds" - ); - assert_eq!( - fx.nodes[P].node.metrics().tree.parent_switches.get(), - guard.p, - "setup: P must not switch parent during the MMP rounds" - ); - assert_eq!( - fx.nodes[M].node.tree_state().my_declaration().parent_id(), - &fx.p, - "setup: M's parent must still be P" - ); + assert_steady(&fx.nodes, M, P, guard, &fx.p); } /// FilterAnnounces M has sent. @@ -1109,16 +1135,25 @@ fn sent_count(fx: &FlipFixture) -> u64 { fx.nodes[M].node.metrics().bloom.sent.get() } -/// Drain the fixture, then run MMP rounds until M has seen a report from P. -async fn start_reports(fx: &mut FlipFixture) { - drain_quiet(&mut fx.nodes).await; +/// Drain `nodes`, then run MMP exchanges between nodes `a` and `b` until `a` +/// has seen a report from `b`. +async fn await_report(nodes: &mut [TestNode], a: usize, b: usize) { + drain_quiet(nodes).await; for _ in 0..10 { - if reports_seen(fx) >= 1 { + if seen_by(nodes, a, b) >= 1 { break; } - mmp_round(&mut fx.nodes).await; + mmp_between(nodes, a, b).await; } - assert!(reports_seen(fx) >= 1, "setup: P must report to M"); + assert!( + seen_by(nodes, a, b) >= 1, + "setup: node {b} must report to node {a}" + ); +} + +/// Drain the fixture, then run MMP rounds until M has seen a report from P. +async fn start_reports(fx: &mut FlipFixture) { + await_report(&mut fx.nodes, M, P).await; } /// Deliver C's filter carrying the marker and send M's announce of it to P. @@ -1159,19 +1194,26 @@ async fn lose_marker(fx: &mut FlipFixture) -> u64 { sent } -/// The first eight bytes of the handshake hash of M's current session with P. -fn link_epoch(fx: &FlipFixture) -> [u8; 8] { - let hash = fx.nodes[M] +/// The first eight bytes of the handshake hash of node `a`'s current session +/// with node `b`. +fn epoch_of(nodes: &[TestNode], a: usize, b: usize) -> [u8; 8] { + let remote = *nodes[b].node.node_addr(); + let hash = nodes[a] .node - .get_peer(&fx.p) + .get_peer(&remote) .and_then(|peer| peer.noise_session()) - .expect("M has a session with P") + .expect("setup: the link has a session") .handshake_hash(); let mut epoch = [0u8; 8]; epoch.copy_from_slice(&hash[..8]); epoch } +/// The first eight bytes of the handshake hash of M's current session with P. +fn link_epoch(fx: &FlipFixture) -> [u8; 8] { + epoch_of(&fx.nodes, M, P) +} + /// A FilterAnnounce lost in transit is resent once a receiver report shows /// the loss, so the peer ends up holding the filter. #[tokio::test] @@ -1221,52 +1263,76 @@ async fn test_bloom_unchanged_filter_with_newer_sequence_marks_no_peer() { cleanup_nodes(&mut fx.nodes).await; } -/// Make M rekey on its next check: one message on a session is enough, time -/// never triggers it, and both ends of M's links are aged past the -/// responder's rekey-acceptance gate so both rekeys are ordinary ones. -fn arm_rekey(fx: &mut FlipFixture) { - fx.nodes[M].node.replace_context(|ctx| { +/// Make node `a` rekey on its next check: one message on a session is +/// enough, time never triggers it, and both ends of every link of `a` are +/// aged past the responder's rekey-acceptance gate so every rekey is an +/// ordinary one. +fn arm_rekeys(nodes: &mut [TestNode], a: usize) { + nodes[a].node.replace_context(|ctx| { let mut cfg = (*ctx.config).clone(); cfg.node.rekey.enabled = true; cfg.node.rekey.after_messages = 1; cfg.node.rekey.after_secs = u64::MAX; ctx.config = std::sync::Arc::new(cfg); }); - let (m, p, c) = (fx.m, fx.p, fx.c); + let local = *nodes[a].node.node_addr(); + let remotes: Vec = nodes[a].node.peers.keys().copied().collect(); let age = Duration::from_secs(31); - for (i, remote) in [(M, p), (P, m), (M, c), (C, m)] { - fx.nodes[i] - .node - .get_peer_mut(&remote) - .expect("setup: link peer present") - .test_backdate_session_established(age); + for remote in remotes { + let b = nodes + .iter() + .position(|tn| *tn.node.node_addr() == remote) + .expect("setup: every peer is a test node"); + for (i, addr) in [(a, remote), (b, local)] { + nodes[i] + .node + .get_peer_mut(&addr) + .expect("setup: link peer present") + .test_backdate_session_established(age); + } } } +/// Make M rekey on its next check, with both ends of both of M's links aged +/// past the responder's rekey-acceptance gate. +fn arm_rekey(fx: &mut FlipFixture) { + arm_rekeys(&mut fx.nodes, M); +} + +/// Drive the real rekey handshake until node `a`'s session with node `b` is +/// cut over. +async fn cutover(nodes: &mut [TestNode], a: usize, b: usize) { + let before = epoch_of(nodes, a, b); + for _ in 0..6 { + nodes[a].node.check_rekey().await; + nodes[b].node.check_rekey().await; + for _ in 0..3 { + tokio::time::sleep(Duration::from_millis(5)).await; + process_available_packets(nodes).await; + } + if epoch_of(nodes, a, b) != before { + break; + } + } + assert_ne!( + epoch_of(nodes, a, b), + before, + "setup: node {a}'s link to node {b} must rekey" + ); + let (addr_a, addr_b) = (*nodes[a].node.node_addr(), *nodes[b].node.node_addr()); + assert!( + !nodes[a].node.get_peer(&addr_b).unwrap().rekey_in_progress(), + "setup: node {a}'s rekey with node {b} must be complete" + ); + assert!( + !nodes[b].node.get_peer(&addr_a).unwrap().rekey_in_progress(), + "setup: node {b}'s rekey with node {a} must be complete" + ); +} + /// Drive the real rekey handshake until M's session with P is cut over. async fn rekey_cutover(fx: &mut FlipFixture) { - let before = link_epoch(fx); - for _ in 0..6 { - fx.nodes[M].node.check_rekey().await; - fx.nodes[P].node.check_rekey().await; - for _ in 0..3 { - tokio::time::sleep(Duration::from_millis(5)).await; - process_available_packets(&mut fx.nodes).await; - } - if link_epoch(fx) != before { - break; - } - } - assert_ne!(link_epoch(fx), before, "setup: M's link to P must rekey"); - let (m, p) = (fx.m, fx.p); - assert!( - !fx.nodes[M].node.get_peer(&p).unwrap().rekey_in_progress(), - "setup: M's rekey with P must be complete" - ); - assert!( - !fx.nodes[P].node.get_peer(&m).unwrap().rekey_in_progress(), - "setup: P's rekey with M must be complete" - ); + cutover(&mut fx.nodes, M, P).await; } /// An announce lost just before a link rekey is resent on the new session. @@ -1574,3 +1640,463 @@ async fn test_bloom_reports_from_the_previous_session_do_not_trigger_resends() { ); cleanup_nodes(&mut fx.nodes).await; } + +// ===== Resend of a tree announce the peer did not receive ===== +// +// Tree announces are confirmed and resent by the same receiver-report check +// as filter announces, so their node tests share this file's MMP, rekey and +// loss helpers. They run on real converged state only: every role is read +// from the converged tree, and nothing in any node's tree state is forged. +// Final assertions read the sequence the receiver stores for the sender, +// never what the sender believes it sent. + +/// A converged line of loopback nodes with a sender S that is not root and +/// its parent R. +/// +/// In a line every neighbour of S other than its parent is its child, whose +/// ancestry contains S, so S has no alternative parent and cannot switch. +struct TreeLine { + nodes: Vec, + /// Index of the sender. + s: usize, + /// Index of the sender's parent. + r: usize, +} + +/// Index of the node whose address is `addr`. +fn index_of(nodes: &[TestNode], addr: &NodeAddr) -> usize { + nodes + .iter() + .position(|tn| tn.node.node_addr() == addr) + .expect("setup: address belongs to a test node") +} + +/// Converge a line of `n` nodes (2 or 4) and pick S and R from the result. +/// +/// With 4 nodes, S is the first of indices 1 and 2 that is not root; with 2, +/// S is the node that is not root. R is S's parent. Every node's per-peer +/// tree announce rate limit is set to 0 and whatever convergence left +/// pending is flushed, so a resend is never held back by the rate limit. +async fn tree_line(n: usize) -> TreeLine { + let edges: Vec<(usize, usize)> = (1..n).map(|i| (i - 1, i)).collect(); + let mut nodes = run_tree_test(n, &edges, false).await; + + let root = (0..n) + .find(|&i| nodes[i].node.tree_state().is_root()) + .expect("setup: the line must have a root"); + let candidates: &[usize] = if n == 2 { &[0, 1] } else { &[1, 2] }; + let s = *candidates + .iter() + .find(|&&i| !nodes[i].node.tree_state().is_root()) + .expect("setup: one candidate sender is not root"); + let s_addr = *nodes[s].node.node_addr(); + let r = index_of( + &nodes, + nodes[s].node.tree_state().my_declaration().parent_id(), + ); + let neighbours: Vec = [s.checked_sub(1), Some(s + 1)] + .into_iter() + .flatten() + .filter(|&i| i < n) + .collect(); + eprintln!( + "tree_line({n}): root index {root}, S {s}, R {r}, shape: {}", + if r == root { + "R is root" + } else { + "R is not root" + } + ); + + assert!( + !nodes[s].node.tree_state().is_root(), + "setup: S is not root" + ); + assert_eq!( + nodes[s].node.peers.len(), + neighbours.len(), + "setup: S has exactly its line neighbours as peers" + ); + assert!(neighbours.contains(&r), "setup: R is S's neighbour"); + for &c in neighbours.iter().filter(|&&i| i != r) { + assert_eq!( + nodes[c].node.tree_state().my_declaration().parent_id(), + &s_addr, + "setup: S's other neighbour declares S as its parent" + ); + } + + for tn in nodes.iter_mut() { + for peer in tn.node.peers.values_mut() { + peer.set_tree_announce_min_interval_ms(0); + } + tn.node.send_pending_tree_announces().await; + } + drain_quiet(&mut nodes).await; + for (i, tn) in nodes.iter().enumerate() { + for peer in tn.node.peers.values() { + assert!( + !peer.has_pending_tree_announce(), + "setup: node {i} must have no pending tree announce" + ); + } + } + TreeLine { nodes, s, r } +} + +/// Turn off the periodic parent re-evaluation on `node`, so its periodic +/// re-broadcast cannot be what delivers a lost announce. +fn no_reeval(node: &mut Node) { + node.replace_context(|ctx| { + let mut cfg = (*ctx.config).clone(); + cfg.node.tree.reeval_interval_secs = 0; + ctx.config = std::sync::Arc::new(cfg); + }); +} + +/// Hold off S's fallback resend. S's announces to a child that never +/// reports stay outstanding, and on a loaded host a test running past the +/// fallback would add an unchecked resend to the child and break the exact +/// send counts. +fn hold_fallback(line: &mut TreeLine) { + line.nodes[line.s] + .node + .tree_state_mut() + .set_fallback(u64::MAX); +} + +/// Whether S's announce to R still awaits confirmation. +fn parent_outstanding(line: &TreeLine) -> bool { + let parent = *line.nodes[line.r].node.node_addr(); + line.nodes[line.s] + .node + .tree_state() + .announce_outstanding(&parent) +} + +/// TreeAnnounces node `i` has sent. +fn tree_sent(nodes: &[TestNode], i: usize) -> u64 { + nodes[i].node.metrics().tree.sent.get() +} + +/// The declaration sequence node `r` stores for node `s`. +fn held_seq(nodes: &[TestNode], r: usize, s: usize) -> Option { + let sender = *nodes[s].node.node_addr(); + nodes[r] + .node + .tree_state() + .peer_declaration(&sender) + .map(|decl| decl.sequence()) +} + +/// Give S a new declaration sequence with the same parent, signed, as a +/// position change would. Returns the new sequence. +fn bump(line: &mut TreeLine) -> u64 { + let (s, r) = (line.s, line.r); + let parent = *line.nodes[r].node.node_addr(); + let identity = line.nodes[s].node.identity().clone(); + let ts = line.nodes[s].node.tree_state_mut(); + let seq = ts.my_declaration().sequence() + 1; + let timestamp = ts.my_declaration().timestamp() + 1; + ts.set_parent(parent, seq, timestamp, crate::time::mono_ms()); + ts.recompute_coords(); + sign_declaration(ts.my_declaration_mut(), &identity).unwrap(); + + let ts = line.nodes[s].node.tree_state(); + assert!(!ts.is_root(), "setup: S is still not root after the bump"); + assert_eq!( + ts.my_declaration().parent_id(), + &parent, + "setup: S's parent is still R after the bump" + ); + assert!( + held_seq(&line.nodes, r, s).is_some_and(|held| seq > held), + "setup: the new sequence is fresher than the one R holds for S" + ); + seq +} + +/// Send S's current announce to R and check exactly one was sent. Returns +/// S's sent count after the send. +async fn send_up(line: &mut TreeLine) -> u64 { + let (s, r) = (line.s, line.r); + let parent = *line.nodes[r].node.node_addr(); + let before = tree_sent(&line.nodes, s); + line.nodes[s] + .node + .send_tree_announce_to_peer(&parent) + .await + .expect("setup: the send must succeed"); + let after = tree_sent(&line.nodes, s); + assert_eq!(after, before + 1, "setup: S must send exactly one announce"); + after +} + +/// Lose the announce just sent to R, and check the loss took. +async fn lose_up(line: &mut TreeLine, old: Option) { + let (s, r) = (line.s, line.r); + assert_eq!( + drop_queued(&mut line.nodes[r]).await, + 1, + "setup: exactly the one announce frame must be lost" + ); + assert_eq!( + held_seq(&line.nodes, r, s), + old, + "control: R must still hold S's old sequence" + ); +} + +/// Bump S's declaration, send it to R and lose it on the way. Returns the new +/// sequence and S's sent count after the send. +async fn lose_bump(line: &mut TreeLine) -> (u64, u64) { + let old = held_seq(&line.nodes, line.r, line.s); + let seq = bump(line); + let sent = send_up(line).await; + lose_up(line, old).await; + (seq, sent) +} + +/// `rounds` rounds of an MMP exchange between S and R followed by S's tree +/// tick, with checks that a report arrived and nothing switched parent. +async fn tree_rounds(line: &mut TreeLine, rounds: usize) { + let (s, r) = (line.s, line.r); + let parent = *line.nodes[r].node.node_addr(); + let seen = seen_by(&line.nodes, s, r); + let guard = guard_of(&line.nodes, s, r); + for _ in 0..rounds { + mmp_between(&mut line.nodes, s, r).await; + line.nodes[s].node.check_tree_state().await; + process_available_packets(&mut line.nodes).await; + } + assert!( + seen_by(&line.nodes, s, r) > seen, + "setup: a receiver report must arrive after the send" + ); + assert_steady(&line.nodes, s, r, &guard, &parent); +} + +/// Whether S has a tree announce pending for R. +fn parent_pending(line: &TreeLine) -> bool { + let parent = *line.nodes[line.r].node.node_addr(); + line.nodes[line.s] + .node + .get_peer(&parent) + .expect("setup: R is S's peer") + .has_pending_tree_announce() +} + +/// A TreeAnnounce lost in transit is resent once a receiver report shows the +/// loss, so the parent ends up holding the new declaration. +#[tokio::test] +async fn test_tree_announce_lost_in_transit_reaches_the_peer_after_a_receiver_report() { + let mut line = tree_line(4).await; + let (s, r) = (line.s, line.r); + await_report(&mut line.nodes, s, r).await; + no_reeval(&mut line.nodes[s].node); + hold_fallback(&mut line); + + let (seq, sent) = lose_bump(&mut line).await; + tree_rounds(&mut line, 3).await; + + assert_eq!( + tree_sent(&line.nodes, s), + sent + 1, + "S must resend to R exactly once after the loss" + ); + assert!( + !parent_pending(&line), + "the rate limit must not be holding the resend" + ); + assert_eq!( + held_seq(&line.nodes, r, s), + Some(seq), + "R must hold the declaration whose announce was lost" + ); + cleanup_nodes(&mut line.nodes).await; +} + +/// A node with a single peer has no periodic re-broadcast, so without a +/// resend a lost announce is never recovered. +#[tokio::test] +async fn test_tree_announce_lost_by_a_node_with_one_peer_reaches_the_peer_after_a_receiver_report() +{ + let mut line = tree_line(2).await; + let (s, r) = (line.s, line.r); + await_report(&mut line.nodes, s, r).await; + + let (seq, _) = lose_bump(&mut line).await; + tree_rounds(&mut line, 3).await; + + assert_eq!( + held_seq(&line.nodes, r, s), + Some(seq), + "R must hold the declaration whose announce was lost" + ); + cleanup_nodes(&mut line.nodes).await; +} + +/// An announce lost just before a link rekey is resent on the new session. +#[tokio::test] +async fn test_tree_announce_lost_before_a_link_rekey_reaches_the_peer_after_the_cutover() { + let mut line = tree_line(4).await; + let (s, r) = (line.s, line.r); + await_report(&mut line.nodes, s, r).await; + no_reeval(&mut line.nodes[s].node); + hold_fallback(&mut line); + + let (seq, sent) = lose_bump(&mut line).await; + arm_rekeys(&mut line.nodes, s); + cutover(&mut line.nodes, s, r).await; + tree_rounds(&mut line, 5).await; + + assert_eq!( + tree_sent(&line.nodes, s), + sent + 1, + "S must resend to R exactly once after the loss" + ); + assert_eq!( + held_seq(&line.nodes, r, s), + Some(seq), + "R must hold the declaration whose announce was lost before the rekey" + ); + assert!( + !parent_outstanding(&line), + "S's resend on the new session must be confirmed" + ); + cleanup_nodes(&mut line.nodes).await; +} + +/// An announce that arrives is confirmed from the receiver reports and never +/// resent. +#[tokio::test] +async fn test_tree_delivered_announce_is_confirmed_without_a_resend() { + let mut line = tree_line(4).await; + let (s, r) = (line.s, line.r); + await_report(&mut line.nodes, s, r).await; + no_reeval(&mut line.nodes[s].node); + hold_fallback(&mut line); + + let seq = bump(&mut line); + let sent = send_up(&mut line).await; + assert!( + parent_outstanding(&line), + "control: the tracker must hold the announce straight after the send" + ); + process_available_packets(&mut line.nodes).await; + assert_eq!( + held_seq(&line.nodes, r, s), + Some(seq), + "control: R must hold the announce" + ); + tree_rounds(&mut line, 3).await; + + assert_eq!( + tree_sent(&line.nodes, s), + sent, + "S must not resend a delivered announce" + ); + assert!(!parent_pending(&line), "R must not be marked"); + assert!( + !parent_outstanding(&line), + "the delivered announce must be confirmed" + ); + cleanup_nodes(&mut line.nodes).await; +} + +/// On a converged mesh with clean links, every tree announce is confirmed +/// from the receiver reports and none is resent. The convergence announces +/// go out before any report, so this checks the zero baseline on real +/// counters. +#[tokio::test] +async fn test_tree_clean_links_confirm_every_announce_without_a_resend() { + let mut nodes = run_tree_test(3, &[(0, 1), (1, 2)], false).await; + for tn in nodes.iter_mut() { + for peer in tn.node.peers.values_mut() { + peer.set_tree_announce_min_interval_ms(0); + } + tn.node.send_pending_tree_announces().await; + no_reeval(&mut tn.node); + } + drain_quiet(&mut nodes).await; + for (i, tn) in nodes.iter().enumerate() { + for peer in tn.node.peers.values() { + assert!( + !peer.has_pending_tree_announce(), + "setup: node {i} must have no pending tree announce" + ); + } + } + let snapshot = |nodes: &[TestNode]| -> Vec<(u64, u64, NodeAddr)> { + nodes + .iter() + .map(|tn| { + ( + tn.node.metrics().tree.sent.get(), + tn.node.metrics().tree.parent_switches.get(), + *tn.node.tree_state().my_declaration().parent_id(), + ) + }) + .collect() + }; + let before = snapshot(&nodes); + assert!( + (0..nodes.len()).all(|i| tree_sent(&nodes, i) > 0), + "setup: every node must have sent tree announces while converging" + ); + + for _ in 0..5 { + for i in 0..nodes.len() { + zero_intervals(&mut nodes[i].node); + nodes[i].node.check_mmp_reports().await; + process_available_packets(&mut nodes).await; + } + for tn in nodes.iter_mut() { + tn.node.check_tree_state().await; + } + process_available_packets(&mut nodes).await; + } + + assert_eq!( + snapshot(&nodes), + before, + "no node may send a tree announce, switch parent or change parent" + ); + for (i, tn) in nodes.iter().enumerate() { + for peer in tn.node.peers.keys() { + assert!( + !tn.node.tree_state().announce_outstanding(peer), + "node {i} must have confirmed its tree announce to every peer" + ); + } + } + cleanup_nodes(&mut nodes).await; +} + +/// With no receiver report at all, a lost tree announce is still resent once +/// the fallback interval passes. +#[tokio::test] +async fn test_tree_lost_announce_is_resent_after_the_fallback_when_no_receiver_report_arrives() { + let mut line = tree_line(4).await; + let (s, r) = (line.s, line.r); + drain_quiet(&mut line.nodes).await; + no_reeval(&mut line.nodes[s].node); + line.nodes[s].node.tree_state_mut().set_fallback(0); + let seen = seen_by(&line.nodes, s, r); + + let (seq, _) = lose_bump(&mut line).await; + line.nodes[s].node.check_tree_state().await; + process_available_packets(&mut line.nodes).await; + + assert_eq!( + seen_by(&line.nodes, s, r), + seen, + "setup: R must send no report" + ); + assert_eq!( + held_seq(&line.nodes, r, s), + Some(seq), + "R must hold the declaration once the fallback resends it" + ); + cleanup_nodes(&mut line.nodes).await; +} diff --git a/src/node/tree.rs b/src/node/tree.rs index 100e28d8..21dadad6 100644 --- a/src/node/tree.rs +++ b/src/node/tree.rs @@ -139,7 +139,26 @@ impl Node { peer.record_tree_announce_sent(now_ms); } - trace!(peer = %self.peer_display_name(peer_addr), "Sent TreeAnnounce"); + // Keep the announce outstanding until the receiver reports confirm + // it. Read after the send: anything else taking a counter in between + // only makes the recorded counter higher, which delays confirmation + // rather than confirming a frame that was never covered. + let seq = announce.declaration.sequence(); + if let Some(link) = self.link_evidence(peer_addr) { + self.tree_state.record_announce( + *peer_addr, + seq, + link.next_counter.saturating_sub(1), + &link, + crate::time::mono_ms(), + ); + } + + trace!( + peer = %self.peer_display_name(peer_addr), + seq = seq, + "Sent TreeAnnounce" + ); Ok(()) } @@ -584,13 +603,52 @@ impl Node { /// Periodic tree maintenance, called from the tick handler. /// - /// Sends pending rate-limited announces and checks for periodic - /// parent re-evaluation based on current MMP link costs. + /// Marks peers whose last announce was not confirmed delivered, sends + /// pending rate-limited announces (so a resend goes out in the same tick + /// through the ordinary send path), and checks for periodic parent + /// re-evaluation based on current MMP link costs. pub(super) async fn check_tree_state(&mut self) { + self.tree_resend(); self.send_pending_tree_announces().await; self.check_periodic_parent_reeval().await; } + /// Mark for resend every peer whose outstanding TreeAnnounce the receiver + /// reports show was lost, or could not confirm in time. + /// + /// The resend is an ordinary announce built from the current declaration, + /// so if our position moved on since the lost one, the peer gets the + /// newer position. + fn tree_resend(&mut self) { + let now_ms = crate::time::mono_ms(); + let waiting: Vec = self + .peers + .keys() + .filter(|addr| self.tree_state.announce_outstanding(addr)) + .copied() + .collect(); + for addr in waiting { + let Some(link) = self.link_evidence(&addr) else { + continue; + }; + let counter = self.tree_state.outstanding_counter(&addr); + let seq = self.tree_state.outstanding_seq(&addr); + let Some(reason) = self.tree_state.check_announce(&addr, &link, now_ms) else { + continue; + }; + if let Some(peer) = self.peers.get_mut(&addr) { + peer.mark_tree_announce_pending(); + } + debug!( + peer = %self.peer_display_name(&addr), + reason = ?reason, + counter = ?counter, + seq = ?seq, + "Resending unconfirmed TreeAnnounce" + ); + } + } + /// Periodic parent re-evaluation based on current MMP link costs. /// /// Self-paces using `last_parent_reeval` and the configured diff --git a/src/packaging_tests.rs b/src/packaging_tests.rs index 7812c7d1..a361af42 100644 --- a/src/packaging_tests.rs +++ b/src/packaging_tests.rs @@ -475,6 +475,54 @@ fn ps_lines(ps1: &str) -> Vec { .collect() } +/// Asserts that line `i` of install-service.ps1's code lines is an `if` whose +/// body is `Write-Error` then `exit 1`, so the condition it tests stops the +/// install rather than only reporting it. +fn refuses_at(lines: &[String], i: usize, what: &str) { + let cond = &lines[i]; + assert!( + cond.starts_with("if (") && cond.ends_with('{'), + "install-service.ps1: the {what} is not an if statement: {cond}" + ); + let body = lines.get(i + 1..i + 3).unwrap_or_default(); + assert!( + body.len() == 2 && body[0].starts_with("Write-Error ") && body[1] == "exit 1", + "install-service.ps1: the {what} is not followed by Write-Error then exit 1: \ + {cond}\n then: {body:?}" + ); +} + +/// Returns the index of the line that closes the block opened on line +/// `start`, found by brace depth. Braces inside single- or double-quoted +/// strings are not counted. +fn block_end(lines: &[String], start: usize) -> usize { + let mut depth = 0i64; + for (i, line) in lines.iter().enumerate().skip(start) { + let mut quote = None; + for c in line.chars() { + match (quote, c) { + (None, '"' | '\'') => quote = Some(c), + (Some(q), _) if c == q => quote = None, + (None, '{') => depth += 1, + (None, '}') => depth -= 1, + _ => {} + } + } + if depth <= 0 { + assert!( + i > start, + "install-service.ps1: no block opens at code line {start}: {}", + lines[start] + ); + return i; + } + } + panic!( + "install-service.ps1: the block at code line {start} never closes: {}", + lines[start] + ) +} + /// Guards the order in which install-service.ps1 secures `C:\ProgramData\fips`. /// /// The directory inherits `C:\ProgramData`'s access, under which any local @@ -664,19 +712,6 @@ fn windows_installer_restricts_config_dir_before_any_path_inside_it() { #[test] fn windows_installer_refusal_conditions_stop_the_install() { let lines = ps_lines(&repo_file("packaging/windows/install-service.ps1")); - let refuses_at = |i: usize, what: &str| { - let cond = &lines[i]; - assert!( - cond.starts_with("if (") && cond.ends_with('{'), - "install-service.ps1: the {what} is not an if statement: {cond}" - ); - let body = lines.get(i + 1..i + 3).unwrap_or_default(); - assert!( - body.len() == 2 && body[0].starts_with("Write-Error ") && body[1] == "exit 1", - "install-service.ps1: the {what} is not followed by Write-Error then exit 1: \ - {cond}\n then: {body:?}" - ); - }; let find = |what: &str, pred: &dyn Fn(&str) -> bool| -> Vec { let found: Vec = lines .iter() @@ -708,7 +743,7 @@ fn windows_installer_refusal_conditions_stop_the_install() { for i in find("owner refusal", &|l| { l == "if ($trustedOwners -notcontains $ownerSid) {" }) { - refuses_at(i, "owner refusal"); + refuses_at(&lines, i, "owner refusal"); } let dir_links = find("refusal of $ConfigDir as a link", &|l| { @@ -722,7 +757,7 @@ fn windows_installer_refusal_conditions_stop_the_install() { dir_links.len() ); for i in dir_links { - refuses_at(i, "refusal of $ConfigDir as a link"); + refuses_at(&lines, i, "refusal of $ConfigDir as a link"); } for i in find("refusal of a link or folder entry", &|l| { @@ -737,7 +772,7 @@ fn windows_installer_refusal_conditions_stop_the_install() { "install-service.ps1: an entry must be refused if it is a link or a folder, \ either one: {cond}" ); - refuses_at(i, "refusal of a link or folder entry"); + refuses_at(&lines, i, "refusal of a link or folder entry"); } } @@ -806,6 +841,163 @@ fn windows_installer_icacls_calls_act_on_links_and_check_exit_codes() { } } +/// Guards the peer ACL files install-service.ps1 creates, and its refusal of +/// legacy ones. +/// +/// Earlier releases read `peers.allow` and `peers.deny` from `\etc\fips` on +/// the system drive, where any local user can create files, and the service +/// still reads a file there when it is missing from `C:\ProgramData\fips`. So +/// for each of the two files the installer must stop when the legacy file +/// exists and the current one does not, which is exactly when the service +/// would enforce the legacy file, and otherwise create the current file empty +/// if it is missing, without truncating one that exists. Every refusal is +/// decided before any file is created, and all of it happens after the config +/// directory is secured and before the binaries are copied or the service is +/// registered, so a refusal leaves an existing install's binaries, config and +/// service as they were. Each check is bounded by its loop's closing brace, +/// since a statement moved out of its loop runs for the last file only. +#[test] +fn windows_installer_creates_empty_peer_acl_files_and_refuses_legacy_ones() { + let lines = ps_lines(&repo_file("packaging/windows/install-service.ps1")); + let all = |pred: &dyn Fn(&str) -> bool| -> Vec { + lines + .iter() + .enumerate() + .filter(|(_, l)| pred(l)) + .map(|(i, _)| i) + .collect() + }; + let first = |what: &str, pred: &dyn Fn(&str) -> bool| -> usize { + all(pred) + .first() + .copied() + .unwrap_or_else(|| panic!("install-service.ps1: no line {what}")) + }; + + let legacy_dir = first("assigning $legacyAclDir", &|l| { + l.starts_with("$legacyAclDir = ") + }); + assert_eq!( + lines[legacy_dir], r#"$legacyAclDir = "$env:SystemDrive\etc\fips""#, + "install-service.ps1: the legacy peer ACL directory is not \\etc\\fips on the \ + system drive" + ); + + let loop_line = r#"foreach ($name in @("peers.allow", "peers.deny")) {"#; + let loops = all(&|l| { + l.starts_with("foreach") && (l.contains("peers.allow") || l.contains("peers.deny")) + }); + assert_eq!( + loops.len(), + 2, + "install-service.ps1: expected two loops over the peer ACL files, the refusal \ + and the creation, found {}", + loops.len() + ); + for &i in &loops { + assert_eq!( + lines[i], loop_line, + "install-service.ps1: a peer ACL loop does not cover both files" + ); + } + let (refusal, creation) = (loops[0], loops[1]); + let (refusal_end, creation_end) = (block_end(&lines, refusal), block_end(&lines, creation)); + let refusal_body = refusal + 1..refusal_end; + let creation_body = creation + 1..creation_end; + + for text in [ + "$legacy = Join-Path $legacyAclDir $name", + r#"$current = "$ConfigDir\$name""#, + ] { + assert!( + lines[refusal_body.clone()].iter().any(|l| l == text), + "install-service.ps1: the refusal loop has no line {text}" + ); + } + let check = refusal_body + .clone() + .find(|&i| lines[i].starts_with("if (") && lines[i].contains("$legacy")) + .unwrap_or_else(|| { + panic!("install-service.ps1: the refusal loop does not test the legacy file") + }); + assert_eq!( + lines[check], + "if ((Test-Path -LiteralPath $legacy) -and -not (Test-Path -LiteralPath $current)) {", + "install-service.ps1: the refusal must hold only when the legacy file exists and \ + the current one does not" + ); + refuses_at(&lines, check, "refusal of a legacy peer ACL file"); + assert!( + block_end(&lines, check) < refusal_end, + "install-service.ps1: the refusal of a legacy peer ACL file does not close inside \ + the refusal loop" + ); + + let aclfile_line = r#"$aclFile = "$ConfigDir\$name""#; + let guard_line = "if (-not (Test-Path -LiteralPath $aclFile)) {"; + let create = creation_body + .clone() + .find(|&i| lines[i].contains("New-Item") && lines[i].contains("-ItemType File")) + .unwrap_or_else(|| panic!("install-service.ps1: the creation loop does not create a file")); + let new_item = &lines[create]; + assert!( + new_item.contains("$aclFile") && !new_item.to_ascii_lowercase().contains("-force"), + "install-service.ps1: the peer ACL file must be created at $aclFile without -Force, \ + which would truncate an existing list: {new_item}" + ); + assert!( + lines[creation + 1..create] + .iter() + .any(|l| l == aclfile_line), + "install-service.ps1: the creation loop does not set $aclFile to the file in $ConfigDir" + ); + let guard = create - 1; + assert!( + lines[guard] == guard_line && block_end(&lines, guard) < creation_end, + "install-service.ps1: the peer ACL file is not created only when it is missing" + ); + for l in &lines[creation_body] { + assert!( + [aclfile_line, guard_line, new_item.as_str(), "}"].contains(&l.as_str()) + || l.starts_with("Write-Host "), + "install-service.ps1: the creation loop does more than create a missing file: {l}" + ); + } + + let is_check = |l: &str| l == "& $refuseEntries"; + let order = [ + ("check after the reset", all(&is_check).get(2).copied()), + ("$legacyAclDir", Some(legacy_dir)), + ("refusal loop", Some(refusal)), + ("refusal of a legacy file", Some(check)), + ("end of the refusal loop", Some(refusal_end)), + ("creation loop", Some(creation)), + ("creation of a peer ACL file", Some(create)), + ("end of the creation loop", Some(creation_end)), + ( + "binary copy", + all(&|l| l.contains("$Binaries")).first().copied(), + ), + ( + "service registration", + all(&|l| l.contains("--install-service")).first().copied(), + ), + ]; + for pair in order.windows(2) { + let [(a, ia), (b, ib)] = pair else { + unreachable!("windows(2) yields pairs") + }; + let (ia, ib) = ( + ia.unwrap_or_else(|| panic!("install-service.ps1: no {a}")), + ib.unwrap_or_else(|| panic!("install-service.ps1: no {b}")), + ); + assert!( + ia < ib, + "install-service.ps1: {a} (code line {ia}) must come before {b} (code line {ib})" + ); + } +} + const COMMON_CONFIG: &str = "packaging/common/fips.yaml"; const OPENWRT_CONFIG: &str = "packaging/openwrt-ipk/files/etc/fips/fips.yaml"; diff --git a/src/proto/bloom/mod.rs b/src/proto/bloom/mod.rs index b58ad251..cf4332e0 100644 --- a/src/proto/bloom/mod.rs +++ b/src/proto/bloom/mod.rs @@ -28,7 +28,7 @@ mod tests; pub use core::BloomFilter; pub use limits::{DEFAULT_FILTER_SIZE_BITS, DEFAULT_HASH_COUNT, V1_SIZE_CLASS}; -pub use state::{BloomState, LinkEvidence, RrCounters}; +pub use state::BloomState; pub use wire::FilterAnnounce; /// Errors related to Bloom filter operations. diff --git a/src/proto/bloom/state.rs b/src/proto/bloom/state.rs index e827c4f9..bdebfb03 100644 --- a/src/proto/bloom/state.rs +++ b/src/proto/bloom/state.rs @@ -4,225 +4,7 @@ use alloc::collections::{BTreeMap, BTreeSet}; use super::BloomFilter; use crate::NodeAddr; - -/// How long an announce the receiver reports cannot check waits before its -/// one unchecked resend, in milliseconds. Equal to the default link dead -/// timeout, so an outage that did not remove the peer has ended by then. -pub const FALLBACK_MS: u64 = 30_000; - -/// The largest gap the per-peer resend backoff imposes, in milliseconds. -pub const MAXGAP_MS: u64 = 60_000; - -/// A run of resends with no resend for this long, in milliseconds, resets the -/// backoff. It must exceed [`MAXGAP_MS`], or a sustained trigger resending at -/// the largest gap would reset its own backoff every time. -pub const QUIET_MS: u64 = 120_000; - -/// Unchecked resends (`Unverified`, `SessionChanged` or `Timeout`) allowed per -/// announce lineage per session. -pub const UNVERIFIED_BUDGET: u8 = 1; - -/// Resends on reported loss allowed per announce lineage per session. -pub const LOSS_BUDGET: u8 = 3; - -/// Highest backoff level. `gap` at this level is already capped at -/// [`MAXGAP_MS`], so a higher level would add nothing; the cap keeps the -/// shift in range. -const MAX_LEVEL: u8 = 7; - -/// The cumulative counters of one ReceiverReport the peer sent about our -/// frames on a link. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct RrCounters { - /// Highest link counter the peer had received from us. - pub highest: u64, - /// Link frames from us the peer had counted, cumulative. - pub received: u64, - /// Of those, frames that arrived below the highest counter, cumulative. - pub reordered: u32, -} - -/// What the shell reads from one peer's link at one moment, for deciding -/// whether an announce reached that peer. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub struct LinkEvidence { - /// Identity of the current link session (from its handshake hash). The - /// send counter restarts at 0 in every session, so counters are compared - /// only within one epoch. - pub epoch: u64, - /// The next send counter the current session will use. - pub next_counter: u64, - /// The last ReceiverReport accepted in the current session, if any. - pub rr: Option, -} - -impl LinkEvidence { - /// The report, when it can describe the current session. - /// - /// The peer cannot have received a counter this session has not used yet, - /// so a report whose highest counter is at or above `next_counter` - /// describes another session. That happens briefly around a rekey, when a - /// report or frame of the old session is counted against the new one, and - /// such a report is no evidence either way. - pub fn usable_rr(&self) -> Option { - self.rr.filter(|rr| rr.highest < self.next_counter) - } -} - -/// Why an announce is being resent, for the shell's log line. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ResendReason { - /// A report covering the announce shows fewer frames arrived since its - /// base than were sent. - Loss, - /// The first usable report already covers an announce it cannot check. - Unverified, - /// The announce was sent on an earlier session and cannot be checked. - SessionChanged, - /// No usable report checked the announce within the fallback interval. - Timeout, -} - -/// What an announce's delivery is measured from. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -enum Base { - /// The last usable report before the announce was sent, and whether it - /// has no holes: every counter up to its highest had arrived. - Counted(RrCounters, bool), - /// Nothing received yet in the peer's first session: every counter from 0 - /// must arrive. - Zero, - /// No base: a later report can become one if it does not yet cover the - /// announce. - Unknown, -} - -/// The one announce to a peer still awaiting confirmation. -#[derive(Clone, Copy, Debug)] -struct SentAnnounce { - /// Link counter the announce was sent with. - counter: u64, - /// Session the counter belongs to (or, once orphaned, the session that - /// orphaned it). - epoch: u64, - /// What delivery is measured from. - base: Base, - /// Sent on an earlier session, so it can never be checked. - orphan: bool, - /// When it was sent or orphaned, for the fallback. - at_ms: u64, -} - -/// Per-peer delivery tracking for filter announces. -#[derive(Clone, Debug)] -struct AckState { - /// Session in which this entry was created; only there does a missing - /// report mean the peer has received nothing yet. - first_epoch: u64, - /// The outstanding announce, if one is unconfirmed. - sent: Option, - /// Backoff level: the number of resends in the current run, capped. - level: u8, - /// When the last resend was triggered. - resent_ms: Option, - /// Session the budgets were last refilled for. - budget_epoch: u64, - /// Unchecked resends left for the current lineage in this session. - unverified_left: u8, - /// Loss resends left for the current lineage in this session. - loss_left: u8, -} - -impl AckState { - /// A fresh entry for a peer first sent to in session `epoch`. - fn new(epoch: u64) -> Self { - Self { - first_epoch: epoch, - sent: None, - level: 0, - resent_ms: None, - budget_epoch: epoch, - unverified_left: UNVERIFIED_BUDGET, - loss_left: LOSS_BUDGET, - } - } - - /// Refill both budgets for session `epoch`. - fn refill(&mut self, epoch: u64) { - self.budget_epoch = epoch; - self.unverified_left = UNVERIFIED_BUDGET; - self.loss_left = LOSS_BUDGET; - } - - /// Whether report `rr`, taken in session `epoch`, shows no holes: every - /// counter up to its highest had arrived. Only in the peer's first session - /// does its cumulative count start at 0, so a later session's report is - /// never known to be whole. - fn whole(&self, epoch: u64, rr: RrCounters) -> bool { - epoch == self.first_epoch && rr.highest.checked_add(1) == Some(rr.received) - } - - /// Whether the backoff allows a resend at `now_ms`, resetting the level - /// after a quiet period. - fn backoff_allows(&mut self, now_ms: u64) -> bool { - let Some(last) = self.resent_ms else { - return true; - }; - if now_ms >= last.saturating_add(QUIET_MS) { - self.level = 0; - } - now_ms >= last.saturating_add(gap(self.level)) - } -} - -/// Minimum time after a resend before the next one, at backoff `level`. -fn gap(level: u8) -> u64 { - match level { - 0 => 0, - n => (1000u64 << (n.min(MAX_LEVEL) - 1)).min(MAXGAP_MS), - } -} - -/// Whether every counter in the report's range since `base` arrived. -/// -/// Within one receiver epoch every frame counted between two reports is a -/// distinct counter at or below `h1`. Those in `(h0, h1]` number at most all -/// receipts, `got`, and at least the non-reorder receipts, `sure`: a frame -/// that arrived after a higher counter is a reorder whether its counter lies -/// inside the window or at or below `h0`. -/// -/// - `got` below the span is a loss: fewer frames arrived than the window -/// holds. -/// - `sure` equal to the span is delivery. -/// - `got` equal to the span is delivery when the base has no holes, since -/// then no counter at or below `h0` is left to arrive late. -/// -/// Anything else is ambiguous and proves nothing, as is an inconsistent pair -/// (a counter went backwards), which means the two reports straddle a -/// receiver reset or another session's frame. Frames reserved but never -/// sent and frames dropped before counting only lower the counts, so a lost -/// frame is confirmed only if the peer overcounts. -fn delivered(base: Base, rr: RrCounters) -> Option { - let (r0, o0, span, complete) = match base { - Base::Counted(b, complete) => ( - b.received, - b.reordered, - rr.highest.checked_sub(b.highest)?, - complete, - ), - Base::Zero => (0, 0, rr.highest.checked_add(1)?, true), - Base::Unknown => return None, - }; - let got = rr.received.checked_sub(r0)?; - let sure = got.checked_sub(u64::from(rr.reordered.checked_sub(o0)?))?; - if got < span { - Some(false) - } else if sure == span || (complete && got == span) { - Some(true) - } else { - None - } -} +use crate::proto::mmp::delivery::{Acks, LinkEvidence, ResendReason}; /// State for managing Bloom filter announcements. /// @@ -245,10 +27,8 @@ pub struct BloomState { sequence: u64, /// Last outgoing filter sent to each peer (for change detection). last_sent_filters: BTreeMap, - /// How long an unchecked announce waits for its fallback resend (ms). - fallback_ms: u64, /// Per-peer delivery tracking for sent announces. - acks: BTreeMap, + acks: Acks, } impl BloomState { @@ -263,8 +43,7 @@ impl BloomState { pending_updates: BTreeSet::new(), sequence: 0, last_sent_filters: BTreeMap::new(), - fallback_ms: FALLBACK_MS, - acks: BTreeMap::new(), + acks: Acks::new(), } } @@ -307,9 +86,10 @@ impl BloomState { } /// Set how long an announce the receiver reports cannot check waits - /// before its fallback resend. Defaults to [`FALLBACK_MS`]. + /// before its fallback resend. Defaults to + /// [`FALLBACK_MS`](crate::proto::mmp::delivery::FALLBACK_MS). pub fn set_fallback(&mut self, ms: u64) { - self.fallback_ms = ms; + self.acks.set_fallback(ms); } /// Add a leaf dependent that we'll include in our filter. @@ -414,28 +194,8 @@ impl BloomState { link: &LinkEvidence, now_ms: u64, ) { - let new_lineage = self.last_sent_filters.get(&peer) != Some(filter); - let ack = self - .acks - .entry(peer) - .or_insert_with(|| AckState::new(link.epoch)); - if new_lineage || link.epoch != ack.budget_epoch { - ack.refill(link.epoch); - } - // With no report yet, the zero baseline holds only in the peer's first - // session: a later session's cumulative count includes earlier ones. - let base = match link.usable_rr() { - Some(rr) if rr.highest < counter => Base::Counted(rr, ack.whole(link.epoch, rr)), - _ if link.rr.is_none() && link.epoch == ack.first_epoch => Base::Zero, - _ => Base::Unknown, - }; - ack.sent = Some(SentAnnounce { - counter, - epoch: link.epoch, - base, - orphan: false, - at_ms: now_ms, - }); + let fresh = self.last_sent_filters.get(&peer) != Some(filter); + self.acks.record(peer, fresh, counter, link, now_ms); } /// Decide whether the outstanding announce to `peer` must be resent. @@ -449,9 +209,11 @@ impl BloomState { /// triggering a resend. A report from another session, or a pair of /// reports that is inconsistent or cannot tell a late frame from before /// the base from one inside the window, is no evidence. Each announce lineage - /// gets [`UNVERIFIED_BUDGET`] unchecked and [`LOSS_BUDGET`] loss resends - /// per session, and a per-peer backoff spaces all resends by 1, 2, 4 ... - /// up to 60 s until [`QUIET_MS`] passes with none. + /// gets [`UNVERIFIED_BUDGET`](crate::proto::mmp::delivery::UNVERIFIED_BUDGET) + /// unchecked and [`LOSS_BUDGET`](crate::proto::mmp::delivery::LOSS_BUDGET) + /// loss resends per session, and a per-peer backoff spaces all resends by + /// 1, 2, 4 ... up to 60 s until + /// [`QUIET_MS`](crate::proto::mmp::delivery::QUIET_MS) passes with none. /// /// On `Some`, the peer has been marked for an update; the ordinary send /// path delivers the resend. @@ -461,69 +223,9 @@ impl BloomState { link: &LinkEvidence, now_ms: u64, ) -> Option { - let fallback_ms = self.fallback_ms; - let ack = self.acks.get_mut(peer)?; - let mut sent = ack.sent?; - - if link.epoch != ack.budget_epoch { - ack.refill(link.epoch); - } - if sent.epoch != link.epoch { - sent.orphan = true; - sent.epoch = link.epoch; - sent.base = Base::Unknown; - sent.at_ms = now_ms; - } - ack.sent = Some(sent); - - let rr = link.usable_rr(); - let due = now_ms >= sent.at_ms.saturating_add(fallback_ms); - let candidate = match (sent.base, rr) { - (Base::Unknown, Some(rr)) => { - if !sent.orphan && rr.highest < sent.counter { - sent.base = Base::Counted(rr, ack.whole(link.epoch, rr)); - ack.sent = Some(sent); - return None; - } - Some(if sent.orphan { - ResendReason::SessionChanged - } else { - ResendReason::Unverified - }) - } - (Base::Unknown, None) => due.then_some(ResendReason::Timeout), - (base, rr) => { - let covering = rr.filter(|rr| rr.highest >= sent.counter); - let loss = match covering.and_then(|rr| delivered(base, rr)) { - Some(true) => { - ack.sent = None; - return None; - } - Some(false) if ack.loss_left > 0 => Some(ResendReason::Loss), - _ => None, - }; - loss.or(due.then_some(ResendReason::Timeout)) - } - }?; - - let loss = candidate == ResendReason::Loss; - let left = if loss { - ack.loss_left - } else { - ack.unverified_left - }; - if left == 0 || !ack.backoff_allows(now_ms) { - return None; - } - if loss { - ack.loss_left -= 1; - } else { - ack.unverified_left -= 1; - } - ack.level = (ack.level + 1).min(MAX_LEVEL); - ack.resent_ms = Some(now_ms); + let reason = self.acks.check(peer, link, now_ms)?; self.mark_update_needed(*peer); - Some(candidate) + Some(reason) } /// Whether an announce to `peer` is still awaiting confirmation. @@ -533,7 +235,7 @@ impl BloomState { /// The link counter of the announce to `peer` awaiting confirmation. pub fn outstanding_counter(&self, peer: &NodeAddr) -> Option { - self.acks.get(peer)?.sent.map(|sent| sent.counter) + self.acks.outstanding(peer) } /// Mark only peers whose outgoing filter has actually changed. diff --git a/src/proto/bloom/tests/state.rs b/src/proto/bloom/tests/state.rs index c5e66e74..9f8c2418 100644 --- a/src/proto/bloom/tests/state.rs +++ b/src/proto/bloom/tests/state.rs @@ -322,7 +322,7 @@ fn test_bloom_state_mark_changed_peers_excludes_source() { // straight after the send. use crate::NodeAddr; -use crate::proto::bloom::state::{ +use crate::proto::mmp::delivery::{ FALLBACK_MS, LOSS_BUDGET, LinkEvidence, QUIET_MS, ResendReason, RrCounters, UNVERIFIED_BUDGET, }; diff --git a/src/proto/mmp/delivery.rs b/src/proto/mmp/delivery.rs new file mode 100644 index 00000000..8adce9b2 --- /dev/null +++ b/src/proto/mmp/delivery.rs @@ -0,0 +1,415 @@ +//! Delivery of announces, inferred from the link's receiver reports. +//! +//! The transport accepting a frame is not delivery: a datagram can still be +//! lost. An announce is therefore held as outstanding until the peer's +//! ReceiverReports show that every link counter since a base, up to and +//! including the announce's own, arrived. The reports carry cumulative +//! counts, so the evidence has an upper and a lower bound: +//! +//! - loss is concluded only when the upper bound, all receipts since the +//! base, falls short of the span of counters; +//! - delivery is concluded only when the lower bound, the receipts that were +//! not reorders, equals the span, or when all receipts equal the span on a +//! base known to have no holes; +//! - the gap between the two bounds is no evidence either way. +//! +//! A report whose highest counter is at or above the session's next send +//! counter describes another session and is no evidence. An announce the +//! reports cannot check gets one unchecked resend; loss and unchecked +//! resends are bounded per announce lineage per session and spaced by a +//! per-peer backoff. +//! +//! Sans-IO: time is injected as `u64` milliseconds and the shell supplies +//! the link counters, so nothing here reads a clock or touches a socket. + +use alloc::collections::BTreeMap; + +use crate::NodeAddr; + +/// How long an announce the receiver reports cannot check waits before its +/// one unchecked resend, in milliseconds. Equal to the default link dead +/// timeout, so an outage that did not remove the peer has ended by then. +pub const FALLBACK_MS: u64 = 30_000; + +/// The largest gap the per-peer resend backoff imposes, in milliseconds. +pub const MAXGAP_MS: u64 = 60_000; + +/// A run of resends with no resend for this long, in milliseconds, resets the +/// backoff. It must exceed [`MAXGAP_MS`], or a sustained trigger resending at +/// the largest gap would reset its own backoff every time. +pub const QUIET_MS: u64 = 120_000; + +/// Unchecked resends (`Unverified`, `SessionChanged` or `Timeout`) allowed per +/// announce lineage per session. +pub const UNVERIFIED_BUDGET: u8 = 1; + +/// Resends on reported loss allowed per announce lineage per session. +pub const LOSS_BUDGET: u8 = 3; + +/// Highest backoff level. `gap` at this level is already capped at +/// [`MAXGAP_MS`], so a higher level would add nothing; the cap keeps the +/// shift in range. +const MAX_LEVEL: u8 = 7; + +/// The cumulative counters of one ReceiverReport the peer sent about our +/// frames on a link. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct RrCounters { + /// Highest link counter the peer had received from us. + pub highest: u64, + /// Link frames from us the peer had counted, cumulative. + pub received: u64, + /// Of those, frames that arrived below the highest counter, cumulative. + pub reordered: u32, +} + +/// What the shell reads from one peer's link at one moment, for deciding +/// whether an announce reached that peer. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct LinkEvidence { + /// Identity of the current link session (from its handshake hash). The + /// send counter restarts at 0 in every session, so counters are compared + /// only within one epoch. + pub epoch: u64, + /// The next send counter the current session will use. + pub next_counter: u64, + /// The last ReceiverReport accepted in the current session, if any. + pub rr: Option, +} + +impl LinkEvidence { + /// The report, when it can describe the current session. + /// + /// The peer cannot have received a counter this session has not used yet, + /// so a report whose highest counter is at or above `next_counter` + /// describes another session. That happens briefly around a rekey, when a + /// report or frame of the old session is counted against the new one, and + /// such a report is no evidence either way. + pub fn usable_rr(&self) -> Option { + self.rr.filter(|rr| rr.highest < self.next_counter) + } +} + +/// Why an announce is being resent, for the shell's log line. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ResendReason { + /// A report covering the announce shows fewer frames arrived since its + /// base than were sent. + Loss, + /// The first usable report already covers an announce it cannot check. + Unverified, + /// The announce was sent on an earlier session and cannot be checked. + SessionChanged, + /// No usable report checked the announce within the fallback interval. + Timeout, +} + +/// What an announce's delivery is measured from. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum Base { + /// The last usable report before the announce was sent, and whether it + /// has no holes: every counter up to its highest had arrived. + Counted(RrCounters, bool), + /// Nothing received yet in the peer's first session: every counter from 0 + /// must arrive. + Zero, + /// No base: a later report can become one if it does not yet cover the + /// announce. + Unknown, +} + +/// The one announce to a peer still awaiting confirmation. +#[derive(Clone, Copy, Debug)] +struct SentAnnounce { + /// Link counter the announce was sent with. + counter: u64, + /// Session the counter belongs to (or, once orphaned, the session that + /// orphaned it). + epoch: u64, + /// What delivery is measured from. + base: Base, + /// Sent on an earlier session, so it can never be checked. + orphan: bool, + /// When it was sent or orphaned, for the fallback. + at_ms: u64, +} + +/// Per-peer delivery tracking for sent announces. +#[derive(Clone, Debug)] +struct AckState { + /// Session in which this entry was created; only there does a missing + /// report mean the peer has received nothing yet. + first_epoch: u64, + /// The outstanding announce, if one is unconfirmed. + sent: Option, + /// Backoff level: the number of resends in the current run, capped. + level: u8, + /// When the last resend was triggered. + resent_ms: Option, + /// Session the budgets were last refilled for. + budget_epoch: u64, + /// Unchecked resends left for the current lineage in this session. + unverified_left: u8, + /// Loss resends left for the current lineage in this session. + loss_left: u8, +} + +impl AckState { + /// A fresh entry for a peer first sent to in session `epoch`. + fn new(epoch: u64) -> Self { + Self { + first_epoch: epoch, + sent: None, + level: 0, + resent_ms: None, + budget_epoch: epoch, + unverified_left: UNVERIFIED_BUDGET, + loss_left: LOSS_BUDGET, + } + } + + /// Refill both budgets for session `epoch`. + fn refill(&mut self, epoch: u64) { + self.budget_epoch = epoch; + self.unverified_left = UNVERIFIED_BUDGET; + self.loss_left = LOSS_BUDGET; + } + + /// Whether report `rr`, taken in session `epoch`, shows no holes: every + /// counter up to its highest had arrived. Only in the peer's first session + /// does its cumulative count start at 0, so a later session's report is + /// never known to be whole. + fn whole(&self, epoch: u64, rr: RrCounters) -> bool { + epoch == self.first_epoch && rr.highest.checked_add(1) == Some(rr.received) + } + + /// Whether the backoff allows a resend at `now_ms`, resetting the level + /// after a quiet period. + fn backoff_allows(&mut self, now_ms: u64) -> bool { + let Some(last) = self.resent_ms else { + return true; + }; + if now_ms >= last.saturating_add(QUIET_MS) { + self.level = 0; + } + now_ms >= last.saturating_add(gap(self.level)) + } +} + +/// Minimum time after a resend before the next one, at backoff `level`. +fn gap(level: u8) -> u64 { + match level { + 0 => 0, + n => (1000u64 << (n.min(MAX_LEVEL) - 1)).min(MAXGAP_MS), + } +} + +/// Whether every counter in the report's range since `base` arrived. +/// +/// Within one receiver epoch every frame counted between two reports is a +/// distinct counter at or below `h1`. Those in `(h0, h1]` number at most all +/// receipts, `got`, and at least the non-reorder receipts, `sure`: a frame +/// that arrived after a higher counter is a reorder whether its counter lies +/// inside the window or at or below `h0`. +/// +/// - `got` below the span is a loss: fewer frames arrived than the window +/// holds. +/// - `sure` equal to the span is delivery. +/// - `got` equal to the span is delivery when the base has no holes, since +/// then no counter at or below `h0` is left to arrive late. +/// +/// Anything else is ambiguous and proves nothing, as is an inconsistent pair +/// (a counter went backwards), which means the two reports straddle a +/// receiver reset or another session's frame. Frames reserved but never +/// sent and frames dropped before counting only lower the counts, so a lost +/// frame is confirmed only if the peer overcounts. +fn delivered(base: Base, rr: RrCounters) -> Option { + let (r0, o0, span, complete) = match base { + Base::Counted(b, complete) => ( + b.received, + b.reordered, + rr.highest.checked_sub(b.highest)?, + complete, + ), + Base::Zero => (0, 0, rr.highest.checked_add(1)?, true), + Base::Unknown => return None, + }; + let got = rr.received.checked_sub(r0)?; + let sure = got.checked_sub(u64::from(rr.reordered.checked_sub(o0)?))?; + if got < span { + Some(false) + } else if sure == span || (complete && got == span) { + Some(true) + } else { + None + } +} + +/// Delivery tracking for one kind of announce, per peer. +/// +/// Each kind of announce keeps its own instance, so the budgets and the +/// backoff of one kind never hold back the other. +#[derive(Clone, Debug)] +pub struct Acks { + /// How long an unchecked announce waits for its fallback resend (ms). + fallback_ms: u64, + /// Per-peer delivery tracking for sent announces. + peers: BTreeMap, +} + +impl Default for Acks { + fn default() -> Self { + Self::new() + } +} + +impl Acks { + /// Tracking with no peer and the default fallback, [`FALLBACK_MS`]. + pub fn new() -> Self { + Self { + fallback_ms: FALLBACK_MS, + peers: BTreeMap::new(), + } + } + + /// Set how long an announce the receiver reports cannot check waits + /// before its fallback resend. + pub fn set_fallback(&mut self, ms: u64) { + self.fallback_ms = ms; + } + + /// Record an announce the transport accepted for `peer`, sent with link + /// counter `counter`, so it stays outstanding until the peer's receiver + /// reports show it arrived. + /// + /// `fresh` says the announce starts a new lineage (new content), which + /// refills its resend budgets. The budgets also refill when the session + /// changes, and never otherwise, so a resend of the same content spends + /// from its lineage's budget. + pub fn record( + &mut self, + peer: NodeAddr, + fresh: bool, + counter: u64, + link: &LinkEvidence, + now_ms: u64, + ) { + let ack = self + .peers + .entry(peer) + .or_insert_with(|| AckState::new(link.epoch)); + if fresh || link.epoch != ack.budget_epoch { + ack.refill(link.epoch); + } + // With no report yet, the zero baseline holds only in the peer's first + // session: a later session's cumulative count includes earlier ones. + let base = match link.usable_rr() { + Some(rr) if rr.highest < counter => Base::Counted(rr, ack.whole(link.epoch, rr)), + _ if link.rr.is_none() && link.epoch == ack.first_epoch => Base::Zero, + _ => Base::Unknown, + }; + ack.sent = Some(SentAnnounce { + counter, + epoch: link.epoch, + base, + orphan: false, + at_ms: now_ms, + }); + } + + /// Decide whether the outstanding announce to `peer` must be resent. + /// + /// Confirms the announce when a usable report covering its counter shows + /// every counter since its base arrived. Resends on a covering report that + /// shows a loss; once, as soon as a usable report arrives, for an announce + /// the reports cannot check; and once after the fallback interval when no + /// usable report checks it. A usable report that does not yet cover an + /// announce sent in the current session becomes its base instead of + /// triggering a resend. A report from another session, or a pair of + /// reports that is inconsistent or cannot tell a late frame from before + /// the base from one inside the window, is no evidence. Each announce lineage + /// gets [`UNVERIFIED_BUDGET`] unchecked and [`LOSS_BUDGET`] loss resends + /// per session, and a per-peer backoff spaces all resends by 1, 2, 4 ... + /// up to 60 s until [`QUIET_MS`] passes with none. + /// + /// On `Some`, the caller must arrange the resend; nothing is marked here. + pub fn check( + &mut self, + peer: &NodeAddr, + link: &LinkEvidence, + now_ms: u64, + ) -> Option { + let fallback_ms = self.fallback_ms; + let ack = self.peers.get_mut(peer)?; + let mut sent = ack.sent?; + + if link.epoch != ack.budget_epoch { + ack.refill(link.epoch); + } + if sent.epoch != link.epoch { + sent.orphan = true; + sent.epoch = link.epoch; + sent.base = Base::Unknown; + sent.at_ms = now_ms; + } + ack.sent = Some(sent); + + let rr = link.usable_rr(); + let due = now_ms >= sent.at_ms.saturating_add(fallback_ms); + let candidate = match (sent.base, rr) { + (Base::Unknown, Some(rr)) => { + if !sent.orphan && rr.highest < sent.counter { + sent.base = Base::Counted(rr, ack.whole(link.epoch, rr)); + ack.sent = Some(sent); + return None; + } + Some(if sent.orphan { + ResendReason::SessionChanged + } else { + ResendReason::Unverified + }) + } + (Base::Unknown, None) => due.then_some(ResendReason::Timeout), + (base, rr) => { + let covering = rr.filter(|rr| rr.highest >= sent.counter); + let loss = match covering.and_then(|rr| delivered(base, rr)) { + Some(true) => { + ack.sent = None; + return None; + } + Some(false) if ack.loss_left > 0 => Some(ResendReason::Loss), + _ => None, + }; + loss.or(due.then_some(ResendReason::Timeout)) + } + }?; + + let loss = candidate == ResendReason::Loss; + let left = if loss { + ack.loss_left + } else { + ack.unverified_left + }; + if left == 0 || !ack.backoff_allows(now_ms) { + return None; + } + if loss { + ack.loss_left -= 1; + } else { + ack.unverified_left -= 1; + } + ack.level = (ack.level + 1).min(MAX_LEVEL); + ack.resent_ms = Some(now_ms); + Some(candidate) + } + + /// The link counter of the announce to `peer` awaiting confirmation. + pub fn outstanding(&self, peer: &NodeAddr) -> Option { + self.peers.get(peer)?.sent.map(|sent| sent.counter) + } + + /// Forget everything about `peer`, which was removed. + pub fn remove(&mut self, peer: &NodeAddr) { + self.peers.remove(peer); + } +} diff --git a/src/proto/mmp/mod.rs b/src/proto/mmp/mod.rs index ded473b3..115f2c71 100644 --- a/src/proto/mmp/mod.rs +++ b/src/proto/mmp/mod.rs @@ -27,6 +27,7 @@ use serde::{Deserialize, Serialize}; mod algorithms; mod core; +pub(crate) mod delivery; mod limits; mod metrics; mod path_mtu; diff --git a/src/proto/stp/state.rs b/src/proto/stp/state.rs index 0c99a665..e84efd46 100644 --- a/src/proto/stp/state.rs +++ b/src/proto/stp/state.rs @@ -7,6 +7,7 @@ use super::core::ParentEval; use super::limits::FlapDampener; use super::{CoordEntry, ParentDeclaration, TreeCoordinate}; use crate::NodeAddr; +use crate::proto::mmp::delivery::{Acks, LinkEvidence, ResendReason}; /// What a parent-loss recovery did: whether the tree state changed, and /// whether the recovery switch was the one that armed a dampening episode. @@ -40,6 +41,10 @@ pub struct TreeState { peer_declarations: BTreeMap, /// Each peer's full ancestry to root. peer_ancestry: BTreeMap, + /// Per-peer delivery tracking for sent TreeAnnounces. + acks: Acks, + /// The declaration sequence last recorded as sent to each peer. + announced: BTreeMap, /// Hysteresis factor for cost-based parent re-selection (0.0-1.0). parent_hysteresis: f64, /// Flap-dampening / hold-down state machine. @@ -64,6 +69,8 @@ impl TreeState { root: my_node_addr, peer_declarations: BTreeMap::new(), peer_ancestry: BTreeMap::new(), + acks: Acks::new(), + announced: BTreeMap::new(), parent_hysteresis: 0.0, flap: FlapDampener::new(), } @@ -149,6 +156,69 @@ impl TreeState { pub fn remove_peer(&mut self, peer_id: &NodeAddr) { self.peer_declarations.remove(peer_id); self.peer_ancestry.remove(peer_id); + self.acks.remove(peer_id); + self.announced.remove(peer_id); + } + + /// Record a TreeAnnounce the transport accepted for `peer`, carrying our + /// declaration sequence `seq` and sent with link counter `counter`, so it + /// stays outstanding until the peer's receiver reports show it arrived. + /// + /// The transport accepting a frame is not delivery. A lost announce would + /// leave the peer on our old tree position until the next announce, which + /// on a node with one peer may never come. The announce's content changes + /// only with the declaration sequence, so a sequence other than the one + /// last recorded for `peer` starts a new lineage with fresh resend + /// budgets, and a resend or periodic re-broadcast of the same declaration + /// spends from its lineage's budget. + pub fn record_announce( + &mut self, + peer: NodeAddr, + seq: u64, + counter: u64, + link: &LinkEvidence, + now_ms: u64, + ) { + let fresh = self.announced.insert(peer, seq) != Some(seq); + self.acks.record(peer, fresh, counter, link, now_ms); + } + + /// Decide whether the outstanding TreeAnnounce to `peer` must be resent, + /// by the shared receiver-report rule ([`Acks::check`]). + /// + /// Marks nothing: on `Some`, the caller marks the peer's announce pending + /// so the ordinary send path delivers the current declaration. + pub fn check_announce( + &mut self, + peer: &NodeAddr, + link: &LinkEvidence, + now_ms: u64, + ) -> Option { + self.acks.check(peer, link, now_ms) + } + + /// Whether a TreeAnnounce to `peer` is still awaiting confirmation. + pub fn announce_outstanding(&self, peer: &NodeAddr) -> bool { + self.outstanding_counter(peer).is_some() + } + + /// The link counter of the TreeAnnounce to `peer` awaiting confirmation. + pub fn outstanding_counter(&self, peer: &NodeAddr) -> Option { + self.acks.outstanding(peer) + } + + /// The declaration sequence of the TreeAnnounce to `peer` awaiting + /// confirmation. + pub fn outstanding_seq(&self, peer: &NodeAddr) -> Option { + self.outstanding_counter(peer)?; + self.announced.get(peer).copied() + } + + /// Set how long a TreeAnnounce the receiver reports cannot check waits + /// before its fallback resend. Defaults to + /// [`FALLBACK_MS`](crate::proto::mmp::delivery::FALLBACK_MS). + pub fn set_fallback(&mut self, ms: u64) { + self.acks.set_fallback(ms); } /// Update this node's parent selection. diff --git a/src/proto/stp/tests/acks.rs b/src/proto/stp/tests/acks.rs new file mode 100644 index 00000000..21be4db7 --- /dev/null +++ b/src/proto/stp/tests/acks.rs @@ -0,0 +1,241 @@ +//! Delivery tracking for sent tree announces. +//! +//! Synthetic milliseconds, counters and receiver reports, no I/O. A report is +//! written `(highest, received, reordered)`. Unless a test says otherwise, a +//! send is recorded with `next_counter = counter + 1`, as the shell reads it +//! straight after the send. The shared rule itself (bases, sessions, budgets, +//! backoff, rekey artifacts) is covered by the filter announce tests; these +//! cover what the tree adds: the declaration sequence as the lineage, and +//! removal with the peer. + +use super::util::make_node_addr; +use crate::NodeAddr; +use crate::proto::mmp::delivery::{FALLBACK_MS, LinkEvidence, ResendReason, RrCounters}; +use crate::proto::stp::TreeState; + +/// First session. +const E1: u64 = 0x0e01; +/// Second session. +const E2: u64 = 0x0e02; + +/// A report's cumulative counters. +fn rr(highest: u64, received: u64, reordered: u32) -> Option { + Some(RrCounters { + highest, + received, + reordered, + }) +} + +/// Link evidence for session `epoch`. +fn link(epoch: u64, next_counter: u64, rr: Option) -> LinkEvidence { + LinkEvidence { + epoch, + next_counter, + rr, + } +} + +/// One peer's tree announces, driven as the shell drives them. +struct Track { + state: TreeState, + peer: NodeAddr, + counter: u64, +} + +impl Track { + /// A tracker with nothing sent yet. + fn new() -> Self { + Self { + state: TreeState::new(make_node_addr(0), 1000), + peer: make_node_addr(1), + counter: 0, + } + } + + /// Record a send of declaration sequence `seq` at `counter`. + fn send(&mut self, seq: u64, counter: u64, link: LinkEvidence, now_ms: u64) { + self.state + .record_announce(self.peer, seq, counter, &link, now_ms); + self.counter = counter; + } + + /// One tick of the tracker. + fn check(&mut self, link: LinkEvidence, now_ms: u64) -> Option { + self.state.check_announce(&self.peer, &link, now_ms) + } + + /// Whether the announce is still unconfirmed. + fn outstanding(&self) -> bool { + self.state.announce_outstanding(&self.peer) + } + + /// Check every 1,000 ms from `from_ms` to `to_ms` inclusive, recording + /// each resend as a send of `seq`, as the shell resends the current + /// declaration. `model` gives the evidence at a time, from the + /// outstanding counter: for a check with `false`, and for recording a + /// resend with `true`, where the resend takes the counter + /// `next_counter - 1` of that evidence. Returns the resends. + fn hold( + &mut self, + seq: u64, + from_ms: u64, + to_ms: u64, + model: impl Fn(u64, bool) -> LinkEvidence, + ) -> Vec<(u64, ResendReason)> { + let mut resends = Vec::new(); + let mut now = from_ms; + while now <= to_ms { + if let Some(reason) = self.check(model(self.counter, false), now) { + resends.push((now, reason)); + let ev = model(self.counter, true); + self.send(seq, ev.next_counter - 1, ev, now); + } + now += 1_000; + } + resends + } +} + +/// Loss on every check: each send is based on a report just below it, and +/// each check sees a report two counters on with one frame missing. +fn lossy(counter: u64, recording: bool) -> LinkEvidence { + if recording { + let n = counter + 1; + link(E1, n + 1, rr(n - 1, n, 0)) + } else { + link(E1, counter + 2, rr(counter + 1, counter + 1, 0)) + } +} + +/// Send sequence 5 into a lossy link and spend its three loss resends. +/// +/// The resends are checked but not recorded, so spending the budget does not +/// depend on the lineage decision; the one record each test then makes is +/// the only lineage decision it observes. The window ends before the +/// backoff would allow a fourth resend (7 s), so the budget limit itself is +/// observed only by the test's final assertion. +fn spend_budget() -> Track { + let mut t = Track::new(); + t.send(5, 12, lossy(11, true), 0); + let mut resends = Vec::new(); + for now in (0..=6_000).step_by(1_000) { + if let Some(reason) = t.check(lossy(12, false), now) { + resends.push((now, reason)); + } + } + assert_eq!( + resends, + vec![ + (0, ResendReason::Loss), + (1_000, ResendReason::Loss), + (3_000, ResendReason::Loss), + ], + "setup: the lineage spends its three loss resends" + ); + assert!( + t.outstanding(), + "setup: the lost announce is still outstanding" + ); + t +} + +/// A new declaration sequence starts a new lineage, whose loss budget is +/// full again. +#[test] +fn test_tree_ack_new_sequence_starts_a_new_lineage() { + let mut t = spend_budget(); + let c = t.counter; + t.send(6, c + 1, lossy(c, true), 10_000); + assert_eq!( + t.check(lossy(t.counter, false), 11_000), + Some(ResendReason::Loss), + "a new sequence must refill the loss budget" + ); +} + +/// The periodic re-broadcast sends the same sequence again, so it stays in +/// the lineage and does not refill the loss budget; the one unchecked +/// resend still comes, 30 s after the re-broadcast. +#[test] +fn test_tree_ack_periodic_resend_of_the_same_sequence_keeps_the_budget() { + let mut t = spend_budget(); + let c = t.counter; + t.send(5, c + 1, lossy(c, true), 10_000); + let resends = t.hold(5, 11_000, 120_000, lossy); + assert_eq!( + resends, + vec![(10_000 + FALLBACK_MS, ResendReason::Timeout)], + "no fourth loss resend, and exactly one unchecked resend" + ); +} + +/// Removing the peer forgets its announce, so the next send starts a fresh +/// entry whose first session measures from counter zero. +#[test] +fn test_tree_ack_removed_peer_starts_fresh() { + // Control: without the removal, an announce in a later session with no + // report has no base, and a first report already covering it cannot + // check it. + let mut t = Track::new(); + t.send(5, 12, link(E1, 13, rr(9, 10, 0)), 0); + t.send(5, 3, link(E2, 4, None), 1_000); + assert_eq!( + t.check(link(E2, 4, rr(3, 4, 0)), 2_000), + Some(ResendReason::Unverified), + "control: the kept entry cannot check the announce" + ); + + let mut t = Track::new(); + t.send(5, 12, link(E1, 13, rr(9, 10, 0)), 0); + t.state.remove_peer(&t.peer); + assert!(!t.outstanding(), "removal must forget the announce"); + t.send(5, 3, link(E2, 4, None), 1_000); + assert_eq!(t.check(link(E2, 4, rr(3, 4, 0)), 2_000), None); + assert!(!t.outstanding(), "the new entry must measure from zero"); +} + +/// In the peer's first session with no report before the send, frames that +/// arrive out of order are still every counter from 0, so the announce +/// confirms. This is the traced shape of a tree announce and a filter +/// announce sent in the same tick arriving swapped. +#[test] +fn test_tree_ack_in_window_reorder_on_the_zero_base_confirms() { + let mut t = Track::new(); + t.send(5, 3, link(E1, 4, None), 0); + // 0..=4 all arrived, two of them after a higher counter. + assert_eq!(t.check(link(E1, 5, rr(4, 5, 2)), 1_000), None); + assert!( + !t.outstanding(), + "every counter arrived, so it must confirm" + ); +} + +/// A covering report whose receipts since the base fall short of the span is +/// a loss, and the announce stays outstanding. +#[test] +fn test_tree_ack_short_upper_bound_is_a_loss() { + let mut t = Track::new(); + // Base (9, 10, 0): all of 0..=9 counted. + t.send(5, 12, link(E1, 13, rr(9, 10, 0)), 0); + // Four of 10..=14 arrived. + assert_eq!( + t.check(link(E1, 15, rr(14, 14, 0)), 1_000), + Some(ResendReason::Loss) + ); + assert!(t.outstanding(), "a lost announce stays outstanding"); +} + +/// With no report ever, the announce gets its one unchecked resend exactly +/// at the default fallback, and none after. +#[test] +fn test_tree_ack_fallback_resends_once_per_lineage() { + let mut t = Track::new(); + t.send(5, 12, link(E1, 13, None), 0); + assert_eq!(t.check(link(E1, 13, None), FALLBACK_MS - 1), None); + assert!(t.outstanding()); + let quiet = |c: u64, recording: bool| link(E1, c + if recording { 2 } else { 1 }, None); + let resends = t.hold(5, FALLBACK_MS, 120_000, quiet); + assert_eq!(resends, vec![(FALLBACK_MS, ResendReason::Timeout)]); + assert_eq!(FALLBACK_MS, 30_000); +} diff --git a/src/proto/stp/tests/mod.rs b/src/proto/stp/tests/mod.rs index 71199998..dc2f9f26 100644 --- a/src/proto/stp/tests/mod.rs +++ b/src/proto/stp/tests/mod.rs @@ -1,5 +1,6 @@ //! STP primitive unit tests. Shared helpers live in `util`. +mod acks; mod coordinate; mod limits; mod state; diff --git a/testing/chaos/scenarios/ethernet-mesh.yaml b/testing/chaos/scenarios/ethernet-mesh.yaml index a980856f..ab48cb5d 100644 --- a/testing/chaos/scenarios/ethernet-mesh.yaml +++ b/testing/chaos/scenarios/ethernet-mesh.yaml @@ -82,6 +82,14 @@ traffic: # happened, not a loss ratio, so a busy host slows it without failing # it. It runs at teardown, after flapped links are restored. # +# Recovery: a tree or bloom announce lost to a flap is resent from the +# link's receiver reports a few seconds after the link returns, or within +# about 30 s when those reports cannot confirm it; only a per-peer backoff +# built up by earlier resends on a lossy link can hold a resend longer, up +# to 60 s after the previous one. The 60 s window therefore no longer sits +# at the ordinary recovery bound, and a red here is investigated, not +# expected. +# # Sizes: payload 0 is the smallest ICMPv6 echo and 1200 is near the # 1280-byte TUN MTU. Neither can reach the receiver's padding trim on a # veth link. Measured on a live run of this scenario (2026-09-19): an diff --git a/testing/check-trailing-log.py b/testing/check-trailing-log.py index 990dc34c..d2592167 100755 --- a/testing/check-trailing-log.py +++ b/testing/check-trailing-log.py @@ -132,7 +132,9 @@ def status_tested(name: str, all_text: str, defining_file: Path) -> list[str]: (rf"^\s*{re.escape(name)}\s*\|\|", " ||"), (rf"^\s*{re.escape(name)}\s+[^\n|&]*&&", " &&"), (rf"^\s*{re.escape(name)}\s*&&", " &&"), - (rf"\bif\s+\S*\s*{re.escape(name)}\b.*;\s*then", "if ... ; then"), + # The lookbehind keeps `\S*` from ending inside a longer name, so + # `if text=$(read_gateway_log ...); then` is not read as a call of `log`. + (rf"\bif\s+\S*\s*(? ; then"), # Command substitution. Found 2026-07-23 while fixing a function this # check reported clean: `total=$(count_log_pattern "$p") || { ... }` # consumes the status, but the line begins with the variable, so none diff --git a/testing/deb-install/test.sh b/testing/deb-install/test.sh index 9b46c0b4..a1419491 100755 --- a/testing/deb-install/test.sh +++ b/testing/deb-install/test.sh @@ -304,6 +304,42 @@ EOF return } +# The installed gateway, on the default config, serves .fips on its default +# listen address. Each check needs the gateway running: a gateway that exited +# fails the first one rather than letting the others pass on nothing. +# Args: , the daemon's npub to resolve through the gateway. +check_gateway_default_listener() { + local name="$1" npub="$2" + local journal="" _i + for _i in $(seq 1 5); do + journal=$(docker exec "$name" journalctl -u fips-gateway.service --no-pager 2>/dev/null) || journal="" + printf '%s\n' "$journal" | grep -q "fips-gateway running" && break + sleep 1 + done + if ! printf '%s\n' "$journal" | grep -q "fips-gateway running"; then + fail "the installed fips-gateway did not reach 'fips-gateway running', so its default listener was not observed" + echo " --- fips-gateway journal ---" + printf '%s\n' "$journal" | tail -15 + return + fi + + local sockets + sockets=$(docker exec "$name" ss -Hulnp 'sport = :5365' 2>/dev/null) || sockets="" + if printf '%s\n' "$sockets" | grep -F '[::1]:5365' | grep -q 'fips-gateway'; then + pass "fips-gateway listens on its default [::1]:5365" + else + fail "no fips-gateway socket on [::1]:5365: '$sockets'" + fi + + local answer + answer=$(docker exec "$name" dig +short +tries=1 +time=3 @::1 -p 5365 AAAA "${npub}.fips" 2>&1) + if printf '%s\n' "$answer" | grep -qE '^fd01::[0-9a-f]{1,4}$'; then + pass "the gateway answers ${npub}.fips on [::1]:5365 from its fd01::/112 pool" + else + fail "the gateway did not answer ${npub}.fips on [::1]:5365 from its pool: '$answer'" + fi +} + # Purge the package with the DNS routing file planted and fips-dns stopped, and # check that postrm removes the file and restarts systemd-resolved. # @@ -709,6 +745,8 @@ DOCKERFILE docker exec "$name" journalctl -u fips-gateway.service --no-pager 2>&1 | tail -15 fi + check_gateway_default_listener "$name" "$npub" + check_purge_clears_dns "$name" "$expected_backend" cleanup_container "$name" diff --git a/testing/dns-resolver/test.sh b/testing/dns-resolver/test.sh index 038a1190..07f72df9 100755 --- a/testing/dns-resolver/test.sh +++ b/testing/dns-resolver/test.sh @@ -701,6 +701,114 @@ DOCKERFILE # prepare_binaries) and are copied into each per-distro runtime image. # ───────────────────────────────────────────────────────────────────── +# Print a fips-gateway log from the container with terminal colour codes +# removed, so structured fields can be matched as plain "key=value" text. +# Fails when the log cannot be read. +read_gateway_log() { + local name="$1" log="$2" + local text + text=$(docker exec "$name" cat "$log" 2>/dev/null) || return 1 + printf '%s\n' "$text" | sed 's/\x1b\[[0-9;]*m//g' + return 0 +} + +# The gateway on its default config binds [::1]:5365 and gets past the +# bind to the NAT step, logging one of the two NAT lines whichever way NAT +# goes in this container. Those are the lines the held-port check requires +# to be absent, so this shows the gateway still emits them in that text. +check_gateway_default_bind() { + local name="$1" + local log=/var/log/fips-gateway.log + local text="" read_ok=0 listening=0 nat_step=0 _i + for _i in $(seq 1 5); do + if text=$(read_gateway_log "$name" "$log"); then + read_ok=1 + if printf '%s\n' "$text" | grep -q 'Gateway DNS resolver listening.*addr=\[::1\]:5365'; then + listening=1 + fi + if printf '%s\n' "$text" | grep -qE 'Created nftables table|Failed to create nftables table'; then + nat_step=1 + break + fi + fi + sleep 1 + done + if [ "$read_ok" = "0" ]; then + fail "could not read $log, so the gateway's default bind was not observed" + return + fi + if [ "$listening" = "1" ]; then + pass "fips-gateway listens on its default [::1]:5365" + else + fail "fips-gateway did not log listening on [::1]:5365" + fi + if [ "$nat_step" = "1" ]; then + pass "fips-gateway gets past the DNS bind to the NAT step" + else + fail "fips-gateway logged neither NAT-step line after the DNS bind" + fi + if [ "$listening" = "0" ] || [ "$nat_step" = "0" ]; then + echo " --- $log ---" + printf '%s\n' "$text" | tail -20 + fi +} + +# The gateway exits at the DNS bind when its listen port is held. The +# daemon in this container holds [::1]:5354, so a gateway configured to +# listen there must exit non-zero with the hint naming the daemon, before +# it creates the address pool or the NAT table. Any build that gets past +# the bind logs one of the pool or NAT lines below, whichever way NAT goes +# in this container, so their absence shows the exit came first. +check_gateway_exits_on_held_port() { + local name="$1" + local log=/var/log/fips-gateway-held.log + local fail_before=$FAIL + docker exec "$name" bash -c 'cat > /tmp/gateway-held.yaml <$log 2>&1; echo \"EXIT=\$?\" >>$log" + + local text + if ! text=$(read_gateway_log "$name" "$log"); then + fail "could not read $log, so the held-port exit was not observed" + return + fi + local rc + rc=$(printf '%s\n' "$text" | sed -n 's/^EXIT=//p' | tail -n 1) + if [ -z "$rc" ]; then + fail "the held-port gateway run left no exit status in $log" + elif [ "$rc" = "0" ] || [ "$rc" = "124" ]; then + fail "fips-gateway on a held DNS port exited $rc (expected non-zero, not the timeout)" + else + pass "fips-gateway on a held DNS port exits $rc" + fi + if printf '%s\n' "$text" | grep -qF "the fips daemon's own DNS responder listens on 5354"; then + pass "the bind error names the daemon as the likely holder of 5354" + else + fail "the bind error does not carry the 5354 hint" + fi + local line + for line in "Failed to create virtual IP pool" "Failed to create nftables table" "Created nftables table"; do + if printf '%s\n' "$text" | grep -qF "$line"; then + fail "fips-gateway reached a step after the DNS bind: '$line'" + else + pass "fips-gateway stopped before '$line'" + fi + done + if [ "$FAIL" -gt "$fail_before" ]; then + echo " --- $log ---" + printf '%s\n' "$text" | tail -20 + fi +} + # Args: # distro_label: short tag for container/image names (e.g. "debian12") # docker_base_image: e.g. "debian:12", "ubuntu:26.04" @@ -913,11 +1021,14 @@ EOF' echo " --- fips-gateway log ---" docker exec "$name" tail -20 /var/log/fips-gateway.log 2>&1 || true fi - # Stop the gateway (it will likely have failed past the upstream - # check on something unrelated in this minimal container — we only - # care that the upstream reachability step succeeded). + check_gateway_default_bind "$name" + + # Stop the gateway (it may have failed after the DNS bind on something + # unrelated in this minimal container). docker exec "$name" pkill -f fips-gateway 2>/dev/null || true + check_gateway_exits_on_held_port "$name" + # Teardown via the script: backend config file must be removed # (path varies by backend selected above). local teardown_path diff --git a/testing/openwrt/fixtures/released-fips.yaml b/testing/openwrt/fixtures/released-fips.yaml new file mode 100644 index 00000000..2f15160d --- /dev/null +++ b/testing/openwrt/fixtures/released-fips.yaml @@ -0,0 +1,190 @@ +# FIPS Node Configuration + +node: + identity: + # By default, a new ephemeral keypair is generated on each start. + # Uncomment persistent to keep the same identity across restarts; + # on first start a keypair is saved to fips.key/fips.pub next to + # this config file (mode 0600/0644). + # persistent: true + # + # Or set an explicit key (overrides persistent): + # nsec: "nsec1..." + # Mesh-lookup protocol (node.lookup.*): the overlay coordinate-lookup engine + # (mesh address -> coordinates). Defaults shown; uncomment to override. + # lookup: + # ttl: 64 + # attempt_timeouts_secs: [1, 2, 4, 8] + # recent_expiry_secs: 10 + # backoff_base_secs: 0 + # backoff_max_secs: 0 + # forward_min_interval_secs: 2 + rendezvous: + # Optional Nostr-mediated overlay endpoint rendezvous. + # nostr: + # enabled: true + # policy: configured_only # disabled | configured_only | open + # open_discovery_max_pending: 64 # caps queued open-rendezvous retries + # app: "fips-overlay-v1" + # advertise: true + # advert_relays: + # - "wss://relay.damus.io" + # - "wss://nos.lol" + # - "wss://offchain.pub" + # dm_relays: + # - "wss://relay.damus.io" + # - "wss://nos.lol" + # - "wss://offchain.pub" + # # Optional override. If omitted, FIPS uses the built-in STUN list. + # # Built-in relay/STUN defaults are best-effort and should be + # # overridden by operators for production use. + # stun_servers: + # - "stun:stun.l.google.com:19302" + # - "stun:stun.cloudflare.com:3478" + # - "stun:global.stun.twilio.com:3478" + + # mDNS/DNS-SD peer rendezvous on the local link. Ships commented (the + # daemon default is off); 'fips-ap-setup' uncomments it when creating + # the access SSID — phone FIPS apps cannot see raw-Ethernet beacons, + # so mDNS is how they find this router's daemon. Daemon-wide switch, + # left enabled on 'fips-ap-setup remove'. + # lan: + # enabled: true + +tun: + enabled: true + name: fips0 + mtu: 1280 + +dns: + enabled: true + # bind_addr defaults to "::1" (IPv6 loopback). The shipped + # fips-dns-setup script configures systemd-resolved with a global + # /etc/systemd/resolved.conf.d/fips.conf drop-in pointing at + # [::1]:5354. + # + # Set "::" to expose the responder to mesh peers as well (e.g. for + # gateway hosts that resolve .fips on behalf of LAN clients). The + # mesh-interface filter in src/upper/dns.rs will still defend + # /etc/fips/hosts aliases from cross-mesh enumeration. + # bind_addr: "::1" + port: 5354 + +transports: + udp: + # Dual-stack wildcard, not "0.0.0.0": access-SSID clients (phones) learn + # this node's addresses from the mDNS advert and prefer the IPv6 + # link-local — a v4-only bind silently drops their Noise msg1. + # OpenWrt is Linux (bindv6only=0), so "[::]" accepts v4 too. + bind_addr: "[::]:2121" + # advertise_on_nostr: true + # public: false # false => advertise udp:nat; true => advertise bound host:port + # accept_connections: true # default; refuse inbound msg1 when false + # outbound_only: false # true => bind ephemeral, no listener on a + # # known port. Forces advertise_on_nostr=false + # # and accept_connections=false. Pure-client + # # posture; bind_addr is ignored. + + tcp: + # Accepts inbound connections. No static outbound peers. + bind_addr: "0.0.0.0:8443" + # advertise_on_nostr: true + + # Ethernet transport — physical port names, NOT bridge names. + # Run 'ip link show' on the router to identify port names. + ethernet: + wan: + interface: "eth0" + listen: true + announce: true + auto_connect: true + accept_connections: true + wwan: + interface: "phy0-sta0" + listen: true + announce: true + auto_connect: true + accept_connections: true + lan: + interface: "br-lan" + listen: true + announce: true + auto_connect: true + accept_connections: true + + # 802.11s mesh backhaul between FIPS routers. These entries ship + # commented out so a stock install that never creates fips-mesh* + # logs no per-boot "interface missing" bind warning. Running + # 'fips-mesh-setup ' creates the interface AND uncomments the + # matching block here (once per radio; radio0 -> fips-mesh0, radio1 -> + # fips-mesh1); 'fips-mesh-setup remove' re-comments it. Restart fips + # after — a transport whose interface is missing at startup is skipped, + # not retried. Dual-band routers can mesh on both bands at once — + # failover, not multipath: FIPS keeps one active link per peer, the + # other band stands by. The mesh runs OPEN (no SAE) with 802.11s + # forwarding off: FIPS's Noise handshake is the encryption and + # authentication, and FIPS is the routing layer. See + # docs/how-to/set-up-80211s-mesh-backhaul.md. + # mesh0: + # interface: "fips-mesh0" + # listen: true + # announce: true + # auto_connect: true + # accept_connections: true + # mesh1: + # interface: "fips-mesh1" + # listen: true + # announce: true + # auto_connect: true + # accept_connections: true + + # Open "!FIPS" access SSID for phones and laptops running FIPS. These + # entries ship commented out so a stock install that never creates + # fips-ap* logs no per-boot "interface missing" bind warning. Running + # 'fips-ap-setup ' creates the interface AND uncomments the + # matching block here (once per radio; radio0 -> fips-ap0, radio1 -> + # fips-ap1); 'fips-ap-setup remove' re-comments it. Restart fips after + # — a transport whose interface is missing at startup is skipped, not + # retried. The SSID is OPEN and isolated on purpose: FIPS's Noise + # handshake is the only security layer, and associated clients reach + # nothing but the FIPS handshake surface. See + # docs/how-to/set-up-open-access-ssid.md. + # ap0: + # interface: "fips-ap0" + # listen: true + # announce: true + # auto_connect: true + # accept_connections: true + # ap1: + # interface: "fips-ap1" + # listen: true + # announce: true + # auto_connect: true + # accept_connections: true + + # No BLE transport: OpenWrt builds target musl, which has no BlueZ backend. + +# Outbound LAN gateway. dnsmasq forwards .fips queries to listen=[::1]:5353 +# while it runs (configured by the fips-gateway init script). Requires IPv6 +# forwarding enabled. +gateway: + enabled: true + pool: "fd01::/112" + lan_interface: "br-lan" + dns: + listen: "[::1]:5353" + upstream: "[::1]:5354" + ttl: 60 + pool_grace_period: 60 + +peers: [] + # Static peers for bootstrapping (UDP or TCP): + # - npub: "npub1qmc3cvfz0yu2hx96nq3gp55zdan2qclealn7xshgr448d3nh6lks7zel98" + # alias: "gateway" + # via_nostr: true + # addresses: + # - transport: udp + # addr: "test-us01.fips.network:2121" # IP or hostname (e.g., "peer.example.com:2121") + # - transport: udp + # addr: "nat" # Use node.rendezvous.nostr for Nostr/STUN hole punching + # connect_policy: auto_connect diff --git a/testing/openwrt/scenarios.sh b/testing/openwrt/scenarios.sh index 6c7f4007..fe78548f 100755 --- a/testing/openwrt/scenarios.sh +++ b/testing/openwrt/scenarios.sh @@ -28,9 +28,22 @@ RELEASED_PRERM="$REPO/testing/openwrt/fixtures/released-prerm" INIT_GATEWAY="$REPO/packaging/openwrt-ipk/files/etc/init.d/fips-gateway" APK_SCRIPTS="${APK_SCRIPTS:-}" SHIPPED_YAML="$REPO/packaging/openwrt-ipk/files/etc/fips/fips.yaml" +# The fips.yaml every release up to 0.5.1 shipped, from before the gateway's +# default DNS port moved. +RELEASED_YAML="$REPO/testing/openwrt/fixtures/released-fips.yaml" +GATEWAY_RS="$REPO/src/config/gateway.rs" +SETUP_SCRIPT="$REPO/packaging/openwrt-ipk/files/etc/uci-defaults/90-fips-setup" +LEGACY_LISTEN=' listen: "[::1]:5353"' +SHIPPED_LISTEN=' # listen: "[::1]:5365" # the default; the init script points dnsmasq at this port' +MIGRATED_MSG='fips: moved gateway.dns.listen off the mDNS port 5353 to the default [::1]:5365' WORK=/tmp/fips-openwrt-scenarios UPGRADE_MARKER=/tmp/fips-prerm-upgrade +# The uci stub's state and the directory holding the executable stubs. Fixed +# paths, because the .apk scripts run with only PATH in their environment. +UCI_DIR=/tmp/fips-openwrt-uci +STUB_BIN=/tmp/fips-openwrt-bin +DNSMASQ_OPT='dhcp.@dnsmasq[0].server' FAILURES=0 CASES=0 @@ -124,6 +137,27 @@ assert_not_called() { return 0 } +assert_none_called_with_prefix() { + # assert_none_called_with_prefix + # Fails if any recorded call begins with , whatever follows it. + if [ ! -r "$CALLS" ]; then + bad "$2 — the call log $CALLS cannot be read" + return 0 + fi + matched="" + while IFS= read -r line; do + case "$line" in + "$1"*) matched="$matched$line;" ;; + esac + done < "$CALLS" + if [ -n "$matched" ]; then + bad "$2 — called: $matched" + else + ok "$2" + fi + return 0 +} + assert_file_is() { # assert_file_is got="$(cat "$1" 2>/dev/null)" @@ -182,6 +216,76 @@ assert_order() { return 0 } +# Install an executable uci that keeps each option as a file of values, one per +# line, under $UCI_DIR. Like the real uci, "get" prints a list on one line +# separated by single spaces and fails for an option with no values, and +# del_list removes every copy of the value. Other commands are only logged. +install_uci_stub() { + rm -rf "$UCI_DIR" + mkdir -p "$UCI_DIR" "$STUB_BIN" + { + echo '#!/bin/sh' + echo "dir=$UCI_DIR" + cat <<'STUB' +[ "${1:-}" = "-q" ] && shift +cmd="${1:-}" +[ $# -gt 0 ] && shift +echo "uci $cmd $*" >> "$dir/log" +file_of() { + printf '%s/%s' "$dir" "$(printf '%s' "$1" | sed 's/[^A-Za-z0-9._-]/_/g')" +} +case "$cmd" in +get) + f="$(file_of "$1")" + [ -s "$f" ] || exit 1 + tr '\n' ' ' < "$f" | sed 's/ $//' + echo + ;; +add_list) + echo "${1#*=}" >> "$(file_of "${1%%=*}")" + ;; +del_list) + f="$(file_of "${1%%=*}")" + if [ -f "$f" ]; then + grep -vxF -- "${1#*=}" "$f" > "$f.new" + mv "$f.new" "$f" + fi + ;; +esac +exit 0 +STUB + } > "$STUB_BIN/uci" + chmod 0755 "$STUB_BIN/uci" + + cat > /etc/init.d/dnsmasq <<'STUB' +#!/bin/sh +echo "dnsmasq $1" >> "$CALLS" +STUB + chmod 0755 /etc/init.d/dnsmasq + return 0 +} + +uci_seed() { + # uci_seed