diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index dbb365c7..28d29130 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -284,6 +284,30 @@ jobs: - name: Install system dependencies run: sudo apt-get update && sudo apt-get install -y libdbus-1-dev + # The same address-less interface the musl leg creates. Pinning the + # contract on both libcs is what turns "glibc and musl agree about + # `getifaddrs`" from an assumption into a checked fact — and makes this + # leg fail first if glibc is the one that changes. + - name: Create an address-less interface for the presence probe + run: | + sudo ip link add fips-probe0 type dummy + # `addrgenmode none` before bringing it up: the kernel hands an IPv6 + # link-local to any interface that comes up, and an interface with a + # link-local is not address-less — the fixture would have quietly + # tested nothing. + sudo ip link set fips-probe0 addrgenmode none + sudo ip link set fips-probe0 up + ip addr show fips-probe0 + # Fail rather than test the wrong thing if it acquired one anyway. + if ip addr show fips-probe0 | grep -qE "inet6? "; then + echo "fips-probe0 has an address; it cannot test the address-less case" >&2 + exit 1 + fi + echo "FIPS_TEST_ADDRLESS_IFACE=fips-probe0" >> "$GITHUB_ENV" + # Declare that this runner has fixtures, so a test that depends on + # one fails when the fixture is missing instead of skipping silently. + echo "FIPS_TEST_REQUIRE_FIXTURES=1" >> "$GITHUB_ENV" + - name: Install Rust toolchain uses: actions-rust-lang/setup-rust-toolchain@166cdcfd11aee3cb47222f9ddb555ce30ddb9659 # v1 with: @@ -304,9 +328,39 @@ jobs: - name: Install cargo-nextest uses: taiki-e/install-action@nextest + - name: Run unit tests run: cargo nextest run --all --profile ci + # The bind-success half. Every other unit-test leg runs unprivileged, so + # `PacketSocket::open` cannot succeed on any of them and everything past + # a successful bind — the post-store shutdown check, the `Present` arm of + # the binder loop, `bind_now` itself — runs nowhere in CI. + # + # Built as the runner user and only *executed* under sudo: `cargo` run as + # root would use root's CARGO_HOME and discard the cache this job just + # restored. + # + # `FIPS_TEST_PRIVILEGED` is what makes this leg honest. A test that needs + # a raw socket skips quietly without it; with it set, a test that cannot + # open one fails and says so, so a runner that stops granting the + # capability shows up as a red leg rather than as silence. + - name: Run interface-binding tests with privilege + run: | + cargo test --lib --no-run + BIN=$(cargo test --lib --no-run --message-format=json \ + | jq -r 'select(.reason == "compiler-artifact") + | select(.executable != null) + | select(.target.kind[0] == "lib") + | .executable' \ + | tail -1) + if [ -z "$BIN" ] || [ ! -x "$BIN" ]; then + echo "could not locate the lib test binary" >&2 + exit 1 + fi + echo "running $BIN as root" + sudo -E env FIPS_TEST_PRIVILEGED=1 "$BIN" transport::ethernet --test-threads=1 + - name: Publish test report (Checks tab) uses: dorny/test-reporter@4a2e97665d5fa767581ef38eca97b9694bd4eef4 # v2 if: always() @@ -363,9 +417,116 @@ jobs: - name: Install cargo-nextest uses: taiki-e/install-action@nextest + # The Darwin half of the address-less presence contract. The Linux legs + # pin that `getifaddrs` reports an interface with no addresses as + # present, on both glibc and musl; without this the same claim on the + # BSD-derived implementation the macOS backend actually calls was + # untested, and the test skipped itself silently on this runner. + # + # `feth` is macOS's fake-Ethernet pseudo-interface. It is created + # address-less, and the check below fails the leg rather than testing the + # wrong thing if this runner hands it one anyway — the same shape as the + # Linux fixture step, which needs `addrgenmode none` for exactly that + # reason. + - name: Create an address-less interface for the presence probe + run: | + sudo ifconfig feth0 create + sudo ifconfig feth0 up + ifconfig feth0 + if ifconfig feth0 | grep -qE "^[[:space:]]*inet6? "; then + echo "feth0 has an address; it cannot test the address-less case" >&2 + exit 1 + fi + echo "FIPS_TEST_ADDRLESS_IFACE=feth0" >> "$GITHUB_ENV" + # Declare that this runner has fixtures, so a test that depends on + # one fails when the fixture is missing instead of skipping silently. + echo "FIPS_TEST_REQUIRE_FIXTURES=1" >> "$GITHUB_ENV" + - name: Run unit tests run: cargo nextest run --all --profile ci +# ───────────────────────────────────────────────────────────────────────────── +# Job 2bb – Unit tests (musl) +# +# OpenWrt — the platform the Ethernet transport's dynamic interface binding +# exists for — is musl, and musl reimplements the libc calls that binding is +# built on rather than sharing glibc's. `interface_present` reads `ifa_flags` +# out of `getifaddrs`, and the interfaces it has to see (`fips-mesh0`, +# `fips-ap0`) are deliberately unbridged with no IP address at all, which is +# exactly where getifaddrs implementations differ. Every other leg is glibc, so +# without this one the presence probe is asserted on a libc no test has ever +# run it against, on the target it was written for. +# +# Built for the musl target on a glibc host rather than inside an Alpine +# container. The test binary links musl statically and runs natively on the +# runner, so musl's `getifaddrs` is the one under test — while the build +# scripts stay host artifacts, which keeps rustables' bindgen on the same +# libclang the glibc leg already builds with. Building inside Alpine put +# bindgen on a musl toolchain it does not work on: statically linked build +# scripts cannot `dlopen` libclang, and turning the static CRT off then left +# it loading libclang but unable to parse. None of that is anything this leg +# is trying to test. +# ───────────────────────────────────────────────────────────────────────────── + test-musl: + name: Unit tests (musl) + runs-on: ubuntu-latest + needs: [build] + steps: + - uses: actions/checkout@d23441a48e516b6c34aea4fa41551a30e30af803 # v6 + + - name: Set SOURCE_DATE_EPOCH from git + run: echo "SOURCE_DATE_EPOCH=$(git log -1 --format=%ct)" >> "$GITHUB_ENV" + + # libdbus for the host build scripts; musl-tools for the musl C + # toolchain the `cc`-driven dependencies link against. BLE is excluded on + # musl by a Cargo.toml cfg, so bluer is not in this build at all. + - name: Install system dependencies + run: sudo apt-get update && sudo apt-get install -y libdbus-1-dev musl-tools + + # The address-less interface the presence probe has to be tested + # against; see the matching step on the glibc leg for why loopback + # cannot stand in for it. + - name: Create an address-less interface for the presence probe + run: | + sudo ip link add fips-probe0 type dummy + # `addrgenmode none` before bringing it up: the kernel hands an IPv6 + # link-local to any interface that comes up, and an interface with a + # link-local is not address-less — the fixture would have quietly + # tested nothing. + sudo ip link set fips-probe0 addrgenmode none + sudo ip link set fips-probe0 up + ip addr show fips-probe0 + # Fail rather than test the wrong thing if it acquired one anyway. + if ip addr show fips-probe0 | grep -qE "inet6? "; then + echo "fips-probe0 has an address; it cannot test the address-less case" >&2 + exit 1 + fi + echo "FIPS_TEST_ADDRLESS_IFACE=fips-probe0" >> "$GITHUB_ENV" + # Declare that this runner has fixtures, so a test that depends on + # one fails when the fixture is missing instead of skipping silently. + echo "FIPS_TEST_REQUIRE_FIXTURES=1" >> "$GITHUB_ENV" + + - name: Install Rust toolchain + uses: actions-rust-lang/setup-rust-toolchain@166cdcfd11aee3cb47222f9ddb555ce30ddb9659 # v1 + with: + cache: false + rustflags: '' + target: x86_64-unknown-linux-musl + + - name: Cache Cargo registry + build + uses: actions/cache@caa296126883cff596d87d8935842f9db880ef25 # v5 + with: + path: | + ~/.cargo/registry + ~/.cargo/git + target + key: musl-cargo-${{ hashFiles('**/Cargo.lock') }} + restore-keys: | + musl-cargo- + + - name: Run library tests + run: cargo test --lib --target x86_64-unknown-linux-musl + # ───────────────────────────────────────────────────────────────────────────── # Job 2c – Unit tests (Windows) # ───────────────────────────────────────────────────────────────────────────── @@ -395,6 +556,7 @@ jobs: - name: Install cargo-nextest uses: taiki-e/install-action@nextest + - name: Run unit tests run: cargo nextest run --all --profile ci diff --git a/CHANGELOG.md b/CHANGELOG.md index 0b4c3bfe..2eb07c40 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,232 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- Dynamic interface binding for the Ethernet transport. An interface-bound + transport is now a long-lived object that is *sometimes bound*: the interface + it names need not exist when the daemon starts, may appear minutes later, and + may vanish and return mid-operation. `start_async` returns `Ok` with the + transport **absent** rather than failing, and a per-transport binder task + binds when the interface appears, unbinds when it goes away, and rebinds when + it returns. Start-time absence and runtime detach are one code path. + Detection is event-driven where the kernel offers a source — netlink + `RTNLGRP_LINK` on Linux, `PF_ROUTE` on macOS and FreeBSD — with a 1 s + `getifaddrs` poll underneath as a backstop. Presence means + `IFF_UP` — the interface exists and is administratively up — and + deliberately not `IFF_RUNNING`. Binding needs no carrier and a socket + outlives a carrier flap, so a bridge with nothing plugged into it (`br-lan` + on a wifi-only router) is bound and healthy rather than permanently + `Degraded`, and starts carrying traffic the moment a port comes up. Whether + an interface has carrier is reported separately as `interface.carrier` in + `show_transports`, never acted on. + + This closes the OpenWrt boot race (procd starts `fips` before wifi has + created `fips-mesh0` / `fips-ap0`; both transports were skipped for the life + of the process while the 802.11s peer link formed anyway, so the node looked + healthy and reached nothing), the intermittent-adapter case, and the + mid-operation `wifi reload` that destroyed and recreated an interface under a + live socket. + +- `transports.ethernet.*.optional` (bool, default `false`). Naming an interface + in configuration is a statement that you expect it, so the default is to + complain: while a required interface is missing the node reports `Degraded` + and logs the edge, at a severity that follows how long the absence lasts + (see below). `optional: true` makes absence silent (`info` on the edge, no + health impact) for hardware that is legitimately not always there. It describes the interface's *presence*, not the transport's + importance — an optional interface that is present is used exactly as hard as + any other — and no value of it makes a missing interface fatal at startup. + +- `fipsctl show transports` reports interface presence per transport under a new + `interface` block: `name`, `presence` (`absent` / `binding` / `present`), + `carrier`, `policy` (`required` / `optional`), `since_secs`, `binds` and + `failed_attempts`. The original boot-race bug was expensive precisely because + nothing an operator could see said the node was deaf. + +- `testing/iface-binding/` integration suite (`ci-local.sh --only + iface-binding`, and a GitHub matrix leg): two daemons whose only transports + are interface-bound, run against a veth pair the harness creates, downs, + deletes and recreates underneath them. Asserts the boot race, the late + attach and peering over it, the flap in both directions, + destroy-and-recreate, that an `optional` interface never moves node health, + and that absence is logged once on the edge rather than once per retry. + +#### Node lifecycle + +- Transport-medium change detection, controlled by the new `node.netmon.*` + block (on by default). For each peer whose transport address is a numeric IP + endpoint, the node asks the kernel which local address it would use to reach + *that peer* — a `connect(2)` on a UDP socket, which resolves the route and + sends nothing — and reports a change once some peer held across two + consecutive samples is reached from a different local address, or has stopped + being reachable at all. Asking the question per peer rather than about the + host is what keeps it quiet: a container bridge, a VPN, a `veth` pair or a + tunnel appearing is not the route to any peer and cannot move the + fingerprint, while a peer on the same LAN — reached by its subnet route, not + the default route — is covered, as is a more specific route moving under a + single peer. Peers joining and leaving are ignored on their own, being + ordinary node behaviour rather than a statement about the medium — except + that a peer seen for the first time is checked against its own + `connect()`-ed socket, and reported if that socket is pinned to a source the + routing table would no longer choose, so a medium change in the window + between a peer authenticating and the next sample is not adopted silently + while that peer sits stranded on the old path. A peer + addressed by MAC, by `.onion` or Nym recipient, by a scoped IPv6 literal, or + by a hostname it has not yet been heard from on, has no route to ask about + and contributes nothing; a node with no peers detects nothing, having nothing + bound to the old path to repair. The peer table is read through the node's + existing lock-free entity snapshot, so the detector stays a detached task + holding no node state. A handover is not atomic (the route goes, briefly + there is none, the new one arrives), so a short debounce coalesces the burst + into one event and a fingerprint that settles back where it started reports + nothing. Linux subscribes to `NETLINK_ROUTE` + multicast (the groups `ip monitor` uses) and macOS and FreeBSD to a + `PF_ROUTE` socket, both reacting to the kernel event in milliseconds; every + other platform samples on a timer at `node.netmon.poll_interval_secs`, which + also runs underneath the kernel sources as a backstop, since a netlink socket + drops messages under memory pressure and the subscription can be refused in a + restricted sandbox. A backend only decides *when to look* — the fingerprint + comparison, the debounce and the settled-back suppression are shared — so the + remaining backends (`NotifyIpInterfaceChange` on Windows, an embedder push on + iOS) land behind the same seam without touching the reaction. Android takes + the netlink source, and falls back to the timer where policy refuses the + group bind. What the node does with the signal is the connected-socket + rebind described under Fixed above. + Bluetooth is not covered: an adapter's state is not an IP attachment and is + invisible to this detector. + **Upgrade note: this couples a new key to one that has already shipped.** A + handover is ridden out for up to eight settling rounds of + `node.netmon.debounce_ms` before a change is reported, and if that worst case + reaches `node.link_dead_timeout_secs` the reaper tears the peering down + before the change is ever acted on, so the node refuses to start rather than + run in that shape. At the shipped defaults the margin is wide (8 × 250ms = 2s + against 30s), but detection is on by default, so **a node that shortened + `node.link_dead_timeout_secs` to 1 or 2 seconds for fast failover will be + refused at startup after the upgrade**, naming a `node.netmon.*` key its + operator never set. Raise the timeout, lower `debounce_ms` so eight rounds + stay under it, or set `node.netmon.enabled: false`. A + `node.link_dead_timeout_secs` of 0 is exempt from the check. + +#### Library surface and internals + +- `TransportError::InterfaceUnavailable { interface }`. A missing interface and + a typo'd interface name were previously the same flat + `StartFailed(String)`; nothing downstream could branch on absence. + +### Changed + +- `Degraded` is now a level rather than a latch. The supervisor's reason set + was monotonic, which was correct while no child could recover; with recovery + it would have meant "something broke at some point since boot" rather than + "something is broken now". Interface absence is tracked in its own reversible + set and node health is recomputed on every transition **in both directions**, + so plugging the WAN back in clears `Degraded` without a restart. A transport + whose interface is absent still counts as up, so a single-interface node that + boots before its wifi degrades rather than exiting on "no transports". + +- The OpenWrt package ships the `mesh0`/`mesh1` and `ap0`/`ap1` Ethernet + transports **enabled** with `optional: true`, instead of commented out. + `fips-mesh-setup` and `fips-ap-setup` no longer comment-toggle blocks in + `fips.yaml`, and no longer tell the operator to restart the daemon after + creating an interface — the daemon binds it on its own. `phy0-sta0` (`wwan`) + is marked `optional: true` for the same reason: it only exists while a radio + is in station mode. + +- The Ethernet receive loop backs off and exits on a dead socket instead of + spinning on `Err` with a `warn!` per iteration, and the ad-hoc ENXIO + socket-reopen in the beacon sender is gone. Both hand recovery to the + presence machine: one mechanism for every cause rather than one hack per + symptom. Beacons pause while an interface is absent. + +- The absence edge is not itself an error, and there is exactly one deadline + after it. An interface missing when the daemon starts logs at `info` — that + is the boot race the mechanism exists to absorb, not a fault — and a runtime + detach at `warn`, because a link coming and going is ordinary weather for a + mesh daemon. Ten seconds is the whole grace: past it, absence is no longer a + race against a radio or a container, so a **required** interface still + missing is reported once at `error`. Start-time absence and a runtime detach + share that one deadline rather than getting one each. An `optional` + interface never reaches `error`. Node health does not wait for any of it, + publishing `Degraded` on the first edge either way. + +- The TUN boundary's TCP MSS clamp now tracks the node's egress MTU at + runtime instead of freezing it at startup. `transport_mtu()` is the minimum + across *bound* transports, so a transport that binds minutes after start can + be the narrow one — but the TUN reader and writer were handed a `u16` + computed once when they spawned, while every other consumer + (`show_status`, the control-socket snapshot, the session-layer fragmentation + check) read it live. A node could therefore report one effective IPv6 MTU + and clamp to another. The ceiling is now shared with those threads and + recomputed whenever the bound set changes, in both directions: a narrow + interface appearing tightens it, and its departure releases it. MSS is + negotiated per connection, so a change applies to connections opened after + it; existing ones are not disturbed. + +- Rebinds that keep succeeding into a socket that dies moments later are + damped: consecutive bindings shorter than ten seconds back off on the + 1 s → 30 s curve, and past three of them the binder stops announcing each + bind as a recovery until one lasts. Undamped, a persistently broken socket + behind a healthy interface produced a log pair and a `Degraded`→`Running` + health flap every second. + +- A bind failure that is **not** absence — no `CAP_NET_RAW`, no readable + `/dev/bpf*`, a buffer the kernel refused — fails the daemon's start as it + always has, rather than being waited out. It is a fault, not a state, and + will not resolve on its own; only a missing interface is retried at start. + A non-absence failure during a later rebind still backs off, since the node + is serving by then. + +- The binder cannot outlive its transport, and a teardown that races a bind + cannot leave a live receive loop on a socket nothing owns. A shared stop flag + is raised before teardown and checked by the binder after it stores a + binding, so whichever order the two interleave exactly one of them cleans up; + `EthernetTransport` gained a `Drop` that raises the flag, aborts the binder + and releases the socket, for handles dropped without `stop_async`. + +- Presence edges are published with `try_send` and retried on the next tick + rather than awaited. A bounded channel could previously park the binder + mid-publish — a health channel able to deadlock the machine whose health it + carries — freezing the interface in whatever state it held. + +- `TransportHandle::is_bound()` joins `is_operational()`: the latter means the + transport was *started*, which for an interface-bound transport no longer + implies a live socket. `Node::transport_mtu` now filters on the former, + because an interface that has never existed was clamping the whole node's + IPv6 MTU to a number derived from absent hardware. + +- An interface deleted and recreated under the same name is detected as a + detach. Both backends bind by device rather than by name, so the old socket + is attached to nothing while the name still resolves — and a stale + `AF_PACKET` socket never becomes readable, so nothing errors and nothing + exits. Detection previously rested entirely on the beacon sender failing, + which a node with `announce: false` does not have. The bound interface index + is now captured at bind and compared on every poll. + +- The link-event watcher distinguishes a genuine receive error from + `WouldBlock`. `try_io` clears readiness only on the latter, so a persistent + error — `ENOBUFS` after a burst of link events overflows the socket buffer — + span a core flat with nothing logged. Errors are now counted, logged once, + backed off, and after five the source is abandoned for the presence poll. + +- Presence probes are coalesced to at most ten a second. Linux netlink is + filtered to `RTNLGRP_LINK`, but `PF_ROUTE` has no group filter, so the macOS + source delivers every routing message on the host — route churn, ARP, DHCP + renewals, a VPN going up and down — and each would otherwise drive a full + `getifaddrs` walk. + +- CI runs the library tests on musl (Alpine) as well as glibc. Presence is + built on `getifaddrs` and `ifa_flags`, musl reimplements both independently, + and the interfaces this feature exists for (`fips-mesh0`, `fips-ap0` on + OpenWrt) are unbridged with no IP address at all — the case where + implementations most plausibly differ. It was previously asserted on a libc + no test had ever run it against, on the target it was written for. + +- Interface presence state ignores lock poisoning. Treating a poisoned lock as + a failure meant reading "no socket, tasks dead", which is the destructive + direction: a transport reporting itself present while every send fails, or a + binder tearing down and rebinding every second while teardown silently + declined to abort anything. + ### Fixed #### Node lifecycle @@ -84,64 +310,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 platforms with the connected-socket fast path); elsewhere the heartbeat alone carries the new address. -### Added - -#### Node lifecycle - -- Transport-medium change detection, controlled by the new `node.netmon.*` - block (on by default). For each peer whose transport address is a numeric IP - endpoint, the node asks the kernel which local address it would use to reach - *that peer* — a `connect(2)` on a UDP socket, which resolves the route and - sends nothing — and reports a change once some peer held across two - consecutive samples is reached from a different local address, or has stopped - being reachable at all. Asking the question per peer rather than about the - host is what keeps it quiet: a container bridge, a VPN, a `veth` pair or a - tunnel appearing is not the route to any peer and cannot move the - fingerprint, while a peer on the same LAN — reached by its subnet route, not - the default route — is covered, as is a more specific route moving under a - single peer. Peers joining and leaving are ignored on their own, being - ordinary node behaviour rather than a statement about the medium — except - that a peer seen for the first time is checked against its own - `connect()`-ed socket, and reported if that socket is pinned to a source the - routing table would no longer choose, so a medium change in the window - between a peer authenticating and the next sample is not adopted silently - while that peer sits stranded on the old path. A peer - addressed by MAC, by `.onion` or Nym recipient, by a scoped IPv6 literal, or - by a hostname it has not yet been heard from on, has no route to ask about - and contributes nothing; a node with no peers detects nothing, having nothing - bound to the old path to repair. The peer table is read through the node's - existing lock-free entity snapshot, so the detector stays a detached task - holding no node state. A handover is not atomic (the route goes, briefly - there is none, the new one arrives), so a short debounce coalesces the burst - into one event and a fingerprint that settles back where it started reports - nothing. Linux subscribes to `NETLINK_ROUTE` - multicast (the groups `ip monitor` uses) and macOS and FreeBSD to a - `PF_ROUTE` socket, both reacting to the kernel event in milliseconds; every - other platform samples on a timer at `node.netmon.poll_interval_secs`, which - also runs underneath the kernel sources as a backstop, since a netlink socket - drops messages under memory pressure and the subscription can be refused in a - restricted sandbox. A backend only decides *when to look* — the fingerprint - comparison, the debounce and the settled-back suppression are shared — so the - remaining backends (`NotifyIpInterfaceChange` on Windows, an embedder push on - iOS) land behind the same seam without touching the reaction. Android takes - the netlink source, and falls back to the timer where policy refuses the - group bind. What the node does with the signal is the connected-socket - rebind described under Fixed above. - Bluetooth is not covered: an adapter's state is not an IP attachment and is - invisible to this detector. - **Upgrade note: this couples a new key to one that has already shipped.** A - handover is ridden out for up to eight settling rounds of - `node.netmon.debounce_ms` before a change is reported, and if that worst case - reaches `node.link_dead_timeout_secs` the reaper tears the peering down - before the change is ever acted on, so the node refuses to start rather than - run in that shape. At the shipped defaults the margin is wide (8 × 250ms = 2s - against 30s), but detection is on by default, so **a node that shortened - `node.link_dead_timeout_secs` to 1 or 2 seconds for fast failover will be - refused at startup after the upgrade**, naming a `node.netmon.*` key its - operator never set. Raise the timeout, lower `debounce_ms` so eight rounds - stay under it, or set `node.netmon.enabled: false`. A - `node.link_dead_timeout_secs` of 0 is exempt from the check. - ### Changed - The lockfile moves `chacha20` from 0.10.1 to 0.10.2, because 0.10.1 is yanked. diff --git a/docs/design/fips-transport-layer.md b/docs/design/fips-transport-layer.md index 53a03eea..a4412f74 100644 --- a/docs/design/fips-transport-layer.md +++ b/docs/design/fips-transport-layer.md @@ -1010,6 +1010,313 @@ Transports begin in `Configured` state with all parameters set. `start()` transitions through `Starting` to `Up` (operational). `stop()` moves to `Down`. Transport failures move to `Failed`. +## Interface Presence + +`Up` describes the *transport*, not the socket. An interface-bound transport +(today: Ethernet) carries a second, orthogonal state — whether it is bound +right now — and the two are independent: a transport is `Up` from the moment +it starts, whether or not the interface it names exists. + +### The Gap This Closes + +Three deployment scenarios exercise one missing mechanism: + +- **Boot ordering.** On OpenWrt, procd starts `fips` before wifi has created + `fips-mesh0` / `fips-ap0`. Both transports were skipped and never retried, + while the 802.11s peer link formed anyway — that is mac80211, not the + daemon — so the node looked healthy and reached nothing. The failure was + expensive precisely because nothing an operator could see said the node was + deaf. +- **Intermittent hardware.** A USB ethernet adapter named in `fips.yaml` is + plugged in some days and not others. Its absence is normal and must be + silent; its arrival must bind without operator action. +- **Mid-operation restart.** `wifi reload` for a channel change destroys and + recreates the mesh interface within a couple of seconds. The socket dies, + the receive loop spun on `Err` with no backoff and no exit, and nothing + rebound. + +These are not three features. They are one presence machine plus one policy +field. Before it existed, the first observation was final: an interface +missing at start was logged once and skipped for the life of the process, and +one that disappeared at runtime published a health change but was never +rebound. + +### The Presence Machine + +```text +Absent ──attach──> Binding ──ok──> Present + ^ │ │ + └──── fail/backoff ─┘ │ + └──────────── detach ──────────────┘ +``` + +`start_async` binds if it can and otherwise returns `Ok` with the transport +`Up` and `Absent`; a per-transport binder task then binds when the interface +appears, tears the socket down when it goes away, and rebinds when it +returns. Two invariants do the work: + +- **The transport object survives detach.** Config, `TransportId`, + statistics and the neighbor buffer persist; only the file descriptor and + its loops go. A transport is never destroyed because its interface went + away. +- **Start-time absence and runtime detach are the same transition.** A node + that boots before its wifi and a node whose wifi reloads at 03:00 take one + code path. The old asymmetry — skip forever at start, publish health at + runtime — is gone. + +`TransportError::InterfaceUnavailable` is what makes absence branchable. A +missing interface and a typo'd interface name were the same flat +`StartFailed(String)`, so nothing downstream could tell a state from a fault. + +### What Counts as Present + +Presence means `IFF_UP` — the interface exists and the operator has enabled +it — and deliberately **not** `IFF_RUNNING`. + +Carrier and bindability are different questions, and only the second belongs +in a bind gate. An `AF_PACKET` socket on a carrier-less bridge is valid and +starts carrying traffic the instant a member port comes up, with no rebind: +the socket outlives the carrier. Gating on `IFF_RUNNING` bought nothing and +cost three things: + +- `br-lan` on a router with nothing in its LAN ports is `UP` with + `NO-CARRIER`, so a healthy wifi-only router reported `Degraded` forever and + errored for a fault it did not have; +- every carrier flap the socket would have survived became an unbind/rebind + cycle — churn the presence machine then has to damp, a mechanism + compensating for a policy error; +- an 802.11s interface that reports `RUNNING` only once it has peered cannot + peer, because peering needs beacons, beacons need a bound socket, and the + gate refuses to bind. A deadlock reachable on the hardware this mechanism + was written for. + +The signal `IFF_RUNNING` carries is not lost: `show_transports` reports +`interface.carrier` beside presence, so an operator can still tell a bound +transport carrying nothing from a working one. It is reported rather than +obeyed. + +The probe is `getifaddrs` plus `ifa_flags` rather than an `SIOCGIFFLAGS` +ioctl: it needs no socket, so the watcher can probe before any file +descriptor exists, and it is spelled the same on Linux and the BSDs. + +### Interface Identity + +The configured name is the key, but a name is not a device. Both backends +bind by *device* — `AF_PACKET` stores `sll_ifindex`, a BPF descriptor follows +the interface it was attached to — so an interface deleted and recreated +under the same name leaves the socket attached to something that no longer +exists while the name resolves perfectly well. + +Nothing else notices. A stale `AF_PACKET` socket never becomes readable, so +the receive loop neither errors nor exits, and send failures go to the caller +rather than to the binder. A listen-only node (`announce: false`, so no +beacon sender to fail) therefore sat `present` and deaf indefinitely after a +`wifi reload` — the original bug wearing a different hat. The bound index is +captured at bind and compared on every poll; a mismatch is a detach. + +Hardware can also change underneath a name. If the name reappears with a MAC +other than the one last bound, that is a different device, so the cached +neighbor entries for that transport are dropped rather than resumed onto, and +the swap is logged at `warn`. Richer selectors (`match: { name | mac | +id_path }`) are deliberately deferred; the requirement here is only that FIPS +never silently resumes onto different hardware. + +### Detection + +| Platform | Source | +| -------- | ------ | +| Linux | netlink `RTNLGRP_LINK` (`RTM_NEWLINK` / `RTM_DELLINK`) | +| macOS, FreeBSD† | `PF_ROUTE` socket, `RTM_IFINFO` | +| Fallback | poll `getifaddrs` + flags, 1 s | + +† Aspirational: the Ethernet transport is +`cfg(any(target_os = "linux", target_os = "macos"))`, so FreeBSD has no +interface-bound transport for a watcher to serve. The `PF_ROUTE` branch +compiles for the BSD family, but only macOS reaches it. + +Where an event source exists, detection is sub-second. The poll stays +underneath as a backstop rather than as the mechanism, and must stay at ~1 s: +the probe is cheap, and letting the interval drift to tens of seconds +reintroduces exactly the latency the event source was added to remove. +Construction is best-effort — a kernel or sandbox that refuses the socket +yields a watcher that never fires, and the binder degrades to its poll. + +Link-event payloads are **not parsed**. An event is a hint to re-run the +presence probe, which is cheap and authoritative; decoding +`nlmsghdr`/`ifinfomsg` to reach the same answer would add a parser whose bugs +would be presence bugs. + +Two rate limits protect the binder from its own event source. Probes are +coalesced to ten a second, because `PF_ROUTE` has no group filter and +delivers every routing message on the host — route churn, ARP, DHCP renewals, +a VPN going up and down — each of which would otherwise drive a full +`getifaddrs` walk. And a persistently failing event source is counted, logged +once, backed off, and after five consecutive errors abandoned for the poll: +losing events is survivable because the poll is the backstop, but burning a +core on a socket that is readable-but-erroring is not. + +Bind failures that are *not* absence back off 1 s → 30 s. Absence itself does +not back off; there is nothing to poll but the probe. + +### Policy: `optional` + +One field per transport, `transports.ethernet.*.optional`, default `false`: + +| | absence | log | retries | +| --- | --- | --- | --- | +| `optional: false` (default) | node reports `Degraded` | `info` at boot / `warn` on a runtime detach, then `error` once if it lasts past 10 s | forever | +| `optional: true` | no health impact | `info`, and nothing after | forever | + +Naming an interface in configuration is a statement that you expect it, so +the default is to complain; silence is opted into. + +`optional` describes **the interface's presence, not the transport's +importance**. An optional interface that is present is used exactly as hard +as any other. No value of it makes a missing interface fatal at startup: the +only fatal case remains "no transports at all came up". If a deployment ever +needs absence to abort startup, that arrives as an explicit `on_absent: exit` +— never as a second meaning for `optional`. + +A bind failure that is not absence — no `CAP_NET_RAW`, no readable +`/dev/bpf*`, a buffer the kernel refused — is a fault, not a state, and still +fails the daemon's start. Retrying those forever would convert a hard, +actionable deployment error into a daemon that retries a socket it can never +open behind a `Degraded` nobody is watching. Only absence is waited out at +start; a non-absence failure during a later *rebind* does back off, since by +then the node is serving and killing it would be the worse answer. + +### Health + +Node health is recomputed on every presence transition **in both +directions**, so a returning interface clears `Degraded` without a restart. +That makes `Degraded` a level rather than a latch: the supervisor's reason set +was monotonic, which was correct while no child could recover, but with +recovery it would have come to mean "something broke at some point since boot" +rather than "something is broken now". Absence lives in its own reversible +set, separate from the one-way `failed` set a start failure enters. + +An absent transport still counts as *up*. It came up — `start_async` returned +`Ok` — so it does not push a single-transport node into the fatal +`NoTransports`, which would make a node that merely booted before its wifi +exit instead of waiting. Absence degrades; it never kills. + +There is deliberately **no restart action in the supervisor FSM.** The +presence watcher and the rebind loop live inside the transport, next to the +file descriptor they manage, and once that exists a supervisor-authored retry +has nothing left to do — it would be a second mechanism racing the first for +the same socket. The supervisor learns about presence +(`Event::ChildAbsent` / `Event::ChildPresent`) and republishes health; it does +not drive rebinding. + +Peer state needs no separate grace period. A send over an absent interface +returns `InterfaceUnavailable` and the peer entry survives untouched, so peers +are already held across a detach and resume when the interface returns; the +liveness reaper is the effective linger bound. A recreated mesh interface +comes back with the same MAC (it is derived from the phy) and Noise sessions +are keyed on the remote peer, so a local rebind is invisible to peers. A +dedicated linger timer would be a second answer to a question already +answered. + +### Logging + +Edges, never attempts. A loop that logs per attempt reproduces the hot log +spin this mechanism removed, at 1–30 s intervals forever on any router with +an unplugged WAN — and operators learn to filter it, which is how the next +real failure gets missed. + +The edge itself is not an error. An interface missing when the daemon starts +and bound a moment later is the ordinary case the mechanism exists to absorb, +so it is `info`; calling it an error at t=0 and "recovered" at t=0.2 s is the +cry-wolf failure this rule exists to prevent. A runtime detach is `warn` — a +link coming and going is ordinary weather for a mesh daemon. + +There is exactly one deadline. Ten seconds is the window in which absence +could still be a race — a radio, a container, a veth arriving late. Past it a +**required** interface is a fault an operator has to fix, and it is reported +once at `error`. Start-time absence and a runtime detach share that deadline +rather than getting one each, for the same reason they share a code path +everywhere else here. An `optional` interface never reaches `error`; that is +what `optional` means. + +Once, not repeated. This was a 1 m / 10 m / 1 h ladder that re-announced the +same fact at rising severity and then went permanently quiet after an hour, +which got both halves wrong: it used the log as a store for something already +published continuously as state, and it stopped mentioning a fault that was +still live. Duration belongs in `interface.since_secs` and in how long +`Degraded` has been held, where a monitor can threshold it per deployment +instead of the daemon compiling one in. + +Node health does not wait for the deadline. `Degraded` publishes on the first +edge, which is the signal an operator actually watches. + +Successful rebinds are damped. Backoff covers failed binds; the opposite and +nastier case is binds that keep *succeeding* into a socket that dies moments +later, which a receive loop giving up on a persistent error while the +interface stays `UP` produces once per second, forever. The binder counts +consecutive bindings that die inside ten seconds, backs off on the same +1 s → 30 s curve, and past three of them stops announcing each bind as a +recovery — holding health where it is until a binding lasts. + +### Egress MTU + +`transport_mtu()` is the minimum across *bound* transports — `is_bound()`, +not `is_operational()`, because an interface-bound transport is operational +from the moment it starts whether or not it holds a socket. Filtering on the +weaker predicate let a transport whose interface had never appeared set the +whole node's IPv6 MTU from hardware that was not present. + +Since a transport can now bind long after start, that minimum moves at +runtime, and every consumer has to read it live. `show_status`, the +control-socket snapshot and the session-layer fragmentation check always did. +The TUN reader and writer did not: they were handed a `u16` at spawn, so a +narrow interface binding later never tightened the TCP MSS clamp and the node +reported one effective MTU while clamping to another. The ceiling is now +shared with those threads — an atomic beside the per-destination +`path_mtu_lookup` they already read on the same packet — and recomputed on +every change to the bound set. + +Both directions, for the same reason `Degraded` is a level rather than a +latch: a narrow interface arriving must tighten the clamp or traffic +egressing over it is clamped too loose, and that interface leaving must +release it or unplugging a low-MTU adapter leaves the node over-clamped until +it restarts. MSS is negotiated per connection at SYN time, so a change binds +connections opened after it and leaves established ones alone. + +### Observability + +`show_transports` carries an `interface` block per interface-bound transport: +netdev name, `presence` (`absent` / `binding` / `present`), `carrier`, +`policy` (`required` / `optional`), `since_secs`, `binds` and +`failed_attempts`. The two counters separate an interface that is flapping +from one that is there and refusing to bind, and `since_secs` measures the +absence *episode* rather than the phase — a bind that fails walks +`Absent → Binding → Absent`, and restarting the clock on those edges would +report a permanently unbindable interface as one second old forever. + +`fipstop`'s transports view names the netdev and the absence policy in their +own columns and shows presence in the State column for these transports, +because `state` reads `up` from the moment the transport starts and is +therefore precisely the wrong answer in the one case someone is scanning that +column for. Both render sites sort by ascending transport id — creation +order, and so grouped by transport type — rather than by `HashMap` iteration +order, which was arbitrary and differed on every daemon restart. + +### What This Retires + +- The `hotplug.d/net` rule that restarted the daemon when the FIPS radio + interfaces appeared, and the `wifi down; wifi up; sleep` dance provisioning + performed to sequence around the race. +- The YAML comment-toggling in `fips-mesh-setup` / `fips-ap-setup` — the mesh + and AP blocks ship enabled with `optional: true` and simply wait. +- The ad-hoc ENXIO socket reopen in the beacon sender: beacons now pause while + absent because the task does not exist then, and recovery is the presence + machine's job. One mechanism for every cause rather than one hack per + symptom. +- The start-time versus runtime asymmetry in the supervisor. + +It also covers the case none of those workarounds did: an interface that flaps +while the daemon is running. + ## Implementation Status | Transport | Status | Notes | diff --git a/docs/reference/cli-fipsctl.md b/docs/reference/cli-fipsctl.md index 2ee7f6f9..00e2c67f 100644 --- a/docs/reference/cli-fipsctl.md +++ b/docs/reference/cli-fipsctl.md @@ -52,7 +52,7 @@ prints the response's `data` object as pretty JSON. | `show mmp` | `show_mmp` | MMP metrics summary: per-peer link-layer metrics and per-session session-layer metrics. | | `show cache` | `show_cache` | Coordinate cache: TTL, fill ratio, per-destination coords and path MTU. | | `show connections` | `show_connections` | Pending handshake connections: state, idle time, resend count. | -| `show transports` | `show_transports` | Transport instances: type, state, MTU, local address, per-transport stats. | +| `show transports` | `show_transports` | Transport instances: type, state, MTU, local address, per-transport stats, and — for interface-bound transports — interface presence (`absent` / `binding` / `present`), carrier, absence policy (`required` / `optional`), time in the current phase, and bind/failed-attempt counts. | | `show routing` | `show_routing` | Routing summary: pending lookups, retry state, forwarding/discovery/error/congestion counters. | | `show identity-cache` | `show_identity_cache` | Cached `(node_addr → npub)` entries with last-seen timestamps. | | `show native-flows` | `show_native_flows` | Native datagram API: open and pending flows with their ports, queue depth and age, bound listeners with their backlog, and the `native` counters. | diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 1cb2bdee..df31aa2f 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -673,6 +673,80 @@ root; on macOS it requires read/write access to a `/dev/bpf*` device. | `auto_connect` | bool | `false` | Auto-connect to discovered peers | | `accept_connections` | bool | `false` | Accept incoming connection attempts from discovered peers | | `beacon_interval_secs` | u64 | `30` | Announcement beacon interval in seconds (minimum 10) | +| `optional` | bool | `false` | Whether absence of the interface is normal. See below | + +**Dynamic binding.** The interface does not have to exist when the daemon +starts. A transport whose interface is missing comes up *absent*: it is not a +start failure, it is not skipped, and it binds on its own the moment the +interface appears — sub-second where the kernel offers link events (netlink on +Linux, `PF_ROUTE` on the BSDs), within a second otherwise. An interface that +goes away at runtime unbinds and rebinds by the same path, so a `wifi reload` +or an unplugged adapter needs no restart. Presence means `IFF_UP` — the +interface exists and is administratively up — and deliberately not +`IFF_RUNNING`: binding needs no carrier, and the socket keeps working across a +carrier flap without rebinding. A bridge with nothing plugged into it, such as +`br-lan` on a wifi-only router, is therefore bound and healthy rather than +permanently `Degraded`, and starts carrying traffic the moment a port comes up. +Whether an interface has carrier is reported separately, as `interface.carrier` +in `show_transports`. + +`optional` selects how that absence is reported: + +| | absence | log | retries | +| --- | --- | --- | --- | +| `optional: false` (default) | node reports `Degraded` | `info` at boot / `warn` on a runtime detach, then `error` once if it lasts past 10 s | forever | +| `optional: true` | no health impact | `info`, and nothing after | forever | + +Naming an interface in configuration is a statement that you expect it, so the +default is to complain; silence is opted into. Set `optional: true` for +hardware that is legitimately not always there — a dock adapter, a radio only +some boards carry. + +`optional` describes **the interface's presence, not the transport's +importance**. An optional interface that is present is used exactly as hard as +any other. No value of it makes a missing interface fatal at startup: the only +fatal case remains "no transports at all came up". + +Either way the edge is logged once, on entering absence and on recovery — +never once per retry. + +The edge itself is not an error. An interface missing when the daemon starts +and bound a moment later is the ordinary boot race this mechanism exists to +absorb, so it is `info`; a runtime detach is `warn`, because a link coming and +going is ordinary weather for a mesh daemon and a cable unplugged for two +seconds does not need a human. + +Ten seconds is the whole grace. Past that it is no longer a race against a +radio or a container coming up, so a **required** interface still missing is +reported once at `error` and stays `Degraded` until it returns. Start-time +absence and a runtime detach share the one deadline — they are the same +transition throughout this mechanism. An `optional` interface never reaches +`error`; that is what `optional` means. + +Said once, not repeated. How long the absence has lasted is a *state*, and it +is published as one: `interface.since_secs` in `show_transports`, and +`Degraded` for as long as it holds. Re-announcing it on a timer would put a +second, lossier copy of that in the log. + +Node health does not wait for the ten seconds — `Degraded` is published on the +first edge, and that is the signal to watch. + +If a binding keeps dying moments after it is established (a socket that errors +persistently while the interface stays up), the binder stops treating each +bind as a recovery: it backs off on the same 1 s → 30 s curve, holds node +health at its degraded reading, and stays quiet until a binding survives ten +seconds. Without that damping a broken socket produces a health flap and a log +pair every second, which is the same cry-wolf failure the edge-only logging +rule exists to prevent. + +A bind failure that is **not** absence — no `CAP_NET_RAW`, no readable +`/dev/bpf*`, a buffer the kernel refused — is a fault, not a state, and fails +the daemon's start as it always has. Only a missing interface is waited out. + +`fipsctl show transports` reports the current state per transport under +`interface`: `presence` (`absent` / `binding` / `present`), `carrier`, +`policy` (`required` / `optional`), `since_secs`, `binds`, and +`failed_attempts`. **Named instances.** Multiple Ethernet interfaces can be configured by using named sub-keys instead of flat parameters: diff --git a/docs/reference/control-socket.md b/docs/reference/control-socket.md index ef51b06b..f4a71943 100644 --- a/docs/reference/control-socket.md +++ b/docs/reference/control-socket.md @@ -124,7 +124,7 @@ table below lists every command currently registered. | `show_mmp` | — | `peers[]` (link-layer per peer), `sessions[]` (session-layer per session). Each entry includes loss/RTT/ETX/goodput, smoothed values, trends. | | `show_cache` | — | `count`, `max_entries`, `fill_ratio`, `default_ttl_ms`, `expired`, `avg_age_ms`, `entries[]` — per-destination coords, depth, age, last-used, optional `path_mtu`. | | `show_connections` | — | `connections[]` — pending handshakes: `link_id`, `direction`, `handshake_state`, `started_at_ms`, `idle_ms`, `resend_count`, optional `expected_peer`. | -| `show_transports` | — | `transports[]` — `transport_id`, `type`, `state`, `mtu`, `name`, `local_addr`, optional `tor_mode`, `onion_address`, `tor_monitoring`, `stats`. | +| `show_transports` | — | `transports[]` — `transport_id`, `type`, `state`, `mtu`, `name`, `local_addr`, optional `tor_mode`, `onion_address`, `tor_monitoring`, `stats`, and `interface` for interface-bound transports. `interface` carries `name` (the configured netdev), `presence` (`absent` / `binding` / `present`), `carrier` (whether the link has `IFF_RUNNING` — reported only, never acted on: presence is `IFF_UP`, so a bound interface with no carrier is normal), `policy` (`required` / `optional`), `since_secs` (how long the current presence phase has been held), `binds` (successful binds since the transport was created — `1` after a clean start, more means it has rebound) and `failed_attempts` (failed binds since the last success). Absent entirely for transports that are not bound to a named interface, rather than reported as a permanently-`present` interface named `""`. Note that `state` describes the *transport* (`up` once started) and `interface.presence` describes the *socket*: an `up` transport whose interface is `absent` is started and waiting, which is a normal state and not a failure. | | `show_routing` | — | `coord_cache_entries`, `identity_cache_entries`, `pending_lookups[]`, `pending_tun_destinations`, `pending_tun_packets`, `recent_requests`, `retries[]`, `forwarding`, `discovery` (request/response sub-counters; includes `req_deduplicated` — requests suppressed as recent duplicates — and `req_dedup_cache_full` — requests admitted because the dedup cache was full), `error_signals`, `congestion`. | | `show_identity_cache` | — | `entries[]`, `count`, `max_entries`. Each entry: `node_addr`, `npub`, `display_name`, `ipv6_addr`, `last_seen_ms`, `age_ms`. | | `show_native_flows` | — | `flows[]`, `listeners[]`, `stats` (the `native` counter family). Each flow: `flow_id`, `peer` (the peer's npub, which is its address; always present, because the flow carries the key its client named or its session authenticated), `peer_addr` (the 16-byte node address in hex — a truncated hash of the same key, kept because it is what `show_sessions` and `show_routing` key on), `local_port`, `remote_port`, `state` (`established` / `pending_accept`), `queued` (datagrams the node is holding for the flow), `age_ms` (time since the flow reached its current state: opened for a flow this node opened, accepted for one taken off a listener, announced for one still pending — accepting a pending flow restarts the clock). Each listener: `local_port`, `backlog`. | diff --git a/docs/tutorials/ground-up-mesh.md b/docs/tutorials/ground-up-mesh.md index 38047758..4c245560 100644 --- a/docs/tutorials/ground-up-mesh.md +++ b/docs/tutorials/ground-up-mesh.md @@ -223,6 +223,7 @@ is "all four flags on both ends." > accept_connections: true > dongle: > interface: "enx00aabbccddee" +> optional: true > announce: true > # ... > ``` @@ -231,6 +232,16 @@ is "all four flags on both ends." > A single ground-up link only needs the flat form shown > first; named instances become useful when the same node > bridges multiple physical segments. +> +> `optional: true` on the dongle says its absence is normal — +> a USB adapter that is plugged in some days and not others. +> Without it, naming an interface is a statement that you +> expect it, and while it is missing the node reports +> `Degraded` and logs at `error`. Either way the interface +> does not have to exist when the daemon starts: a transport +> whose interface is missing waits and binds when it appears, +> and rebinds if it later goes away. Watch that with +> `fipsctl show transports`. ## Step 3: Grant the daemon permission to open raw sockets diff --git a/src/config/mod.rs b/src/config/mod.rs index bef599ed..b61bf975 100644 --- a/src/config/mod.rs +++ b/src/config/mod.rs @@ -1042,6 +1042,68 @@ impl Config { /// Validate cross-field configuration invariants. pub fn validate(&self) -> Result<(), ConfigError> { + self.validate_ethernet_interfaces()?; + self.validate_rendezvous() + } + + /// Reject interface names no kernel could ever hand back. + /// + /// The presence machine deliberately cannot tell a typo from an interface + /// that has not been created yet — both are simply absent, and waiting is + /// the right answer for the second. That is what makes this check worth + /// having: a name that is *impossible* is the one case still separable + /// from "not there yet", and without it a typo costs a permanently + /// `Degraded` node whose only symptom is an interface that never arrives. + /// + /// Syntax only. Whether a well-formed name exists is the binder's + /// question, asked once a second, forever. + fn validate_ethernet_interfaces(&self) -> Result<(), ConfigError> { + // Kernel limit: `IFNAMSIZ` is 16 including the terminating NUL, on + // both Linux and the BSDs. + const MAX_INTERFACE_NAME: usize = 15; + + let mut seen: std::collections::HashMap<&str, &str> = std::collections::HashMap::new(); + + for (name, cfg) in self.transports.ethernet.iter() { + let label = name.unwrap_or("ethernet"); + let iface = cfg.interface.as_str(); + + if iface.is_empty() { + return Err(ConfigError::Validation(format!( + "transport `{label}` has an empty `interface`" + ))); + } + if iface.len() > MAX_INTERFACE_NAME { + return Err(ConfigError::Validation(format!( + "transport `{label}` interface `{iface}` is {} bytes; \ + the kernel limit is {MAX_INTERFACE_NAME}, so no such \ + interface can exist", + iface.len() + ))); + } + if iface.contains('/') || iface.chars().any(char::is_whitespace) { + return Err(ConfigError::Validation(format!( + "transport `{label}` interface `{iface}` contains a \ + character no interface name may hold" + ))); + } + + // Two transports on one netdev means two sockets on the same + // device at the same ethertype, each receiving every frame the + // other does. + if let Some(prior) = seen.insert(iface, label) { + return Err(ConfigError::Validation(format!( + "transports `{prior}` and `{label}` both bind interface \ + `{iface}`" + ))); + } + } + + Ok(()) + } + + /// Cross-checks between transports, peers and the Nostr rendezvous. + fn validate_rendezvous(&self) -> Result<(), ConfigError> { let nostr = &self.node.rendezvous.nostr; let any_transport_advertises_on_nostr = self diff --git a/src/config/transport.rs b/src/config/transport.rs index 6bc3d380..5850fb17 100644 --- a/src/config/transport.rs +++ b/src/config/transport.rs @@ -302,6 +302,25 @@ pub struct EthernetConfig { /// Announcement beacon interval in seconds. Default: 30. #[serde(default, skip_serializing_if = "Option::is_none")] pub beacon_interval_secs: Option, + + /// Whether the absence of this interface is normal. Default: false. + /// + /// Naming an interface in configuration is a statement that you expect it, + /// so the default is to complain: while the interface is missing the node + /// reports `Degraded`, the edge is logged (`info` at startup, `warn` on a + /// runtime detach), and an absence that outlasts the bring-up window — ten + /// seconds, past which it is no longer a race against a radio or a + /// container — is reported once at `error`. Set `optional: true` for + /// hardware that is legitimately not always there — a dock adapter, a + /// radio that only some boards carry — and its absence becomes silent + /// (`info` on the edge, no health impact, no error). + /// + /// This describes **the interface's presence, not the transport's + /// importance**. An optional interface that is present is used exactly as + /// hard as any other; setting it does not deprioritize the transport, and + /// no value of this field makes a missing interface fatal at startup. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub optional: Option, } impl EthernetConfig { @@ -340,6 +359,11 @@ impl EthernetConfig { self.accept_connections.unwrap_or(false) } + /// Whether absence of the interface is normal. Default: false. + pub fn optional(&self) -> bool { + self.optional.unwrap_or(false) + } + /// Get the beacon interval, clamped to minimum. Default: 30s. pub fn beacon_interval_secs(&self) -> u64 { self.beacon_interval_secs @@ -1071,4 +1095,40 @@ mod tests { serde_yaml::from_str("interface: eth0\nbogus: true\n"); assert!(bogus.is_err()); } + + #[test] + fn ethernet_absence_is_an_error_unless_opted_out() { + // Naming an interface is a statement that you expect it, so the + // default has to be the loud one. A default of `true` here would make + // every missing interface silent, which is the failure mode the + // whole presence mechanism exists to stop hiding. + let bare: EthernetConfig = serde_yaml::from_str("interface: eth0\n").unwrap(); + assert_eq!(bare.optional, None); + assert!(!bare.optional(), "absence must default to an error"); + + let opted: EthernetConfig = + serde_yaml::from_str("interface: enx00e04c680001\noptional: true\n").unwrap(); + assert!(opted.optional()); + + let explicit: EthernetConfig = + serde_yaml::from_str("interface: eth0\noptional: false\n").unwrap(); + assert!(!explicit.optional()); + } + + #[test] + fn ethernet_optional_survives_a_round_trip() { + // The packaged OpenWrt config ships `optional: true` on the mesh and + // access blocks; a serializer that dropped it would silently turn + // every stock router Degraded. + let opted: EthernetConfig = + serde_yaml::from_str("interface: fips-mesh0\noptional: true\n").unwrap(); + let round: EthernetConfig = + serde_yaml::from_str(&serde_yaml::to_string(&opted).unwrap()).unwrap(); + assert!(round.optional()); + + // ... and the default stays absent from the output rather than being + // written back as an explicit `false`. + let bare: EthernetConfig = serde_yaml::from_str("interface: eth0\n").unwrap(); + assert!(!serde_yaml::to_string(&bare).unwrap().contains("optional")); + } } diff --git a/src/control/queries.rs b/src/control/queries.rs index 83fe847a..2bdccd36 100644 --- a/src/control/queries.rs +++ b/src/control/queries.rs @@ -1361,8 +1361,17 @@ pub(crate) fn show_connections_from_handle( /// `show_transports` — Transport instances. pub fn show_transports(node: &Node) -> Value { - let transports: Vec = node - .transport_ids() + // Ascending id, which is creation order: UDP, then Ethernet, then TCP, + // Tor, Nym, BLE, with each type's instances in config order. The map + // behind `transport_ids` is a `HashMap`, so without this the array order + // is whatever the hash seed produced — arbitrary, and different on every + // daemon restart. Anything scripting against this output, and every view + // rendering it, inherits that. Sorting by id groups the list by transport + // type for free, because the ids were handed out that way. + let mut ids: Vec<_> = node.transport_ids().copied().collect(); + ids.sort_by_key(|id| id.as_u32()); + let transports: Vec = ids + .iter() .map(|id| { let handle = node.get_transport(id).unwrap(); let mut t_json = json!({ @@ -1390,6 +1399,21 @@ pub fn show_transports(node: &Node) -> Value { t_json["tor_monitoring"] = serde_json::to_value(&monitoring).unwrap_or_default(); } + // Interface presence, for the transports that have an interface. + // Absent from the payload entirely for the ones that do not, rather + // than reported as a permanently-`present` interface named "". + if let Some(p) = handle.interface_presence() { + t_json["interface"] = json!({ + "name": handle.interface_name().unwrap_or_default(), + "presence": p.presence, + "carrier": p.carrier, + "policy": p.policy, + "since_secs": p.since_secs, + "binds": p.binds, + "failed_attempts": p.failed_attempts, + }); + } + t_json["stats"] = handle.transport_stats(); t_json @@ -1434,6 +1458,18 @@ pub(crate) fn show_transports_from_handle(handle: &super::read_handle::ControlRe t_json["tor_monitoring"] = monitoring.clone(); } + if let Some(iface) = &t.interface { + t_json["interface"] = json!({ + "name": iface.name, + "presence": iface.presence, + "carrier": iface.carrier, + "policy": iface.policy, + "since_secs": iface.since_secs, + "binds": iface.binds, + "failed_attempts": iface.failed_attempts, + }); + } + t_json["stats"] = t.stats.clone(); t_json diff --git a/src/control/snapshot.rs b/src/control/snapshot.rs index bf7c4074..a5e7dec6 100644 --- a/src/control/snapshot.rs +++ b/src/control/snapshot.rs @@ -778,6 +778,21 @@ pub(crate) struct TransportRow { pub onion_address: Option, pub tor_monitoring: Option, pub stats: serde_json::Value, + /// Interface presence for interface-bound transports; `None` for the rest. + pub interface: Option, +} + +/// Interface name, presence and policy for an interface-bound transport, as +/// `show_transports` renders it. +#[derive(Clone, PartialEq)] +pub(crate) struct InterfaceRow { + pub name: String, + pub presence: &'static str, + pub carrier: bool, + pub policy: &'static str, + pub since_secs: u64, + pub binds: u64, + pub failed_attempts: u32, } /// MMP trend labels for a peer's link-layer block in `show_mmp` (each present diff --git a/src/node/dataplane/rx_loop.rs b/src/node/dataplane/rx_loop.rs index 4fec454d..c81b31ab 100644 --- a/src/node/dataplane/rx_loop.rs +++ b/src/node/dataplane/rx_loop.rs @@ -109,6 +109,16 @@ impl Node { } }; + // Interface-presence receiver, or a dummy channel — same pattern and + // same reason as the child-liveness receiver above. + let (mut presence_rx, _presence_guard) = match self.transport_presence_rx.take() { + Some(rx) => (rx, None), + None => { + let (tx, rx) = tokio::sync::mpsc::channel(1); + (rx, Some(tx)) + } + }; + let tick_period = Duration::from_secs(self.config().node.tick_interval_secs); let mut tick = tokio::time::interval(tick_period); @@ -346,6 +356,49 @@ impl Node { self.supervisor.state = ns; } } + // A transport child exiting leaves the bound set, so + // it can be the one that was holding the node's egress + // MTU down. `is_bound()` is `is_operational()` plus the + // presence refinement, and this moves the first half. + self.refresh_tun_mss_ceiling(); + } + } + // Interface presence. An interface-bound transport's binder + // reports attach and detach; the FSM folds it into health. + // Unlike `ChildExited` this is reversible in both directions — + // the interface coming back republishes `Running` — which is + // the whole point of `Degraded` being a level rather than a + // latch. + maybe_presence = presence_rx.recv() => { + if let Some(edge) = maybe_presence { + // Health is policy-filtered; the MTU floor below is + // not. An `optional` interface's absence is normal and + // must not move the node off `Full`, but it changes + // the bound set all the same. + if edge.health_relevant { + let child = crate::node::lifecycle::supervisor::Child::Transport( + edge.transport_id, + ); + let event = if edge.present { + crate::node::lifecycle::supervisor::Event::ChildPresent { child } + } else { + crate::node::lifecycle::supervisor::Event::ChildAbsent { child } + }; + let actions = self.supervisor.fsm.step(event); + for action in actions { + if let crate::node::lifecycle::supervisor::Action::PublishState( + ns, + ) = action + { + self.supervisor.state = ns; + } + } + } + // The bound set just changed, so the node's egress MTU + // floor may have. Both directions: an interface that + // binds can be the narrow one, and one that detaches + // can be the reason the clamp was tight. + self.refresh_tun_mss_ceiling(); } } Some(ipv6_packet) = tun_outbound_rx.recv() => { diff --git a/src/node/lifecycle/mod.rs b/src/node/lifecycle/mod.rs index 41a3a67f..586e3de6 100644 --- a/src/node/lifecycle/mod.rs +++ b/src/node/lifecycle/mod.rs @@ -1381,6 +1381,17 @@ impl Node { self.child_exit_tx = Some(child_exit_tx); self.child_exit_rx = Some(child_exit_rx); + // Interface-presence channel. Created before `create_transports` so + // every interface-bound transport gets the sender at construction and + // its very first bind attempt — the one `start_async` makes inline — + // is already reportable. A boot race therefore reaches the FSM while + // it is still `Starting`, and start-completion health resolves to + // `Degraded` on the first publish rather than publishing `Full` and + // correcting it a moment later. + let (presence_tx, presence_rx) = tokio::sync::mpsc::channel(16); + self.transport_presence_tx = Some(presence_tx); + self.transport_presence_rx = Some(presence_rx); + // Initialize transports first (before TUN, before Nostr discovery). // Creation allocates each transport's id; the supervisor FSM authors // the start order over those ids. @@ -1642,12 +1653,20 @@ impl Node { info!(" address: {}", device.address()); info!(" mtu: {}", mtu); - // Calculate max MSS for TCP clamping + // Seed the shared MSS ceiling from whatever is bound + // right now. Both TUN threads read it live from here + // on, so a transport binding or unbinding later moves + // the clamp instead of leaving it at this instant's + // value — see `crate::upper::tun::MssCeiling`. + self.refresh_tun_mss_ceiling(); + let max_mss = self.tun_mss_ceiling.clone(); let effective_mtu = self.effective_ipv6_mtu(); - let max_mss = effective_mtu.saturating_sub(40).saturating_sub(20); // IPv6 + TCP headers info!("effective MTU: {} bytes", effective_mtu); - debug!(" max TCP MSS: {} bytes", max_mss); + debug!( + " max TCP MSS: {} bytes", + max_mss.load(std::sync::atomic::Ordering::Relaxed) + ); // On macOS and FreeBSD, create a shutdown pipe. Writing to it // unblocks the reader thread's select() loop without closing @@ -1671,8 +1690,8 @@ impl Node { // Create writer (dups the fd for independent write access). // Pass path_mtu_lookup so inbound SYN-ACK clamp can read // per-destination path MTU learned via discovery. - let (writer, tun_tx) = - device.create_writer(max_mss, self.path_mtu_lookup.clone())?; + let (writer, tun_tx) = device + .create_writer(max_mss.clone(), self.path_mtu_lookup.clone())?; // Spawn writer thread. On exit it self-reports // `Child::Tun` (sync context → `blocking_send`); TUN @@ -1698,7 +1717,6 @@ impl Node { // self-reports `Child::Tun` on exit (sync context → // `blocking_send`). Exactly one cfg variant compiles, // so the single clone is moved into that closure. - let transport_mtu = self.transport_mtu(); let path_mtu_lookup = self.path_mtu_lookup.clone(); let reader_child_tx = self.child_exit_tx.clone(); #[cfg(any(target_os = "macos", target_os = "freebsd"))] @@ -1709,7 +1727,7 @@ impl Node { our_addr, reader_tun_tx, outbound_tx, - transport_mtu, + max_mss, path_mtu_lookup, shutdown_read_fd, ); @@ -1725,7 +1743,7 @@ impl Node { our_addr, reader_tun_tx, outbound_tx, - transport_mtu, + max_mss, path_mtu_lookup, ); if let Some(tx) = &reader_child_tx { @@ -1856,6 +1874,16 @@ impl Node { } }; + // Drain any presence edges this child's start produced *before* + // reporting the child itself. An interface-bound transport whose + // interface is missing reports absence from inside `start_async` + // and then reports `SubstrateUp` (absence is a state, not a start + // failure), so ordering the drain first means start-completion + // health already knows about the absence when `pending` empties. + // Otherwise a boot race publishes `Full` and corrects itself a + // moment later, and every consumer sees a spurious transition. + let _ = self.drain_transport_presence(); + let feedback_actions = self.supervisor.fsm.step(feedback); for action in &feedback_actions { if let Action::PublishState(ns) = action { @@ -1864,6 +1892,12 @@ impl Node { } } + // Late edges: a transport that bound after its `SubstrateUp` was + // reported, or one that detached during a later child's bring-up. + if let Some(ns) = self.drain_transport_presence() { + start_outcome = Some(ns); + } + // Seams that never triggered inside the loop: the "Transports // initialized" info! when there was no non-transport child, and the // peer-connect when there was no Tun/Dns child (today it still runs, @@ -1902,8 +1936,10 @@ impl Node { // children. Enumerate them for the operator, then proceed — // a degraded node serves traffic. warn!( - degraded_children = ?self.supervisor.fsm.failed(), - "Node started DEGRADED: one or more configured optional children failed to start" + degraded_children = ?self.supervisor.fsm.degraded_children(), + absent_interfaces = ?self.supervisor.fsm.absent(), + "Node started DEGRADED: one or more configured optional children failed to \ + start, or a configured interface is absent" ); } _ => {} @@ -2262,6 +2298,50 @@ impl Node { } } + /// Feed every queued interface-presence edge to the supervisor FSM, + /// returning the last [`NodeState`] it asked to publish (if any). + /// + /// Non-blocking: it drains what is already queued and returns. Used during + /// bring-up, where the rx_loop's presence arm is not running yet — from + /// then on that arm owns the same translation. + pub(in crate::node) fn drain_transport_presence(&mut self) -> Option { + let mut edges = Vec::new(); + if let Some(rx) = self.transport_presence_rx.as_mut() { + while let Ok(edge) = rx.try_recv() { + edges.push(edge); + } + } + + let mut published = None; + let saw_edge = !edges.is_empty(); + for edge in edges { + // Health is policy-filtered; the MTU refresh below is not. See + // `saw_edge`. + if !edge.health_relevant { + continue; + } + let child = Child::Transport(edge.transport_id); + let event = if edge.present { + Event::ChildPresent { child } + } else { + Event::ChildAbsent { child } + }; + for action in self.supervisor.fsm.step(event) { + if let Action::PublishState(ns) = action { + published = Some(ns); + } + } + } + if saw_edge { + // A bind or unbind changes which transports are bound, and so the + // node's egress MTU floor. During bring-up this runs before the + // TUN threads exist, which is exactly when it must: they read the + // ceiling this leaves behind. + self.refresh_tun_mss_ceiling(); + } + published + } + /// Reconstruct the supervised up-set from observed runtime presence, so the /// FSM authors the teardown order regardless of how the node reached /// `Running`. Worker pools are deliberately excluded: today's teardown never diff --git a/src/node/lifecycle/supervisor.rs b/src/node/lifecycle/supervisor.rs index 31a3bb76..af960219 100644 --- a/src/node/lifecycle/supervisor.rs +++ b/src/node/lifecycle/supervisor.rs @@ -67,6 +67,39 @@ //! when a task/thread dies at runtime) is **deferred**: start-completion health //! resolution is start-framed, and liveness monitoring is a substantial unbuilt //! mechanism. This commit is start-time health only. +//! +//! ## Scope: interface presence, and `Degraded` as a level (this commit) +//! +//! Interface-bound transports are now *sometimes bound*: a transport whose +//! interface is missing at start comes up [`Absent`] and binds later, and one +//! whose interface goes away at runtime unbinds and rebinds when it returns. +//! Two things follow for this machine. +//! +//! - **A second reason set.** [`Event::ChildAbsent`] / [`Event::ChildPresent`] +//! move a child in and out of `absent`, which feeds `Degraded` exactly like +//! `failed` does. It is kept separate because it is *reversible* and `failed` +//! is not: a child that failed to start stays failed for the bring-up, while +//! an absent interface is expected to come back. +//! - **`Degraded` is a level, not a latch.** Nothing ever removed from `failed`, +//! which was correct while no child could recover — a monotonic set accurately +//! described a one-way door. Once recovery exists the assumption inverts: plug +//! the WAN back in and the node would stay `Degraded` until the process +//! restarted, and `Degraded` would come to mean "something broke at some point +//! since boot" rather than "something is broken now". +//! [`Self::classify_health`](SupervisorFsm::classify_health) is therefore +//! recomputed on every transition **in both directions**. +//! +//! An absent transport still counts as *up*. It came up — `start_async` returns +//! `Ok` with the transport absent — so it does not push a single-transport node +//! into the fatal [`FailReason::NoTransports`], which would make a node that +//! merely booted before its wifi exit instead of waiting. Absence degrades; it +//! never kills. +//! +//! There is deliberately **no restart action**. Rebinding is owned by the +//! transport's own binder task, which is where the file descriptor and the +//! presence watcher live; the FSM is told what happened and republishes health. +//! +//! [`Absent`]: crate::transport::ethernet::Presence::Absent use std::collections::HashSet; use std::sync::Arc; @@ -174,6 +207,22 @@ pub(crate) enum Event { /// The child whose task/thread exited. child: Child, }, + /// An interface-bound child lost its interface — it was never there at + /// start, or it went away at runtime. The child stays *up* (the transport + /// object survives detach, only its socket goes) but contributes + /// `Degraded`. Valid while `Running`; while `Starting` the edge is recorded + /// so start-completion health already reflects it. + ChildAbsent { + /// The child whose interface is absent. + child: Child, + }, + /// An interface-bound child's interface came back and it rebound. Clears + /// the absence and republishes health, which is how `Degraded` becomes + /// reversible. + ChildPresent { + /// The child whose interface is present again. + child: Child, + }, } /// A driver-scheduled timer the supervisor can arm. Only the @@ -320,7 +369,16 @@ pub(crate) struct SupervisorFsm { up: HashSet, /// Configured children that failed to start during the current bring-up. /// Feeds the `Degraded` health determination when `pending` empties. + /// + /// One-way within a bring-up: a start failure is not recoverable, so + /// nothing removes from this set until the next `Start`. failed: HashSet, + /// Children that are up but whose network interface is currently absent. + /// + /// Reversible, unlike [`Self::failed`] — that is the whole reason it is a + /// second set rather than more entries in the first. Feeds `Degraded` the + /// same way, and empties as interfaces come back. + absent: HashSet, } impl SupervisorFsm { @@ -330,6 +388,7 @@ impl SupervisorFsm { state: SupState::Created, up: HashSet::new(), failed: HashSet::new(), + absent: HashSet::new(), } } @@ -348,6 +407,7 @@ impl SupervisorFsm { }, up: up.into_iter().collect(), failed: HashSet::new(), + absent: HashSet::new(), } } @@ -357,13 +417,26 @@ impl SupervisorFsm { &self.state } - /// The configured children that failed to start during bring-up. The driver - /// reads this on the `Degraded` start outcome to enumerate the degraded - /// children in an operator-visible `warn!`. + /// The configured children that failed to start during bring-up. Kept for + /// the tests that pin the failure-vs-absence split; the driver reports + /// [`Self::degraded_children`], which is the union of the two. + #[cfg(test)] pub(in crate::node) fn failed(&self) -> &HashSet { &self.failed } + /// Children whose interface is currently absent. + pub(in crate::node) fn absent(&self) -> &HashSet { + &self.absent + } + + /// Every child currently contributing `Degraded` — the ones that failed to + /// start plus the ones whose interface is away. This is what an operator + /// wants named when the node reports `Degraded`. + pub(in crate::node) fn degraded_children(&self) -> HashSet { + self.failed.union(&self.absent).copied().collect() + } + /// Whether the machine is in the bounded-drain window. The driver uses this /// after the rx loop returns to decide between the drain-teardown path and /// the immediate-`stop()` fallback. @@ -398,6 +471,8 @@ impl SupervisorFsm { Event::DrainDeadlineElapsed => self.on_drain_deadline_elapsed(), Event::ChildStopped { child } => self.on_child_stopped(child), Event::ChildExited { child } => self.on_child_exited(child), + Event::ChildAbsent { child } => self.on_child_absent(child), + Event::ChildPresent { child } => self.on_child_present(child), } } @@ -444,6 +519,7 @@ impl SupervisorFsm { self.up.clear(); self.failed.clear(); + self.absent.clear(); // A node with no children at all resolves health immediately. Zero // transports up → `Failed` (this is the behavioral @@ -497,19 +573,30 @@ impl SupervisorFsm { self.classify_health() } - /// Classify health from the current `up` / `failed` sets and set the - /// resulting state, returning the [`NodeState`] the driver should publish. - /// Shared by start-completion ([`Self::resolve_start_health`]) and runtime - /// child-exit ([`Self::on_child_exited`]): + /// Classify health from the current `up` / `failed` / `absent` sets and set + /// the resulting state, returning the [`NodeState`] the driver should + /// publish. Shared by start-completion ([`Self::resolve_start_health`]), + /// runtime child-exit ([`Self::on_child_exited`]) and the presence edges + /// ([`Self::on_child_absent`] / [`Self::on_child_present`]): /// /// - zero transports up → [`SupState::Failed`] / [`NodeState::Failed`]; - /// - ≥1 transport up but some child in `failed` → [`Health::Degraded`] / - /// [`NodeState::Degraded`]; - /// - everything up and nothing failed → [`Health::Full`] / [`NodeState::Running`]. + /// - ≥1 transport up but some child in `failed` or `absent` → + /// [`Health::Degraded`] / [`NodeState::Degraded`]; + /// - everything up, nothing failed, nothing absent → [`Health::Full`] / + /// [`NodeState::Running`]. /// /// Worker-pool failures are captured in `failed` like any other optional /// child, so they contribute `Degraded` (never `Failed`) — the inline crypto /// fallback keeps the node correct without the pools. + /// + /// A transport whose interface is absent is still counted among + /// `transports_up`: it came up, it is simply not bound. Excluding it would + /// make a single-ethernet node that booted before its wifi resolve to the + /// fatal [`FailReason::NoTransports`] and exit — which is the failure this + /// whole mechanism exists to remove. Absence degrades; it never kills. + /// + /// Recomputed in **both** directions. This function is the reason + /// `Degraded` is a level rather than a latch. fn classify_health(&mut self) -> NodeState { let transports_up = self .up @@ -521,10 +608,10 @@ impl SupervisorFsm { reason: FailReason::NoTransports, }; NodeState::Failed - } else if !self.failed.is_empty() { + } else if !self.failed.is_empty() || !self.absent.is_empty() { self.state = SupState::Running { health: Health::Degraded { - reasons: self.failed.clone(), + reasons: self.degraded_children(), }, }; NodeState::Degraded @@ -536,6 +623,53 @@ impl SupervisorFsm { } } + /// An interface-bound child lost its interface. + /// + /// The child stays in the up-set: the transport object survives detach — + /// config, id, statistics and neighbor buffer persist, only the descriptor + /// and its loops go. Only the reason set changes. + /// + /// While `Starting` the edge is recorded silently; start-completion health + /// picks it up when `pending` empties, so a boot race resolves to + /// `Degraded` on the first publish rather than publishing `Full` and + /// immediately correcting it. Inert outside `Starting` / `Running`: a + /// teardown in flight owns its own bookkeeping. + fn on_child_absent(&mut self, child: Child) -> Vec { + match self.state { + SupState::Starting { .. } => { + self.absent.insert(child); + Vec::new() + } + SupState::Running { .. } => { + if !self.absent.insert(child) { + return Vec::new(); + } + vec![Action::PublishState(self.classify_health())] + } + _ => Vec::new(), + } + } + + /// An interface-bound child's interface came back. + /// + /// Clears the absence and republishes. A child that is not currently + /// recorded absent produces nothing — a duplicate edge is not an event. + fn on_child_present(&mut self, child: Child) -> Vec { + match self.state { + SupState::Starting { .. } => { + self.absent.remove(&child); + Vec::new() + } + SupState::Running { .. } => { + if !self.absent.remove(&child) { + return Vec::new(); + } + vec![Action::PublishState(self.classify_health())] + } + _ => Vec::new(), + } + } + fn on_stop(&mut self) -> Vec { if !matches!(self.state, SupState::Running { .. }) { return Vec::new(); @@ -588,6 +722,7 @@ impl SupervisorFsm { if let SupState::Stopping { pending } = &mut self.state { pending.remove(&child); self.up.remove(&child); + self.absent.remove(&child); if pending.is_empty() { self.state = SupState::Stopped; } @@ -617,6 +752,8 @@ impl SupervisorFsm { if !self.up.remove(&child) { return Vec::new(); } + // An exit supersedes an absence: the child is gone, not waiting. + self.absent.remove(&child); self.failed.insert(child); vec![Action::PublishState(self.classify_health())] } @@ -1393,4 +1530,289 @@ mod tests { ); assert!(s.failed().is_empty()); } + + // ── Interface presence ──────────────────────────────────────────────── + // + // The absence set is separate from `failed` because it is reversible, and + // reversibility is what turns `Degraded` from a latch into a level. + + /// Helper: bring a full node up cleanly and leave it `Running{Full}`. + fn running_node() -> SupervisorFsm { + let mut s = SupervisorFsm::new(); + s.step(start_full()); + for child in [ + Child::Transport(tid(1)), + Child::Transport(tid(2)), + Child::EncryptWorkers, + Child::DecryptWorkers, + Child::Nostr, + Child::Mdns, + Child::Tun, + Child::Dns, + ] { + s.step(Event::SubstrateUp { child }); + } + assert_eq!( + s.state(), + &SupState::Running { + health: Health::Full + } + ); + s + } + + #[test] + fn an_absent_interface_degrades_a_running_node() { + let mut s = running_node(); + assert_eq!( + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)) + }), + vec![Action::PublishState(NodeState::Degraded)] + ); + assert!(s.absent().contains(&Child::Transport(tid(1)))); + // Absence is not failure: the two sets stay distinct. + assert!(s.failed().is_empty()); + } + + #[test] + fn a_returning_interface_clears_degraded() { + // The regression this whole design turns on: plug the WAN back in and + // the node must leave `Degraded`, not carry it until the process + // restarts. `Degraded` is a level, not a latch. + let mut s = running_node(); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)), + }); + assert_eq!( + s.step(Event::ChildPresent { + child: Child::Transport(tid(1)) + }), + vec![Action::PublishState(NodeState::Running)] + ); + assert_eq!( + s.state(), + &SupState::Running { + health: Health::Full + } + ); + assert!(s.absent().is_empty()); + } + + #[test] + fn health_reflects_the_last_interface_still_away() { + // Two interfaces away, one returns: still Degraded. Only the empty + // absence set publishes Full. + let mut s = running_node(); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)), + }); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(2)), + }); + assert_eq!( + s.step(Event::ChildPresent { + child: Child::Transport(tid(1)) + }), + vec![Action::PublishState(NodeState::Degraded)] + ); + assert_eq!( + s.step(Event::ChildPresent { + child: Child::Transport(tid(2)) + }), + vec![Action::PublishState(NodeState::Running)] + ); + } + + #[test] + fn a_duplicate_presence_edge_is_not_an_event() { + let mut s = running_node(); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)), + }); + assert_eq!( + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)) + }), + vec![], + "re-reporting the same absence must not republish" + ); + s.step(Event::ChildPresent { + child: Child::Transport(tid(1)), + }); + assert_eq!( + s.step(Event::ChildPresent { + child: Child::Transport(tid(1)) + }), + vec![], + "re-reporting the same return must not republish" + ); + } + + #[test] + fn absence_during_bringup_resolves_degraded_on_the_first_publish() { + // The boot race. The transport reports absence from inside its own + // start, then reports up. Start-completion health must already know, + // so the node never publishes `Full` and corrects itself. + let mut s = SupervisorFsm::new(); + s.step(start_full()); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)), + }); + for child in [ + Child::Transport(tid(1)), + Child::Transport(tid(2)), + Child::EncryptWorkers, + Child::DecryptWorkers, + Child::Nostr, + Child::Mdns, + Child::Tun, + ] { + assert_eq!(s.step(Event::SubstrateUp { child }), vec![]); + } + assert_eq!( + s.step(Event::SubstrateUp { child: Child::Dns }), + vec![Action::PublishState(NodeState::Degraded)] + ); + let mut reasons = HashSet::new(); + reasons.insert(Child::Transport(tid(1))); + assert_eq!( + s.state(), + &SupState::Running { + health: Health::Degraded { reasons } + } + ); + } + + #[test] + fn a_lone_absent_transport_degrades_rather_than_fails() { + // A single-ethernet node that boots before its wifi. The transport + // came up — absence is a state, not a start failure — so it counts + // among the transports up and the node serves. Resolving to `Failed` + // here would make the daemon exit on the very race this mechanism + // exists to survive. + let mut s = SupervisorFsm::new(); + s.step(Event::Start { + transports: vec![tid(1)], + encrypt_workers: false, + decrypt_workers: false, + nostr: false, + mdns: false, + tun: false, + dns: false, + }); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)), + }); + assert_eq!( + s.step(Event::SubstrateUp { + child: Child::Transport(tid(1)) + }), + vec![Action::PublishState(NodeState::Degraded)] + ); + assert!( + !matches!(s.state(), SupState::Failed { .. }), + "an absent interface must never be the fatal no-transports case" + ); + } + + #[test] + fn an_exit_supersedes_an_absence() { + // A transport that is away and then exits is gone, not waiting: it + // leaves the up-set and moves from `absent` to `failed`, so a later + // spurious `ChildPresent` cannot resurrect it into `Full`. + let mut s = running_node(); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)), + }); + s.step(Event::ChildExited { + child: Child::Transport(tid(1)), + }); + assert!(s.absent().is_empty()); + assert!(s.failed().contains(&Child::Transport(tid(1)))); + assert_eq!( + s.step(Event::ChildPresent { + child: Child::Transport(tid(1)) + }), + vec![] + ); + assert!(matches!( + s.state(), + SupState::Running { + health: Health::Degraded { .. } + } + )); + } + + #[test] + fn presence_edges_are_inert_outside_starting_and_running() { + // Teardown and drain own their own bookkeeping; a late edge from a + // binder that has not noticed the stop must not author actions. + let mut created = SupervisorFsm::new(); + assert_eq!( + created.step(Event::ChildAbsent { + child: Child::Transport(tid(1)) + }), + vec![] + ); + + let mut draining = running_node(); + draining.step(Event::Drain { deadline_ms: 1 }); + assert_eq!( + draining.step(Event::ChildAbsent { + child: Child::Transport(tid(1)) + }), + vec![] + ); + assert!( + draining.is_draining(), + "a presence edge must not end a drain" + ); + } + + #[test] + fn a_new_start_clears_the_previous_absences() { + let mut s = running_node(); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)), + }); + s.step(Event::Stop); + for child in [ + Child::Dns, + Child::Nostr, + Child::Mdns, + Child::Transport(tid(1)), + Child::Transport(tid(2)), + Child::Tun, + ] { + s.step(Event::ChildStopped { child }); + } + s.step(start_full()); + assert!(s.absent().is_empty()); + } + + #[test] + fn degraded_children_is_the_union_of_both_reason_sets() { + let mut s = SupervisorFsm::new(); + s.step(start_full()); + s.step(Event::ChildAbsent { + child: Child::Transport(tid(1)), + }); + s.step(Event::SubstrateFailed { child: Child::Mdns }); + for child in [ + Child::Transport(tid(1)), + Child::Transport(tid(2)), + Child::EncryptWorkers, + Child::DecryptWorkers, + Child::Nostr, + Child::Tun, + Child::Dns, + ] { + s.step(Event::SubstrateUp { child }); + } + let degraded = s.degraded_children(); + assert!(degraded.contains(&Child::Mdns)); + assert!(degraded.contains(&Child::Transport(tid(1)))); + assert_eq!(degraded.len(), 2); + } } diff --git a/src/node/mod.rs b/src/node/mod.rs index adc7dc84..893bffeb 100644 --- a/src/node/mod.rs +++ b/src/node/mod.rs @@ -375,6 +375,16 @@ pub struct Node { /// SYN/SYN-ACK clamp can use the smaller of the local-egress floor /// and the learned per-destination path MTU. path_mtu_lookup: crate::upper::tun::PathMtuLookup, + /// Node-global TCP MSS ceiling, shared live with the TUN reader and writer + /// threads and recomputed whenever the set of *bound* transports changes. + /// + /// Sits beside `path_mtu_lookup` because it answers the other half of the + /// same question at the same moment: that map supplies the per-destination + /// ceiling, this supplies the local-egress one, and the clamp takes the + /// smaller. Both have to be read live — a transport that binds after start + /// can be the narrow one, and one that unbinds can be the reason the node + /// was clamped at all. + tun_mss_ceiling: crate::upper::tun::MssCeiling, /// Which transport last supplied a *link seed* into `path_mtu_lookup`, /// per destination. /// @@ -419,6 +429,20 @@ pub struct Node { /// rx_loop select arm that feeds `Event::ChildExited` to the supervisor FSM. child_exit_rx: Option>, + // === Interface Presence Channel === + /// Sender half of the interface-presence channel, cloned into every + /// interface-bound transport so its binder task can report attach and + /// detach. Held on `self` for the rx_loop's lifetime as the keep-alive + /// sender, exactly like [`Self::child_exit_tx`]. + /// + /// Separate from the child-exit channel because presence is *reversible*: + /// an exit is one-way, an interface comes back. + transport_presence_tx: Option, + /// Receiver half of the interface-presence channel, `take()`-en by the + /// rx_loop select arm that feeds `Event::ChildAbsent` / `Event::ChildPresent` + /// to the supervisor FSM. + transport_presence_rx: Option, + // === Per-Peer Control Machines === /// Per-peer lifecycle control FSMs, keyed by the stable `LinkId` that spans /// the handshake→active lifetime. Each machine owns its handshake crypto @@ -814,6 +838,8 @@ impl Node { packet_rx: None, child_exit_tx: None, child_exit_rx: None, + transport_presence_tx: None, + transport_presence_rx: None, peer_machines: HashMap::new(), peer_timers: HashMap::new(), peers: HashMap::new(), @@ -881,6 +907,12 @@ impl Node { peer_acl, host_map, path_mtu_lookup: Arc::new(std::sync::RwLock::new(HashMap::new())), + // Seeded at the IPv6 minimum, which is what `transport_mtu()` + // itself falls back to when nothing is bound. Refreshed before + // the TUN threads start and on every change to the bound set. + tun_mss_ceiling: Arc::new(std::sync::atomic::AtomicU16::new( + crate::upper::icmp::mss_ceiling(crate::upper::tun::IPV6_MIN_MTU), + )), path_mtu_seeded_by: Arc::new(std::sync::RwLock::new(HashMap::new())), #[cfg(unix)] decrypt_registered_sessions: std::collections::HashSet::new(), @@ -979,6 +1011,8 @@ impl Node { packet_rx: None, child_exit_tx: None, child_exit_rx: None, + transport_presence_tx: None, + transport_presence_rx: None, peer_machines: HashMap::new(), peer_timers: HashMap::new(), peers: HashMap::new(), @@ -1043,6 +1077,12 @@ impl Node { peer_acl, host_map, path_mtu_lookup: Arc::new(std::sync::RwLock::new(HashMap::new())), + // Seeded at the IPv6 minimum, which is what `transport_mtu()` + // itself falls back to when nothing is bound. Refreshed before + // the TUN threads start and on every change to the bound set. + tun_mss_ceiling: Arc::new(std::sync::atomic::AtomicU16::new( + crate::upper::icmp::mss_ceiling(crate::upper::tun::IPV6_MIN_MTU), + )), path_mtu_seeded_by: Arc::new(std::sync::RwLock::new(HashMap::new())), #[cfg(unix)] decrypt_registered_sessions: std::collections::HashSet::new(), @@ -1099,6 +1139,11 @@ impl Node { let mut eth = EthernetTransport::new(transport_id, name, eth_config, packet_tx.clone()); eth.set_local_pubkey(xonly); + // The binder task reports attach and detach here, so node + // health tracks the interface in both directions. + if let Some(tx) = self.transport_presence_tx.clone() { + eth.set_presence_tx(tx); + } transports.push(TransportHandle::Ethernet(eth)); } } @@ -1412,6 +1457,46 @@ impl Node { crate::upper::icmp::effective_ipv6_mtu(self.transport_mtu()) } + /// The TCP MSS ceiling the TUN threads are currently clamping to. + #[cfg(test)] + pub(crate) fn tun_mss_ceiling(&self) -> u16 { + self.tun_mss_ceiling + .load(std::sync::atomic::Ordering::Relaxed) + } + + /// Recompute the shared TUN MSS ceiling from the currently bound + /// transports, and log it if it moved. + /// + /// Called wherever the bound set can change — the presence edges that + /// bind and unbind an interface-bound transport, and a child exiting — + /// so the clamp the TUN threads apply keeps agreeing with the + /// `effective_ipv6_mtu` this node reports in `show_status`. + /// + /// Moves in **both** directions, deliberately. A narrow interface + /// appearing has to tighten the ceiling or the clamp is wrong for + /// traffic that will egress over it; that same interface going away has + /// to release it, or unplugging a low-MTU adapter leaves the node + /// over-clamped until it restarts. It is the same argument that makes + /// `Degraded` a level rather than a latch: nothing here is one-way once + /// a transport can come back. + /// + /// Existing flows are not re-clamped — MSS is negotiated per connection + /// at SYN time, so a change applies to connections opened after it. + pub(crate) fn refresh_tun_mss_ceiling(&self) { + use std::sync::atomic::Ordering; + + let ceiling = crate::upper::icmp::mss_ceiling(self.transport_mtu()); + let previous = self.tun_mss_ceiling.swap(ceiling, Ordering::Relaxed); + if previous != ceiling { + tracing::info!( + previous_max_mss = previous, + max_mss = ceiling, + effective_ipv6_mtu = self.effective_ipv6_mtu(), + "Node egress MTU changed; TCP MSS ceiling updated for new connections" + ); + } + } + /// Get the transport MTU governing the global TUN-boundary MSS clamp. /// /// Returns the **minimum** MTU across all operational transports, or @@ -1427,10 +1512,17 @@ impl Node { /// to vary across HashMap iteration order + async-startup race) makes /// the clamp deterministic across daemon restarts. pub fn transport_mtu(&self) -> u16 { + // `is_bound`, not `is_operational`. An interface-bound transport is + // "operational" from the moment it starts, whether or not its + // interface exists — so filtering on that let a transport whose + // interface has never appeared clamp the whole node's IPv6 MTU to a + // number derived from hardware that is not present. Before dynamic + // binding a transport that could not bind was never inserted here at + // all, so the distinction did not exist to get wrong. let min_operational = self .transports .values() - .filter(|h| h.is_operational()) + .filter(|h| h.is_bound()) .map(|h| h.mtu()) .min(); if let Some(mtu) = min_operational { @@ -2336,8 +2428,13 @@ impl Node { .collect(); // --- transports (show_transports) --- - let transport_rows: Vec = self - .transport_ids() + // Ascending id, matching `show_transports`; see the note there. The + // off-loop renderer reads this table verbatim, so the two paths would + // otherwise disagree about ordering as well as being arbitrary. + let mut transport_ids: Vec<_> = self.transport_ids().copied().collect(); + transport_ids.sort_by_key(|id| id.as_u32()); + let transport_rows: Vec = transport_ids + .iter() .map(|id| { let handle = self.get_transport(id).unwrap(); snap::TransportRow { @@ -2353,6 +2450,15 @@ impl Node { .tor_monitoring() .map(|m| serde_json::to_value(&m).unwrap_or_default()), stats: handle.transport_stats(), + interface: handle.interface_presence().map(|p| snap::InterfaceRow { + name: handle.interface_name().unwrap_or_default().to_string(), + presence: p.presence, + carrier: p.carrier, + policy: p.policy, + since_secs: p.since_secs, + binds: p.binds, + failed_attempts: p.failed_attempts, + }), } }) .collect(); diff --git a/src/node/tests/unit.rs b/src/node/tests/unit.rs index 12b0ff7b..92cf8d01 100644 --- a/src/node/tests/unit.rs +++ b/src/node/tests/unit.rs @@ -1842,6 +1842,115 @@ async fn test_transport_mtu_returns_min_across_operational() { } } +#[tokio::test] +async fn the_tun_mss_ceiling_follows_a_transport_arriving_and_leaving() { + // The TUN reader and writer used to be handed a `u16` computed once at + // spawn. Every other consumer of `transport_mtu()` reads it live, so once + // a transport could bind minutes after start the daemon reported one + // effective MTU in `show_status` and clamped to another. + // + // Both directions. A narrow transport arriving has to tighten the ceiling + // or traffic egressing over it is clamped too loose; the same transport + // leaving has to release it, or unplugging a low-MTU adapter leaves the + // node over-clamped until it restarts. + let mut node = make_node(); + let (packet_tx, packet_rx) = packet_channel(64); + node.supervisor.packet_tx = Some(packet_tx); + node.packet_rx = Some(packet_rx); + + let wide = make_udp_transport_with_mtu(1, 1452).await; + node.transports.insert(TransportId::new(1), wide); + node.refresh_tun_mss_ceiling(); + let wide_ceiling = node.tun_mss_ceiling(); + assert_eq!( + wide_ceiling, + crate::upper::icmp::mss_ceiling(1452), + "the seeded ceiling must match the only bound transport" + ); + + // A narrower transport arrives after the TUN threads would already be + // running. The shared ceiling has to tighten. + let narrow = make_udp_transport_with_mtu(2, 1280).await; + node.transports.insert(TransportId::new(2), narrow); + node.refresh_tun_mss_ceiling(); + let narrow_ceiling = node.tun_mss_ceiling(); + assert_eq!(narrow_ceiling, crate::upper::icmp::mss_ceiling(1280)); + assert!( + narrow_ceiling < wide_ceiling, + "a narrower transport must tighten the clamp, not be ignored" + ); + assert_eq!( + narrow_ceiling, + crate::upper::icmp::mss_ceiling(node.transport_mtu()), + "the clamp and the reported MTU must not disagree" + ); + + // ...and leaving has to release it again. + if let Some(mut gone) = node.transports.remove(&TransportId::new(2)) { + gone.stop().await.ok(); + } + node.refresh_tun_mss_ceiling(); + assert_eq!( + node.tun_mss_ceiling(), + wide_ceiling, + "the ceiling must rise again when the narrow transport goes away" + ); + + for transport in node.transports.values_mut() { + transport.stop().await.ok(); + } +} + +#[tokio::test] +async fn a_presence_edge_refreshes_the_tun_mss_ceiling_without_being_asked() { + // The one above pins the arithmetic; this pins the wiring. A presence + // edge arriving has to refresh the ceiling on its own — if the refresh is + // dropped from the edge handlers the value silently stops tracking, which + // is the defect in its original form. + let mut node = make_node(); + let (packet_tx, packet_rx) = packet_channel(64); + node.supervisor.packet_tx = Some(packet_tx); + node.packet_rx = Some(packet_rx); + + let (presence_tx, presence_rx) = tokio::sync::mpsc::channel(4); + node.transport_presence_tx = Some(presence_tx.clone()); + node.transport_presence_rx = Some(presence_rx); + + // Nothing bound: the conservative seed. + node.refresh_tun_mss_ceiling(); + let seeded = node.tun_mss_ceiling(); + assert_eq!(seeded, crate::upper::icmp::mss_ceiling(1280)); + + // A wide transport appears, and an edge announces it. No explicit + // refresh call here — draining the edge is the whole trigger. + let wide = make_udp_transport_with_mtu(1, 1452).await; + node.transports.insert(TransportId::new(1), wide); + presence_tx + .send(crate::transport::TransportPresence { + transport_id: TransportId::new(1), + present: true, + health_relevant: true, + }) + .await + .expect("presence edge queued"); + node.drain_transport_presence(); + + assert_eq!( + node.tun_mss_ceiling(), + crate::upper::icmp::mss_ceiling(1452), + "draining a presence edge must refresh the ceiling on its own" + ); + assert_ne!( + node.tun_mss_ceiling(), + seeded, + "the ceiling stayed at its seed, so the edge did not refresh it" + ); + + for transport in node.transports.values_mut() { + transport.stop().await.ok(); + } +} + #[tokio::test] async fn test_transport_mtu_fallback_when_no_operational_transports() { // No transports configured at all → falls back to 1280 (IPv6 minimum). diff --git a/src/transport/ethernet/io.rs b/src/transport/ethernet/io.rs index c60b37dd..fd7e30df 100644 --- a/src/transport/ethernet/io.rs +++ b/src/transport/ethernet/io.rs @@ -9,6 +9,120 @@ use crate::transport::TransportError; /// Broadcast MAC address. pub const ETHERNET_BROADCAST: [u8; 6] = [0xff; 6]; +/// Whether the named interface exists and is administratively up. +/// +/// Presence is `IFF_UP` — the interface exists and the operator has enabled +/// it — and deliberately **not** `IFF_RUNNING`. +/// +/// Carrier is a different question from bindability, and only the second one +/// belongs in a bind gate. An `AF_PACKET` socket on a carrier-less bridge is +/// perfectly valid and starts carrying traffic the instant a member port comes +/// up, with no rebind: the socket outlives the carrier. Gating on `IFF_RUNNING` +/// bought nothing and cost three things — +/// +/// - `br-lan` on a router with nothing plugged into its LAN ports is `UP` with +/// `NO-CARRIER`, so a perfectly healthy wifi-only router reported `Degraded` +/// forever; +/// - every carrier flap the socket would have survived became an unbind / +/// rebind cycle, which is churn the presence machine then has to damp; +/// - an 802.11s mesh interface that reports `RUNNING` only once it has peered +/// cannot peer, because peering needs beacons, which need a bound socket, +/// which the gate refuses. A deadlock reachable on shipped hardware. +/// +/// The signal `IFF_RUNNING` does carry — "is anything plugged in" — is not +/// lost; it is reported alongside presence by [`interface_carrier`] and +/// surfaced in `show_transports`, where an operator can read it without it +/// steering the daemon. +/// +/// `getifaddrs` rather than an `SIOCGIFFLAGS` ioctl: it needs no socket, so +/// the presence watcher can poll before any file descriptor exists, and it is +/// spelled the same on Linux and the BSDs. +#[cfg(unix)] +pub fn interface_present(interface: &str) -> bool { + interface_present_probe(interface).unwrap_or(false) +} + +/// [`interface_present`], keeping "the probe failed" distinct from "absent". +/// +/// `None` means the kernel would not answer. A caller deciding whether to +/// *bind* can treat that as absence and retry on the next tick, which is what +/// [`interface_present`] does. A caller deciding whether to *unbind* must not: +/// see [`interface_has_flags`]. +#[cfg(unix)] +pub fn interface_present_probe(interface: &str) -> Option { + interface_has_flags(interface, libc::IFF_UP as u32) +} + +/// The kernel's index for the named interface, or `None` if it does not exist. +/// +/// A name is not a device, and neither is a name that is still there. Both +/// backends bind by index — `AF_PACKET` stores `sll_ifindex`, and a BPF +/// descriptor follows the device it was attached to — so an interface deleted +/// and recreated under the same name leaves the socket attached to a device +/// that no longer exists while the *name* resolves perfectly well. Comparing +/// the live index against the one captured at bind is what tells those apart. +#[cfg(unix)] +pub fn interface_index(interface: &str) -> Option { + let c_name = std::ffi::CString::new(interface).ok()?; + // Cheaper than `getifaddrs`: one syscall, no allocation, no walk. + match unsafe { libc::if_nametoindex(c_name.as_ptr()) } { + 0 => None, + idx => Some(idx), + } +} + +/// Whether the named interface currently has carrier (`IFF_RUNNING`). +/// +/// Reported, never acted on — see [`interface_present`]. `false` for an +/// interface that does not exist, which keeps "no carrier" and "no interface" +/// from being told apart here; presence answers that. +#[cfg(unix)] +pub fn interface_carrier(interface: &str) -> bool { + // Report-only, so a probe failure reads the same as no carrier. + interface_has_flags(interface, (libc::IFF_UP | libc::IFF_RUNNING) as u32).unwrap_or(false) +} + +/// Whether the named interface exists and has every flag in `wanted` set, or +/// `None` if the question could not be asked. +/// +/// The `None` matters. `getifaddrs` is a netlink dump on Linux and it does +/// fail for reasons that have nothing to do with the interface — `ENOBUFS` +/// under memory pressure or a busy netlink socket, `EMFILE`/`ENFILE` under fd +/// exhaustion, since it opens a socket of its own. Answering `false` there +/// reports a present interface as gone, and a caller holding a live binding +/// would tear a working socket down over a transient syscall failure. Callers +/// that can tell the two apart should. +#[cfg(unix)] +fn interface_has_flags(interface: &str, wanted: u32) -> Option { + let Ok(c_name) = std::ffi::CString::new(interface) else { + // An interior NUL is not a probe failure — no such interface can + // exist, and no retry will change that. + return Some(false); + }; + + let mut addrs: *mut libc::ifaddrs = std::ptr::null_mut(); + if unsafe { libc::getifaddrs(&mut addrs) } != 0 { + return None; + } + + let mut matched = false; + let mut cur = addrs; + while !cur.is_null() { + let entry = unsafe { &*cur }; + if !entry.ifa_name.is_null() + && unsafe { libc::strcmp(entry.ifa_name, c_name.as_ptr()) } == 0 + && entry.ifa_flags & wanted == wanted + { + matched = true; + break; + } + cur = entry.ifa_next; + } + + unsafe { libc::freeifaddrs(addrs) }; + Some(matched) +} + // Platform-specific PacketSocket implementation. #[cfg(target_os = "linux")] #[path = "io_linux.rs"] @@ -193,6 +307,12 @@ mod async_impl { /// A received frame: (payload, source_mac). type Frame = (Vec, [u8; 6]); + /// Consecutive failed BPF reads before the reader thread gives up. + /// + /// Mirrors the receive loop's own error threshold: the point is not to + /// tolerate errors but to end the task so the binder can rebind. + const READ_ERROR_EXIT_THRESHOLD: u32 = 5; + pub struct AsyncPacketSocket { inner: Arc, /// `None` once shutdown has taken the receiver, which is what makes @@ -219,6 +339,9 @@ mod async_impl { let mut parse_buf = vec![0u8; bpf_buflen]; let mut parse_offset: usize = 0; let mut parse_len: usize = 0; + // Consecutive failed reads, to bound a socket whose + // interface went away underneath it. + let mut read_errors: u32 = 0; let nfds = bpf_fd.max(shutdown_fd) + 1; loop { @@ -286,11 +409,31 @@ mod async_impl { if err.raw_os_error() == Some(libc::EBADF) { break; } + if err.kind() == std::io::ErrorKind::Interrupted { + continue; + } + } + // Anything else — `ENXIO` is the one that matters, + // which is what BPF answers once the interface it + // was attached to is torn away — used to loop here + // forever. That mattered beyond the spin: the + // binder's detach check asks whether this thread is + // still running, so a thread that never returns + // reports a dead socket as a live one, and the + // transport sits `present` and deaf until the name + // or index happens to change too. Give up after a + // streak and let the return close the channel, + // which fails `recv_from`, which ends the tokio + // task the binder is actually watching. + read_errors += 1; + if read_errors >= READ_ERROR_EXIT_THRESHOLD { + break; } parse_len = 0; parse_offset = 0; continue; } + read_errors = 0; parse_len = ret as usize; parse_offset = 0; } diff --git a/src/transport/ethernet/io_linux.rs b/src/transport/ethernet/io_linux.rs index 806e3269..cc0bae0c 100644 --- a/src/transport/ethernet/io_linux.rs +++ b/src/transport/ethernet/io_linux.rs @@ -59,6 +59,14 @@ impl PacketSocket { if ret < 0 { let err = std::io::Error::last_os_error(); unsafe { libc::close(fd) }; + // The interface can disappear between the index lookup and the + // bind. That is absence arriving a few microseconds late, not a + // configuration fault, so it reports as absence. + if matches!(err.raw_os_error(), Some(libc::ENODEV) | Some(libc::ENXIO)) { + return Err(TransportError::InterfaceUnavailable { + interface: interface.to_string(), + }); + } return Err(TransportError::StartFailed(format!( "bind(AF_PACKET, {}) failed: {}", interface, err @@ -231,11 +239,11 @@ fn get_if_index(_fd: RawFd, interface: &str) -> Result { let idx = unsafe { libc::if_nametoindex(c_name.as_ptr()) }; if idx == 0 { - return Err(TransportError::StartFailed(format!( - "interface not found: {} ({})", - interface, - std::io::Error::last_os_error() - ))); + // Absence, not a fault: the caller's presence watcher rebinds when the + // interface shows up. See `TransportError::InterfaceUnavailable`. + return Err(TransportError::InterfaceUnavailable { + interface: interface.to_string(), + }); } Ok(idx as i32) } diff --git a/src/transport/ethernet/io_macos.rs b/src/transport/ethernet/io_macos.rs index 3c8cab1f..10c806ea 100644 --- a/src/transport/ethernet/io_macos.rs +++ b/src/transport/ethernet/io_macos.rs @@ -393,10 +393,18 @@ fn bind_to_interface(fd: RawFd, interface: &str) -> Result<(), TransportError> { let ret = unsafe { libc::ioctl(fd, BIOCSETIF, ifreq.as_ptr()) }; if ret < 0 { + let err = std::io::Error::last_os_error(); + // BIOCSETIF answers ENXIO for an interface that is not there. The + // interface can also vanish between the index lookup and this ioctl, + // so absence is reported as absence rather than as a bind fault. + if matches!(err.raw_os_error(), Some(libc::ENXIO) | Some(libc::ENODEV)) { + return Err(TransportError::InterfaceUnavailable { + interface: interface.to_string(), + }); + } return Err(TransportError::StartFailed(format!( "BIOCSETIF({}) failed: {}", - interface, - std::io::Error::last_os_error() + interface, err ))); } Ok(()) @@ -498,11 +506,11 @@ fn get_if_index(interface: &str) -> Result { let idx = unsafe { libc::if_nametoindex(c_name.as_ptr()) }; if idx == 0 { - return Err(TransportError::StartFailed(format!( - "interface not found: {} ({})", - interface, - std::io::Error::last_os_error() - ))); + // Absence, not a fault — see the Linux twin and + // `TransportError::InterfaceUnavailable`. + return Err(TransportError::InterfaceUnavailable { + interface: interface.to_string(), + }); } Ok(idx as i32) } diff --git a/src/transport/ethernet/mod.rs b/src/transport/ethernet/mod.rs index f134205a..6e459a01 100644 --- a/src/transport/ethernet/mod.rs +++ b/src/transport/ethernet/mod.rs @@ -4,27 +4,225 @@ //! uses AF_PACKET/SOCK_DGRAM sockets; on macOS, uses BPF devices (`/dev/bpf*`). //! Works on wired Ethernet and WiFi interfaces (kernel mac80211 abstracts //! 802.11 transparently on Linux). +//! +//! ## Dynamic interface binding +//! +//! The interface a transport names is not required to exist when the daemon +//! starts. `start_async` binds if it can and otherwise returns `Ok` with the +//! transport [`Absent`](presence::Presence::Absent); a binder task then waits +//! for the interface, binds when it appears, tears down when it goes away, and +//! rebinds when it returns. Start-time absence and runtime detach are the same +//! transition, so a node that boots before wifi and a node whose wifi reloads +//! at 03:00 take one code path. See [`presence`] for the state machine. pub mod addr; pub mod io; pub mod neighbor; +pub mod presence; pub mod stats; +mod watcher; pub use addr::parse_mac_string; +pub use presence::{AbsencePolicy, Presence}; use super::{ - DiscoveredPeer, PacketTx, ReceivedPacket, Transport, TransportAddr, TransportError, - TransportId, TransportState, TransportType, + DiscoveredPeer, PacketTx, PresenceTx, ReceivedPacket, Transport, TransportAddr, TransportError, + TransportId, TransportPresence, TransportState, TransportType, }; use crate::config::EthernetConfig; -use io::{AsyncPacketSocket, ETHERNET_BROADCAST, PacketSocket}; +use io::{ + AsyncPacketSocket, ETHERNET_BROADCAST, PacketSocket, interface_present, interface_present_probe, +}; use neighbor::{FRAME_TYPE_BEACON, FRAME_TYPE_DATA, NeighborBuffer, build_beacon, parse_beacon}; +use presence::{ABSENCE_ERROR_AFTER, ChurnGuard, PresenceState, bind_backoff}; use stats::EthernetStats; +use watcher::LinkWatcher; use secp256k1::XOnlyPublicKey; -use std::sync::Arc; +use std::sync::atomic::{AtomicBool, AtomicU16, AtomicU32, Ordering}; +use std::sync::{Arc, Mutex, PoisonError, RwLock}; +use std::time::{Duration, Instant}; use tokio::task::JoinHandle; -use tracing::{debug, info, trace, warn}; +use tracing::{debug, error, info, trace, warn}; + +/// Presence poll interval for the fallback watcher. +/// +/// Deliberately small. `getifaddrs` plus a flags read is cheap, and letting +/// this drift to tens of seconds reintroduces exactly the boot-race latency +/// the mechanism exists to remove. +const WATCH_INTERVAL: Duration = Duration::from_secs(1); + +/// Floor on how often the presence probe may run. +/// +/// The Linux event source is bound to `RTNLGRP_LINK` and speaks only when a +/// link changes. `PF_ROUTE` has no group filter, so the macOS source delivers +/// every routing message on the box — route churn, ARP, DHCP renewals, a VPN +/// going up and down — and each one would otherwise wake the binder into a +/// full `getifaddrs` walk. Coalescing to ten probes a second keeps detection +/// sub-second while bounding the work an unrelated chatty network can cause. +const MIN_PROBE_INTERVAL: Duration = Duration::from_millis(100); + +/// Consecutive receive errors tolerated before the receive loop gives up and +/// hands the transport back to the binder. +/// +/// The loop used to `warn!` every iteration with no backoff, so a dead +/// descriptor was a hot log spin. Exiting is the correct response: a socket +/// that errors repeatedly is a socket to rebind, not one to keep reading. +const RECV_ERROR_EXIT_THRESHOLD: u32 = 5; + +/// Consecutive beacon send failures tolerated before the beacon sender exits +/// and lets the binder rebind. +/// +/// This replaces the ad-hoc ENXIO socket-reopen that used to live in +/// `beacon_sender_loop` — the only recovery logic in the tree, and in the +/// wrong layer. Recovery is the presence machine's job. +const BEACON_ERROR_EXIT_THRESHOLD: u32 = 3; + +/// The mutable half of a bound interface: the socket and everything derived +/// from it. +/// +/// Split out of [`EthernetTransport`] so the binder task can replace it +/// wholesale on a rebind while `send`, `mtu()` and the control plane keep +/// reading through a shared `Arc`. `None` in `socket` *is* absence. +struct Binding { + /// The live socket, or `None` while absent. + socket: RwLock>>, + /// Effective payload MTU of the current binding. + mtu: AtomicU16, + /// Local MAC of the current binding. + local_mac: RwLock>, + /// Kernel index of the device the current binding is attached to. `0` when + /// unbound. Both backends bind by device, not by name, so this is the + /// identity that has to keep matching — see [`io::interface_index`]. + bound_index: AtomicU32, + /// Receive and beacon tasks belonging to the current binding. + tasks: Mutex>>, +} + +impl Binding { + fn new(default_mtu: u16) -> Self { + Self { + socket: RwLock::new(None), + mtu: AtomicU16::new(default_mtu), + local_mac: RwLock::new(None), + bound_index: AtomicU32::new(0), + tasks: Mutex::new(Vec::new()), + } + } + + fn socket(&self) -> Option> { + self.socket + .read() + .unwrap_or_else(PoisonError::into_inner) + .clone() + } + + fn mtu(&self) -> u16 { + self.mtu.load(Ordering::Relaxed) + } + + fn local_mac(&self) -> Option<[u8; 6]> { + *self + .local_mac + .read() + .unwrap_or_else(PoisonError::into_inner) + } + + /// Whether `interface` still names the device this binding is attached to. + /// + /// A recreated netdev takes a new kernel index, so the old socket is bound + /// to nothing while the name resolves fine. Nothing else notices: the + /// receive loop on a stale `AF_PACKET` socket never becomes readable, so + /// it never errors and never exits, and send failures are reported to the + /// caller rather than to the binder. Without this check a listen-only node + /// — one with `announce: false`, and so no beacon sender to fail — sits + /// `present` and deaf indefinitely after a `wifi reload`. + fn device_replaced(&self, interface: &str) -> bool { + let bound = self.bound_index.load(Ordering::Relaxed); + if bound == 0 { + return false; + } + match io::interface_index(interface) { + Some(current) => current != bound, + // The name is gone; the caller's presence probe reports that more + // precisely, so this is not the check that should claim it. + None => false, + } + } + + /// Whether every task of the current binding is still running. + /// + /// A finished task means the socket underneath it died — a recreated veth + /// handing out ENXIO, a BPF device torn away — even when the interface + /// name is still present. The binder treats that as a detach. + fn tasks_alive(&self) -> bool { + let tasks = self.tasks.lock().unwrap_or_else(PoisonError::into_inner); + !tasks.is_empty() && tasks.iter().all(|h| !h.is_finished()) + } + + /// Drop the socket and stop its loops. Idempotent. + /// + /// Every lock here ignores poisoning. Declining to abort a task because a + /// mutex was poisoned would leak a receive loop per rebind while the + /// binder, seeing no live tasks, rebound once a second forever — a stuck + /// state strictly worse than touching data a panicking thread had + /// written. + fn tear_down(&self) { + if let Some(socket) = self.socket() { + // Wakes the macOS reader thread's `select()`; a no-op on Linux, + // where `AsyncFd` cancellation is enough. + socket.shutdown(); + } + for task in self + .tasks + .lock() + .unwrap_or_else(PoisonError::into_inner) + .drain(..) + { + task.abort(); + } + *self.socket.write().unwrap_or_else(PoisonError::into_inner) = None; + *self + .local_mac + .write() + .unwrap_or_else(PoisonError::into_inner) = None; + // A stale index left here would make the next binding look replaced + // the moment it came up. + self.bound_index.store(0, Ordering::Relaxed); + } +} + +/// Everything the binder task needs to bind, run, and rebind. +/// +/// The transport object survives detach: this context, the neighbor buffer, +/// the statistics and the `TransportId` all persist across an interface going +/// away and coming back. Only the file descriptor and its loops go. +struct BinderContext { + /// Set when the transport is stopping or being dropped. The binder must + /// not publish a new binding past this point. + /// + /// Teardown aborts the binder, but `bind_now` contains no await points, so + /// a cancellation issued while it runs takes effect only at the *next* + /// await — after the socket and task handles have been stored. Without + /// this flag the sequence "tear_down clears an empty binding, binder + /// stores a fresh one, binder is cancelled" leaves a live receive loop on + /// a socket nothing owns. The flag is set before `tear_down`, and checked + /// by the binder *after* it stores, so whichever order the two interleave + /// exactly one of them cleans up. + shutdown: Arc, + transport_id: TransportId, + name: Option, + interface: String, + config: EthernetConfig, + policy: AbsencePolicy, + packet_tx: PacketTx, + neighbor_buffer: Arc, + stats: Arc, + binding: Arc, + presence: Arc, + local_pubkey: Option, + presence_tx: Option, +} /// Ethernet transport for FIPS. /// @@ -40,28 +238,29 @@ pub struct EthernetTransport { config: EthernetConfig, /// Current state. state: TransportState, - /// Async socket (None until started). - socket: Option>, + /// The socket and its derived values, replaced wholesale on each rebind. + binding: Arc, + /// Interface presence, published to the control plane and the supervisor. + presence: Arc, + /// How absence of this interface is reported. + policy: AbsencePolicy, /// Channel for delivering received packets to Node. packet_tx: PacketTx, - /// Receive loop task handle. - recv_task: Option>, - /// Beacon sender task handle. - beacon_task: Option>, - /// Local MAC address (after start). - local_mac: Option<[u8; 6]>, + /// Binder task: binds, watches, and rebinds for the transport's lifetime. + binder_task: Option>, + /// Shared stop flag, honoured by the binder. See [`BinderContext`]. + shutdown: Arc, /// Interface name (from config). interface: String, - /// Effective payload MTU: interface MTU minus 3 bytes of frame header - /// (`[type:1][length:2 LE][payload]`). The 2-byte length field is required - /// to trim NIC minimum-frame padding before AEAD verification. - effective_mtu: u16, /// Neighbor buffer for discovered peers. neighbor_buffer: Arc, /// Transport-level statistics. stats: Arc, /// Node's public key for beacon construction. local_pubkey: Option, + /// Presence edges are published here so node health tracks absence in both + /// directions. + presence_tx: Option, } impl EthernetTransport { @@ -75,22 +274,26 @@ impl EthernetTransport { let interface = config.interface.clone(); let neighbor_buffer = Arc::new(NeighborBuffer::new(transport_id)); let stats = Arc::new(EthernetStats::new()); + let policy = AbsencePolicy::from_optional(config.optional()); Self { transport_id, name, config, state: TransportState::Configured, - socket: None, + // 1499 = the common 1500-byte interface MTU minus the 3-byte frame + // header; replaced by the real value at the first bind. + binding: Arc::new(Binding::new(1499)), + presence: Arc::new(PresenceState::new()), + policy, packet_tx, - recv_task: None, - beacon_task: None, - local_mac: None, + binder_task: None, + shutdown: Arc::new(AtomicBool::new(false)), interface, - effective_mtu: 1499, // default, updated on start neighbor_buffer, stats, local_pubkey: None, + presence_tx: None, } } @@ -104,9 +307,30 @@ impl EthernetTransport { &self.interface } - /// Get the local MAC address (only valid after start). + /// Get the local MAC address (only valid while bound). pub fn local_mac(&self) -> Option<[u8; 6]> { - self.local_mac + self.binding.local_mac() + } + + /// Current interface presence. + pub fn presence(&self) -> Presence { + self.presence.presence() + } + + /// Whether the interface currently has carrier. Reported, never acted on — + /// see [`io::interface_present`]. + pub fn has_carrier(&self) -> bool { + io::interface_carrier(&self.interface) + } + + /// Shared presence state, for the control plane and for tests. + pub fn presence_state(&self) -> &Arc { + &self.presence + } + + /// How absence of this interface is reported. + pub fn absence_policy(&self) -> AbsencePolicy { + self.policy } /// Set the node's public key for beacon construction. @@ -116,6 +340,14 @@ impl EthernetTransport { self.local_pubkey = Some(pubkey); } + /// Install the channel presence edges are published on. + /// + /// Must be called before start. Without it the transport still binds and + /// rebinds; only the health reporting is lost. + pub fn set_presence_tx(&mut self, tx: PresenceTx) { + self.presence_tx = Some(tx); + } + /// Get a reference to the statistics. pub fn stats(&self) -> &Arc { &self.stats @@ -123,8 +355,11 @@ impl EthernetTransport { /// Start the transport asynchronously. /// - /// Creates the AF_PACKET socket, spawns the receive loop, and - /// optionally spawns the beacon sender task. + /// Binds the interface if it is present, then hands the transport to a + /// binder task that owns every later bind and unbind. **Returns `Ok` when + /// the interface is absent**: absence is a state the transport tracks, not + /// a start failure. A node that boots before its wifi comes up is degraded + /// (or, for an `optional` interface, unremarkable) — not permanently deaf. pub async fn start_async(&mut self) -> Result<(), TransportError> { if !self.state.can_start() { return Err(TransportError::AlreadyStarted); @@ -132,109 +367,61 @@ impl EthernetTransport { self.state = TransportState::Starting; - // Create and bind AF_PACKET socket - let raw_socket = PacketSocket::open(&self.config.interface, self.config.ethertype())?; + // A restart reuses the transport object, so clear any stop from the + // previous run before the binder can observe it. + self.shutdown.store(false, Ordering::SeqCst); - // Get local MAC and MTU - let local_mac = raw_socket.local_mac()?; - let if_mtu = raw_socket.interface_mtu()?; + // The bring-up window is measured from here, not from whenever this + // object happened to be constructed. + self.presence.mark_starting(); - // Effective MTU: interface MTU minus 3 bytes for frame header - // (1 byte frame type + 2 bytes LE payload length) - let effective_mtu = if let Some(configured_mtu) = self.config.mtu { - // Config MTU cannot exceed interface MTU - 3 - configured_mtu.min(if_mtu.saturating_sub(3)) - } else { - if_mtu.saturating_sub(3) - }; - self.effective_mtu = effective_mtu; - self.local_mac = Some(local_mac); - - // Set buffer sizes - raw_socket.set_recv_buffer_size(self.config.recv_buf_size())?; - raw_socket.set_send_buffer_size(self.config.send_buf_size())?; - - // Wrap in async - let async_socket = raw_socket.into_async()?; - let socket = Arc::new(async_socket); - self.socket = Some(socket.clone()); - - // Spawn receive loop - let transport_id = self.transport_id; - let packet_tx = self.packet_tx.clone(); - let mtu = self.effective_mtu; - let listen_enabled = self.config.listen(); - let neighbor_buffer = self.neighbor_buffer.clone(); - let stats = self.stats.clone(); - let recv_socket = socket.clone(); - - let recv_task = tokio::spawn(async move { - ethernet_receive_loop( - recv_socket, - transport_id, - packet_tx, - mtu, - listen_enabled, - neighbor_buffer, - stats, - ) - .await; + let ctx = Arc::new(BinderContext { + shutdown: self.shutdown.clone(), + transport_id: self.transport_id, + name: self.name.clone(), + interface: self.interface.clone(), + config: self.config.clone(), + policy: self.policy, + packet_tx: self.packet_tx.clone(), + neighbor_buffer: self.neighbor_buffer.clone(), + stats: self.stats.clone(), + binding: self.binding.clone(), + presence: self.presence.clone(), + local_pubkey: self.local_pubkey, + presence_tx: self.presence_tx.clone(), }); - self.recv_task = Some(recv_task); - // Spawn beacon sender if announce is enabled - if self.config.announce() { - if let Some(pubkey) = self.local_pubkey { - let beacon_socket = socket.clone(); - let interval_secs = self.config.beacon_interval_secs(); - let beacon_stats = self.stats.clone(); - let beacon_transport_id = self.transport_id; - - let beacon_interface = self.config.interface.clone(); - let beacon_ethertype = self.config.ethertype(); - - let beacon_task = tokio::spawn(async move { - beacon_sender_loop( - beacon_socket, - pubkey, - interval_secs, - beacon_stats, - beacon_transport_id, - beacon_interface, - beacon_ethertype, - ) - .await; - }); - self.beacon_task = Some(beacon_task); - } else { - warn!( - transport_id = %self.transport_id, - "Announce enabled but no local pubkey set; beacons disabled" - ); + // First attempt inline, so the common case (interface already there) + // keeps its ordering: the transport is bound and logged before + // `start_async` returns, exactly as before this mechanism existed. + // An edge the channel refused here is handed to the binder to retry, + // rather than dropped: it is the only edge either consumer will ever + // see for this transport until the interface next changes state. + let initial_unpublished: Option = match bind_and_spawn(&ctx).await { + Ok(()) => publish_presence(&ctx, true), + Err(TransportError::InterfaceUnavailable { .. }) => { + // Absence is a state, not a start failure. Come up and wait. + log_initial_absence(&ctx); + publish_presence(&ctx, false) } - } + // Anything else — no CAP_NET_RAW, no free BPF device, a buffer + // the kernel refused — is a fault, not a state, and it will not + // resolve on its own. It fails the start exactly as it did before + // this mechanism existed, so a node deployed without the + // capability still dies loudly at boot instead of retrying + // forever behind a `Degraded` nobody is watching. + Err(e) => { + self.state = TransportState::Configured; + return Err(e); + } + }; + + let watcher_ctx = ctx.clone(); + self.binder_task = Some(tokio::spawn(async move { + binder_loop(watcher_ctx, initial_unpublished).await; + })); self.state = TransportState::Up; - - if let Some(ref name) = self.name { - info!( - name = %name, - interface = %self.interface, - mac = %format_mac(&local_mac), - mtu = effective_mtu, - if_mtu = if_mtu, - "Ethernet transport started" - ); - } else { - info!( - interface = %self.interface, - mac = %format_mac(&local_mac), - mtu = effective_mtu, - if_mtu = if_mtu, - "Ethernet transport started" - ); - } - Ok(()) } @@ -244,25 +431,12 @@ impl EthernetTransport { return Err(TransportError::NotStarted); } - // Signal the socket to shut down. On macOS this writes to the - // shutdown pipe, waking the reader thread's select() immediately. - // On Linux this is a no-op (AsyncFd cancellation handles it). - if let Some(ref socket) = self.socket { - socket.shutdown(); - } + // The flag first, then the abort. The binder checks it after storing + // a binding, so a bind that completes between here and `tear_down` + // cleans up after itself instead of outliving the transport. + self.shutdown.store(true, Ordering::SeqCst); - // Abort tasks. On Linux, safe to await since all I/O is - // AsyncFd-based and cancellation-safe. On macOS, do NOT await — - // on a current_thread runtime the aborted task can't be polled - // while we're blocked on the JoinHandle, causing a deadlock. - if let Some(task) = self.beacon_task.take() { - task.abort(); - #[cfg(not(target_os = "macos"))] - { - let _ = task.await; - } - } - if let Some(task) = self.recv_task.take() { + if let Some(task) = self.binder_task.take() { task.abort(); #[cfg(not(target_os = "macos"))] { @@ -270,9 +444,24 @@ impl EthernetTransport { } } - // Drop socket - self.socket.take(); - self.local_mac = None; + // `tear_down` shuts the socket down and aborts the loops. On macOS the + // aborted tasks are deliberately not awaited: on a current_thread + // runtime an aborted task cannot be polled while we are blocked on its + // `JoinHandle`, which deadlocks. + let was_present = self.presence.is_present(); + self.binding.tear_down(); + self.presence.transition(Presence::Absent); + + // Retract the edge the binder left standing. Every other transition + // pairs its edges; a transport stopped while `Present` would otherwise + // leave health reading `present: true` for a socket that is gone. + if was_present && let Some(tx) = &self.presence_tx { + let _ = tx.try_send(TransportPresence { + transport_id: self.transport_id, + present: false, + health_relevant: !self.policy.is_optional(), + }); + } self.state = TransportState::Down; @@ -298,15 +487,25 @@ impl EthernetTransport { return Err(TransportError::NotStarted); } - if data.len() > self.effective_mtu as usize { + // Absence is reported as absence, not as "not started": the caller can + // tell an interface that is away from a transport that was never + // brought up. + let socket = self + .binding + .socket() + .ok_or_else(|| TransportError::InterfaceUnavailable { + interface: self.interface.clone(), + })?; + + let mtu = self.binding.mtu(); + if data.len() > mtu as usize { return Err(TransportError::MtuExceeded { packet_size: data.len(), - mtu: self.effective_mtu, + mtu, }); } let dest_mac = parse_mac_addr(addr)?; - let socket = self.socket.as_ref().ok_or(TransportError::NotStarted)?; // Prepend frame type prefix and 2-byte LE payload length. // The length field lets the receiver trim Ethernet minimum-frame padding @@ -332,6 +531,28 @@ impl EthernetTransport { } } +/// Stop the binder and release the socket when the transport is dropped +/// without `stop_async`. +/// +/// A `TransportHandle` can be dropped without ever being stopped — a pending +/// handle the supervisor never spawns, an unwind through `start()`. Without +/// this, the binder outlives the object that owns it: not merely a leaked +/// task, but one that goes on opening raw sockets on an interface nobody is +/// reading, once per attach, for the life of the process. +/// +/// `Drop` cannot await, so this does what it can synchronously — raise the +/// stop flag, abort the binder, release the socket and its loops — which is +/// all `stop_async` does beyond awaiting the abort and logging. +impl Drop for EthernetTransport { + fn drop(&mut self) { + self.shutdown.store(true, Ordering::SeqCst); + if let Some(task) = self.binder_task.take() { + task.abort(); + } + self.binding.tear_down(); + } +} + impl Transport for EthernetTransport { fn transport_id(&self) -> TransportId { self.transport_id @@ -346,7 +567,7 @@ impl Transport for EthernetTransport { } fn mtu(&self) -> u16 { - self.effective_mtu + self.binding.mtu() } fn start(&mut self) -> Result<(), TransportError> { @@ -380,11 +601,619 @@ impl Transport for EthernetTransport { } } +// ============================================================================ +// Binder +// ============================================================================ + +/// Bind the interface and spawn the receive and beacon loops. +/// +/// On success the binding's socket slot, MTU and MAC are published and the +/// presence state moves to [`Presence::Present`]. On failure nothing is +/// published and the presence state is left for the caller to classify. +async fn bind_and_spawn(ctx: &Arc) -> Result<(), TransportError> { + // Probe before opening anything. The socket is created before the + // interface is named on both backends, so without this an unprivileged + // process would report "permission denied" for an interface that is + // simply not there — and absence would become indistinguishable from a + // missing capability, which is exactly the distinction this mechanism + // needs to make. + // + // Deliberately ahead of the `Binding` transition: absence is not an + // attempt, and announcing one would put a permanently-missing interface + // into `Binding` on every poll. + if !interface_present(&ctx.interface) { + return Err(TransportError::InterfaceUnavailable { + interface: ctx.interface.clone(), + }); + } + + // `Binding` is owned here, both edges of it. Leaving the revert to callers + // meant a caller that returned early on failure — the fail-fast start + // path — left the transport reporting `binding` for the rest of its life. + ctx.presence.transition(Presence::Binding); + let bound = bind_now(ctx); + if bound.is_err() { + ctx.presence.transition(Presence::Absent); + return bound; + } + + // Checked *after* the store, and set before teardown's own `tear_down`, + // so a stop that lands anywhere inside `bind_now` is cleaned up by + // exactly one of the two. Dropping this check is what leaves a receive + // loop reading a socket belonging to a stopped transport. + if ctx.shutdown.load(Ordering::SeqCst) { + ctx.binding.tear_down(); + ctx.presence.transition(Presence::Absent); + return Err(TransportError::NotStarted); + } + + bound +} + +/// Open the socket and start the loops. Every exit is an error the caller +/// converts back to [`Presence::Absent`]; nothing here is left half-published. +fn bind_now(ctx: &Arc) -> Result<(), TransportError> { + let raw_socket = PacketSocket::open(&ctx.interface, ctx.config.ethertype())?; + let local_mac = raw_socket.local_mac()?; + let if_mtu = raw_socket.interface_mtu()?; + + // Effective MTU: interface MTU minus 3 bytes for frame header + // (1 byte frame type + 2 bytes LE payload length) + let effective_mtu = if let Some(configured_mtu) = ctx.config.mtu { + // Config MTU cannot exceed interface MTU - 3 + configured_mtu.min(if_mtu.saturating_sub(3)) + } else { + if_mtu.saturating_sub(3) + }; + + raw_socket.set_recv_buffer_size(ctx.config.recv_buf_size())?; + raw_socket.set_send_buffer_size(ctx.config.send_buf_size())?; + + // Captured before the socket is wrapped: this is the device identity the + // binding is attached to, and the thing that has to keep matching. + let bound_index = raw_socket.if_index() as u32; + + let socket = Arc::new(raw_socket.into_async()?); + + let mut tasks = Vec::new(); + + let recv_task = { + let socket = socket.clone(); + let transport_id = ctx.transport_id; + let packet_tx = ctx.packet_tx.clone(); + let listen_enabled = ctx.config.listen(); + let neighbor_buffer = ctx.neighbor_buffer.clone(); + let stats = ctx.stats.clone(); + tokio::spawn(async move { + ethernet_receive_loop( + socket, + transport_id, + packet_tx, + effective_mtu, + listen_enabled, + neighbor_buffer, + stats, + ) + .await; + }) + }; + tasks.push(recv_task); + + if ctx.config.announce() { + if let Some(pubkey) = ctx.local_pubkey { + let socket = socket.clone(); + let interval_secs = ctx.config.beacon_interval_secs(); + let stats = ctx.stats.clone(); + let transport_id = ctx.transport_id; + tasks.push(tokio::spawn(async move { + beacon_sender_loop(socket, pubkey, interval_secs, stats, transport_id).await; + })); + } else { + warn!( + transport_id = %ctx.transport_id, + "Announce enabled but no local pubkey set; beacons disabled" + ); + } + } + + *ctx.binding + .socket + .write() + .unwrap_or_else(PoisonError::into_inner) = Some(socket); + *ctx.binding + .local_mac + .write() + .unwrap_or_else(PoisonError::into_inner) = Some(local_mac); + ctx.binding.mtu.store(effective_mtu, Ordering::Relaxed); + ctx.binding + .bound_index + .store(bound_index, Ordering::Relaxed); + *ctx.binding + .tasks + .lock() + .unwrap_or_else(PoisonError::into_inner) = tasks; + + // Only now does presence read `Present`. Flipping it before the socket + // slot is filled would let a concurrent send see a bound transport and be + // told its interface is unavailable. + // + // A name is not a device: if the interface reappeared with a MAC other + // than the one last bound, this is different hardware wearing the same + // name, so drop the cached neighbors rather than silently resuming onto + // it. + let hardware_changed = ctx.presence.record_bind(local_mac); + if hardware_changed { + ctx.neighbor_buffer.take(); + warn!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + mac = %format_mac(&local_mac), + "Interface reappeared with a different MAC; treating as new hardware \ + and dropping cached neighbors" + ); + } + + let rebind = ctx.presence.binds() > 1; + if let Some(ref name) = ctx.name { + info!( + name = %name, + interface = %ctx.interface, + mac = %format_mac(&local_mac), + mtu = effective_mtu, + if_mtu = if_mtu, + rebind, + "Ethernet transport started" + ); + } else { + info!( + interface = %ctx.interface, + mac = %format_mac(&local_mac), + mtu = effective_mtu, + if_mtu = if_mtu, + rebind, + "Ethernet transport started" + ); + } + + Ok(()) +} + +/// Log the edge into a *start-time* absence, once. +/// +/// At `info`, for both policies. An interface that is not there when the +/// daemon starts and binds a moment later is the ordinary case this whole +/// mechanism exists to absorb — on a router the daemon routinely wins the +/// race against the radio — so it is not an error yet. It becomes one by +/// outlasting the race: [`report_sustained_absence`] says so at `error`, once, +/// after [`ABSENCE_ERROR_AFTER`]. Node health does not wait for that; it reads +/// `Degraded` from the first edge. +/// +/// A runtime detach is logged separately, at `warn`, and reaches the same +/// deadline by the same route — start-time absence and runtime detach are one +/// transition here as everywhere else in this module. Nothing reaches `error` +/// for merely happening, only for outlasting the window in which it could +/// still have been a race. +/// +/// Never per retry attempt either way. A loop that logs per attempt +/// reproduces the hot log spin this mechanism removed, at 1–30 s intervals +/// forever on any router with an unplugged WAN — and operators learn to +/// filter it, which is how the next real failure gets missed. +fn log_initial_absence(ctx: &Arc) { + if ctx.policy.is_optional() { + info!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + "Ethernet interface absent; waiting for it to appear (optional)" + ); + } else { + info!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + "Ethernet interface absent; waiting for it to appear" + ); + } +} + +/// Publish a presence edge to node health. Returns the edge health is still +/// owed: `None` when it went out, `Some(present)` when the channel refused it +/// and the caller must retry it on the next tick. +/// +/// **The return is the caller's pending slot, not a status.** Health is a +/// level, so an edge that reaches the consumer supersedes whatever earlier +/// edge was still waiting to be retried, and writing this result over the +/// pending slot is what makes that true. Reporting bare success and leaving +/// the slot to the caller let a landed edge sit behind a stale one: a `false` +/// republished over a transport that had since bound, or — the worse +/// direction — a `true` republished for an interface that had since detached. +/// The consumer does not dedup (`node::dataplane::rx_loop`), so that second +/// one leaves node health reading present for an interface that is gone, and +/// no reap running for the peers that can no longer be reached. +/// +/// **Never blocks.** This used to `await` the send, which handed a bounded +/// 16-slot channel the power to stop the machine reporting into it: a +/// receiver that is slow, not yet running, or gone leaves the binder parked +/// mid-publish, and an interface frozen in whatever state it happened to +/// hold. A health channel must not be able to deadlock the thing whose health +/// it carries. The caller retries a refused edge on its next tick, so a +/// momentarily full channel costs latency rather than correctness. +/// +/// An `optional` interface publishes its edge with `health_relevant: false`. +/// That is the entire health half of the policy: absence of an interface +/// whose absence is normal must not move the node off `Full`. +/// +/// It is deliberately *not* silence. The edge still goes out, because the +/// consumer derives two things from it and only one of them is about health: +/// the node's egress MTU floor is a function of the bound set, and an +/// `optional` interface binding or unbinding changes that set exactly as a +/// `required` one does. Filtering the edge here — which is what this used to +/// do — left the TUN MSS clamp stale for every `optional` transport, which on +/// the shipped OpenWrt config is five of seven. +#[must_use] +fn publish_presence(ctx: &Arc, present: bool) -> Option { + let Some(tx) = &ctx.presence_tx else { + return None; + }; + match tx.try_send(TransportPresence { + transport_id: ctx.transport_id, + present, + health_relevant: !ctx.policy.is_optional(), + }) { + Ok(()) => None, + Err(tokio::sync::mpsc::error::TrySendError::Closed(_)) => { + // Nobody is listening and nobody will be. Owing nothing stops the + // caller retrying an edge that can never land. + None + } + Err(tokio::sync::mpsc::error::TrySendError::Full(_)) => Some(present), + } +} + +/// The binder: one task per interface-bound transport, alive for the +/// transport's whole lifetime. +/// +/// Present → watch for detach; Absent → watch for attach. This is the polling +/// fallback; on platforms with a link-event source the same loop is driven by +/// events instead (see `docs`/the netlink watcher), and the interval below is +/// what "immediate" degrades to without one. +async fn binder_loop(ctx: Arc, initial_unpublished: Option) { + let watcher = LinkWatcher::new(); + let mut ticker = tokio::time::interval(WATCH_INTERVAL); + ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + // Consume the immediate first tick: `start_async` has just probed. + ticker.tick().await; + + debug!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + event_driven = watcher.is_event_driven(), + "Interface binder started" + ); + + // Whether this absence episode has already outlasted the bring-up window + // and been reported. Reset on every entry into absence, so each outage is + // judged on its own duration rather than inheriting the last one's. + let mut absence_reported = false; + // Damping for bindings that keep succeeding and then dying young. + let mut churn = ChurnGuard::new(); + // `start_async` binds inline and publishes that edge itself, outside this + // guard. Seed the guard from that bind, or the guard believes it has + // announced nothing: `detached` takes `announced` to decide whether an + // edge is owed, so the first detach after a clean start would retract + // nothing and health would keep reading `Full` for as long as the + // interface stayed away. `stabilized` cannot repair it either — it + // early-returns while `bound_at` is `None`, which it is for a bind this + // loop did not perform. + if ctx.presence.is_present() { + churn.bound(Instant::now()); + } + // A presence edge the channel refused, to retry on the next tick. Health + // is a level, so only the most recent value matters — an older pending + // edge is superseded rather than queued. Supersession is `publish_presence`'s + // job and not this loop's: every publish below writes its return over this + // slot, so one that lands clears whatever was waiting in the same + // statement. Seeded with the start edge if that one was refused. + let mut unpublished: Option = initial_unpublished; + // When the presence probe last ran, for the coalescing floor. + let mut last_probe = Instant::now(); + + loop { + // Whichever comes first. With an event source the tick is a backstop + // and detection is sub-second; without one the watcher never fires and + // the tick is the whole mechanism. + tokio::select! { + _ = ticker.tick() => {} + _ = watcher.changed() => {} + } + + // Coalesce wake-ups that arrive faster than the probe floor. The tick + // never trips this; a firehose event source does. + if let Some(wait) = probe_delay(last_probe.elapsed()) { + tokio::time::sleep(wait).await; + } + last_probe = Instant::now(); + + if let Some(pending) = unpublished { + unpublished = publish_presence(&ctx, pending); + } + + if ctx.presence.is_present() { + // Detach is either the interface going away or the socket under it + // dying while the name stays (a recreated veth, a reloaded phy). + // + // A probe that could not run is not an interface that went away. + // Hold the binding and re-ask next tick, rather than tearing a + // working socket down because `getifaddrs` hit `ENOBUFS` or the + // process ran out of descriptors — the latter being a state the + // rebind could not recover from anyway. + let gone = match interface_present_probe(&ctx.interface) { + Some(present) => !present, + None => { + debug!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + "Interface presence probe failed; holding the binding" + ); + false + } + }; + let replaced = !gone && ctx.binding.device_replaced(&ctx.interface); + let dead = !ctx.binding.tasks_alive(); + let detach = classify_detach(gone, replaced, dead); + if detach.is_none() { + // Still bound. A binding held back during a churn streak is + // announced here, once it has proved it will last. + if churn.stabilized(Instant::now()) { + info!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + "Ethernet interface stable again" + ); + unpublished = publish_presence(&ctx, true); + } + continue; + } + + ctx.binding.tear_down(); + ctx.presence.transition(Presence::Absent); + absence_reported = false; + + let outcome = churn.detached(Instant::now()); + if outcome.log_edge { + let reason = detach + .expect("classify_detach returned Some above") + .as_str(); + // `warn`, not `error`. A link coming and going is the weather + // in a mesh daemon, and a cable unplugged for two seconds does + // not need a human. If it stays away, + // `report_sustained_absence` says so at `error` once the + // bring-up window is out. Health is the immediate signal and + // does not wait — `Degraded` publishes on this edge. + match ctx.policy { + AbsencePolicy::Optional => info!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + reason, + "Ethernet interface detached (optional)" + ), + AbsencePolicy::Required => warn!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + reason, + "Ethernet interface detached" + ), + } + } + if outcome.entered_churn { + warn!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + consecutive = churn.streak(), + "Ethernet binding keeps dying immediately after binding; \ + backing off and holding health until one lasts" + ); + } + // Only when an edge is actually owed. `retract` is + // `mem::take(&mut announced)`, so it is false exactly when this + // binding was never announced as present — there is then nothing + // for a `false` to supersede, and settling the slot here would + // drop a refused edge that is still owed rather than clear a + // stale one. + if outcome.retract { + unpublished = publish_presence(&ctx, false); + } + if let Some(backoff) = outcome.backoff { + tokio::time::sleep(backoff).await; + } + continue; + } + + // Absent. Absence itself does not back off — there is nothing to poll + // but the cheap presence probe. Only a *failed bind* backs off. + // + // The deadline check sits ahead of the presence probe so it covers + // both shapes of absence: an interface that is not there, and one that + // is there and refuses to bind. The second is the one that will not + // fix itself, so it is the one that most needs saying out loud. + if !absence_reported { + absence_reported = report_sustained_absence(&ctx); + } + + if !interface_present(&ctx.interface) { + continue; + } + + match bind_and_spawn(&ctx).await { + Ok(()) => { + if churn.bound(Instant::now()).announce { + info!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + "Ethernet interface recovered" + ); + unpublished = publish_presence(&ctx, true); + } + } + Err(e) => { + // `bind_and_spawn` has already reverted presence to Absent. + if matches!(e, TransportError::InterfaceUnavailable { .. }) { + // Raced with a detach between the probe and the bind. No + // backoff and no log: the next tick re-probes. + // + // And no `record_attempt` either — the counter below gates + // the only log a genuine bind fault ever gets, on + // `attempts == 1`. Charging absence races to it means a + // flap followed by a real fault (CAP_NET_RAW dropped, no + // free BPF device) is never reported at all, and the + // operator gets `report_sustained_absence`'s "still + // missing" instead — for an interface that is present. + continue; + } + let attempts = ctx.presence.record_attempt(); + if attempts == 1 { + error!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + error = %e, + "Ethernet interface is present but will not bind; retrying" + ); + } + tokio::time::sleep(bind_backoff(attempts)).await; + } + } + } +} + +/// How long to wait before the next presence probe, given how long ago the +/// last one ran. +/// +/// The floor exists for `PF_ROUTE`, which has no group filter and delivers +/// every routing message on the host: without it a busy machine wakes the +/// binder far faster than the interface can actually change, and each wake-up +/// is a `getifaddrs` walk. It is a coalescing floor, not a delay — an isolated +/// wake-up after a quiet period waits for nothing, which is what keeps +/// event-driven detection sub-second. +fn probe_delay(since_last_probe: Duration) -> Option { + // `filter` rather than bare `checked_sub`: at exact equality the + // subtraction yields `Some(0)`, and the loop this replaced used a strict + // `<`, so a zero wait must be "no wait" and not a zero-length sleep. + MIN_PROBE_INTERVAL + .checked_sub(since_last_probe) + .filter(|remaining| !remaining.is_zero()) +} + +/// Why a bound interface stopped being usable. +/// +/// The three probes the binder runs while `Present` are independent, and the +/// operator-facing distinction between them is the whole diagnostic value of +/// the detach line: "the cable is out" and "your netdev was recreated under +/// you" want different responses. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +enum DetachReason { + /// The name no longer resolves, or resolves to an interface that is down. + Gone, + /// The name still resolves, but to a *different device* than the one bound + /// — a recreated veth, a reloaded phy. The socket is attached to something + /// that no longer exists while the name looks fine. + Replaced, + /// The socket's loops have exited, so the descriptor under them is dead + /// even though the interface is present and unchanged. + SocketDied, +} + +impl DetachReason { + /// The `reason=` field of the detach log line. + /// + /// These strings are load-bearing: `testing/iface-binding/test.sh` greps + /// them, and an operator greps them. Changing one is a change to an + /// interface, not a wording tweak. + fn as_str(self) -> &'static str { + match self { + Self::Gone => "interface down", + Self::Replaced => "interface replaced", + Self::SocketDied => "socket died", + } + } +} + +/// Classify a bound transport's liveness from the three probe answers, or +/// `None` while it is still bound and healthy. +/// +/// Extracted from the binder loop so the decision is testable without a +/// privileged bind. Reaching `Replaced` or `SocketDied` in a live daemon +/// requires a successful bind and then a specific external event — a netdev +/// recreated under the same name, or five consecutive receive errors — which +/// no unit test can arrange and which the integration suite cannot arrange +/// deterministically either (the netlink event from a delete is acted on +/// within microseconds, so `Gone` wins that race in practice). The inputs are +/// each tested on their own; this makes the branch between them total and +/// exhaustively tested, which is the half that was reachable. +/// +/// Precedence is `Gone` before `Replaced` before `SocketDied`, and it matters: +/// an interface that has gone away has also, trivially, been "replaced" and +/// its socket is also dead, so the most specific true statement about *why* +/// has to win. Callers pass `replaced` already masked by `!gone`, so the +/// first arm is belt-and-braces rather than load-bearing — but a total +/// function should not depend on its caller having masked correctly. +fn classify_detach(gone: bool, replaced: bool, dead: bool) -> Option { + if gone { + Some(DetachReason::Gone) + } else if replaced { + Some(DetachReason::Replaced) + } else if dead { + Some(DetachReason::SocketDied) + } else { + None + } +} + +/// Report an absence that has outlasted [`ABSENCE_ERROR_AFTER`], once. +/// +/// Returns `true` once the report has been made, so the caller stops asking. +/// +/// A required interface still missing past the window is no longer a race +/// against a radio or a container: it is a fault an operator has to fix, and +/// it reads as one. Optional interfaces are silent here by definition — +/// `optional` is the statement that this interface's absence is normal. +/// +/// Said **once**. This used to be a 1 m / 10 m / 1 h ladder that re-announced +/// the same fact at rising severity and then went permanently quiet, which +/// got both halves wrong: it used the log as a state store for something +/// `show_transports` and node health already publish continuously, and it +/// stopped mentioning a live fault after an hour. Duration belongs in +/// `interface.since_secs`; the log's job is to say the thing once, when it +/// becomes true. +fn report_sustained_absence(ctx: &Arc) -> bool { + let absent_for = ctx.presence.since(); + if absent_for < ABSENCE_ERROR_AFTER { + return false; + } + if ctx.policy.is_optional() { + // The window passed; there is simply nothing to say. Return `true` so + // the caller stops re-checking the clock for the rest of the episode. + return true; + } + // Deliberately not prefixed "Ethernet interface absent": that is the edge + // line's opening, and sharing it makes the two indistinguishable to + // anything grepping the log — an operator, or the suite that asserts the + // edge is logged once. + error!( + transport_id = %ctx.transport_id, + interface = %ctx.interface, + absent_secs = absent_for.as_secs(), + "Ethernet interface still missing past the bring-up window; \ + node is degraded until it returns" + ); + true +} + // ============================================================================ // Receive Loop // ============================================================================ /// Ethernet receive loop — runs as a spawned task. +/// +/// Returns on a dead socket (see [`RECV_ERROR_EXIT_THRESHOLD`]); the binder +/// notices the finished task, tears the binding down, and rebinds. async fn ethernet_receive_loop( socket: Arc, transport_id: TransportId, @@ -396,12 +1225,14 @@ async fn ethernet_receive_loop( ) { // Buffer with headroom: frame type prefix + MTU + some extra let mut buf = vec![0u8; mtu as usize + 100]; + let mut consecutive_errors: u32 = 0; debug!(transport_id = %transport_id, "Ethernet receive loop starting"); loop { match socket.recv_from(&mut buf).await { Ok((len, src_mac)) => { + consecutive_errors = 0; if len == 0 { continue; } @@ -475,11 +1306,28 @@ async fn ethernet_receive_loop( } Err(e) => { stats.record_recv_error(); - warn!( - transport_id = %transport_id, - error = %e, - "Ethernet receive error" - ); + consecutive_errors += 1; + // First of a streak only: the rest are the same fact repeated. + if consecutive_errors == 1 { + warn!( + transport_id = %transport_id, + error = %e, + "Ethernet receive error" + ); + } + if consecutive_errors >= RECV_ERROR_EXIT_THRESHOLD { + warn!( + transport_id = %transport_id, + errors = consecutive_errors, + "Ethernet receive loop giving up on this socket; \ + handing back to the interface binder" + ); + break; + } + tokio::time::sleep(Duration::from_millis( + 100 * u64::from(consecutive_errors.min(10)), + )) + .await; } } } @@ -493,22 +1341,19 @@ async fn ethernet_receive_loop( /// Periodic beacon sender loop. /// -/// Detects stale AF_PACKET sockets (ENXIO / os error 6) that occur when -/// the underlying veth interface is destroyed and recreated (e.g., during -/// node churn in chaos tests). After `REOPEN_THRESHOLD` consecutive send -/// failures, attempts to open a fresh socket on the same interface. +/// Exits after [`BEACON_ERROR_EXIT_THRESHOLD`] consecutive send failures. The +/// stale-socket reopen this loop used to perform itself (an ENXIO special case +/// for veth pairs destroyed and recreated under a live socket) now belongs to +/// the binder: the loop's exit is the detach signal, and the rebind is one +/// mechanism for every cause rather than one hack per symptom. Beacons pause +/// while the interface is absent because the task simply does not exist then. async fn beacon_sender_loop( - mut socket: Arc, + socket: Arc, pubkey: XOnlyPublicKey, interval_secs: u64, stats: Arc, transport_id: TransportId, - interface: String, - ethertype: u16, ) { - /// Number of consecutive ENXIO errors before attempting socket reopen. - const REOPEN_THRESHOLD: u32 = 3; - let beacon = build_beacon(&pubkey); let interval = tokio::time::Duration::from_secs(interval_secs); @@ -555,8 +1400,6 @@ async fn beacon_sender_loop( consecutive_errors += 1; stats.record_send_error(); - let is_enxio = format!("{e}").contains("os error 6"); - // Log only the first error in a streak to avoid log spam if consecutive_errors == 1 { warn!( @@ -566,48 +1409,19 @@ async fn beacon_sender_loop( ); } - if is_enxio && consecutive_errors >= REOPEN_THRESHOLD { + if consecutive_errors >= BEACON_ERROR_EXIT_THRESHOLD { info!( transport_id = %transport_id, consecutive_errors, - interface = %interface, - "Stale veth detected (ENXIO), attempting socket reopen" + "Beacon socket looks dead; handing back to the interface binder" ); - match reopen_beacon_socket(&interface, ethertype) { - Ok(new_socket) => { - socket = Arc::new(new_socket); - consecutive_errors = 0; - info!( - transport_id = %transport_id, - interface = %interface, - "Beacon socket reopened successfully" - ); - } - Err(e) => { - warn!( - transport_id = %transport_id, - error = %e, - interface = %interface, - "Failed to reopen beacon socket, will retry" - ); - } - } + break; } } } } -} -/// Attempt to open a fresh AF_PACKET socket for beacon sending. -/// -/// This is called when the beacon sender detects that the underlying veth -/// has been recreated and the old socket FD is stale (ENXIO). -fn reopen_beacon_socket( - interface: &str, - ethertype: u16, -) -> Result { - let raw_socket = PacketSocket::open(interface, ethertype)?; - raw_socket.into_async() + debug!(transport_id = %transport_id, "Beacon sender stopped"); } // ============================================================================ @@ -743,4 +1557,1055 @@ mod tests { fn test_beacon_size() { assert_eq!(neighbor::BEACON_SIZE, 34); } + + // ── Dynamic interface binding ───────────────────────────────────────── + + fn absent_transport(optional: bool) -> (EthernetTransport, super::super::PacketRx) { + // A name no host has. `fips` is not a valid netdev prefix anywhere and + // the suffix keeps it clear of the test harness's own veth pairs. + let config = EthernetConfig { + interface: "fips-absent-x0".to_string(), + ethertype: None, + mtu: None, + recv_buf_size: None, + send_buf_size: None, + listen: Some(true), + announce: Some(false), + auto_connect: None, + accept_connections: None, + beacon_interval_secs: None, + optional: Some(optional), + }; + let (tx, rx) = super::super::packet_channel(8); + ( + EthernetTransport::new(TransportId::new(1), Some("lab".into()), config, tx), + rx, + ) + } + + #[tokio::test] + async fn a_missing_interface_starts_absent_rather_than_failing() { + // The boot race, at the mechanism level: `start_async` succeeds with + // the transport absent. Failing here is what made an OpenWrt node that + // booted before wifi stay deaf for the life of the process. + let (mut eth, _rx) = absent_transport(false); + assert_eq!(eth.presence(), Presence::Absent); + + eth.start_async() + .await + .expect("absence is not a start failure"); + + assert_eq!(eth.state(), TransportState::Up); + assert_eq!(eth.presence(), Presence::Absent); + assert!(eth.local_mac().is_none()); + assert_eq!(eth.presence_state().binds(), 0); + + eth.stop_async().await.expect("stop"); + } + + #[tokio::test] + async fn sending_while_absent_reports_absence_not_not_started() { + // A caller must be able to tell an interface that is away from a + // transport that was never brought up. + let (mut eth, _rx) = absent_transport(false); + eth.start_async().await.expect("start"); + + let addr = TransportAddr::from_bytes(&[0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff]); + let err = eth.send_async(&addr, b"hello").await.unwrap_err(); + assert!( + matches!(err, TransportError::InterfaceUnavailable { ref interface } + if interface == "fips-absent-x0"), + "expected InterfaceUnavailable, got {err:?}" + ); + + eth.stop_async().await.expect("stop"); + } + + #[tokio::test] + async fn the_absence_policy_comes_from_config() { + let (required, _a) = absent_transport(false); + assert_eq!(required.absence_policy(), AbsencePolicy::Required); + let (optional, _b) = absent_transport(true); + assert_eq!(optional.absence_policy(), AbsencePolicy::Optional); + } + + #[tokio::test] + async fn absence_is_published_to_node_health() { + // The edge the supervisor turns into `Degraded`. + let (mut eth, _rx) = absent_transport(false); + let (tx, mut presence_rx) = tokio::sync::mpsc::channel(4); + eth.set_presence_tx(tx); + + eth.start_async().await.expect("start"); + + let edge = presence_rx + .try_recv() + .expect("an absence edge is published"); + assert_eq!(edge.transport_id, TransportId::new(1)); + assert!(!edge.present); + + eth.stop_async().await.expect("stop"); + } + + #[tokio::test] + async fn an_optional_interface_publishes_an_edge_that_does_not_move_health() { + // `optional: true` is a statement about presence, and its whole health + // effect is this: a dock adapter that is not plugged in must not make + // the node report Degraded. + // + // It is not a statement about the *edge*. The edge still goes out, + // carrying `health_relevant: false`, because the consumer derives the + // node's egress MTU floor from the same channel and an optional + // interface changes the bound set exactly as a required one does. + // Suppressing the edge here — which is what this used to do — left the + // TUN MSS clamp stale for every optional transport. + let (mut eth, _rx) = absent_transport(true); + let (tx, mut presence_rx) = tokio::sync::mpsc::channel(4); + eth.set_presence_tx(tx); + + eth.start_async().await.expect("start"); + + let edge = presence_rx + .try_recv() + .expect("an optional interface still publishes its edge"); + assert_eq!(edge.transport_id, TransportId::new(1)); + assert!(!edge.present); + assert!( + !edge.health_relevant, + "an optional interface must not report absence to node health" + ); + + eth.stop_async().await.expect("stop"); + } + + #[tokio::test] + async fn a_required_interface_publishes_an_edge_that_moves_health() { + // The other half of the policy, pinned alongside it so the pair cannot + // drift into agreeing. + let (mut eth, _rx) = absent_transport(false); + let (tx, mut presence_rx) = tokio::sync::mpsc::channel(4); + eth.set_presence_tx(tx); + + eth.start_async().await.expect("start"); + + let edge = presence_rx + .try_recv() + .expect("an absence edge is published"); + assert!(!edge.present); + assert!(edge.health_relevant); + + eth.stop_async().await.expect("stop"); + } + + #[tokio::test] + async fn the_binder_stops_with_the_transport() { + // Teardown must abort the binder first: a rebind racing a stop would + // hand the node a socket nothing is going to read. + let (mut eth, _rx) = absent_transport(true); + eth.start_async().await.expect("start"); + eth.stop_async().await.expect("stop"); + assert_eq!(eth.state(), TransportState::Down); + assert_eq!(eth.presence(), Presence::Absent); + assert!( + eth.stop_async().await.is_err(), + "stopping twice is an error" + ); + } + + #[tokio::test] + async fn a_bind_fault_still_fails_the_start() { + // Absence is a state the transport waits out; a fault is not. No + // CAP_NET_RAW (Linux) or no readable /dev/bpf* (macOS) will not fix + // itself, so it must fail the start exactly as it did before dynamic + // binding existed — otherwise a node deployed without the capability + // retries a socket it can never open, forever, behind a `Degraded` + // nobody is watching. + // + // Loopback is the vehicle: it is present on every host, so the + // presence probe passes and the *bind* is what fails. + // + // Whether it fails is a property of the host, not of the code: an + // unprivileged CI runner is refused, while root — or a developer + // machine whose /dev/bpf* is group-readable — is not. Establish that + // as a precondition by opening the socket directly rather than + // branching inside the assertion, so this test either exercises the + // fail-fast path or declares itself inapplicable. + let loopback = if cfg!(target_os = "macos") { + "lo0" + } else { + "lo" + }; + assert!( + io::interface_present(loopback), + "loopback must be present for this test to mean anything" + ); + // Whether the socket opens splits the test into two halves, and both + // assert. Returning early on the privileged host — which is what this + // used to do — made the test vacuous as root and on any developer + // machine with a group-readable /dev/bpf*, so the fail-fast path it is + // named for went unchecked exactly where someone was most likely to be + // running it. + let can_open = PacketSocket::open(loopback, 0x2121).is_ok(); + + let config = EthernetConfig { + interface: loopback.to_string(), + ethertype: None, + mtu: None, + recv_buf_size: None, + send_buf_size: None, + listen: Some(true), + announce: Some(false), + auto_connect: None, + accept_connections: None, + beacon_interval_secs: None, + optional: None, + }; + let (tx, _rx) = super::super::packet_channel(8); + let mut eth = EthernetTransport::new(TransportId::new(9), None, config, tx); + + if can_open { + // The privileged half. A present, bindable interface binds inline + // and reports itself bound before `start_async` returns — which is + // the ordinary case on a booted router, and which no other unit + // test reaches: every other one here uses an interface that does + // not exist, so the bind-success path has no unit coverage at all + // without this branch. + eth.start_async() + .await + .expect("a present, bindable interface must start"); + assert_eq!( + eth.presence(), + Presence::Present, + "a bind that succeeded must leave the transport present" + ); + assert!( + eth.binding.socket().is_some(), + "a present transport must hold its socket" + ); + eth.stop_async().await.expect("stop"); + return; + } + + let err = eth + .start_async() + .await + .expect_err("an unprivileged raw socket must not start"); + assert!( + !matches!(err, TransportError::InterfaceUnavailable { .. }), + "a permission fault must not be reported as absence: {err:?}" + ); + assert_eq!( + eth.state(), + TransportState::Configured, + "a failed start must not leave the transport half-up" + ); + assert_eq!( + eth.presence(), + Presence::Absent, + "a failed bind must not leave presence parked in `binding`" + ); + } + + #[tokio::test] + async fn waiting_for_an_interface_never_looks_like_binding() { + // `Binding` means an attempt is in flight. An interface that is simply + // not there is not an attempt, so the probe sits ahead of the + // transition and a permanently-absent transport must read `absent` on + // every poll — never a `binding` that never resolves, and never a + // climbing failed-attempt count for binds it never tried. + let (mut eth, _rx) = absent_transport(false); + eth.start_async().await.expect("start"); + + // Long enough for the binder to have polled several times. + tokio::time::sleep(Duration::from_millis(2500)).await; + + assert_eq!(eth.presence(), Presence::Absent); + assert_eq!(eth.presence_state().binds(), 0); + assert_eq!( + eth.presence_state().attempts(), + 0, + "absence is not a failed attempt" + ); + + eth.stop_async().await.expect("stop"); + } + + #[tokio::test] + async fn a_poisoned_binding_does_not_strand_the_transport() { + // The transport-side twin of the presence-state poisoning test: + // `.ok()` on these locks would report the socket as gone and the + // tasks as dead, which is the destructive direction — a binder that + // tears down and rebinds every second while `tear_down` silently + // declines to abort anything, leaking a receive loop per cycle. + let (mut eth, _rx) = absent_transport(false); + eth.start_async().await.expect("start"); + + let binding = Arc::clone(ð.binding); + let panicked = std::thread::spawn(move || { + let _guard = binding.tasks.lock().unwrap(); + panic!("poison the task list while holding it"); + }) + .join(); + assert!(panicked.is_err(), "the helper thread was supposed to panic"); + assert!(eth.binding.tasks.is_poisoned()); + + // Still answerable, and teardown still runs to completion. + let _ = eth.binding.tasks_alive(); + eth.stop_async().await.expect("stop"); + assert_eq!(eth.state(), TransportState::Down); + } + + #[test] + fn the_probe_floor_sits_under_the_poll_interval() { + // The floor coalesces a firehose event source (`PF_ROUTE` on macOS has + // no group filter, so it delivers every routing message on the box) + // without becoming the thing that governs detection latency. If it + // ever reached the poll interval it would be the cadence rather than a + // bound on it. + assert!(MIN_PROBE_INTERVAL < WATCH_INTERVAL); + assert!( + !MIN_PROBE_INTERVAL.is_zero(), + "a zero floor coalesces nothing" + ); + } + + #[test] + fn a_present_probe_answers_for_loopback_and_refuses_a_fiction() { + // `lo` is up on every host this test runs on, and the fictional name + // is the negative control. Together they pin that the probe reads + // flags rather than merely answering "the call worked". + assert!(io::interface_present("lo") || io::interface_present("lo0")); + assert!(!io::interface_present("fips-absent-x0")); + } + + #[test] + fn a_device_index_identifies_the_interface_a_binding_holds() { + // A name is not a device. Both backends bind by index, so a netdev + // deleted and recreated under the same name leaves the socket attached + // to nothing while the name resolves perfectly well — and nothing else + // notices: a stale AF_PACKET socket never becomes readable, so the + // receive loop never errors and never exits, and send failures go to + // the caller rather than to the binder. Comparing the index is the + // only thing standing between a listen-only node and sitting + // `present` and deaf after a `wifi reload`. + let lo = if io::interface_present("lo") { + "lo" + } else { + "lo0" + }; + let index = io::interface_index(lo).expect("loopback has an index"); + assert!(index > 0); + assert_eq!(io::interface_index("fips-absent-x0"), None); + + let binding = Binding::new(1499); + // Nothing bound: there is no identity to have changed. + assert!(!binding.device_replaced(lo)); + + binding.bound_index.store(index, Ordering::Relaxed); + assert!(!binding.device_replaced(lo), "same device, same index"); + + binding.bound_index.store(index + 1000, Ordering::Relaxed); + assert!( + binding.device_replaced(lo), + "a different index under the same name is different hardware" + ); + + // A name that has gone is absence, which the presence probe reports + // more precisely; this check must not also claim it. + assert!(!binding.device_replaced("fips-absent-x0")); + } + + #[tokio::test] + async fn teardown_forgets_the_device_it_was_bound_to() { + // A stale index surviving teardown would make the next binding look + // replaced the moment it came up. + let (mut eth, _rx) = absent_transport(false); + eth.binding.bound_index.store(12345, Ordering::Relaxed); + eth.start_async().await.expect("start"); + eth.stop_async().await.expect("stop"); + assert_eq!(eth.binding.bound_index.load(Ordering::Relaxed), 0); + } + + /// An interface with **no addresses at all** is still visible to the + /// presence probe. + /// + /// This is the assumption the whole mechanism rests on and the one no + /// other test reaches. `interface_present` walks `getifaddrs` and reads + /// `ifa_flags`; neither the presence of a link-level entry for an + /// address-less interface nor the flags on it are specified anywhere — + /// `getifaddrs` is not in POSIX, glibc synthesizes an `AF_PACKET` entry + /// per interface from netlink, and musl reimplements the whole call + /// independently. Loopback cannot test this: it has `127.0.0.1`, so + /// probing it asks "does `getifaddrs` work", which was never in doubt. + /// + /// And FIPS is squarely in that corner by design. `fips-mesh0` and + /// `fips-ap0` on OpenWrt are deliberately unbridged with no IP + /// configuration — the transport speaks raw frames and never wants an + /// address — and OpenWrt is musl. If an address-less interface is + /// invisible here, those transports never bind, and because they ship + /// `optional: true` they never say so: the router reports `Running`, the + /// 802.11s link forms anyway because that is mac80211 rather than the + /// daemon, and the node reaches nothing. That is the original bug, whole, + /// on the platform this was written for. + /// + /// Creating such an interface needs `CAP_NET_ADMIN`, so CI makes one and + /// names it here rather than the test conjuring it. Absent the variable + /// there is nothing to assert — which is why the CI step that provides it + /// fails loudly rather than skipping, and why it checks that the interface + /// really has no address: the kernel hands an IPv6 link-local to anything + /// that comes up, and the first version of that fixture tested a + /// perfectly well-addressed interface without anyone noticing. + #[test] + fn an_interface_with_no_addresses_is_still_present() { + // A silent skip is how this test spent its life green without ever + // running: no fixture, early return, pass. It still has to skip on a + // developer machine that has no address-less interface, so the guard + // is the runner declaring that it *does* — if the fixture step is + // removed or renamed, this fails instead of quietly covering nothing. + let Ok(iface) = std::env::var("FIPS_TEST_ADDRLESS_IFACE") else { + assert!( + std::env::var_os("FIPS_TEST_REQUIRE_FIXTURES").is_none(), + "this runner sets FIPS_TEST_REQUIRE_FIXTURES but not \ + FIPS_TEST_ADDRLESS_IFACE: the fixture step did not run, and \ + the musl/glibc getifaddrs contract this pins went unchecked" + ); + return; + }; + assert!( + io::interface_present(&iface), + "interface {iface} has no addresses and must still be visible to \ + the presence probe; if this fails on musl, every address-less \ + interface on OpenWrt is invisible to interface binding" + ); + // Carrier is asserted rather than discarded, but the expected answer + // is `true`, not `false`: a Linux `dummy` brought up reports + // `UP,LOWER_UP`, so `IFF_RUNNING` is set and it has carrier. The + // comment this replaces claimed the opposite — "up but not running" — + // which is why the result was discarded rather than checked. + // + // So this fixture pins address-less *presence*, and cannot demonstrate + // the presence-vs-carrier split; an interface that is up with no + // carrier is a bridge with nothing plugged in, which no fixture here + // creates. `carrier_is_reported_separately_from_presence` pins that + // split from the other side, on a missing interface having neither. + // + // Linux only, because the expected answer is a property of the fixture + // device rather than of the code: a `dummy` that is up reports + // `IFF_RUNNING`, and macOS's `feth` has its own semantics that are not + // pinned here. What both platforms do assert is the part that matters + // — an interface with no addresses is still *present*. + #[cfg(target_os = "linux")] + assert!( + io::interface_carrier(&iface), + "a dummy interface that is up reports IFF_RUNNING; if this fails, \ + the fixture is no longer a dummy and what it pins has changed" + ); + // And it resolves to an index, which is what a bind would attach to. + assert!(io::interface_index(&iface).is_some()); + } + + #[test] + fn carrier_is_reported_separately_from_presence() { + // Presence is `IFF_UP`; carrier is `IFF_RUNNING` and is reported, not + // acted on. Loopback carries both, which pins that the two probes read + // different flags rather than one calling the other. The negative + // control pins that a missing interface has neither — "no carrier" and + // "no interface" must not be confused, and presence is what tells them + // apart. + let lo = if io::interface_present("lo") { + "lo" + } else { + "lo0" + }; + assert!(io::interface_present(lo)); + assert!(io::interface_carrier(lo)); + assert!(!io::interface_carrier("fips-absent-x0")); + } + + #[tokio::test] + async fn an_absent_interface_does_not_clamp_the_node_mtu() { + // `is_operational` means started, not bound. A transport whose + // interface has never existed reports its configured MTU, so a caller + // that filters on `is_operational` lets absent hardware set a value + // for the whole node — `transport_mtu`, and with it the node's IPv6 + // MTU, was doing exactly that. + let (mut eth, _rx) = absent_transport(false); + eth.start_async().await.expect("start"); + + let handle = super::super::TransportHandle::Ethernet(eth); + assert!( + handle.is_operational(), + "an absent transport is still started" + ); + assert!( + !handle.is_bound(), + "an absent transport must not count as usable" + ); + assert!( + handle.mtu() > 0, + "it still reports an MTU, which is the trap" + ); + } + + #[tokio::test] + async fn a_dropped_transport_does_not_leave_its_binder_running() { + // A handle can be dropped without ever being stopped — one the + // supervisor never spawns, an unwind through start(). Without `Drop` + // the binder outlives the object and goes on opening raw sockets on an + // interface nobody reads, for the life of the process. + let (mut eth, _rx) = absent_transport(false); + eth.start_async().await.expect("start"); + + let shutdown = Arc::clone(ð.shutdown); + let binder = eth + .binder_task + .as_ref() + .expect("binder spawned") + .abort_handle(); + assert!(!shutdown.load(Ordering::SeqCst)); + + drop(eth); + + assert!( + shutdown.load(Ordering::SeqCst), + "drop must raise the stop flag" + ); + // Give the aborted task a moment to be reaped by the runtime. + tokio::time::sleep(Duration::from_millis(50)).await; + assert!(binder.is_finished(), "drop must stop the binder"); + } + + #[tokio::test] + async fn a_stop_racing_a_bind_leaves_nothing_behind() { + // `bind_now` has no await points, so an abort issued while it runs + // takes effect only afterwards — after the socket and tasks are + // stored. The stop flag is what makes "teardown cleared an empty + // binding, then the binder filled it" impossible to end in a live + // receive loop on a socket nothing owns. + let (mut eth, _rx) = absent_transport(false); + eth.start_async().await.expect("start"); + eth.stop_async().await.expect("stop"); + + assert!(eth.shutdown.load(Ordering::SeqCst)); + assert!( + eth.binding.socket().is_none(), + "no socket may survive teardown" + ); + assert!(!eth.binding.tasks_alive(), "no loop may survive teardown"); + + // And a bind attempted after the stop refuses rather than publishing. + let ctx = Arc::new(BinderContext { + shutdown: Arc::clone(ð.shutdown), + transport_id: eth.transport_id, + name: None, + interface: eth.interface.clone(), + config: eth.config.clone(), + policy: eth.policy, + packet_tx: eth.packet_tx.clone(), + neighbor_buffer: eth.neighbor_buffer.clone(), + stats: eth.stats.clone(), + binding: eth.binding.clone(), + presence: eth.presence.clone(), + local_pubkey: None, + presence_tx: None, + }); + assert!( + bind_and_spawn(&ctx).await.is_err(), + "a bind must not complete for a stopped transport" + ); + assert!(eth.binding.socket().is_none()); + } + + /// A stop that lands while a bind is in flight is cleaned up by the bind, + /// not left running. + /// + /// `bind_now` has no await points, so an abort issued while it runs takes + /// effect only afterwards — after the socket and tasks are stored. The + /// post-store check is what makes "teardown cleared an empty binding, then + /// the binder filled it" impossible to end in a live receive loop on a + /// socket nothing owns. + /// + /// This needs a bind that *succeeds*, which needs privilege, so it is the + /// one branch `a_stop_racing_a_bind_leaves_nothing_behind` cannot reach: + /// that test's interface does not exist, so `bind_and_spawn` refuses at the + /// presence probe long before this check. + #[tokio::test] + async fn a_bind_that_completes_after_a_stop_undoes_itself() { + let loopback = if cfg!(target_os = "macos") { + "lo0" + } else { + "lo" + }; + + if PacketSocket::open(loopback, 0x2121).is_err() { + // Unprivileged: no bind can succeed, so there is no post-store + // state to observe. Loud on a runner that claims otherwise, so + // this cannot go quiet the way the fixture tests did. + assert!( + std::env::var_os("FIPS_TEST_PRIVILEGED").is_none(), + "this runner declares FIPS_TEST_PRIVILEGED but cannot open a \ + raw socket on {loopback}: the post-store shutdown check went \ + unexercised" + ); + return; + } + + let (mut eth, _rx) = absent_transport(false); + eth.interface = loopback.to_string(); + eth.shutdown.store(true, Ordering::SeqCst); + + let ctx = Arc::new(BinderContext { + shutdown: Arc::clone(ð.shutdown), + transport_id: eth.transport_id, + name: None, + interface: eth.interface.clone(), + config: eth.config.clone(), + policy: eth.policy, + packet_tx: eth.packet_tx.clone(), + neighbor_buffer: eth.neighbor_buffer.clone(), + stats: eth.stats.clone(), + binding: eth.binding.clone(), + presence: eth.presence.clone(), + local_pubkey: None, + presence_tx: None, + }); + + let err = bind_and_spawn(&ctx) + .await + .expect_err("a bind must not stand for a stopped transport"); + assert!( + matches!(err, TransportError::NotStarted), + "expected the shutdown refusal, got {err:?}" + ); + assert!( + eth.binding.socket().is_none(), + "the bind must undo its own socket when it loses the stop race" + ); + assert!( + !eth.binding.tasks_alive(), + "and its own loops, or a receive task outlives the transport" + ); + assert_eq!(eth.presence.presence(), Presence::Absent); + } + + #[test] + fn the_probe_floor_coalesces_a_burst_but_never_delays_a_lone_wake_up() { + // The floor's whole job is `PF_ROUTE`, which has no group filter and + // delivers every routing message on the host. Only the *constants* + // were asserted before, which says nothing about the behaviour. + // + // A wake-up that arrives hard on the last probe waits out the + // remainder... + assert_eq!( + probe_delay(Duration::ZERO), + Some(MIN_PROBE_INTERVAL), + "a wake-up with no gap waits the whole floor" + ); + let half = MIN_PROBE_INTERVAL / 2; + assert_eq!( + probe_delay(half), + Some(MIN_PROBE_INTERVAL - half), + "a wake-up mid-floor waits only the remainder" + ); + + // ...and one after a quiet period waits for nothing, which is what + // keeps event-driven detection sub-second rather than floor-limited. + assert_eq!(probe_delay(MIN_PROBE_INTERVAL), None); + assert_eq!(probe_delay(Duration::from_secs(3600)), None); + } + + #[tokio::test] + async fn the_absence_deadline_fires_once_and_only_for_required() { + // `report_sustained_absence` decides whether an absence has outlasted + // the window in which it could still have been an ordinary bring-up + // race. Nothing tested it: the integration suite infers it from + // counting ERROR lines, and ten seconds of real time is how a deadline + // ends up asserted by proxy rather than directly. + for (optional, label) in [(false, "required"), (true, "optional")] { + let (eth, _rx) = absent_transport(optional); + let ctx = Arc::new(BinderContext { + shutdown: Arc::clone(ð.shutdown), + transport_id: eth.transport_id, + name: None, + interface: eth.interface.clone(), + config: eth.config.clone(), + policy: eth.policy, + packet_tx: eth.packet_tx.clone(), + neighbor_buffer: eth.neighbor_buffer.clone(), + stats: eth.stats.clone(), + binding: eth.binding.clone(), + presence: eth.presence.clone(), + local_pubkey: None, + presence_tx: None, + }); + + // Inside the window: nothing to say yet, either way. + assert!( + !report_sustained_absence(&ctx), + "{label}: a fresh absence is still a bring-up race" + ); + + // Past it: both policies report handled, so the caller stops + // re-checking the clock for the rest of the episode. The + // difference is whether anything was said, not whether the + // deadline passed. + ctx.presence + .backdate_for_test(ABSENCE_ERROR_AFTER + Duration::from_secs(1)); + assert!( + report_sustained_absence(&ctx), + "{label}: past the window the deadline is handled" + ); + } + } + + /// A netdev recreated under the same name produces the exact probe pair + /// that classifies as `Replaced` — present, but a different device. + /// + /// This is the `wifi reload` case from #125, and the one branch the + /// integration suite cannot reach: with link events live the kernel's + /// `RTM_DELLINK` is acted on within microseconds, so the binder observes + /// "gone" before the recreate lands and takes the branch already covered. + /// + /// Rather than race the binder, this asserts the *inputs* directly. Paired + /// with `every_detach_classification_is_total_and_ordered`, which pins that + /// `(gone: false, replaced: true)` maps to `Replaced`, the path is covered + /// end to end without depending on scheduling. + #[cfg(target_os = "linux")] + #[tokio::test] + async fn a_netdev_recreated_under_its_name_reads_as_replaced() { + use std::process::Command; + + let iface = "fips-swap0"; + let peer = "fips-swap0p"; + let ip = |args: &[&str]| { + Command::new("ip") + .args(args) + .status() + .map(|s| s.success()) + .unwrap_or(false) + }; + let make = || { + let _ = ip(&["link", "del", iface]); + ip(&["link", "add", iface, "type", "veth", "peer", "name", peer]) + && ip(&["link", "set", iface, "up"]) + }; + + if !make() { + assert!( + std::env::var_os("FIPS_TEST_PRIVILEGED").is_none(), + "this runner declares FIPS_TEST_PRIVILEGED but cannot create a \ + veth: the replaced-device path went unexercised" + ); + return; + } + + let config = EthernetConfig { + interface: iface.to_string(), + ethertype: None, + mtu: None, + recv_buf_size: None, + send_buf_size: None, + listen: Some(true), + announce: Some(false), + auto_connect: None, + accept_connections: None, + beacon_interval_secs: None, + optional: Some(true), + }; + let (tx, _rx) = super::super::packet_channel(8); + let mut eth = EthernetTransport::new(TransportId::new(11), None, config, tx); + + if eth.start_async().await.is_err() { + let _ = ip(&["link", "del", iface]); + assert!( + std::env::var_os("FIPS_TEST_PRIVILEGED").is_none(), + "this runner declares FIPS_TEST_PRIVILEGED but cannot bind a \ + raw socket to {iface}" + ); + return; + } + assert_eq!(eth.presence(), Presence::Present, "the veth must bind"); + assert!(!eth.binding.device_replaced(iface), "freshly bound"); + + // Swap the device: same name, new kernel index. + assert!(make(), "recreate the veth under the same name"); + + // The two probes that together mean `Replaced`. + assert!( + io::interface_present(iface), + "the name still resolves after the swap — that is what makes this \ + different from `gone`" + ); + assert!( + eth.binding.device_replaced(iface), + "the bound index must no longer match the name, or a recreated \ + netdev leaves the socket attached to a device that is gone while \ + the name looks fine" + ); + assert_eq!( + classify_detach(false, true, eth.binding.tasks_alive()), + Some(DetachReason::Replaced) + ); + + eth.stop_async().await.expect("stop"); + let _ = ip(&["link", "del", iface]); + } + + #[test] + fn every_detach_classification_is_total_and_ordered() { + // All eight probe combinations, because the branch between them was + // the untested part. `Replaced` and `SocketDied` are otherwise + // unreachable from a test: both need a bind that succeeded and then a + // specific external event, and the integration suite cannot arrange + // the recreate deterministically either — the netlink event from a + // delete is acted on within microseconds, so `Gone` wins that race. + use DetachReason::{Gone, Replaced, SocketDied}; + + // (gone, replaced, dead) -> expected + let cases = [ + ((false, false, false), None), + ((false, false, true), Some(SocketDied)), + ((false, true, false), Some(Replaced)), + ((false, true, true), Some(Replaced)), + ((true, false, false), Some(Gone)), + ((true, false, true), Some(Gone)), + ((true, true, false), Some(Gone)), + ((true, true, true), Some(Gone)), + ]; + for ((gone, replaced, dead), expected) in cases { + assert_eq!( + classify_detach(gone, replaced, dead), + expected, + "classify_detach({gone}, {replaced}, {dead})" + ); + } + } + + #[test] + fn a_healthy_binding_is_the_only_unclassified_state() { + // The gate the loop actually uses: anything other than all-three-false + // is a detach. Stated separately from the table so a future fourth + // probe cannot be added and silently default to "still bound". + assert!(classify_detach(false, false, false).is_none()); + for (gone, replaced, dead) in [ + (true, false, false), + (false, true, false), + (false, false, true), + ] { + assert!( + classify_detach(gone, replaced, dead).is_some(), + "a failing probe must detach" + ); + } + } + + #[test] + fn detach_reason_labels_are_the_ones_greppers_expect() { + // These strings are an interface. `testing/iface-binding/test.sh` + // greps `reason="interface replaced"`, and operators grep the others. + assert_eq!(DetachReason::Gone.as_str(), "interface down"); + assert_eq!(DetachReason::Replaced.as_str(), "interface replaced"); + assert_eq!(DetachReason::SocketDied.as_str(), "socket died"); + + // And they are distinct, or a grep cannot tell two causes apart. + let labels = std::collections::HashSet::from([ + DetachReason::Gone.as_str(), + DetachReason::Replaced.as_str(), + DetachReason::SocketDied.as_str(), + ]); + assert_eq!(labels.len(), 3); + } + + #[tokio::test] + async fn a_restarted_transport_starts_its_binder_again() { + // `start_async` clears the stop flag so a restart is not immediately + // undone by the previous run's shutdown. Nothing tested the second + // start at all: `the_binder_stops_with_the_transport` only asserts a + // second *stop* errors, so a transport that could never be restarted + // would have passed everything here. + let (mut eth, _rx) = absent_transport(false); + + eth.start_async().await.expect("first start"); + eth.stop_async().await.expect("stop"); + assert!(eth.shutdown.load(Ordering::SeqCst), "stop raises the flag"); + + eth.start_async() + .await + .expect("a stopped transport restarts"); + assert!( + !eth.shutdown.load(Ordering::SeqCst), + "the restart must clear the previous run's stop, or the new binder \ + tears its own binding down on its first pass" + ); + assert_eq!(eth.state(), TransportState::Up); + + eth.stop_async().await.expect("stop again"); + } + + #[tokio::test] + async fn the_absence_deadline_is_measured_from_the_start() { + // `PresenceState::new` stamps the episode clock at construction, but + // construction and `start_async` need not be adjacent — config load and + // supervisor staging sit between them. Without the restamp a transport + // staged for longer than the window reports sustained absence on its + // very first binder tick, having given the interface no bring-up window + // at all, which is the one thing the window exists to provide. + let (mut eth, _rx) = absent_transport(false); + + // Stand in for a slow bring-up by ageing the clock past the deadline. + std::thread::sleep(Duration::from_millis(20)); + let staged_for = eth.presence.since(); + + eth.start_async().await.expect("start"); + assert!( + eth.presence.since() < staged_for, + "the episode clock must restart at start, not run from whenever \ + the object happened to be constructed" + ); + + eth.stop_async().await.expect("stop"); + } + + #[tokio::test] + async fn a_refused_presence_edge_is_delivered_once_there_is_room() { + // The retry slot, which nothing followed through. The existing test + // fills the channel and asserts the binder keeps running, then drops + // the receiver — so a slot that captured the edge and never re-sent it + // would pass, and health would sit on a stale level forever. + let (mut eth, _rx) = absent_transport(false); + let (tx, mut presence_rx) = tokio::sync::mpsc::channel(1); + eth.set_presence_tx(tx.clone()); + + // Occupy the only slot, so the start edge is refused on its way out. + tx.try_send(TransportPresence { + transport_id: TransportId::new(99), + present: true, + health_relevant: true, + }) + .expect("the one slot"); + + eth.start_async().await.expect("start"); + + // Drain the squatter. The binder now has room on its next pass. + let squatter = presence_rx.recv().await.expect("squatter"); + assert_eq!(squatter.transport_id, TransportId::new(99)); + + let edge = tokio::time::timeout(Duration::from_secs(5), presence_rx.recv()) + .await + .expect("the refused edge must be retried, not dropped") + .expect("channel open"); + assert_eq!(edge.transport_id, TransportId::new(1)); + assert!(!edge.present, "the retried edge is the absence it refused"); + + eth.stop_async().await.expect("stop"); + } + + #[tokio::test] + async fn an_edge_that_lands_supersedes_the_one_the_channel_refused() { + // The pending slot holds the single edge health is owed, not a queue, + // so a publish that lands has to clear it whatever value it was + // holding: health is a level, and the value that just reached the + // consumer is the current one. + // + // Both directions are asserted, because a fix aimed at one leaves the + // other standing. A stale `false` left pending is republished over a + // transport that has since bound. A stale `true` left pending is + // republished for an interface that has since detached, and the + // consumer does not dedup (`node::dataplane::rx_loop`), so node health + // reads present for an interface that is gone and no reap runs for the + // peers that can no longer be reached. + let (eth, _rx) = absent_transport(false); + let (tx, mut presence_rx) = tokio::sync::mpsc::channel(1); + let ctx = Arc::new(BinderContext { + shutdown: Arc::clone(ð.shutdown), + transport_id: eth.transport_id, + name: None, + interface: eth.interface.clone(), + config: eth.config.clone(), + policy: eth.policy, + packet_tx: eth.packet_tx.clone(), + neighbor_buffer: eth.neighbor_buffer.clone(), + stats: eth.stats.clone(), + binding: eth.binding.clone(), + presence: eth.presence.clone(), + local_pubkey: None, + presence_tx: Some(tx.clone()), + }); + + // Occupy the only slot, so the next edge is refused on its way out. + tx.try_send(TransportPresence { + transport_id: TransportId::new(99), + present: true, + health_relevant: true, + }) + .expect("the one slot"); + + // A refused edge is owed, and a refusal must not clear the slot — + // that is what makes the retry happen at all. A fix that settles + // unconditionally fails here rather than passing the two below. + assert_eq!( + publish_presence(&ctx, true), + Some(true), + "a refused edge stays owed" + ); + + // The consumer catches up, and a detach lands. It supersedes the + // `true` that never went out. + presence_rx.recv().await.expect("the squatter drains"); + assert_eq!( + publish_presence(&ctx, false), + None, + "a detach that reached the consumer supersedes the pending \ + `true`; republishing that `true` reports an interface that is \ + gone as present" + ); + + // The mirror. The `false` just sent is the squatter now, so the next + // edge is refused in turn. + assert_eq!( + publish_presence(&ctx, false), + Some(false), + "a refused edge stays owed in this direction too" + ); + presence_rx.recv().await.expect("the detach drains"); + assert_eq!( + publish_presence(&ctx, true), + None, + "a bind that reached the consumer supersedes the pending \ + `false`; republishing that `false` reports a live transport as \ + absent" + ); + } + + #[tokio::test] + async fn a_full_presence_channel_does_not_block_the_binder() { + // The health channel must not be able to deadlock the machine whose + // health it carries. `send().await` on a bounded channel could: a + // receiver that is slow, not yet running, or gone parked the binder + // mid-publish and froze the interface in whatever state it held. + let (mut eth, _rx) = absent_transport(false); + let (tx, presence_rx) = tokio::sync::mpsc::channel(1); + eth.set_presence_tx(tx.clone()); + + // Fill the channel, then never drain it. + tx.try_send(TransportPresence { + transport_id: TransportId::new(1), + present: true, + health_relevant: true, + }) + .expect("first slot"); + + // With a blocking publish this start would hang forever. + let started = tokio::time::timeout(Duration::from_secs(5), eth.start_async()) + .await + .expect("start must not block on a full presence channel"); + started.expect("start"); + + // The binder keeps polling despite the refused edge. + tokio::time::sleep(Duration::from_millis(1500)).await; + assert_eq!(eth.presence(), Presence::Absent); + + drop(presence_rx); + eth.stop_async().await.expect("stop"); + } } diff --git a/src/transport/ethernet/presence.rs b/src/transport/ethernet/presence.rs new file mode 100644 index 00000000..36b3c091 --- /dev/null +++ b/src/transport/ethernet/presence.rs @@ -0,0 +1,851 @@ +//! Interface presence state for interface-bound transports. +//! +//! A transport bound to a network interface is a long-lived object that is +//! *sometimes bound*. The interface it names may not exist when the daemon +//! starts, may appear minutes later, may vanish and return mid-operation, and +//! may never appear at all. This module holds the state that makes that +//! observable and drives the rebind loop in [`super`]. +//! +//! ```text +//! Absent ──attach──> Binding ──ok──> Present +//! ^ │ │ +//! └──── fail/backoff ─┘ │ +//! └──────────── detach ──────────────┘ +//! ``` +//! +//! Two invariants do the work: +//! +//! - **The transport object survives detach.** Config, `TransportId`, +//! statistics, and the neighbor buffer persist; only the file descriptor and +//! its loops go. A transport is never destroyed because its interface went +//! away. +//! - **Start-time absence and runtime detach are the same transition.** A node +//! that boots before wifi and a node whose wifi reloads at 03:00 take one +//! code path. +//! +//! Presence tracks `IFF_UP` — the interface exists and the operator has +//! enabled it — and deliberately not `IFF_RUNNING`: binding needs no carrier, +//! and a socket outlives a carrier flap. Carrier is reported alongside it +//! rather than steering it. See +//! [`interface_present`](super::io::interface_present) for why. + +use std::sync::atomic::{AtomicU8, AtomicU32, AtomicU64, Ordering}; +use std::sync::{PoisonError, RwLock, RwLockReadGuard, RwLockWriteGuard}; +use std::time::{Duration, Instant}; + +/// Read a lock, ignoring poisoning. +/// +/// Every value guarded in this module is plain data — an `Instant`, an +/// `Option<[u8; 6]>` — that a panic mid-write cannot leave logically +/// inconsistent, so poisoning carries no information worth propagating. +/// Treating it as a failure is what would hurt: the callers here are on the +/// presence path, and "assume the worst" there means a transport that reports +/// itself bound while every send fails, or a binder that tears down and +/// rebinds every second forever. A stuck state is a worse outcome than +/// reading a byte written by a thread that later panicked. +fn read(lock: &RwLock) -> RwLockReadGuard<'_, T> { + lock.read().unwrap_or_else(PoisonError::into_inner) +} + +/// Write a lock, ignoring poisoning. See [`read`]. +fn write(lock: &RwLock) -> RwLockWriteGuard<'_, T> { + lock.write().unwrap_or_else(PoisonError::into_inner) +} + +/// How absence of the configured interface is reported. +/// +/// Describes *the interface's presence*, not the transport's importance: an +/// optional interface that is present is used exactly as hard as any other. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum AbsencePolicy { + /// Naming an interface in configuration is a statement that you expect it, + /// so the default is to complain: absence degrades node health from the + /// first edge, and if it outlasts [`ABSENCE_ERROR_AFTER`] — the window in + /// which it could still have been an ordinary bring-up race — it is + /// reported once at `error`. + Required, + /// Absence is normal for this interface (a dock adapter, a radio that only + /// exists on some hardware): no health impact, `info` on the edge. + Optional, +} + +impl AbsencePolicy { + /// `optional: true` in configuration selects [`AbsencePolicy::Optional`]. + pub fn from_optional(optional: bool) -> Self { + if optional { + Self::Optional + } else { + Self::Required + } + } + + /// Whether absence should be hidden from node health. + pub fn is_optional(self) -> bool { + matches!(self, Self::Optional) + } + + /// Operator-facing label, used by `show_transports`. + pub fn as_str(self) -> &'static str { + match self { + Self::Required => "required", + Self::Optional => "optional", + } + } +} + +/// Where a transport sits in the presence cycle. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum Presence { + /// The interface is not there (or is there without carrier). No socket, no + /// loops; the watcher is waiting. + Absent, + /// The interface appeared and a bind is in flight, or a non-absence bind + /// failure is backing off. + Binding, + /// Bound, with a live socket and running loops. + Present, +} + +impl Presence { + fn from_u8(v: u8) -> Self { + match v { + 1 => Self::Binding, + 2 => Self::Present, + _ => Self::Absent, + } + } + + fn as_u8(self) -> u8 { + match self { + Self::Absent => 0, + Self::Binding => 1, + Self::Present => 2, + } + } + + /// Operator-facing label, used by `show_transports`. + pub fn as_str(self) -> &'static str { + match self { + Self::Absent => "absent", + Self::Binding => "binding", + Self::Present => "present", + } + } +} + +impl std::fmt::Display for Presence { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(self.as_str()) + } +} + +/// Shared, lock-light presence state. +/// +/// Written by the transport's binder task, read by `send`, by the control +/// plane, and by tests. Held behind an `Arc` so the binder task can outlive +/// any particular borrow of the transport. +#[derive(Debug)] +pub struct PresenceState { + phase: AtomicU8, + /// Successful binds since the transport was created. `1` after a clean + /// start; every increment past that is a rebind. + binds: AtomicU64, + /// Failed bind attempts since the last successful bind. Reset on bind. + attempts: AtomicU32, + /// When the current phase was entered. + since: RwLock, + /// MAC observed at the last successful bind. A name reappearing with a + /// different MAC is different hardware, not the same device returning. + last_mac: RwLock>, +} + +impl Default for PresenceState { + fn default() -> Self { + Self::new() + } +} + +impl PresenceState { + /// A fresh tracker in [`Presence::Absent`]. + pub fn new() -> Self { + Self { + phase: AtomicU8::new(Presence::Absent.as_u8()), + binds: AtomicU64::new(0), + attempts: AtomicU32::new(0), + since: RwLock::new(Instant::now()), + last_mac: RwLock::new(None), + } + } + + /// Current phase. + pub fn presence(&self) -> Presence { + Presence::from_u8(self.phase.load(Ordering::Acquire)) + } + + /// Whether the transport currently holds a bound socket. + pub fn is_present(&self) -> bool { + self.presence() == Presence::Present + } + + /// How long the current presence *episode* has been held. + /// + /// Episode, not phase: `Binding` is part of the absence episode until it + /// succeeds. An interface that has been gone for a week while a bind is + /// retried and refused every second must report a week, not one second — + /// otherwise [`ABSENCE_ERROR_AFTER`] is never reached and the + /// operator-facing `since_secs` reads as a healthy young absence forever. + pub fn since(&self) -> Duration { + read(&self.since).elapsed() + } + + /// Restart the episode clock, for a transport that is about to start. + /// + /// `new()` stamps the clock at construction, but construction and + /// `start_async` need not be adjacent — config load and supervisor staging + /// sit between them. Left alone, a transport staged for longer than + /// [`ABSENCE_ERROR_AFTER`] logs the sustained-absence error on its very + /// first binder tick, having given the interface no bring-up window at + /// all. The window is supposed to absorb exactly that race. + pub fn mark_starting(&self) { + *write(&self.since) = Instant::now(); + } + + /// Test-only: age the episode clock, so deadline behaviour can be tested + /// without sleeping through it. + /// + /// `ABSENCE_ERROR_AFTER` is ten seconds. A test that waited it out would + /// be ten seconds of nothing, which is how deadline logic ends up + /// untested. + #[cfg(test)] + pub(crate) fn backdate_for_test(&self, by: Duration) { + let mut since = write(&self.since); + *since = since.checked_sub(by).unwrap_or(*since); + } + + /// Successful binds since creation (`1` after a clean start). + pub fn binds(&self) -> u64 { + self.binds.load(Ordering::Relaxed) + } + + /// Failed bind attempts since the last successful bind. + pub fn attempts(&self) -> u32 { + self.attempts.load(Ordering::Relaxed) + } + + /// MAC observed at the last successful bind, if any. + pub fn last_mac(&self) -> Option<[u8; 6]> { + *read(&self.last_mac) + } + + /// Move to `phase`, returning `true` if this was an actual edge. + /// + /// Edge-vs-level is what the logging policy keys on: logged once on + /// entering absence and once on recovery, never per retry attempt. + /// + /// The episode clock ([`Self::since`]) restarts only when *boundness* + /// changes — Present↔not-Present. A failed bind walks + /// `Absent → Binding → Absent`, and resetting the clock on those would + /// hide a permanent absence behind a timer that never gets past one + /// second. + pub fn transition(&self, phase: Presence) -> bool { + let prev = Presence::from_u8(self.phase.swap(phase.as_u8(), Ordering::AcqRel)); + if prev == phase { + return false; + } + if (prev == Presence::Present) != (phase == Presence::Present) { + *write(&self.since) = Instant::now(); + } + true + } + + /// Record a successful bind at `mac`. + /// + /// Returns `true` when the interface came back as *different hardware* — + /// the name reappeared with a MAC other than the one last bound. The + /// caller drops cached neighbor state rather than silently resuming onto + /// a different adapter. + pub fn record_bind(&self, mac: [u8; 6]) -> bool { + let changed = match *read(&self.last_mac) { + Some(prev) => prev != mac, + None => false, + }; + *write(&self.last_mac) = Some(mac); + self.binds.fetch_add(1, Ordering::Relaxed); + self.attempts.store(0, Ordering::Relaxed); + self.transition(Presence::Present); + changed + } + + /// Record a failed bind attempt, returning the new attempt count. + pub fn record_attempt(&self) -> u32 { + self.attempts.fetch_add(1, Ordering::Relaxed) + 1 + } +} + +/// Backoff for bind failures that are *not* absence — permission denied, +/// buffer sizing, a BPF device shortage. Absence itself does not back off +/// where an event source is available: there is nothing to poll. +/// +/// 1 s doubling to a 30 s ceiling. +pub fn bind_backoff(attempts: u32) -> Duration { + const BASE_SECS: u64 = 1; + const CEILING_SECS: u64 = 30; + let shift = attempts.saturating_sub(1).min(5); + Duration::from_secs((BASE_SECS << shift).min(CEILING_SECS)) +} + +/// How long a *required* interface may be absent before it is an error. +/// +/// Absence is a state the presence machine handles, so it is not an error for +/// happening — a daemon that wins the race against its own radio, or a cable +/// out for two seconds, is the ordinary case this mechanism exists to absorb, +/// and calling that an error at t=0 and "recovered" at t=0.2 s is cry-wolf. +/// Past this window it is no longer a race: something an operator has to fix +/// is wrong, and the log should say so once. +/// +/// One window for both shapes of absence. A node that boots before its wifi +/// and a node whose wifi reloads at 03:00 take one code path everywhere else +/// in this module; giving them different deadlines would reintroduce exactly +/// the start-versus-runtime asymmetry the presence machine removed. +/// +/// Tuned against the platforms this exists for: comfortably past a veth or a +/// container coming up, short enough that a mesh radio which never appears is +/// named while somebody is still watching the boot. Raising it hides a real +/// fault for longer; lowering it starts reporting ordinary bring-up races. +pub const ABSENCE_ERROR_AFTER: Duration = Duration::from_secs(10); + +/// Minimum lifetime for a binding to count as a real recovery. +/// +/// A socket that dies sooner than this never really came back. +pub const MIN_STABLE_BINDING: Duration = Duration::from_secs(10); + +/// Consecutive short-lived bindings before the binder stops treating a +/// successful bind as a recovery. +pub const CHURN_THRESHOLD: u32 = 3; + +/// Damping for the rebind loop. +/// +/// Backoff covers *failed* binds; this covers the opposite and nastier case — +/// binds that keep **succeeding** into a socket that dies moments later. A +/// receive loop that gives up on a persistent error while the interface stays +/// `UP` produces exactly that: tear down, rebind, succeed, fail again, once +/// per second, forever. Undamped it is an `error!`/`info!` pair and a +/// `Degraded`→`Running` health flap every cycle, which defeats both the +/// "log edges, not attempts" rule and the meaning of `Degraded`. +/// +/// So: count consecutive bindings that die young, back off between them on +/// the same 1 s → 30 s curve, and once the streak reaches +/// [`CHURN_THRESHOLD`] stop announcing each bind as a recovery — hold the +/// node at its degraded reading until a binding actually survives +/// [`MIN_STABLE_BINDING`]. A binding that holds ends the streak. +/// +/// Pure state, driven by an injected clock, so the policy is testable without +/// a network interface. +#[derive(Debug, Default)] +pub struct ChurnGuard { + /// Consecutive bindings that died younger than [`MIN_STABLE_BINDING`]. + streak: u32, + /// When the current binding was established. + bound_at: Option, + /// Whether the current binding was announced as a recovery. + announced: bool, +} + +/// What the caller should do about a successful bind. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct BindOutcome { + /// Log the recovery and publish presence now. `false` while churning: + /// the bind is held back until it proves it will last. + pub announce: bool, +} + +/// What the caller should do about a detach. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct DetachOutcome { + /// Publish absence. `false` when the binding that just died was never + /// announced, so there is nothing to retract. + pub retract: bool, + /// Log the detach edge. `false` once churning — the fact has been said. + pub log_edge: bool, + /// This detach is the one that crossed [`CHURN_THRESHOLD`]; say so once. + pub entered_churn: bool, + /// Wait this long before trying to bind again. + pub backoff: Option, +} + +impl ChurnGuard { + /// A guard with no history. + pub fn new() -> Self { + Self::default() + } + + /// Consecutive short-lived bindings, for logging and tests. + pub fn streak(&self) -> u32 { + self.streak + } + + /// Record a successful bind. + pub fn bound(&mut self, now: Instant) -> BindOutcome { + self.bound_at = Some(now); + let announce = self.streak < CHURN_THRESHOLD; + if announce { + self.announced = true; + } + BindOutcome { announce } + } + + /// Called on every tick while bound. Returns `true` exactly once, at the + /// moment a held-back binding has proved stable and should be announced. + pub fn stabilized(&mut self, now: Instant) -> bool { + let Some(bound_at) = self.bound_at else { + return false; + }; + if now.duration_since(bound_at) < MIN_STABLE_BINDING { + return false; + } + let newly_announced = !self.announced; + self.streak = 0; + self.announced = true; + newly_announced + } + + /// Record a detach. + pub fn detached(&mut self, now: Instant) -> DetachOutcome { + let young = self + .bound_at + .is_some_and(|t| now.duration_since(t) < MIN_STABLE_BINDING); + self.bound_at = None; + + if young { + self.streak += 1; + } else { + self.streak = 0; + } + + DetachOutcome { + retract: std::mem::take(&mut self.announced), + log_edge: self.streak < CHURN_THRESHOLD, + entered_churn: self.streak == CHURN_THRESHOLD, + backoff: (self.streak > 0).then(|| bind_backoff(self.streak)), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::sync::Arc; + + #[test] + fn presence_starts_absent() { + let p = PresenceState::new(); + assert_eq!(p.presence(), Presence::Absent); + assert_eq!(p.binds(), 0); + assert_eq!(p.attempts(), 0); + assert!(p.last_mac().is_none()); + } + + #[test] + fn transition_reports_edges_only() { + let p = PresenceState::new(); + assert!(p.transition(Presence::Binding)); + assert!(!p.transition(Presence::Binding)); + assert!(p.transition(Presence::Present)); + } + + #[test] + fn bind_records_mac_and_clears_attempts() { + let p = PresenceState::new(); + p.record_attempt(); + p.record_attempt(); + assert_eq!(p.attempts(), 2); + + let mac = [0x02, 0, 0, 0, 0, 1]; + assert!(!p.record_bind(mac), "first bind is not a hardware change"); + assert_eq!(p.attempts(), 0); + assert_eq!(p.binds(), 1); + assert_eq!(p.presence(), Presence::Present); + assert_eq!(p.last_mac(), Some(mac)); + } + + #[test] + fn rebind_on_same_mac_is_not_a_hardware_change() { + let p = PresenceState::new(); + let mac = [0x02, 0, 0, 0, 0, 1]; + p.record_bind(mac); + p.transition(Presence::Absent); + assert!(!p.record_bind(mac)); + assert_eq!(p.binds(), 2); + } + + #[test] + fn rebind_on_different_mac_is_a_hardware_change() { + let p = PresenceState::new(); + p.record_bind([0x02, 0, 0, 0, 0, 1]); + p.transition(Presence::Absent); + assert!(p.record_bind([0x02, 0, 0, 0, 0, 2])); + } + + #[test] + fn backoff_climbs_to_a_ceiling() { + assert_eq!(bind_backoff(0), Duration::from_secs(1)); + assert_eq!(bind_backoff(1), Duration::from_secs(1)); + assert_eq!(bind_backoff(2), Duration::from_secs(2)); + assert_eq!(bind_backoff(3), Duration::from_secs(4)); + assert_eq!(bind_backoff(6), Duration::from_secs(30)); + assert_eq!(bind_backoff(u32::MAX), Duration::from_secs(30)); + } + + #[test] + fn the_error_deadline_outlasts_an_ordinary_bring_up_race() { + // The window has to clear the races it exists to absorb — a veth + // arriving a fraction of a second late, a container starting — while + // staying short enough that a radio which never appears is named + // during the boot somebody is watching. It is also the one deadline: + // start-time absence and a runtime detach share it. + assert!( + ABSENCE_ERROR_AFTER >= Duration::from_secs(5), + "shorter than a bring-up race would report the ordinary case" + ); + assert!( + ABSENCE_ERROR_AFTER <= Duration::from_secs(60), + "longer and a required interface that never appears goes unsaid \ + for the whole boot" + ); + } + + // ── The absence clock measures an episode, not a phase ──────────────── + + #[test] + fn a_failed_bind_does_not_restart_the_absence_clock() { + // An interface that is present but refuses to bind walks + // Absent → Binding → Absent on every retry. If those edges reset the + // clock, `since_secs` reads as a one-second-old absence forever and + // the error deadline is never reached — so a permission error would + // sit silently behind a healthy-looking counter. + let p = PresenceState::new(); + std::thread::sleep(Duration::from_millis(30)); + let before = p.since(); + + assert!(p.transition(Presence::Binding)); + assert!(p.transition(Presence::Absent)); + assert!(p.transition(Presence::Binding)); + assert!(p.transition(Presence::Absent)); + + assert!( + p.since() >= before, + "the absence clock ran backwards across failed binds" + ); + } + + #[test] + fn the_clock_restarts_only_when_boundness_changes() { + let p = PresenceState::new(); + std::thread::sleep(Duration::from_millis(30)); + + // Absent → Present restarts it: a new episode began. + p.record_bind([0x02, 0, 0, 0, 0, 1]); + assert!(p.since() < Duration::from_millis(30)); + + std::thread::sleep(Duration::from_millis(30)); + let bound_for = p.since(); + + // Present → Present is not an edge at all. + assert!(!p.transition(Presence::Present)); + assert!(p.since() >= bound_for); + + // Present → Absent restarts it: the episode ended. + assert!(p.transition(Presence::Absent)); + assert!(p.since() < Duration::from_millis(30)); + } + + // ── Poisoning must not be a stuck state ─────────────────────────────── + + #[test] + fn a_poisoned_lock_still_reports_presence() { + // `.ok()`-style handling would make a poisoned lock read as "no MAC, + // no socket, tasks dead" — a transport reporting itself present while + // every send fails, and a binder rebinding once a second forever. + // Poisoning carries no information about plain data, so it is ignored. + let p = Arc::new(PresenceState::new()); + p.record_bind([0x02, 0, 0, 0, 0, 7]); + + let poisoner = Arc::clone(&p); + let panicked = std::thread::spawn(move || { + let _guard = poisoner.last_mac.write().unwrap(); + panic!("poison the lock while holding it"); + }) + .join(); + assert!(panicked.is_err(), "the helper thread was supposed to panic"); + assert!( + p.last_mac.is_poisoned(), + "the lock was supposed to be poisoned" + ); + + assert_eq!( + p.last_mac(), + Some([0x02, 0, 0, 0, 0, 7]), + "a poisoned lock must not erase the binding" + ); + // And the clock still answers rather than collapsing to zero. + let _ = p.since(); + } + + // ── Rebind churn ────────────────────────────────────────────────────── + + #[test] + fn a_healthy_bind_and_detach_is_not_churn() { + let mut g = ChurnGuard::new(); + let t0 = Instant::now(); + assert!(g.bound(t0).announce, "a first bind is a recovery"); + + // Held well past the stability floor, then lost. + let out = g.detached(t0 + MIN_STABLE_BINDING + Duration::from_secs(60)); + assert!(out.retract, "an announced binding must be retracted"); + assert!(out.log_edge, "an isolated detach is worth a line"); + assert!(!out.entered_churn); + assert_eq!(out.backoff, None, "one clean outage must not back off"); + assert_eq!(g.streak(), 0); + } + + #[test] + fn an_unseeded_guard_retracts_nothing_and_never_repairs_itself() { + // Why `binder_loop` seeds the guard when it inherits a binding from + // `start_async`, rather than leaving it fresh. + // + // A guard that was never told about a bind believes it has announced + // nothing, so it asks for no retraction — and health, which learned + // `present: true` from the inline bind, would keep reading `Full` with + // the interface gone. `stabilized` cannot rescue it either: with no + // `bound_at` there is nothing for it to judge stable. + let mut g = ChurnGuard::new(); + let t0 = Instant::now(); + + assert!( + !g.stabilized(t0 + MIN_STABLE_BINDING + Duration::from_secs(60)), + "a guard with no recorded bind has nothing to stabilize" + ); + + let out = g.detached(t0 + Duration::from_secs(60)); + assert!( + !out.retract, + "an unseeded guard retracts nothing — which is exactly why the \ + binder must seed it from the inline bind" + ); + } + + #[test] + fn a_seeded_guard_retracts_the_edge_the_inline_bind_published() { + // The fix, from the binder's angle: seeding with `bound` is what makes + // the first detach after a clean start reach node health. + let mut g = ChurnGuard::new(); + let t0 = Instant::now(); + g.bound(t0); + + let out = g.detached(t0 + MIN_STABLE_BINDING + Duration::from_secs(60)); + assert!( + out.retract, + "the edge `start_async` published must be retracted on detach" + ); + assert!(out.log_edge); + } + + #[test] + fn short_lived_bindings_back_off() { + // The failure this guards: a receive loop that gives up on a + // persistent error while the interface stays UP. Bind succeeds, dies, + // rebinds, dies — once per second, forever, undamped. + let mut g = ChurnGuard::new(); + let mut t = Instant::now(); + + for expected in [1u64, 2, 4] { + g.bound(t); + t += Duration::from_secs(1); + let out = g.detached(t); + assert_eq!( + out.backoff, + Some(Duration::from_secs(expected)), + "streak {} should back off {expected}s", + g.streak() + ); + } + assert_eq!(g.streak(), 3); + } + + #[test] + fn churn_stops_announcing_and_stops_logging() { + let mut g = ChurnGuard::new(); + let mut t = Instant::now(); + + // The detaches below the threshold are still news and still logged. + for _ in 0..CHURN_THRESHOLD - 1 { + assert!(g.bound(t).announce); + t += Duration::from_secs(1); + let out = g.detached(t); + assert!(out.log_edge, "the first few detaches are still news"); + assert!(!out.entered_churn); + } + + // The detach that crosses the threshold reports the churn instead of + // the edge: one line saying "this keeps happening", not two saying + // "it happened" and "it keeps happening". + assert!(g.bound(t).announce); + t += Duration::from_secs(1); + let crossing = g.detached(t); + assert!(crossing.entered_churn, "crossing must be announced once"); + assert!(!crossing.log_edge, "the churn line replaces the edge line"); + + // Past the threshold: bindings are no longer announced as recoveries, + // so node health stays put instead of flapping every second, and the + // edges stop being logged. + for _ in 0..5 { + assert!(!g.bound(t).announce, "a churning bind is not a recovery"); + t += Duration::from_secs(1); + let out = g.detached(t); + assert!(!out.log_edge, "churn must not log per cycle"); + assert!(!out.retract, "nothing was announced, so nothing to retract"); + assert!( + !out.entered_churn, + "the threshold is crossed once, not repeatedly" + ); + } + } + + #[test] + fn backoff_during_churn_is_capped() { + let mut g = ChurnGuard::new(); + let mut t = Instant::now(); + let mut last = None; + for _ in 0..12 { + g.bound(t); + t += Duration::from_secs(1); + last = g.detached(t).backoff; + } + assert_eq!( + last, + Some(Duration::from_secs(30)), + "churn backoff must climb to the ceiling and stop" + ); + } + + #[test] + fn a_binding_that_lasts_ends_the_streak_and_announces_once() { + let mut g = ChurnGuard::new(); + let mut t = Instant::now(); + + // Churn into the held-back state. + for _ in 0..CHURN_THRESHOLD + 1 { + g.bound(t); + t += Duration::from_secs(1); + g.detached(t); + } + assert!(!g.bound(t).announce); + + // Not yet stable: still nothing to say. + assert!(!g.stabilized(t + Duration::from_secs(1))); + + // Survived the floor: announce exactly once, and the streak is over. + let stable_at = t + MIN_STABLE_BINDING; + assert!( + g.stabilized(stable_at), + "a binding that lasts is a recovery" + ); + assert!( + !g.stabilized(stable_at + Duration::from_secs(60)), + "recovery is announced once, not on every tick" + ); + assert_eq!(g.streak(), 0); + + // And the next detach behaves like an ordinary one again. + let out = g.detached(stable_at + Duration::from_secs(60)); + assert!(out.retract); + assert!(out.log_edge); + assert_eq!(out.backoff, None); + } + + #[test] + fn stabilized_is_silent_for_an_ordinary_binding() { + // A bind that was announced immediately must not be announced again + // when it passes the stability floor. + let mut g = ChurnGuard::new(); + let t = Instant::now(); + assert!(g.bound(t).announce); + assert!(!g.stabilized(t + MIN_STABLE_BINDING + Duration::from_secs(1))); + } + + #[test] + fn policy_labels() { + assert_eq!(AbsencePolicy::from_optional(true), AbsencePolicy::Optional); + assert_eq!(AbsencePolicy::from_optional(false), AbsencePolicy::Required); + assert!(AbsencePolicy::Optional.is_optional()); + assert!(!AbsencePolicy::Required.is_optional()); + assert_eq!(AbsencePolicy::Required.as_str(), "required"); + // Both labels, not just one. `show_transports` renders this string and + // fipstop's severity split keys on it, so a swapped pair would paint + // every expected interface as the tolerated kind and vice versa — + // while a test that checks only `Required` stays green through it. + assert_eq!(AbsencePolicy::Optional.as_str(), "optional"); + } + + #[test] + fn a_first_bind_is_not_a_hardware_change() { + // The boundary the flush hangs off. `record_bind` returns "different + // hardware", and on the very first bind there is no previous MAC to + // differ from — so it must answer false, or every clean start would + // drop a neighbour cache it had just built and log a hardware swap + // that never happened. + let state = PresenceState::new(); + assert!( + !state.record_bind([1, 2, 3, 4, 5, 6]), + "the first bind has nothing to differ from" + ); + assert_eq!(state.binds(), 1); + assert_eq!(state.presence(), Presence::Present); + } + + #[test] + fn a_rebind_on_new_hardware_reports_the_change_once() { + // And it reports the change once, not on every subsequent bind: the + // caller drops its cached neighbours on a `true`, so a sticky answer + // would flush the cache on every rebind forever. + let state = PresenceState::new(); + state.record_bind([1, 2, 3, 4, 5, 6]); + + assert!( + state.record_bind([0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff]), + "a name returning on a different MAC is different hardware" + ); + assert!( + !state.record_bind([0xaa, 0xbb, 0xcc, 0xdd, 0xee, 0xff]), + "the same hardware rebinding is not a change" + ); + assert_eq!(state.binds(), 3); + } + + #[test] + fn every_presence_label_is_distinct_and_round_trips() { + // The labels are the operator-facing vocabulary — `show_transports` + // emits them and the fipstop State column renders them — and + // `Presence::as_str` had no test at all, so `binding` in particular was + // never observed by anything. + use std::collections::HashSet; + let all = [Presence::Absent, Presence::Binding, Presence::Present]; + let labels: HashSet<&str> = all.iter().map(|p| p.as_str()).collect(); + assert_eq!(labels.len(), 3, "each phase needs its own label"); + assert!(labels.contains("binding")); + + for phase in all { + assert_eq!( + Presence::from_u8(phase.as_u8()), + phase, + "{phase} must survive the atomic round trip the state uses" + ); + assert_eq!(phase.to_string(), phase.as_str(), "Display must agree"); + } + + // Anything outside the enum reads as absent rather than panicking: the + // byte comes out of an AtomicU8 that a torn write could leave at any + // value, and the safe answer there is "not bound". + assert_eq!(Presence::from_u8(99), Presence::Absent); + } +} diff --git a/src/transport/ethernet/watcher.rs b/src/transport/ethernet/watcher.rs new file mode 100644 index 00000000..5e00b6bf --- /dev/null +++ b/src/transport/ethernet/watcher.rs @@ -0,0 +1,337 @@ +//! Link-event sources for the interface presence watcher. +//! +//! The presence machine works on a 1-second poll alone. This module removes +//! the latency: where the kernel offers a link-event source, the binder blocks +//! on it and reacts in sub-second time, and the poll stays underneath as a +//! backstop rather than as the mechanism. +//! +//! | Platform | Source | +//! | -------- | ------ | +//! | Linux | netlink `RTNLGRP_LINK` (`RTM_NEWLINK` / `RTM_DELLINK`) | +//! | macOS, FreeBSD | `PF_ROUTE` socket, `RTM_IFINFO` | +//! | Fallback | poll `getifaddrs` + flags, 1 s | +//! +//! The messages themselves are deliberately **not parsed**. A link event is a +//! hint to re-run the presence probe, which is cheap and authoritative; +//! decoding `nlmsghdr`/`ifinfomsg` payloads to reach the same answer would add +//! a parser whose bugs would be presence bugs. Any event on the socket wakes +//! the binder, which then asks +//! [`interface_present`](super::io::interface_present). +//! +//! Construction is best-effort. A kernel or sandbox that refuses the socket +//! yields a watcher that never fires, and the binder degrades to its poll. + +use std::os::unix::io::{AsRawFd, RawFd}; +use std::sync::atomic::{AtomicBool, AtomicU32, Ordering}; +use std::time::Duration; + +use tokio::io::unix::AsyncFd; +use tracing::{debug, warn}; + +/// Consecutive receive errors before the event source is abandoned for the +/// caller's poll. +const ERROR_GIVE_UP: u32 = 5; + +/// An owned link-event socket. Closes its descriptor on drop. +struct LinkEventSocket { + fd: RawFd, +} + +impl AsRawFd for LinkEventSocket { + fn as_raw_fd(&self) -> RawFd { + self.fd + } +} + +impl Drop for LinkEventSocket { + fn drop(&mut self) { + unsafe { libc::close(self.fd) }; + } +} + +impl LinkEventSocket { + fn recv(&self, buf: &mut [u8]) -> std::io::Result { + let n = unsafe { libc::recv(self.fd, buf.as_mut_ptr() as *mut libc::c_void, buf.len(), 0) }; + if n < 0 { + Err(std::io::Error::last_os_error()) + } else { + Ok(n as usize) + } + } +} + +/// Open the platform's link-event socket, non-blocking. +#[cfg(target_os = "linux")] +fn open_link_socket() -> std::io::Result { + // RTMGRP_LINK. Spelled as a literal because the constant's name and + // availability differ across libc versions; the value is ABI. + const RTMGRP_LINK: u32 = 1; + + let fd = unsafe { + libc::socket( + libc::AF_NETLINK, + libc::SOCK_RAW | libc::SOCK_NONBLOCK | libc::SOCK_CLOEXEC, + libc::NETLINK_ROUTE, + ) + }; + if fd < 0 { + return Err(std::io::Error::last_os_error()); + } + let socket = LinkEventSocket { fd }; + + let mut sa: libc::sockaddr_nl = unsafe { std::mem::zeroed() }; + sa.nl_family = libc::AF_NETLINK as u16; + sa.nl_groups = RTMGRP_LINK; + let ret = unsafe { + libc::bind( + fd, + &sa as *const libc::sockaddr_nl as *const libc::sockaddr, + std::mem::size_of::() as libc::socklen_t, + ) + }; + if ret < 0 { + return Err(std::io::Error::last_os_error()); + } + Ok(socket) +} + +/// Open the platform's link-event socket, non-blocking. +#[cfg(not(target_os = "linux"))] +fn open_link_socket() -> std::io::Result { + // PF_ROUTE delivers RTM_IFINFO (and the rest of the routing messages) to + // every reader; no bind and no group selection exist for it. + let fd = unsafe { libc::socket(libc::PF_ROUTE, libc::SOCK_RAW, libc::AF_UNSPEC) }; + if fd < 0 { + return Err(std::io::Error::last_os_error()); + } + let socket = LinkEventSocket { fd }; + + let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) }; + if flags < 0 { + return Err(std::io::Error::last_os_error()); + } + if unsafe { libc::fcntl(fd, libc::F_SETFL, flags | libc::O_NONBLOCK) } < 0 { + return Err(std::io::Error::last_os_error()); + } + Ok(socket) +} + +/// A source of "something about the links changed" wake-ups. +pub(crate) struct LinkWatcher { + /// `None` when no event source could be opened — the caller's poll is then + /// the whole mechanism, which is exactly the documented fallback. + inner: Option>, + /// Consecutive receive errors. Reset by any successful read. + errors: AtomicU32, + /// Set once the source has been abandoned for good. + /// + /// Abandonment has to outlive the future that decided it. `changed()` is + /// called fresh on every pass of the caller's `select!` and dropped + /// whenever the poll ticker wins, so a `pending()` inside that future + /// parks nothing beyond the current pass — without this flag the next pass + /// re-reads the dead socket, re-counts the error, and re-logs the + /// give-up warning, once per wake-up, forever. + given_up: AtomicBool, +} + +impl LinkWatcher { + /// Open the platform link-event source, falling back to nothing. + pub(crate) fn new() -> Self { + let inner = match open_link_socket() { + Ok(socket) => match AsyncFd::new(socket) { + Ok(afd) => Some(afd), + Err(e) => { + debug!(error = %e, "Link event socket not registrable; polling instead"); + None + } + }, + Err(e) => { + debug!(error = %e, "No link event source available; polling instead"); + None + } + }; + Self { + inner, + errors: AtomicU32::new(0), + given_up: AtomicBool::new(false), + } + } + + /// Whether an event source is actually backing this watcher. + pub(crate) fn is_event_driven(&self) -> bool { + self.inner.is_some() + } + + /// Resolve when the kernel reports a link change. + /// + /// Never resolves when no event source is available, which makes it safe + /// to `select!` against the poll ticker: the ticker simply always wins. + pub(crate) async fn changed(&self) { + let Some(afd) = &self.inner else { + std::future::pending::<()>().await; + unreachable!("pending never resolves") + }; + + // Already abandoned on an earlier pass. Park without touching the + // socket, so giving up costs one syscall in total rather than one per + // caller wake-up for the life of the process. + if self.given_up.load(Ordering::Relaxed) { + std::future::pending::<()>().await; + unreachable!("pending never resolves") + } + + loop { + let Ok(mut guard) = afd.readable().await else { + // The registration died. Stop firing rather than spinning; the + // caller's poll continues to cover presence. + std::future::pending::<()>().await; + unreachable!("pending never resolves") + }; + + // Drain to WouldBlock so a burst of link messages is one wake-up + // and the socket buffer does not fill behind us. + let mut buf = [0u8; 4096]; + let mut saw_event = false; + let mut failure = None; + loop { + match guard.try_io(|inner| inner.get_ref().recv(&mut buf)) { + Ok(Ok(n)) if n > 0 => saw_event = true, + // A zero-length read. Readiness is *not* cleared by + // `try_io` here — it clears only on `WouldBlock` — so + // breaking out plainly would leave `readable()` instantly + // ready with nothing to read, and this loop would spin + // without ever returning `Pending`. That starves the + // caller's `select!` of its poll ticker entirely, which + // takes presence detection down with it. Clear it by hand + // and treat it as a fault, so the give-up path applies. + Ok(Ok(_)) => { + guard.clear_ready(); + failure = Some(std::io::Error::from(std::io::ErrorKind::UnexpectedEof)); + break; + } + // A genuine socket error. Distinct from WouldBlock, and + // the distinction is the whole point: `try_io` clears + // readiness only on WouldBlock, so breaking out of a real + // error leaves `readable()` instantly ready, `recv` + // failing again, and the loop spinning a core flat with + // nothing logged. Clear it by hand and back off. + Ok(Err(e)) => { + guard.clear_ready(); + failure = Some(e); + break; + } + // WouldBlock — readiness is cleared, drain complete. + Err(_) => break, + } + } + + if saw_event { + self.errors.store(0, Ordering::Relaxed); + return; + } + + if let Some(e) = failure { + let errors = self.errors.fetch_add(1, Ordering::Relaxed) + 1; + if errors == 1 { + // ENOBUFS is the realistic one: a burst of link events + // overflowed the socket buffer, so the kernel dropped some. + // Losing events is survivable — the caller polls — but the + // spin is not, and neither is doing it silently. + warn!(error = %e, "Link event source read failed"); + } + if errors >= ERROR_GIVE_UP { + self.given_up.store(true, Ordering::Relaxed); + warn!( + errors, + "Link event source is not recoverable; falling back to \ + polling for interface presence" + ); + std::future::pending::<()>().await; + unreachable!("pending never resolves") + } + tokio::time::sleep(Duration::from_millis(100) * errors).await; + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The watcher must construct on any host, with or without a usable event + /// source, because the binder builds one unconditionally. + #[tokio::test] + async fn watcher_constructs_and_reports_its_backing() { + let w = LinkWatcher::new(); + // Both answers are legitimate — a sandbox may refuse the socket — so + // this pins that asking is safe, not which answer comes back. + let _ = w.is_event_driven(); + } + + /// A descriptor whose `recv` always fails must not become a busy loop. + /// + /// `try_io` clears readiness only on `WouldBlock`. Breaking out of a real + /// error left `readable()` instantly ready, `recv` failing again, and the + /// loop spinning a core flat with nothing logged — the realistic trigger + /// being `ENOBUFS` when a burst of link events overflows the socket + /// buffer. A pipe stands in for that here: `recv` on one answers + /// `ENOTSOCK`, every time, which is exactly the shape of a persistent + /// error. + #[tokio::test] + async fn a_persistently_failing_source_gives_up_instead_of_spinning() { + let mut fds = [0i32; 2]; + assert_eq!(unsafe { libc::pipe(fds.as_mut_ptr()) }, 0, "pipe()"); + let (read_fd, write_fd) = (fds[0], fds[1]); + + // AsyncFd requires a non-blocking descriptor. + let flags = unsafe { libc::fcntl(read_fd, libc::F_GETFL) }; + assert!(unsafe { libc::fcntl(read_fd, libc::F_SETFL, flags | libc::O_NONBLOCK) } >= 0); + + let watcher = LinkWatcher { + inner: Some(AsyncFd::new(LinkEventSocket { fd: read_fd }).expect("register")), + errors: AtomicU32::new(0), + given_up: AtomicBool::new(false), + }; + + // Keep producing readiness edges. The error arm calls `clear_ready`, + // and a descriptor that was already readable before re-registration + // may never deliver another edge on its own — which would stall the + // loop at one error and hide whether the give-up path works. A steady + // trickle stands in for the burst of link events that provokes the + // real failure. + let writer = tokio::task::spawn_blocking(move || { + for _ in 0..200 { + if unsafe { libc::write(write_fd, b"x".as_ptr().cast(), 1) } < 0 { + break; + } + std::thread::sleep(Duration::from_millis(25)); + } + unsafe { libc::close(write_fd) }; + }); + + // Never resolves — there is no event to report — but it must reach the + // give-up state rather than burn until the timeout. + let fired = tokio::time::timeout(Duration::from_secs(5), watcher.changed()).await; + assert!(fired.is_err(), "a failing source must not report an event"); + assert!( + watcher.errors.load(Ordering::Relaxed) >= ERROR_GIVE_UP, + "the error path must count, back off and stop, not spin silently" + ); + + writer.abort(); + } + + /// A watcher with no event source must never resolve, so a `select!` + /// against the poll ticker degrades cleanly instead of spinning. + #[tokio::test] + async fn a_sourceless_watcher_never_fires() { + let w = LinkWatcher { + inner: None, + errors: AtomicU32::new(0), + given_up: AtomicBool::new(false), + }; + let fired = tokio::time::timeout(std::time::Duration::from_millis(50), w.changed()).await; + assert!(fired.is_err(), "sourceless watcher resolved"); + } +} diff --git a/src/transport/mod.rs b/src/transport/mod.rs index 6156cdcb..088c959c 100644 --- a/src/transport/mod.rs +++ b/src/transport/mod.rs @@ -112,6 +112,62 @@ pub fn packet_channel(buffer: usize) -> (PacketTx, PacketRx) { tokio::sync::mpsc::channel(buffer) } +/// Operator-visible interface presence, rendered by `show_transports`. +/// +/// Worth as much as the retry itself. The original boot-race bug was expensive +/// precisely because the 802.11s peer link formed regardless of the daemon, so +/// nothing an operator could see said the node was deaf. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct InterfacePresence { + /// `absent`, `binding`, or `present`. + pub presence: &'static str, + /// Whether the interface currently has carrier (`IFF_RUNNING`). + /// + /// Reported, never acted on. Presence is `IFF_UP`, because binding does + /// not need carrier and a socket outlives a carrier flap — but "is + /// anything plugged in" is still what an operator wants to know when a + /// bound transport is carrying nothing, so it is reported here instead of + /// steering the daemon. + pub carrier: bool, + /// `required` or `optional`. + pub policy: &'static str, + /// How long the current phase has been held. + pub since_secs: u64, + /// Successful binds since the transport was created (`1` after a clean + /// start; more means it has rebound). + pub binds: u64, + /// Failed bind attempts since the last successful bind. + pub failed_attempts: u32, +} + +/// A presence edge published by an interface-bound transport. +/// +/// Absence and return are the same transition seen from two sides, so one +/// event type carries both: `present: false` on detach (including a start +/// where the interface was never there), `present: true` on every successful +/// bind after the first observation. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub struct TransportPresence { + /// The transport whose interface changed presence. + pub transport_id: TransportId, + /// Whether the interface is now bound. + pub present: bool, + /// Whether this edge should move node health. + /// + /// `false` for an `optional` interface, whose absence is normal and must + /// not take the node off `Full`. The edge is still published, because the + /// bound set changed either way and the node's egress MTU floor is derived + /// from it — health and MTU are two different questions riding one + /// channel, and only the first one is policy-filtered. + pub health_relevant: bool, +} + +/// Channel sender for transport presence edges. +pub type PresenceTx = tokio::sync::mpsc::Sender; + +/// Channel receiver for transport presence edges. +pub type PresenceRx = tokio::sync::mpsc::Receiver; + // ============================================================================ // Errors // ============================================================================ @@ -128,6 +184,20 @@ pub enum TransportError { #[error("transport failed to start: {0}")] StartFailed(String), + /// The named network interface is not usable right now: it does not exist, + /// or it exists but is administratively down (no `IFF_UP`). + /// + /// Distinct from [`TransportError::StartFailed`] because absence is a + /// *state*, not a fault. Interface-bound transports treat it as "not bound + /// yet" and keep a presence watcher running; a `StartFailed` carrying the + /// same text could not be told apart from a typo'd interface name or a + /// missing capability. + #[error("interface unavailable: {interface}")] + InterfaceUnavailable { + /// The configured interface name. + interface: String, + }, + #[error("transport shutdown failed: {0}")] ShutdownFailed(String), @@ -874,6 +944,52 @@ impl TransportHandle { } } + /// Interface presence for interface-bound transports: the phase label, the + /// absence policy, how long the phase has been held, and the failed-bind + /// count since the last successful bind. + /// + /// `None` for transports that are not bound to a named interface — for + /// those, presence is not a concept and an operator should not be shown an + /// always-`present` column. + pub fn interface_presence(&self) -> Option { + match self { + #[cfg(any(target_os = "linux", target_os = "macos"))] + TransportHandle::Ethernet(t) => { + let state = t.presence_state(); + Some(InterfacePresence { + presence: t.presence().as_str(), + carrier: t.has_carrier(), + policy: t.absence_policy().as_str(), + since_secs: state.since().as_secs(), + binds: state.binds(), + failed_attempts: state.attempts(), + }) + } + _ => None, + } + } + + /// Whether this transport can actually put a frame on the wire *now*. + /// + /// [`Self::is_operational`] answers a different question: it means the + /// transport was started, which for an interface-bound transport no longer + /// implies a live socket — that is the whole point of presence. Callers + /// that are choosing a transport to use, or deriving a value from one, + /// want this; callers reasoning about lifecycle want `is_operational`. + /// + /// `true` for every transport that is not interface-bound, so this is + /// `is_operational` with the presence refinement applied where it exists. + pub fn is_bound(&self) -> bool { + if !self.is_operational() { + return false; + } + match self { + #[cfg(any(target_os = "linux", target_os = "macos"))] + TransportHandle::Ethernet(t) => t.presence() == ethernet::Presence::Present, + _ => true, + } + } + /// Get the interface name (Ethernet only, returns None for other transports). pub fn interface_name(&self) -> Option<&str> { match self { diff --git a/src/upper/tun.rs b/src/upper/tun.rs index 45262aeb..cf08537d 100644 --- a/src/upper/tun.rs +++ b/src/upper/tun.rs @@ -80,6 +80,35 @@ impl PathMtuEntry { /// address). pub type PathMtuLookup = Arc>>; +/// The node-global TCP MSS ceiling, shared live with the TUN reader and +/// writer threads. +/// +/// Shared rather than copied because the value it is derived from moves at +/// runtime. `Node::transport_mtu()` is the minimum across *bound* transports, +/// and since dynamic interface binding a transport can bind minutes after +/// start or unbind mid-operation — so a narrow interface appearing must +/// tighten the clamp, and its departure must release it. Every other consumer +/// of `transport_mtu()` already reads it live (`show_status`, the snapshot, +/// the session-layer fragmentation check); these two threads captured a `u16` +/// at spawn and were the only place left where the daemon could report one +/// effective MTU and clamp to another. +/// +/// A relaxed load per packet, beside the `PathMtuLookup` read that already +/// happens on the same packet — strictly the cheaper of the two. Ordering is +/// irrelevant: this is a clamp, and a packet that reads the previous value +/// during the store is clamped by the ceiling that was correct a microsecond +/// earlier. The per-flow ceiling has always had that property. +/// The IPv6 minimum link MTU (RFC 8200): every compliant path accepts a packet +/// this large, so an MSS derived from it fits anywhere. +/// +/// Two callers, and they must not disagree: the cold-flow fallback in +/// [`per_flow_max_mss`], and the seed for the node's [`MssCeiling`] before any +/// transport has bound — which is the same value `Node::transport_mtu()` falls +/// back to when nothing is bound. +pub const IPV6_MIN_MTU: u16 = 1280; + +pub type MssCeiling = Arc; + /// Compute the effective TCP MSS ceiling for a packet given its peer /// address bytes (a 16-byte IPv6 destination on outbound, source on /// inbound). Returns `min(global_max_mss, learned_path_max_mss)` when @@ -115,7 +144,6 @@ pub(crate) fn per_flow_max_mss( // RFC 8200 IPv6-minimum MTU (1280) → effective FIPS-encapsulated // payload (1203) → TCP segment after IPv6+TCP headers (1143). // Used as the conservative ceiling for empty-lookup destinations. - const IPV6_MIN_MTU: u16 = 1280; let conservative_max_mss = mss_ceiling(IPV6_MIN_MTU); let empty_lookup_ceiling = std::cmp::min(global_max_mss, conservative_max_mss); @@ -435,13 +463,14 @@ impl TunDevice { /// a channel sender for submitting packets to be written. /// /// `max_mss` is the global TCP MSS ceiling derived from the local - /// `transport_mtu()` floor. `path_mtu_lookup` is a read-only handle to - /// the per-destination path MTU map populated by discovery; the writer - /// reads it on each inbound SYN-ACK to compute a per-flow ceiling that - /// honors learned narrow paths through the mesh. + /// `transport_mtu()` floor, shared live so a transport that binds or + /// unbinds after start moves it (see [`MssCeiling`]). `path_mtu_lookup` + /// is a read-only handle to the per-destination path MTU map populated by + /// discovery; the writer reads both on each inbound SYN-ACK to compute a + /// per-flow ceiling that honors learned narrow paths through the mesh. pub fn create_writer( &self, - max_mss: u16, + max_mss: MssCeiling, path_mtu_lookup: PathMtuLookup, ) -> Result<(TunWriter, TunTx), TunError> { let fd = self.device.as_raw_fd(); @@ -517,7 +546,7 @@ pub struct TunWriter { file: File, rx: mpsc::Receiver>, name: String, - max_mss: u16, + max_mss: MssCeiling, path_mtu_lookup: PathMtuLookup, } @@ -531,16 +560,23 @@ impl TunWriter { pub fn run(mut self) { use super::tcp_mss::clamp_tcp_mss; - debug!(name = %self.name, max_mss = self.max_mss, "TUN writer starting"); + debug!( + name = %self.name, + max_mss = self.max_mss.load(std::sync::atomic::Ordering::Relaxed), + "TUN writer starting" + ); for mut packet in self.rx { + // Read per packet, not once: a transport binding or unbinding + // moves the node's egress floor at runtime. See `MssCeiling`. + let global_max_mss = self.max_mss.load(std::sync::atomic::Ordering::Relaxed); // Per-destination clamp: peer IPv6 source address (bytes 8..24) // identifies the flow's remote end. If discovery has learned a // smaller path MTU for that peer, tighten the ceiling. let effective_max_mss = if packet.len() >= 24 { - per_flow_max_mss(&self.path_mtu_lookup, &packet[8..24], self.max_mss) + per_flow_max_mss(&self.path_mtu_lookup, &packet[8..24], global_max_mss) } else { - self.max_mss + global_max_mss }; // Clamp TCP MSS on inbound SYN-ACK packets if clamp_tcp_mss(&mut packet, effective_max_mss) { @@ -622,17 +658,17 @@ pub fn run_tun_reader( our_addr: FipsAddress, tun_tx: TunTx, outbound_tx: TunOutboundTx, - transport_mtu: u16, + max_mss: MssCeiling, path_mtu_lookup: PathMtuLookup, ) { - let (name, mut buf, max_mss) = tun_reader_setup(device.name(), mtu, transport_mtu); + let (name, mut buf) = tun_reader_setup(device.name(), mtu, &max_mss); loop { match device.read_packet(&mut buf) { Ok(n) if n > 0 => { if !handle_tun_packet( &mut buf[..n], - max_mss, + max_mss.load(std::sync::atomic::Ordering::Relaxed), &name, our_addr, &tun_tx, @@ -685,13 +721,13 @@ pub fn run_tun_reader( our_addr: FipsAddress, tun_tx: TunTx, outbound_tx: TunOutboundTx, - transport_mtu: u16, + max_mss: MssCeiling, path_mtu_lookup: PathMtuLookup, shutdown_fd: std::os::unix::io::RawFd, ) { let _shutdown_fd = ShutdownFd(shutdown_fd); let tun_fd = device.device().as_raw_fd(); - let (name, mut buf, max_mss) = tun_reader_setup(device.name(), mtu, transport_mtu); + let (name, mut buf) = tun_reader_setup(device.name(), mtu, &max_mss); // Set TUN fd to non-blocking so we can use select + read without blocking // past the point where select returns readable. @@ -741,7 +777,7 @@ pub fn run_tun_reader( Ok(n) if n > 0 => { if !handle_tun_packet( &mut buf[..n], - max_mss, + max_mss.load(std::sync::atomic::Ordering::Relaxed), &name, our_addr, &tun_tx, @@ -774,30 +810,25 @@ pub fn run_tun_reader( // _shutdown_fd closes on drop } -/// Common setup for TUN reader: allocates buffer, computes max MSS. -fn tun_reader_setup(device_name: &str, mtu: u16, transport_mtu: u16) -> (String, Vec, u16) { - use super::icmp::effective_ipv6_mtu; - +/// Common setup for TUN reader: allocates the buffer and names the device. +/// +/// The MSS ceiling is deliberately *not* returned. It is read from the shared +/// [`MssCeiling`] on every packet, because a transport binding or unbinding +/// moves it after this function has run; returning it here is what let the +/// reader clamp to a floor derived from the transports that happened to be +/// bound at spawn. +fn tun_reader_setup(device_name: &str, mtu: u16, max_mss: &MssCeiling) -> (String, Vec) { let name = device_name.to_string(); let buf = vec![0u8; mtu as usize + 100]; - const IPV6_HEADER: u16 = 40; - const TCP_HEADER: u16 = 20; - let effective_mtu = effective_ipv6_mtu(transport_mtu); - let max_mss = effective_mtu - .saturating_sub(IPV6_HEADER) - .saturating_sub(TCP_HEADER); - debug!( name = %name, tun_mtu = mtu, - transport_mtu = transport_mtu, - effective_mtu = effective_mtu, - max_mss = max_mss, + max_mss = max_mss.load(std::sync::atomic::Ordering::Relaxed), "TUN reader starting" ); - (name, buf, max_mss) + (name, buf) } /// Process a single TUN packet. Returns `false` if the reader should exit. @@ -1076,12 +1107,13 @@ mod windows_tun { /// packets independently. Returns the writer and a channel sender for /// submitting packets to be written. /// - /// `max_mss` is the global TCP MSS ceiling. `path_mtu_lookup` is a - /// read-only handle to per-destination path MTU learned via - /// discovery. + /// `max_mss` is the global TCP MSS ceiling, shared live so a transport + /// that binds or unbinds after start moves it (see [`MssCeiling`]). + /// `path_mtu_lookup` is a read-only handle to per-destination path MTU + /// learned via discovery. pub fn create_writer( &self, - max_mss: u16, + max_mss: MssCeiling, path_mtu_lookup: PathMtuLookup, ) -> Result<(TunWriter, TunTx), TunError> { let (tx, rx) = mpsc::channel(); @@ -1119,7 +1151,7 @@ mod windows_tun { session: Arc, rx: mpsc::Receiver>, name: String, - max_mss: u16, + max_mss: MssCeiling, path_mtu_lookup: PathMtuLookup, } @@ -1132,14 +1164,21 @@ mod windows_tun { use super::per_flow_max_mss; use crate::upper::tcp_mss::clamp_tcp_mss; - debug!(name = %self.name, max_mss = self.max_mss, "TUN writer starting"); + debug!( + name = %self.name, + max_mss = self.max_mss.load(std::sync::atomic::Ordering::Relaxed), + "TUN writer starting" + ); for mut packet in self.rx { + // Read per packet, not once: a transport binding or unbinding + // moves the node's egress floor. See `MssCeiling`. + let global_max_mss = self.max_mss.load(std::sync::atomic::Ordering::Relaxed); // Per-destination clamp (peer source IPv6 = bytes 8..24) let effective_max_mss = if packet.len() >= 24 { - per_flow_max_mss(&self.path_mtu_lookup, &packet[8..24], self.max_mss) + per_flow_max_mss(&self.path_mtu_lookup, &packet[8..24], global_max_mss) } else { - self.max_mss + global_max_mss }; // Clamp TCP MSS on inbound SYN-ACK packets if clamp_tcp_mss(&mut packet, effective_max_mss) { @@ -1188,17 +1227,17 @@ mod windows_tun { our_addr: FipsAddress, tun_tx: TunTx, outbound_tx: TunOutboundTx, - transport_mtu: u16, + max_mss: MssCeiling, path_mtu_lookup: PathMtuLookup, ) { - let (name, mut buf, max_mss) = super::tun_reader_setup(device.name(), mtu, transport_mtu); + let (name, mut buf) = super::tun_reader_setup(device.name(), mtu, &max_mss); loop { match device.read_packet(&mut buf) { Ok(n) if n > 0 => { if !super::handle_tun_packet( &mut buf[..n], - max_mss, + max_mss.load(std::sync::atomic::Ordering::Relaxed), &name, our_addr, &tun_tx,