diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3f389450..af1b3f08 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -221,7 +221,11 @@ jobs: ${{ runner.os }}-cargo- - name: Build - run: cargo build --release + # --bins --examples rather than the bare default: the native datagram + # API's echo server is a cargo example, and the integration image needs + # it. Naming --bins keeps the daemon and its tools in the build, which + # --examples alone would drop. Mirrors testing/ci-local.sh. + run: cargo build --release --bins --examples - name: SHA-256 hashes (Linux) if: runner.os == 'Linux' @@ -236,6 +240,15 @@ jobs: shell: pwsh run: Get-FileHash target\release\fips.exe, target\release\fipsctl.exe, target\release\fipstop.exe -Algorithm SHA256 + # Cargo puts an example under target/release/examples. Staging them + # beside the bins keeps the artifact's common root at target/release, so + # every existing consumer still finds its file at _bin/. + - name: Stage the native API examples beside the release binaries + if: matrix.os == 'ubuntu-latest' + run: | + cp target/release/examples/native-echo target/release/native-echo + cp target/release/examples/native-surface target/release/native-surface + # Upload the Linux binary so integration jobs can use it without rebuilding - name: Upload Linux binary if: matrix.os == 'ubuntu-latest' @@ -247,6 +260,8 @@ jobs: target/release/fipsctl target/release/fipstop target/release/fips-gateway + target/release/native-echo + target/release/native-surface retention-days: 1 # ───────────────────────────────────────────────────────────────────────────── @@ -521,6 +536,14 @@ jobs: # loopback-delivered query was once misattributed to the mesh # interface and dropped. Single matrix entry runs all 13 # scenarios sequentially; ~7-12 min warm, ~12-15 min cold. + # Native datagram API: a client process opening a pubkey-to-pubkey + # flow over the daemon's Unix socket. One single-node leg covering + # the socket, its access mode and the command surface, plus a + # two-node pair that sends a real datagram end to end. Fast: no + # per-distro images and no TUN. ~2-3 min. + - suite: native-api + type: native-api + - suite: dns-resolver type: dns-resolver @@ -544,6 +567,13 @@ jobs: cp _bin/fipsctl testing/docker/fipsctl [ -f _bin/fipstop ] && cp _bin/fipstop testing/docker/fipstop || true [ -f _bin/fips-gateway ] && cp _bin/fips-gateway testing/docker/fips-gateway || true + # Not optional: the Dockerfile COPYs both native API examples + # unconditionally, and a missing source there fails the shared image + # build for every leg, not just native-api. Fail here instead, where + # the cause is legible. + chmod +x _bin/native-echo _bin/native-surface + cp _bin/native-echo testing/docker/native-echo + cp _bin/native-surface testing/docker/native-surface docker build -t fips-test:latest testing/docker docker build -t fips-test-app:latest -f testing/docker/Dockerfile.app testing/docker @@ -734,6 +764,32 @@ jobs: docker rm -f "$c" >/dev/null 2>&1 || true done + # ── Native datagram API ───────────────────────────────────────────── + # Reads FIPS_TEST_IMAGE rather than defaulting to a name, so it runs + # against the image this workflow built. The two-node check creates and + # removes its own docker network. + - name: Run native-api test + if: matrix.type == 'native-api' + timeout-minutes: 15 + env: + FIPS_TEST_IMAGE: fips-test:latest + run: bash testing/native-api/test.sh + + - name: Collect logs on failure (native-api) + if: matrix.type == 'native-api' && failure() + run: | + docker ps -a --filter "name=fips-native" --format '{{.Names}}' | while read -r c; do + echo "--- ${c} ---" + docker logs "$c" 2>&1 | tail -100 || true + done + + - name: Stop containers (native-api) + if: matrix.type == 'native-api' && always() + run: | + docker ps -a --filter "name=fips-native" --format '{{.Names}}' | while read -r c; do + docker rm -f "$c" >/dev/null 2>&1 || true + done + # ── DNS resolver multi-backend integration ────────────────────────── # The dns-resolver harness builds its own fips binary from source in a # Debian 12 builder image (shared cache layout with deb-install). Runs diff --git a/CHANGELOG.md b/CHANGELOG.md index 82e6b512..cb148ae3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -221,6 +221,33 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 terminal leaves behind. `--json` emits exactly one document at the end, so a script parsing the report does not have to skip past progress output. +- An **experimental** native datagram API addressed by public key, off by + default and not a stable interface. A client process opens a flow to a + peer's public key on a chosen port and sends and receives datagrams on a + file descriptor the daemon hands it: no IPv6 emulation, no TUN device and no + DNS, a datagram travelling from key to key. **The wire needs no change and + gets none.** Every FSP data packet has carried a port pair inside its AEAD + envelope since v0.2.0 and port 256 is simply the IPv6 shim, so what was + missing was a way for a program to ask for a port of its own and be handed + the traffic. The x-only public key is the address and an npub is that key + written in bech32, so converting between them is a local encoding rather + than a lookup or a name service; the 16-byte node address that travels on + the wire is a truncated hash of the key, does not invert, and appears + nowhere a client can see. A listener is a descriptor: the daemon writes one + message per arrival to it, carrying the new flow's descriptor and the peer's + address, so poll, select and epoll work on a listener and accepting is a + `recvmsg`. There is no accept command and no reject command, and refusing a + flow is closing the descriptor you were handed. The Rust surface mirrors + `std::net`, with `FipsStream::connect`, `FipsListener::bind`, `incoming`, + `accept`, `io::Result` and an errno mapping rather than a bespoke error + type, plus `set_nonblocking`, `AsFd` and four deadline methods under the + names and signatures `std::net` uses for the same jobs. One rule has no + counterpart in Berkeley sockets and a client author must know it: the v1 + wire carries no half-close, so nothing peer-driven ever closes a flow, and a + server written to read until the flow ends waits for a signal that cannot + arrive. The listener is built on `SOCK_SEQPACKET`, so it is available on + Linux and FreeBSD. + ### Changed - Node health is determined at start completion instead of unconditionally diff --git a/docs/design/README.md b/docs/design/README.md index 015372c3..92e931d2 100644 --- a/docs/design/README.md +++ b/docs/design/README.md @@ -33,6 +33,7 @@ documents cover specific subsystems in detail. | [fips-mesh-layer.md](fips-mesh-layer.md) | FIPS Mesh Protocol (FMP): peer authentication, link encryption, forwarding | | [fips-session-layer.md](fips-session-layer.md) | FIPS Session Protocol (FSP): end-to-end encryption, sessions | | [fips-ipv6-adapter.md](fips-ipv6-adapter.md) | IPv6 adaptation: TUN interface, DNS, MTU enforcement | +| [fips-native-api.md](fips-native-api.md) | Native datagram API: pubkey-addressed flows over FSP, and what it is instead of the TUN path | ### Cross-Cutting diff --git a/docs/design/diagrams/fips-native-api-stack-comparison.svg b/docs/design/diagrams/fips-native-api-stack-comparison.svg new file mode 100644 index 00000000..f63acec9 --- /dev/null +++ b/docs/design/diagrams/fips-native-api-stack-comparison.svg @@ -0,0 +1,119 @@ + + + + + + + + + + + + + + + + Application + unmodified on the left, written against the key on the right + + + socket API + https://<npub>.fips:443 + a name, resolved to an address + + native datagram API + fips://<npub>:443 + the key itself, no lookup + + + + + + + + + HTTP + a format every client already speaks + + + your own format + you design it; 1 send = 1 datagram + + + + TLS + trust anchored in a CA certificate + + + FSP + end-to-end AEAD; trust in the key itself + + + + TCP + reliability, ordering, flow control + + + ROD + Reliable Object Delivery — not in v1 + a v2 capability, may come forward + + + + IPv6 + routing by address prefix + + + FMP + spanning tree, bloom, per-hop auth + + + + + + + eth0 + the ordinary internet + + + fips0 + IPv6 adapter: TUN + DNS + + + Transport + UDP, Ethernet, WiFi, BLE, Tor, serial + + + + tunnelled into FSP + + Both endpoints name the same node and port, but not the same kind of port: on the left it is a TCP port inside the + tunnel, on the right an FSP port, whose tier rules are expected to change. The adapter is not a bottom layer — an + IPv6 packet reaching fips0 becomes an FSP payload, so the whole left stack runs inside the right one, TCP included. + + Two ways into one mesh: tunnel the IP stack, or address the key directly + diff --git a/docs/design/fips-architecture.md b/docs/design/fips-architecture.md index d7828131..1942dc16 100644 --- a/docs/design/fips-architecture.md +++ b/docs/design/fips-architecture.md @@ -83,6 +83,13 @@ addresses with header compression so unmodified IP applications can use the network transparently, while the native datagram API addresses destinations directly by npub. +The native datagram API is **experimental** and off by default. It is +not a stable API surface and carries no compatibility promise. See +[../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md) +for enabling it and writing against it, and +[../reference/security.md](../reference/security.md#native-datagram-api) +for what enabling it grants to the `fips` group. + ![Node Architecture](diagrams/fips-node-architecture.svg) The mesh routes application traffic across heterogeneous transports diff --git a/docs/design/fips-concepts.md b/docs/design/fips-concepts.md index cb6ab027..39a74f38 100644 --- a/docs/design/fips-concepts.md +++ b/docs/design/fips-concepts.md @@ -24,6 +24,17 @@ a native FIPS datagram service, or through an IPv6 adaptation layer that presents each node as an IPv6 endpoint for compatibility with existing IP-based applications. +![Two ways into the mesh](diagrams/fips-native-api-stack-comparison.svg) + +Both columns are the same mesh. On the left an unmodified program keeps +the stack it already has, and reaches the mesh through `fips0`, a +virtual network interface that carries its IPv6 packets. On the right a +FIPS-aware program names the far node by its key and skips the IP layers +altogether. The two protocols in the middle are the mesh's own: **FSP** +encrypts end to end between the two nodes, and **FMP** authenticates +each hop and decides where a packet goes next. +[fips-architecture.md](fips-architecture.md) takes them in order. + ## Why FIPS? **Self-sovereign identity**: FIPS nodes generate their own addresses, diff --git a/docs/design/fips-ipv6-adapter.md b/docs/design/fips-ipv6-adapter.md index f86cbae9..50b2a4d4 100644 --- a/docs/design/fips-ipv6-adapter.md +++ b/docs/design/fips-ipv6-adapter.md @@ -18,6 +18,15 @@ connects to the kernel's IPv6 stack. Applications that are FIPS-aware can bypass the adapter entirely and use the native FIPS datagram API, addressing destinations directly by npub. +![Where the adapter sits](diagrams/fips-native-api-stack-comparison.svg) + +The adapter is the `fips0` box, and the arrow leaving it is the point: an IPv6 +packet arriving there does not go out to a wire. It becomes an FSP payload, so +the whole IP stack above runs inside the mesh's own stack — TCP included, which +is what lets an unmodified `ssh` or `curl` work across the mesh. The right-hand +column is the same mesh reached without the adapter, and +[fips-native-api.md](fips-native-api.md) covers that path. + ## DNS Integration ### The Problem diff --git a/docs/design/fips-native-api.md b/docs/design/fips-native-api.md new file mode 100644 index 00000000..4fabe951 --- /dev/null +++ b/docs/design/fips-native-api.md @@ -0,0 +1,147 @@ +# Native Datagram API + +The native datagram API lets a local program move bytes between two public keys +over FSP, with no IPv6 emulation and no TUN device in the path. A program calls +`connect` for a flow to a public key and a port, or `bind` for a port to receive +flows on, and from then on uses ordinary socket calls. + +This document explains what the interface is for and where its edges are. For +the surface itself — every type, method, errno and command — see +[../reference/native-api.md](../reference/native-api.md). For the steps to enable +it and write a program, see +[../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md). + +## Where it sits + +![Stack comparison](diagrams/fips-native-api-stack-comparison.svg) + +The two endpoints at the top are the same node reached two ways. **The +`fips://` form is illustrative**: no code in this repository parses it, nothing +registers the scheme, and the API takes a key and a port as separate arguments +rather than a URL. It is drawn because it is the shape an address takes on that +side, against a `.fips` name the adapter's DNS really does resolve. + +Read row by row, the native path replaces three layers and declines to replace a +fourth. FSP takes TLS's place and anchors trust in the key rather than in a +certificate authority. FMP takes IPv6's place and routes by spanning tree and +bloom filter rather than by address prefix, with the address derived from the +key. The transport layer takes the medium's place and can be several media at +once. + +**There is nothing where TCP was**, and on the native path that is the single +most consequential row today. No acknowledgement, no retransmission, no ordering +and no flow control: a program that needs any of them builds it into its own +payload. + +That row is marked **ROD — Reliable Object Delivery**, which is where the +capability is expected to land. ROD is a v2 capability and is not in v1; it may +be pulled forward. Until it is, treat the row as empty and design around it, +because a program written against a reliability layer that is not there yet +fails in the ways this document's "not a reliability layer" section describes. + +**The two paths are not alternatives at the bottom.** They converge. An +unmodified IPv6 program does not stop at a wire: its packets reach `fips0`, and +the adapter hands each one to FSP as a payload. That is the arrow running up the +middle of the diagram, and it is why the left stack is drawn ending at an +interface rather than at Ethernet. + +So the whole left column runs *inside* the right one. **TCP included** — which +is the practical answer to the empty row above it. A program that needs a +reliable ordered stream over the mesh already has one: run it over `fips0` and +let TCP do what TCP does, inside FSP's encryption. What the native API offers +instead is the same mesh with four layers of machinery removed, for a program +willing to do without them. + +The bottom of the diagram is not always the bottom of the stack either. When +FIPS overlays an existing network its transport is UDP, which still rides IP and +Ethernet beneath; when the mesh *is* the network, a transport sits on a link +directly. + +## What it is instead of + +The fastest way to place the interface is by contrast with the TUN device, which +is the other way a program gets FIPS traffic. + +| | TUN interface | Native datagram API | +| --- | ------------- | ------------------- | +| Addressing | IPv6 address | public key, written as an npub | +| Name resolution | DNS over the mesh | none: the program supplies the key | +| Kernel object | TUN device, routes | a `FipsStream` per peer | +| Encapsulation | IPv6 emulated over FSP | FSP port pair, no IP layer | +| Program sees | an IP network | a `FipsStream` | +| Privilege | `CAP_NET_ADMIN` to create the device | membership of group `fips` | +| Demultiplexing | by address and port | by flow, one stream each | + +**The IPv6 emulation is not removed by this interface.** It continues to run +beside it on FSP port 256, which is why that port and the tier around it are +refused to a program. What the native API removes is a program's *dependence* on +it: a program that wants to move bytes between two known public keys no longer +has to acquire an IPv6 address, resolve a name, and hand its payload to a +protocol stack that will encapsulate it again. + +Both paths reach the same place. A native datagram and an emulated IPv6 packet +are both FSP payloads with a port pair, carried in the same encrypted session to +the same peer. The difference is entirely on the local side of the daemon. + +## Status + +**The wire is connected**: a datagram sent on a flow leaves the node over FSP, +and one arriving on a held port reaches its flow. + +**The interface around it is experimental.** It is not versioned, it has no +compatibility promise, and three of its five commands exist only to let the +daemon's own checks drive the receive path without a peer. It is Linux and +FreeBSD only, and it is off by default. + +## What this is not + +**Not a stable interface.** It is an experiment on the v1 wire. Names, fields, +reply shapes and the command set may change without a deprecation cycle. + +**Not the v2 process API.** The v2 external process API is a separate and later +design, which retires ports entirely in favour of a listener, connection and +stream model. Nothing here governs it and nothing there governs this. The one +thing this interface takes from that work is the FSP port tiers, because port +256 already carries the IPv6 shim on the deployed wire and a new service must not +collide with it. + +**Not a reliability layer.** There is no acknowledgement, no retransmission, no +ordering guarantee and no flow control between the two ends. A datagram is +carried or it is dropped. Some drops are counted inside the daemon and none are +reported to a program for real traffic. A program that needs delivery guarantees +builds them itself, on top, in the payload — or runs over `fips0` and lets TCP +provide them. + +Reliable Object Delivery (ROD) is the v2 capability intended to fill this gap, +and it may be pulled forward into v1. **Nothing here anticipates it**: no field, +reply shape or command on this surface is reserved for it, and a program written +today should assume it does not exist. + +**Not an authorization boundary.** The socket's group ownership is the whole of +the access control. Any process that can open it can send as this node's identity +and can receive mesh traffic on any port it can claim, and there is no per-program +separation beyond the port registry. Because the descriptor carries the flow, a +process handed one over `SCM_RIGHTS` can send as this node on that flow without +ever opening the socket. See +[../reference/security.md](../reference/security.md#native-datagram-api). + +**Not multi-tenant.** `max_flows` is node-wide with no per-program share, so one +program can exhaust it, and every other program then sees `EMFILE` on `connect` +and silent drops on its listeners. + +**Not a connection in the TCP sense.** A successful `connect` is a local +registration and contacts no peer. There is no handshake, no keepalive and no +notification that a peer went away. A flow ends when its descriptor closes, and +in no other way. In particular **a peer cannot end your flow: it has no close to +send.** That single fact shapes every program written against this interface, +and the consequences are drawn out in +[../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md#four-things-that-will-bite-you). + +## See also + +- [fips-session-layer.md](fips-session-layer.md) — FSP, which carries the + datagrams and owns the port pair +- [fips-ipv6-adapter.md](fips-ipv6-adapter.md) — the other consumer of FSP, and + what this interface is an alternative to +- [../reference/native-api.md](../reference/native-api.md) — the surface, the + line protocol and the command reference diff --git a/docs/how-to/README.md b/docs/how-to/README.md index 1774a9ae..74965ad5 100644 --- a/docs/how-to/README.md +++ b/docs/how-to/README.md @@ -28,3 +28,6 @@ X" to "X is done". | [set-up-80211s-mesh-backhaul.md](set-up-80211s-mesh-backhaul.md) | Link OpenWrt FIPS routers over an open 802.11s radio backhaul (FIPS provides encryption, authentication, and routing) | | [set-up-open-access-ssid.md](set-up-open-access-ssid.md) | Broadcast the open `!FIPS` access SSID so phones and laptops roam onto the mesh (one ESS: save once, roam every FIPS router) | | [diagnose-mtu-issues.md](diagnose-mtu-issues.md) | Triage MTU-shaped failures and rule out their imposters (bufferbloat, transport saturation) | +| [use-the-native-datagram-api.md](use-the-native-datagram-api.md) | Enable the experimental native datagram API and write a program that sends and receives datagrams by pubkey and port (no IPv6 emulation, no TUN). Read the `fips` group warning first | +| [write-a-native-api-client.md](write-a-native-api-client.md) | Speak the native datagram API's line protocol directly from C, Python or Go, where there is no client library | +| [serve-many-peers-on-one-thread.md](serve-many-peers-on-one-thread.md) | Handle every native API flow from one `poll` loop instead of a thread per peer | diff --git a/docs/how-to/serve-many-peers-on-one-thread.md b/docs/how-to/serve-many-peers-on-one-thread.md new file mode 100644 index 00000000..3130a272 --- /dev/null +++ b/docs/how-to/serve-many-peers-on-one-thread.md @@ -0,0 +1,196 @@ +# Serve Many Peers on One Thread + +**Goal:** handle every native datagram API flow from a single `poll` loop, +instead of dedicating a thread to each peer. + +The straightforward listening program spawns a thread per flow. That is fine for +a handful of peers and wrong at the node's ceiling of 256, where it costs 256 +threads mostly parked in `recv`. + +Every object on this surface is a descriptor, so there is nothing to integrate: +both types implement `AsFd` and `AsRawFd` and go straight into a `poll`, +`select` or `epoll` set. **A listener is readable exactly when `accept` would not +block**, which is the property the whole shape rests on, and the crate asserts it +as a test rather than claiming it. + +Read [use-the-native-datagram-api.md](use-the-native-datagram-api.md) first if +you have not opened a flow before. + +## The whole program + +One dependency beyond the crate, `libc`, for `poll` itself: + +```toml +[dependencies] +fips = { git = "https://github.com/jmcorgan/fips" } +libc = "0.2" +``` + +```rust +//! Serve many flows from one poll loop, with no thread per peer. +//! +//! ```text +//! eventloop /run/fips/api.sock 4242 +//! ``` + +use fips::native::client::{FipsListener, FipsStream}; +use std::env; +use std::error::Error; +use std::os::fd::{AsRawFd, RawFd}; +use std::path::Path; +use std::process::ExitCode; +use std::time::{Duration, Instant}; + +/// How long a flow may go without a datagram before this program closes it. +/// +/// Nothing else will end one. The wire carries no far-end close, so a peer that +/// has stopped sending is indistinguishable from one that is thinking, and a +/// reactor with no deadline holds every flow it ever accepted until it exits. +const IDLE: Duration = Duration::from_secs(30); + +/// Hold the port named on the command line and serve every flow from one loop. +fn run() -> Result<(), Box> { + let mut args = env::args().skip(1); + let (Some(socket), Some(port)) = (args.next(), args.next()) else { + return Err("usage: eventloop ".into()); + }; + let port: u16 = port.parse()?; + + let listener = FipsListener::bind_at(Path::new(&socket), port)?; + println!("holding {}", listener.local_addr()); + + // Each flow with the time its last datagram arrived, which is what the + // deadline is measured against. + let mut flows: Vec<(FipsStream, Instant)> = Vec::new(); + loop { + let mut fds = vec![watch(listener.as_raw_fd())]; + fds.extend(flows.iter().map(|(flow, _)| watch(flow.as_raw_fd()))); + + let count = fds.len() as libc::nfds_t; + // The timeout is what makes the deadline reachable: with no events at + // all the loop must still wake to notice a flow that has gone quiet. + let timeout = IDLE.as_millis() as libc::c_int; + // SAFETY: `fds` is a live slice of `pollfd` for the whole call, and + // `count` is its length. + if unsafe { libc::poll(fds.as_mut_ptr(), count, timeout) } < 0 { + return Err(std::io::Error::last_os_error().into()); + } + + // The established flows first, and backwards: removing a closed one + // must not renumber one not yet examined, and accepting below would + // otherwise push a flow this pass has no `revents` for. The listener + // occupies slot 0, hence the offset. + let now = Instant::now(); + for index in (0..flows.len()).rev() { + if ready(&fds[index + 1]) { + if echo(&flows[index].0) { + flows[index].1 = now; + } else { + flows.remove(index); // dropping it releases the flow + } + } else if now.duration_since(flows[index].1) >= IDLE { + println!("closing an idle flow from {}", flows[index].0.peer_addr()); + flows.remove(index); + } + } + + // Once per readiness rather than in a loop: the descriptor is blocking, + // so a second `accept` with nothing queued would stall the whole loop. + if ready(&fds[0]) { + let (flow, peer) = listener.accept()?; + println!("flow from {peer}"); + flows.push((flow, now)); + } + } +} + +/// A `pollfd` asking for readability on `fd`. +/// +/// `POLLHUP` needs no asking for: it is reported in `revents` whether or not +/// it was requested, which is what lets one mask serve both cases. +fn watch(fd: RawFd) -> libc::pollfd { + libc::pollfd { + fd, + events: libc::POLLIN, + revents: 0, + } +} + +/// Whether this descriptor has something to read or has hung up. +fn ready(poll: &libc::pollfd) -> bool { + poll.revents & (libc::POLLIN | libc::POLLHUP) != 0 +} + +/// Return one datagram, reporting whether the flow is still usable. +fn echo(flow: &FipsStream) -> bool { + let mut buf = vec![0u8; flow.max_payload()]; + match flow.recv(&mut buf) { + // `Ok(0)` is an empty datagram and not a close, so it is echoed like + // any other. `EPIPE` is the daemon gone: no peer can close a flow. + Ok(len) => flow.send(&buf[..len]).is_ok(), + Err(_) => false, + } +} + +/// Report a failure on stderr and exit non-zero. +fn main() -> ExitCode { + match run() { + Ok(()) => ExitCode::SUCCESS, + Err(error) => { + eprintln!("eventloop: {error}"); + ExitCode::FAILURE + } + } +} +``` + +## The four rules + +**Register the listener for readability only.** There is nothing else to ask it +for, and `POLLHUP` arrives in `revents` whether or not it was requested. + +**Accept once per readiness, not in a loop.** The descriptor is blocking, so a +second `accept` with nothing queued stalls the whole loop. Looping until +`WouldBlock` is correct only after `listener.set_nonblocking(true)`, which is +also what an edge-triggered `epoll` requires. + +**Give every flow a deadline, and the poll a timeout that makes the deadline +reachable.** This is the most important of the four. Nothing will tell a reactor +that a peer is finished, so a flow that goes quiet stays in the poll set for ever +unless the program removes it — and with no events at all the loop must still +wake in order to notice. **A reactor with a deadline but no poll timeout has a +deadline it can never reach.** + +**A flow from `accept` arrives blocking, whatever the listener was set to.** +They are separate sockets and the daemon hands over a fresh one. The program +above deliberately leaves them blocking and makes exactly one `recv` per +readiness, which is safe on a blocking descriptor and is why it needs no flags at +all. If you want them otherwise, call `set_nonblocking` on the flow. + +## Where this reaches past the client module + +This program needs `libc` and an `unsafe` block, and it is the only one of the +API's example programs that needs anything. + +That is a narrower gap than it once was. Readiness and a bounded wait were both +missing from the surface; `set_nonblocking` closed the first and +`set_read_timeout` the second, and a program wanting an option on a flow now has +a method for it. What is left is a different kind of thing: **an option on a flow +is something a surface can supply, and a reactor's own polling mechanism is +not.** No surface that stops at the descriptor can supply `poll`. + +`AsRawFd` rather than `AsFd` here is deliberate: `libc::poll` takes a raw +descriptor. Prefer `AsFd` anywhere you **hold** a registration, because its +borrow cannot outlive the stream; this loop rebuilds its `pollfd` set from live +references on every pass, so it holds nothing across an iteration. + +## See also + +- [use-the-native-datagram-api.md](use-the-native-datagram-api.md) — opening + flows and receiving them, and the traps that apply to any program here +- [write-a-native-api-client.md](write-a-native-api-client.md) — the same loop + in a language with no client library +- [../reference/native-api.md](../reference/native-api.md#fipslistener) — what + `accept`, `incoming` and `set_nonblocking` guarantee +- [../reference/native-api.md](../reference/native-api.md#what-a-daemon-restart-costs) + — what a reactor sees when the daemon goes away diff --git a/docs/how-to/use-the-native-datagram-api.md b/docs/how-to/use-the-native-datagram-api.md new file mode 100644 index 00000000..cd65e605 --- /dev/null +++ b/docs/how-to/use-the-native-datagram-api.md @@ -0,0 +1,298 @@ +# Use the Native Datagram API + +**Experimental.** The native datagram API lets a local program send +and receive datagrams addressed by pubkey and FSP port, with no IPv6 +emulation and no TUN device in the path. A client connects to a Unix +socket, opens a flow to a remote pubkey, and is handed a file +descriptor it reads and writes datagrams on. + +It is not a stable API surface, not a reliability layer, and not the +v2 external process API. No compatibility promise is made: the socket +protocol, the Rust client, and the configuration keys may change or be +withdrawn in any release. It is built on Linux and FreeBSD only. + +Before enabling it, read the security posture below. It is short and +it is the whole of the access control. + +## Before you start: what enabling this grants + +The API socket is mode `0770` owned by group `fips`, the same group as +the control socket, and that is the entire authorization model. + +**Any user in the `fips` group can impersonate the node on the mesh.** +A process that can open the socket can send datagrams under this +node's identity to any peer it names, and can hold a port and receive +mesh traffic addressed to this node. Peers authenticate those +datagrams as coming from this node, because they did. + +On a node with the API enabled, treat `fips` group membership exactly +as you would treat `/etc/fips/fips.key`. If the group has been handed +out so that people can run `fipsctl`, enabling the API upgrades every +one of those accounts from "can read node state" to "can speak as the +node". See +[../reference/security.md](../reference/security.md#native-datagram-api). + +The API is disabled by default, and nothing in the packaging turns it +on. + +## Step 1: Enable it on the daemon + +Add to `/etc/fips/fips.yaml` (or a drop-in under `/etc/fips/fips.d/`): + +```yaml +node: + native_api: + enabled: true +``` + +Every other key has a working default; see +[../reference/configuration.md](../reference/configuration.md#native-datagram-api-nodenative_api) +for the full list, including the flow ceiling and the per-flow queue +depth. +Leave `debug_commands` alone — it is off by default and belongs to the +test harness. + +Restart the daemon and confirm the socket came up: + +```sh +sudo systemctl restart fips +sudo journalctl -u fips | grep 'Native API socket listening' +``` + +The log line carries the path the daemon resolved, normally +`/run/fips/api.sock`. + +## Step 2: Link the crate + +The client is a module of the `fips` crate itself, so a program links +the crate and uses `fips::native::client`: + +```toml +[dependencies] +fips = { git = "https://github.com/jmcorgan/fips" } +``` + +A checkout on the same machine can use a path dependency instead: + +```toml +[dependencies] +fips = { path = "../fips" } +``` + +The client is blocking and std-only. It brings no async runtime, so a +plain `fn main` is enough, and a program using it does not need +`serde_json`, `libc`, or any knowledge of the socket's line protocol. + +## Step 3: Open a flow and exchange datagrams + +The surface is shaped like `std::net`: `FipsStream::connect` opens a +flow the way `TcpStream::connect` opens a connection, and the stream it +returns carries the datagrams. + +An address is a public key and a port. The npub is that key written +down, and converting between the two is bech32 and nothing else: no +lookup, no resolution, no name service. So `("npub1...", 4600)` and +`"npub1...:4600"` name the same address, and one parameter takes either, +exactly as `ToSocketAddrs` does: + +```rust +use fips::native::client::FipsStream; +use std::time::Duration; + +fn main() -> Result<(), Box> { + let flow = FipsStream::connect(("npub1...", 4600))?; + + flow.send(b"hello")?; + + // Nothing tells you a peer is never going to answer, so bound the + // wait yourself. Without this the recv below blocks for ever + // against a peer that is offline or listening elsewhere. + flow.set_read_timeout(Some(Duration::from_secs(5)))?; + + // Size the buffer at the flow's own limit so no datagram it can + // carry is truncated on the way in. + let mut buf = vec![0u8; flow.max_payload()]; + let len = flow.recv(&mut buf)?; + println!("{len} bytes back from {}", flow.peer_addr()); + Ok(()) +} +``` + +**`connect` contacts no peer.** It is a local registration at the +daemon, and nothing about it proves the peer exists, is reachable, or is +listening. The Berkeley shape invites the opposite reading, which is why +it is said here as well as in the API documentation. + +`FipsStream::max_payload` is the daemon's answer for that flow: the +transport MTU less the FIPS encapsulation and the four-byte port header. +`FipsStream::send` refuses anything larger locally rather than letting it +be dropped further along. + +`connect` uses the packaged socket path, `/run/fips/api.sock`, and an +ephemeral local port from 49152 upward. Use `connect_from(port, addr)` +when the local port matters, such as when the far end has been told it in +advance, and `connect_at(path, port, addr)` when the daemon's socket is +somewhere else. + +**Bounding a wait.** `set_read_timeout` and `set_write_timeout` bound one +`recv` or one `send`, as their `TcpStream` counterparts do, and +`read_timeout` and `write_timeout` read them back. An expiring deadline +reports `WouldBlock`. `None` clears a deadline; a zero duration is +refused with `EINVAL`, because the kernel reads a zero timeout as "wait +for ever" and a caller passing zero means the opposite. + +## Step 4: Receive flows from peers + +`FipsListener::bind` holds a port, and `accept` blocks until a flow +arrives on it. `incoming()` is the same thing as an iterator, as it is on +`TcpListener`: + +```rust +use fips::native::client::FipsListener; + +fn main() -> Result<(), Box> { + let listener = FipsListener::bind(4600)?; + println!("holding {}", listener.local_addr()); + + for arrival in listener.incoming() { + let flow = arrival?; + std::thread::spawn(move || { + let mut buf = vec![0u8; flow.max_payload()]; + if let Ok(len) = flow.recv(&mut buf) { + let _ = flow.send(&buf[..len]); + } + }); + } + Ok(()) +} +``` + +`bind(0)` asks the daemon to pick the port, and `local_addr()` reports +the one it actually held: `getsockname` after `bind(2)` with port 0. Use +`bind_at(path, port)` when the daemon's socket is somewhere else. A +program wanting two independent accept loops binds two listeners. + +The thread above serves **one datagram and then drops the flow**. That is +deliberate, and the fourth note below says why. + +An accepted flow's `peer_addr()` names the far end by npub and port, +because the key is the address and it is the one that peer's session +authenticated. The 16-byte node address that travels on the wire is a +truncated hash of that key; it does not invert, and it appears nowhere on +this surface. + +Both types implement `AsFd` and `AsRawFd`, which is the point of the +listener being a descriptor: it is readable exactly when `accept` would +not block, so a program with its own `poll`, `select` or `epoll` loop adds +it to that loop rather than dedicating a thread to blocking in `accept`. +**Prefer `AsFd`**: its borrow cannot outlive the stream, so a reactor +cannot hold a registration for a descriptor that has since been closed and +its number reissued to the next `connect`. + +`set_nonblocking(true)` on either type turns a blocking call into +`WouldBlock`, which is the other way to drive a reactor. **A flow from +`accept` is blocking however its listener was set**: they are separate +sockets, so set it on the flow if the flow is what you poll. + +## Four things that will bite you + +**Setup leaves you nothing to keep alive.** `connect` and `bind` each +open a connection to the daemon socket, send one command, take the +descriptor off the reply and close that connection before returning. What +you hold afterwards is that descriptor and plain copies of what the reply +said, so a `FipsStream` and a `FipsListener` are `Send` and `'static`, +borrow nothing, and outlive nothing. A flow lasts exactly as long as its +own descriptor. + +**Dropping a stream is how you close it.** There is no close command. +The daemon watches the descriptor and releases the flow and its port +when it goes away. Dropping a listener unbinds its port the same way, and +leaves the flows already accepted from it untouched. A program that parks +streams in a `Vec` and never removes them holds ports and flow slots +exactly as if it had leaked descriptors. + +**Nothing peer-driven ever ends a flow, so your program has to.** The v1 +wire carries no half-close. Nothing closes the daemon's half of a live +accepted flow, so a loop written as "echo until the flow closes", or one +that breaks on `POLLHUP`, waits for a signal that cannot arrive. The +mistake compiles, reads naturally, and passes every test that does not +involve a real daemon. What it costs is one blocked thread and one held +flow per peer, until the process dies; the node reaches its ceiling of 256 +flows one silent peer at a time, and after that every `connect` on that +node, from any program, returns `EMFILE`. + +Two shapes are correct, and a program that receives at all needs both. +**One exchange per flow** decides how many datagrams a flow carries, so +its end is decided rather than waited for. **A deadline you impose +yourself** — `set_read_timeout`, or a poll timeout in a reactor — bounds +a wait on a peer that may never speak again. The first decides when you +have said enough; the second decides when you have waited long enough. +A longer conversation needs an end-of-conversation marker in the payload, +because the protocol will not supply one. + +Two signals are **not** a close. `Ok(0)` from `recv` is an empty datagram +and only that, so do not write `if n == 0 { break }` out of TCP habit. +`EPIPE` is real, and means the daemon went away — never that a peer +finished. + +**Ports below 1024 are refused.** 0 through 255 are reserved for +protocol use and 256 through 1023 for FIPS standard services, the IPv6 +shim among them. A client may hold 1024 through 65535. The refusal +applies to the remote port too, so a peer listening below 1024 is +unreachable from here. + +## Step 5: Run the worked example + +`examples/native-echo.rs` in the source tree is an echo server built +on nothing but this client. It holds a port and returns each datagram +to whoever sent it: + +```sh +cargo run --example native-echo -- /run/fips/api.sock 4600 +``` + +It prints `native-echo: holding port 4600` once the port is held, then +a line per datagram returned. It serves one datagram per flow by +design, for the reason the third note above gives. + +## Step 6: Inspect what the node is holding + +```sh +fipsctl show native-flows +``` + +This reports every flow the node holds — established and pending +accept — with its ports, its queue depth and its age, every bound +listener with its backlog, and the `native` counter family. The +counters are also in `fipsctl stats metrics` under `native`, where the +`drop_*` fields separate a datagram refused for having no listening +port from one dropped because a client was not reading fast enough. +For the response shape, see +[../reference/control-socket.md](../reference/control-socket.md#read-only-queries). + +Reach for this when datagrams go missing. **Four places lose data with +nothing reported to your program**: a full per-flow queue, a listener that +does not accept fast enough, an outbound datagram sent before a session +exists, and an outbound datagram after the transport MTU has fallen. +[../reference/native-api.md](../reference/native-api.md#where-data-disappears) +describes each and what bounds it. + +## See also + +- [../reference/native-api.md](../reference/native-api.md) + — the whole surface: every type and method, the errno table, the + ceilings, the line protocol and the command reference +- [write-a-native-api-client.md](write-a-native-api-client.md) + — speaking the line protocol directly from another language +- [serve-many-peers-on-one-thread.md](serve-many-peers-on-one-thread.md) + — one `poll` loop instead of the thread per flow this guide spawns +- [../design/fips-native-api.md](../design/fips-native-api.md) + — why this exists beside the TUN path, and what it is not +- [../reference/configuration.md](../reference/configuration.md#native-datagram-api-nodenative_api) + — every `node.native_api.*` key and its default +- [../reference/security.md](../reference/security.md#native-datagram-api) + — what `fips` group membership grants once the API is on +- [../reference/control-socket.md](../reference/control-socket.md) + — the `show_native_flows` response shape +- [../reference/cli-fipsctl.md](../reference/cli-fipsctl.md) + — `fipsctl show native-flows` diff --git a/docs/how-to/write-a-native-api-client.md b/docs/how-to/write-a-native-api-client.md new file mode 100644 index 00000000..268d99e7 --- /dev/null +++ b/docs/how-to/write-a-native-api-client.md @@ -0,0 +1,188 @@ +# Write a Native API Client in Another Language + +**Goal:** speak the native datagram API's line protocol directly, from C, +Python, Go or anything else, without the Rust client module. + +Everything here is something the shipped Rust library already does. It is +written out so an author working where there is no such library knows what they +are reproducing. **Each of these was a real defect before it was a rule, and +each fails intermittently rather than outright.** + +If you are writing Rust, you do not need this guide. Use +`fips::native::client` and see +[use-the-native-datagram-api.md](use-the-native-datagram-api.md). + +For the protocol itself — the framing, the reply shapes, the commands and their +refusals — see +[../reference/native-api.md](../reference/native-api.md#the-line-protocol). + +## Step 1: Read the setup connection with recvmsg, never with a buffered reader + +**Every read on the setup connection is a `recvmsg` with an ancillary buffer.** +A plain `read` consumes a descriptor-bearing message's bytes with no control +buffer, and the kernel then closes the descriptor rather than queueing it. The +reply looks perfectly correct and the flow is silently gone. + +This applies to both setup commands: a `listen` reply carries a descriptor as +much as a `connect` reply does. In any language it means no buffered reader, no +`BufReader`, no `readline`, and no library that wraps the socket in a stream +abstraction. Keep the line buffering in your own code, over `recvmsg`. + +Read a listener's own descriptor the same way, for the ancillary data and the +close-on-exec flag. One message there is one arrival carrying exactly its own +descriptor, so there is nothing to associate. + +## Step 2: Attach a descriptor to the last complete line of its read + +**A descriptor belongs to the last complete line of the read that carried it, +never to the next line the reader assembles.** A `recvmsg` returning ancillary +data ends exactly at the end of the `sendmsg` that carried it, but it may begin +with any amount of data written before it. + +A client that sends one command per connection reads one line and cannot hit +this. A client that pipelines two setup commands on one connection can: the +first reply and the second, descriptor-bearing reply arrive as one read, and a +reader that attached the descriptor to the first would hand the flow to the +wrong caller. + +Two corollaries: + +- A read that carries a descriptor and completes no line must be **reported** + rather than held. Holding it means guessing which later line it belongs to. +- A descriptor that arrives with a line you are going to discard must still be + **closed**, or the flow leaks. + +Neither can happen while the daemon writes exactly one whole line per `sendmsg` +and treats a short write as an error, which it does. That is an invariant of +two programs, though, not of the socket type. + +## Step 3: Treat a zero-byte read as end of file only when POLLHUP is set + +**An empty datagram and a closed peer both produce a zero-byte read, and +`MSG_EOR` does not tell them apart.** On Linux 6.8, `recvmsg` on an `AF_UNIX` +`SOCK_SEQPACKET` socket returns `msg_flags == 0` for a normal message, an empty +message and end of file alike, so the flag carries no information. + +`POLLHUP` does discriminate. After a zero-byte read, a queued empty datagram +leaves the socket with no events pending, while a closed peer leaves `POLLHUP` +set and latched. Poll with an events mask of **zero**, because `POLLHUP` is +reported in `revents` whether or not it was requested. The poll costs nothing: +it runs only on the zero-byte path and does not block. + +Both directions of the mistake are real. Reading an empty datagram as a close +lets a peer tear down a live flow by sending nothing, and presents as a +spurious disconnect. Reading a close as an empty datagram leaves the caller +spinning on a dead flow. + +Note what a close here means: the daemon went away, never a peer finishing. + +## Step 4: Send with MSG_NOSIGNAL + +A datagram written to a flow whose daemon half has gone, or a command written +to a daemon that has exited, raises `SIGPIPE`, whose default disposition kills +the process. + +Rust ignores the signal at startup, and CPython sets it to `SIG_IGN`, so a +program in either language sees `EPIPE`. **A C or C++ client that has not +changed the disposition simply dies.** The daemon and the shipped client pass +the flag on every send. + +## Step 5: Set a deadline on the setup socket + +Set `SO_RCVTIMEO` on the setup connection and rewrite the resulting would-block +into `ETIMEDOUT`. The shipped client uses five seconds. + +Without it, a daemon that accepted your connection and then stopped answering +blocks the setup call forever. This is also what keeps `ETIMEDOUT` to exactly +one producer on the surface, which is what lets a caller read it. + +## Step 6: Keep descriptor hygiene + +Five rules. Each one leaks a flow or loses one when broken. + +**Request close-on-exec** with `MSG_CMSG_CLOEXEC` on the `recvmsg`, rather than +setting it afterwards. Without it the descriptor survives an `exec` into a +child, the child's reference holds the flow open after this process closes its +own, and the flow keeps its slot against the node's ceiling until the child +exits. + +**Walk the whole control buffer**, not only the first header. Close extra +descriptors rather than dropping them on the floor. + +**Check for truncation after taking the descriptors, not before.** A +`MSG_CTRUNC` test that returns early leaks whatever did arrive. + +**Lift the descriptor out of an arrival you cannot parse** before discarding +the message. Refusing a flow is closing its descriptor; discarding the message +without taking it leaks the flow instead. + +**Bound the partial line.** A daemon that stopped sending newlines would +otherwise grow your buffer without end. The shipped client caps it at 64 KiB, +well above any reply. + +## Step 7: Read the errno name, never the message + +The refusal's `data.errno` is the contract. The `message` is for an operator +reading a log. + +A client that matched on English would break on a wording change. The shipped +client discards the message entirely so that no caller can come to depend on +it, and the daemon's own match over its error types is exhaustive precisely so +a new refusal cannot reach a client without a code. + +A reply carrying no `errno` at all should be read as `ECONNREFUSED`, which +covers a daemon older than the field. The errno table is in +[../reference/native-api.md](../reference/native-api.md#the-errno-table). + +## Step 8: Size the receive buffer, and add no framing + +Size every receive buffer at the flow's `max_payload`, read from the reply and +never computed. + +`SOCK_SEQPACKET` truncates a longer datagram, discards the remainder and +reports success. It is detectable: `recvmsg` sets `MSG_TRUNC` in `msg_flags` +when it dropped part of a message. A plain `recv` discards `msg_flags` and so +sees none of it, which is where the belief that truncation is silent comes +from. Test `MSG_TRUNC` as well, and a stale or misread `max_payload` is caught +rather than quietly corrupting a payload. + +Do not add framing. There is no header and no length prefix in either +direction. One send is one datagram. + +## Step 9: Decide when a flow is over, because nothing else will + +Everything above is the library's job. **The termination condition is not**, in +any language. + +The v1 wire carries no half-close. Nothing peer-driven closes the daemon's half +of a live flow, so a loop that reads until the flow ends does not terminate. +Your program decides when a flow is over, or nothing does. + +The two shapes that work are a bounded exchange, where the program serves a +known number of datagrams per flow and then drops it, and an idle deadline, +where the program sets a read timeout and treats its expiry as the end. A +server that reads "until the flow closes" holds a thread per peer forever and +holds every flow against the node's `max_flows` ceiling. + +## Verify it + +The repository's test harness drives the line protocol from Python and is the +closest thing to a second implementation: + +- `testing/native-api/client.py` — a thin RPC client that runs a script of + steps over one connection and checks the replies. It implements every rule + above. +- `testing/native-api/control.py` — reads `show_native_flows` back over the + control socket while a flow is open. + +Check what the node actually holds with `fipsctl show native-flows`, and read +the per-cause drop counters with `fipsctl stats metrics` under `native`. + +## See also + +- [use-the-native-datagram-api.md](use-the-native-datagram-api.md) — the Rust + path, where none of this is your problem +- [../reference/native-api.md](../reference/native-api.md) — the surface, the + line protocol, the command reference and the errno table +- [../reference/control-socket.md](../reference/control-socket.md) — the same + line framing, for the control socket diff --git a/docs/reference/README.md b/docs/reference/README.md index 1127216a..eae33c1f 100644 --- a/docs/reference/README.md +++ b/docs/reference/README.md @@ -19,6 +19,7 @@ guidance on when to use a feature. The "why" lives in design/; the | [nostr-events.md](nostr-events.md) | Kind 37195 advert, Kind 21059 traversal signaling, Kind 10050 inbox relays | | [transports.md](transports.md) | Per-transport statistics counter inventory | | [control-socket.md](control-socket.md) | Line-delimited JSON control protocol for the daemon and gateway | +| [native-api.md](native-api.md) | Native datagram API: the Rust surface, addressing and ports, errno table, ceilings, line protocol, command reference | | [cli-fips.md](cli-fips.md) | `fips` daemon CLI: options, exit codes, environment, files | | [cli-fipsctl.md](cli-fipsctl.md) | `fipsctl` control-client: subcommands, options, exit codes | | [cli-fipstop.md](cli-fipstop.md) | `fipstop` live-status TUI: tabs, keybindings | diff --git a/docs/reference/cli-fipsctl.md b/docs/reference/cli-fipsctl.md index 2a4df1a0..f6f5a195 100644 --- a/docs/reference/cli-fipsctl.md +++ b/docs/reference/cli-fipsctl.md @@ -55,6 +55,7 @@ prints the response's `data` object as pretty JSON. | `show transports` | `show_transports` | Transport instances: type, state, MTU, local address, per-transport stats. | | `show routing` | `show_routing` | Routing summary: pending lookups, retry state, forwarding/discovery/error/congestion counters. | | `show identity-cache` | `show_identity_cache` | Cached `(node_addr → npub)` entries with last-seen timestamps. | +| `show native-flows` | `show_native_flows` | Native datagram API: open and pending flows with their ports, queue depth and age, bound listeners with their backlog, and the `native` counters. | ### `acl ` @@ -69,7 +70,7 @@ Time-series metrics from the in-process history rings. | Subcommand | Control-socket command | Description | | ---------- | ---------------------- | ----------- | | `stats list` | `show_stats_list` | Enumerate available metrics, their units, and the per-ring retention windows. | -| `stats metrics` | `show_metrics` | Dump current counter values for every protocol metric family (`forwarding`, `discovery`, `tree`, `bloom`, `congestion`, `errors`). | +| `stats metrics` | `show_metrics` | Dump current counter values for every protocol metric family (`forwarding`, `discovery`, `tree`, `bloom`, `congestion`, `errors`, `native`). | | `stats peers` | `show_stats_peers` | List peers tracked in stats history (active or recently active). | | `stats history [options]` | `show_stats_history` | Fetch a time-series window for one metric. | diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 765efbf4..a31b61e1 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -402,6 +402,67 @@ tuning under high load or on memory-constrained devices. | `node.buffers.tun_channel` | usize | `1024` | TUN to Node outbound channel capacity | | `node.buffers.dns_channel` | usize | `64` | DNS to Node identity channel capacity | +### Native Datagram API (`node.native_api.*`) + +**Experimental, off by default, and built on Linux and FreeBSD only.** A +client process connects to a Unix socket and asks either to open a flow to a +remote pubkey or to hold a local port. Both answers carry a file descriptor: +a flow's, which the client sends and receives datagrams on, or a listener's, +which arriving flows are delivered on. There is no IPv6 emulation and no TUN +device on this path. + +The surface is not stable, is not a reliability layer, and is not the v2 +external process API. No compatibility promise is made about it: the keys +below, the line protocol behind them, and the Rust client that hides it may +change or be withdrawn in any release. + +The listener is not built on macOS or Windows, and this section is ignored +there. Two separate things bound that: Windows has no `SCM_RIGHTS` and so no +way to hand a file descriptor to another process at all, while macOS has +`SCM_RIGHTS` but does not implement `SOCK_SEQPACKET` for `AF_UNIX`. + +| Parameter | Type | Default | Description | +|-----------|------|---------|-------------| +| `node.native_api.enabled` | bool | `false` | Enable the native API socket | +| `node.native_api.socket_path` | string | *(auto)* | Socket file path. Resolved the same way as the control socket, with the filename `api.sock`: `/run/fips/api.sock` when `/run/fips` exists; then `/var/run/fips/api.sock` on FreeBSD when its private directory exists; then `$XDG_RUNTIME_DIR/fips/api.sock`; finally `/tmp/fips-api.sock` | +| `node.native_api.pending_per_flow` | usize | `16` | Datagrams held for one flow while it waits to be accepted, or while an established flow's client is slow to read. Refused above 64 at startup: the whole batch is written onto a socket pair the client cannot read yet. Refused below 1: a flow that can hold nothing loses its peer's opening datagram between the arrival being announced and the client taking the flow | +| `node.native_api.backlog` | usize | `16` | Flows announced on one listener and not yet taken by its task. Refused below 1 at startup: a listener with no backlog admits no flow, so every arrival would be dropped | +| `node.native_api.max_flows` | usize | `256` | Flows this node holds at once | +| `node.native_api.debug_commands` | bool | `false` | Answer the `inject`, `stats` and `arrive` debug commands. Not a supported interface | + +> **Security note:** the socket is mode `0770`, owned by group `fips`, and +> that is the whole of the authorization model. **Any user in the `fips` group +> can impersonate this node on the mesh.** A process that can open the socket +> can send datagrams under this node's identity to any peer it names, and can +> hold a port and receive mesh traffic addressed to this node on it. There is +> no per-client authentication, no capability check and no audit trail beyond +> the daemon's own logs. On a node with the native API enabled, treat `fips` +> group membership exactly as you would treat the node's private key. This is +> why the API is disabled by default, and why enabling it is an explicit +> operator decision rather than something a package turns on. See +> [security.md](security.md#native-datagram-api). + +A client may hold ports 1024 through 65535. Ports 0 through 255 are reserved +for protocol use and 256 through 1023 for FIPS standard services (the IPv6 +shim among them), and the daemon refuses both ranges by name. A client that +names no local port is given one from 49152 upward. + +`debug_commands` is a separate gate on three commands that exist only so the +test harness can drive the receive and dispatch paths without a wire. +`inject` makes the daemon write bytes the client chose into one of that +client's own flows, and `arrive` makes it dispatch a datagram as though a peer +had sent it, reaching any listener this node holds. A node with the key off +refuses each by name, so a client can tell "this node will not do that" from +"this build has no such command". Leave it off outside the test harness. + +`fipsctl show native-flows` reports the open and pending flows, the bound +listeners and the `native` counters; see the [`fipsctl` +reference](cli-fipsctl.md) and +[control-socket.md](control-socket.md#read-only-queries). A Rust program links +the crate and speaks the API through the `fips::native::client` module, which +hides the line protocol; see +[../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md). + ## TUN Interface (`tun.*`) | Parameter | Type | Default | Description | @@ -1010,6 +1071,13 @@ node: control: enabled: true socket_path: null # null = auto (platform runtime dir → XDG → /tmp) + # native_api: # uncomment to enable the experimental native datagram API + # enabled: true # opt-in, default false; Linux and FreeBSD only + # socket_path: /run/fips/api.sock # omit the key for the resolution above + # pending_per_flow: 16 # datagrams held for one flow; 1..=64 + # backlog: 16 # flows announced on one listener, awaiting its task; at least 1 + # max_flows: 256 # flows this node holds at once + # debug_commands: false # inject/stats/arrive; test harness only buffers: packet_channel: 1024 tun_channel: 1024 diff --git a/docs/reference/control-socket.md b/docs/reference/control-socket.md index 509fd54d..9b87dfdc 100644 --- a/docs/reference/control-socket.md +++ b/docs/reference/control-socket.md @@ -125,9 +125,10 @@ table below lists every command currently registered. | `show_transports` | — | `transports[]` — `transport_id`, `type`, `state`, `mtu`, `name`, `local_addr`, optional `tor_mode`, `onion_address`, `tor_monitoring`, `stats`. | | `show_routing` | — | `coord_cache_entries`, `identity_cache_entries`, `pending_lookups[]`, `pending_tun_destinations`, `pending_tun_packets`, `recent_requests`, `retries[]`, `forwarding`, `discovery` (request/response sub-counters; includes `req_deduplicated` — requests suppressed as recent duplicates — and `req_dedup_cache_full` — requests admitted because the dedup cache was full), `error_signals`, `congestion`. | | `show_identity_cache` | — | `entries[]`, `count`, `max_entries`. Each entry: `node_addr`, `npub`, `display_name`, `ipv6_addr`, `last_seen_ms`, `age_ms`. | +| `show_native_flows` | — | `flows[]`, `listeners[]`, `stats` (the `native` counter family). Each flow: `flow_id`, `peer` (the peer's npub, which is its address; always present, because the flow carries the key its client named or its session authenticated), `peer_addr` (the 16-byte node address in hex — a truncated hash of the same key, kept because it is what `show_sessions` and `show_routing` key on), `local_port`, `remote_port`, `state` (`established` / `pending_accept`), `queued` (datagrams the node is holding for the flow), `age_ms` (time since the flow reached its current state: opened for a flow this node opened, accepted for one taken off a listener, announced for one still pending — accepting a pending flow restarts the clock). Each listener: `local_port`, `backlog`. | | `show_listening_sockets` | — | `fips0_addr`, `firewall_active` (bool — `inet fips` table loaded), `sockets[]`. Each entry: `proto` (`tcp` / `udp`), `local_addr` (`::` or the node's fd00::/8 address), `port`, `pid` (nullable), `process` (nullable), `wildcard_bind` (bool — `local_addr == ::`), `filter` (`accept` / `drop` / `unknown` / `no_firewall`). Linux-only; returns an empty `sockets[]` on other platforms. | | `show_stats_list` | — | `metrics[]` (each with `name`, `unit`, `scope`), `fast_ring_seconds`, `slow_ring_minutes`, `peer_retention_seconds`. | -| `show_metrics` | — | Flat snapshot of every counter family in the metrics registry: `forwarding`, `discovery`, `tree`, `bloom`, `congestion`, `errors`. Each value is that family's counter snapshot object. Counter-only — gauges/histograms that need the live node are excluded. Served off the main loop. Silent-rejection sites classify their reason as a typed `RejectReason` and increment the matching per-family counter exposed here — see [Rejection reasons](#rejection-reasons). | +| `show_metrics` | — | Flat snapshot of every counter family in the metrics registry: `forwarding`, `discovery`, `tree`, `bloom`, `congestion`, `errors`, `native`. Each value is that family's counter snapshot object. Counter-only — gauges/histograms that need the live node are excluded. Served off the main loop. Silent-rejection sites classify their reason as a typed `RejectReason` and increment the matching per-family counter exposed here — see [Rejection reasons](#rejection-reasons). | | `show_stats_history` | `metric` (req), `peer` (req for per-peer metrics), `window` (`s` / `m` / `h`, default `10m`), `granularity` (`1s` / `1m`, default `1s`) | A single `Series`: `metric`, `unit`, `granularity_seconds`, `values[]`. | | `show_stats_all_history` | `peer` (optional npub), `window`, `granularity` | `granularity_seconds`, `window_seconds`, `peer`, `series[]` (one per metric). | | `show_stats_peers` | — | `peers[]`, `count`. Each entry: `npub`, `node_addr`, `display_name`, `is_active`, `first_seen_secs_ago`, `last_contact_secs_ago`. | diff --git a/docs/reference/native-api.md b/docs/reference/native-api.md new file mode 100644 index 00000000..7f0614c4 --- /dev/null +++ b/docs/reference/native-api.md @@ -0,0 +1,739 @@ +# Native Datagram API + +The native datagram API lets a program send and receive datagrams over the +mesh addressed by public key and port, with no IPv6 emulation and no TUN +device. A program opens a flow to a peer, receives a socket descriptor, and +uses ordinary socket calls on it. + +This document describes the surface. For the steps to enable it and write a +first program, see +[../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md). + +**Status: experimental.** The Rust surface and the line protocol may both +change. The API is off by default and is Unix only. + +## Enabling it + +Every bound is a key under `node.native_api`. The defaults are the values +compiled into `src/config/node.rs`. + +| Key | Default | What it bounds | +| --- | ------- | -------------- | +| `enabled` | `false` | Whether the socket is bound at all. | +| `socket_path` | resolved | See [Socket path](#socket-path). | +| `pending_per_flow` | `16` | Datagrams held for one flow, whether it awaits hand-off or its program is slow to read. Ceiling 64, floor 1, both applied when the configuration loads. | +| `backlog` | `16` | Flows announced on one listener and not yet taken. Floor 1. | +| `max_flows` | `256` | Flows this node holds at once, across every program. Established and pending flows both count. | +| `debug_commands` | `false` | Whether the three debug commands are answered. | + +There is no cap on flows per program, and no notion of a program to hang one +on. One program can reach `max_flows` by itself, and the ceiling is shared +with every other program on that node. + +See [configuration.md](configuration.md#native-datagram-api-nodenative_api) +for the full YAML reference, and [security.md](security.md#native-datagram-api) +for what `fips` group membership grants once the API is on. + +## Socket path + +`node.native_api.socket_path` has a default resolved at startup rather than a +fixed string. The resolver takes the first of these that applies and appends +`api.sock`: + +1. `/run/fips/api.sock`, when `/run/fips` exists as a directory. This is the + packaged Linux convention and the constant the shipped Rust client compiles + in. +2. `/var/run/fips/api.sock` on FreeBSD, whose service scripts create that + directory. Linux skips this step. +3. `$XDG_RUNTIME_DIR/fips/api.sock`, when that variable names an existing + directory. A development run usually gets this one. +4. `/tmp/fips-api.sock`, the last resort. + +Selection is by existence of the directory, not by whether the client can +write to it. A client should therefore take the path as an argument with these +as its defaults, rather than compile one in. + +## Addressing + +An address is an x-only secp256k1 public key and a port. Nothing else +identifies an end. There is no node identifier, no address family, and no +scope. + +The npub is the key written down. Conversion between the two is bech32 and +nothing else: no lookup, no resolution, no name service. The conversion is +local, needs no daemon, and fails identically whether or not one runs. + +The 16-byte node address that the wire carries never appears on this surface. +It is a truncated hash of the public key, it does not invert, and no call here +accepts or returns one. + +### Ports + +Ports are FSP ports, carried inside the encrypted FSP envelope. They are +tiered. + +| Range | Use | What a program may do | +| ----- | --- | --------------------- | +| 0-255 | Protocol use | Refused. | +| 256-1023 | FIPS standard services | Refused. Port 256 is the IPv6 shim. | +| 1024-49151 | Application, well known | Name it explicitly. | +| 49152-65535 | Application, ephemeral | Name it explicitly, or ask for 0 and be given one from this range. | + +Nothing reserves the upper range against explicit use. A service that wants a +fixed port above 49151 may hold one, and ephemeral allocation will not take it +away, because the sweep skips a port already held. + +**The refusal applies to the remote port as well as the local one.** A program +cannot send to a peer's port 256 and inject into its IPv6 plane. By the same +rule, a peer listening below 1024 is unreachable from here: the `connect` is +refused with `EADDRNOTAVAIL` before any node state is touched. + +**Port 0 means "any port"**, for both calls, and it is the only spelling of +that. This matches `bind(2)` with port 0. `local_addr()` afterwards reports the +port actually held, which is `getsockname`. + +A local port has exactly one owner, a listener or one connected flow, never +both. A connected flow owns its local port and frees it when the flow closes. +A flow accepted from a listener shares the listener's port and does not own it, +so closing an accepted flow frees no port; the port returns when the listener +drops. Ephemeral allocation sweeps forward from the last claim and wraps once, +so a port is not immediately reused after release. + +## The Rust surface + +The crate module is `fips::native::client`. It is gated to Unix targets. + +```rust +use fips::native::client::{FipsAddr, FipsListener, FipsStream, ToFipsAddr}; +``` + +`SOCKET` is a `&str` constant holding `/run/fips/api.sock`, the packaged path. +`XOnlyPublicKey` is re-exported from `secp256k1` so a caller needs no direct +dependency on that crate. + +### The Berkeley mapping + +The descriptor is a real socket. Everything a program does with a flow after +setup is the operating system, not this API. + +| Berkeley | Here | Notes | +| -------- | ---- | ----- | +| `socket()` | none | There is no unbound object to make. `connect` and `bind` each return one already bound. | +| `bind()` | `FipsListener::bind` | One setup call. Port 0 asks the daemon to choose. | +| `listen()` | none | Folded into `bind`. The backlog is node configuration, not a caller's argument. | +| `connect()` | `FipsStream::connect` | One setup call. A local registration that contacts no peer. | +| `accept()` | `FipsListener::accept` | Blocks until a flow arrives. One `recvmsg`, no round trip to the daemon. | +| `send()` | `FipsStream::send` | One datagram, whole or not at all. One local size check, then one syscall. | +| `recv()` | `FipsStream::recv` | One datagram. `Ok(0)` is an empty one and only that. | +| `poll()`, `select()`, `epoll` | the same calls | Both types are `AsFd` and `AsRawFd`. | +| `fcntl(O_NONBLOCK)` | `set_nonblocking()` | On both types. Read-modify-write, so a flag the caller set survives. | +| `close()` | drop | Dropping releases the flow or unbinds the port. There is no method. | +| `getsockname()` | `local_addr()` | A field read of what setup reported. It cannot fail. | +| `getpeername()` | `peer_addr()` | The same, for the far end. | +| `SO_RCVTIMEO` | `set_read_timeout()` | On a flow, with `set_write_timeout` and both getters. | +| `SO_RCVBUF`, others | via `AsFd` | `setsockopt` on the descriptor. The surface exposes no option beyond the deadlines. | +| `inet_pton()` | npub decode | bech32, locally, with no lookup. | +| `inet_ntop()` | npub encode | Its exact inverse. | + +Four differences from Berkeley remain. There is no `socket()` and no +`listen()`, because a descriptor cannot exist before the daemon has agreed to +make one. A successful `connect` proves nothing about the peer. `Ok(0)` from +`recv` is an empty datagram rather than end of file. And there is a payload +ceiling, reported per flow, which TCP has no equivalent of. + +### FipsAddr + +One end of a flow. The fields are private, so an address in hand always names +a valid key. + +`Copy`, `Debug`, `Clone`, `PartialEq`, `Eq`. **It derives neither `Hash` nor +`Ord`**, so it cannot key a `HashMap` or a `BTreeMap`. Key on +`addr.key().serialize()` or on `addr.to_string()` instead. +`std::net::SocketAddr` derives both, so the omission is a surprise rather than +a convention. + +| Call | Returns, and how it fails | +| ---- | ------------------------- | +| `new(key, port)` | The address. Infallible: both arguments are already well typed. Port 0 and the reserved tiers are representable and are refused later, by the daemon, at setup. | +| `key()` | The public key, by value. Infallible. | +| `port()` | The port. Infallible. | +| `Display` | Writes `npub1...:4242`. Pure bech32, no daemon, no lookup. | +| `FromStr` | Reads that back. `EINVAL` for no colon, a port that is not a `u16`, or a head that is not an npub. An nsec is refused despite being bech32 of the right length. | + +`Display` and `FromStr` are exact inverses. Parsing splits on the last colon +and is purely local, so it fails identically with no daemon running. The colon +is unambiguous because the bech32 character set does not contain one. + +### ToFipsAddr + +The mirror of `std::net::ToSocketAddrs`. One generic parameter takes every +spelling of one address. It is taken by `&self`, so a caller never gives up +ownership, and **no implementation opens a socket or contacts the daemon**. The +only error any of them produces is `EINVAL`. + +| Implementation | Notes | +| -------------- | ----- | +| `FipsAddr` | Returns a copy. Structurally infallible. | +| `(XOnlyPublicKey, u16)` | The binary address, already parsed. Structurally infallible. | +| `([u8; 32], u16)` | 32 raw bytes. The one binary form that can be rejected: `EINVAL` if the bytes are not a valid x-only public key, which `[0u8; 32]` is not. | +| `(&str, u16)` | An npub and a port. `EINVAL` if the npub does not decode. The port is taken as given and range-checked later. | +| `(String, u16)` | The same, for a string from `argv` or a config file. | +| `str` | The whole address as `"npub1...:4242"`. Delegates to `FromStr`. | +| `String` | The same. | +| `&T where T: ToFipsAddr` | A blanket implementation over references, so `&addr` and `&&str` work at all. | + +There is **no associated iterator type**, unlike `ToSocketAddrs`. An address +here resolves to exactly one endpoint. + +The concrete implementations are on the unsized `str` rather than on `&str`. +That is what lets the blanket reference implementation cover `&str` and +`&&str` alike. It makes no difference at a call site. + +### FipsStream + +One datagram flow, and the descriptor it rides on. A flow is an exact match of +both ends and both ports. **The descriptor is the flow**: it lives while a +process holds that descriptor and ends when the last one closes it. + +`Send + Sync + 'static`, with no `Arc` and no borrow. There is **no `Clone` and +no `try_clone`**. Because `send` and `recv` both take `&self`, a shared borrow +across two scoped threads is a full-duplex reader and writer. A program wanting +an owned writer handle reaches for `Arc`. + +#### Constructors + +| Call | What it does | +| ---- | ------------ | +| `connect(addr)` | A flow to `addr` from an ephemeral local port, at the packaged socket path. | +| `connect_from(local, addr)` | The same from a named local port. `connect_from(0, addr)` is exactly `connect(addr)`. | +| `connect_at(sock, local, addr)` | The full form, naming the daemon's socket. What a development node and a test harness need. | + +All three resolve the address first, so a bad address is `EINVAL` with no +socket opened. Then, in this order: whatever `UnixStream::connect` gives for +the socket path (`NotFound` when no daemon runs or the API is disabled, +`PermissionDenied` on socket permissions, `ECONNREFUSED` when the path exists +but nothing is accepting); `EPIPE` if the daemon vanishes mid-write; +`ETIMEDOUT` if no answer arrives inside the five-second setup deadline; then +the daemon's own refusals; then `InvalidData` for a reply this client cannot +parse. + +A stream does not outlive its setup connection, because the connection is not +an object at all. It is a local binding inside the constructor, dropped before +the call returns. What comes back holds an owned descriptor and four `Copy` +scalars. + +#### Methods + +**`send(&self, buf: &[u8]) -> io::Result<()>`** sends one datagram. There is no +byte count, because a `SOCK_SEQPACKET` message is delivered whole or not at +all. `buf.len() > max_payload()` is `EMSGSIZE` before any syscall, so a caller +learns which datagram was too large rather than finding a gap at the far end. +`EPIPE` means the daemon has gone. Takes `&self`, so it can be called +concurrently with `recv` from another thread. + +The library sends with `MSG_NOSIGNAL` rather than writing to the descriptor. A +Rust binary ignores `SIGPIPE` at startup anyway, but a C program that has +loaded this crate would otherwise be killed by a write to a departed daemon +rather than told about it. `EINTR` is retried internally. + +`Ok(())` means only that the local socket pair accepted the bytes. See +[Where data disappears](#where-data-disappears). + +**`recv(&self, buf: &mut [u8]) -> io::Result`** receives one datagram +and returns its length. It **blocks by default**: the descriptor arrives +blocking and no timeout is set on it. `set_nonblocking(true)` makes it return +`WouldBlock` instead. `set_read_timeout` bounds a single wait without going +non-blocking. A datagram longer than `buf` is truncated and the remainder +discarded, which is `SOCK_SEQPACKET` behaviour and is not reported; size `buf` +at `max_payload()` and it cannot happen. `EINTR` is retried. + +**`Ok(0)` is an empty datagram and only that.** An empty datagram and a closed +peer both produce a zero-byte read, and `MSG_EOR` does not tell them apart. On +the zero-byte path only, the library polls the descriptor with an events mask +of zero and reads `POLLHUP` from `revents`, returning `EPIPE` for a genuine +close and `Ok(0)` otherwise. Without that step a peer could tear down a live +flow by sending nothing. The `EPIPE` that `recv` returns means the daemon went +away, never that a peer finished. + +**`peer_addr()`** and **`local_addr()`** return `FipsAddr`, not +`io::Result`, unlike their `TcpStream` counterparts. These are field +reads of what the setup reply already carried. `local_addr()` carries the +node's own public key, which on an accepted stream comes from the arrival +message rather than from the listener, so a node holding more than one identity +still answers correctly. + +**`max_payload()`** returns the largest datagram this flow will carry. +Infallible. **It is a snapshot** taken when the flow opened and does not track +a later MTU change. + +**`set_read_timeout(Option)`** and **`set_write_timeout`** bound one +`recv` or one `send`, as their `TcpStream` counterparts do. `None` clears. +**A zero duration is `EINVAL`**, because the kernel reads a zero timeout as +"wait for ever" and a caller passing zero means the opposite. An expiring +deadline reports `WouldBlock`, the same answer a non-blocking descriptor gives. +**`read_timeout()`** and **`write_timeout()`** read them back, `None` when +unset. The two directions are separate options, and setting one leaves the +other alone. + +**`set_nonblocking(bool)`** puts the flow in or out of non-blocking mode. In +that mode `recv` returns `WouldBlock` rather than waiting, and so does `send` +when the daemon is not draining the flow fast enough. **A flow from `accept` is +blocking however its listener was set**: they are separate sockets and the +daemon hands over a fresh one. + +**`AsFd`** and **`AsRawFd`** both return the flow's descriptor. Prefer `AsFd`: +its borrow cannot outlive the stream, so a reactor cannot hold a registration +for a descriptor that has since been closed and its number reissued. **The +stream keeps ownership** either way. Do not close the descriptor and do not +wrap it in anything that takes ownership, because the stream's own drop is what +releases the flow. This is the route to `SO_RCVBUF` and every other socket +option the surface does not expose. + +A descriptor that survives an `exec` into a child holds the flow open after +this process closes its own copy, and the flow keeps its slot against the +node's ceiling until the child exits. The library requests close-on-exec when +it takes the descriptor, so this happens only if a program deliberately clears +the flag. + +### FipsListener + +A held local port, and the descriptor arriving flows are delivered on. **The +listener is a descriptor**, so it joins an existing `poll`, `select` or `epoll` +loop with no new mechanism, and `accept` is one `recvmsg` on it. + +`Send + Sync + 'static`, no `Clone`, no `try_clone`, no explicit `Drop`. + +**`bind(port)`** holds `port`, or an ephemeral one when `port` is 0, at the +packaged path. **`bind_at(sock, port)`** is the same, naming the daemon's +socket. Failures are the socket-path and setup-deadline rows above, plus +`EADDRINUSE` when a listener or a flow already holds the port, and +`EADDRNOTAVAIL` for a reserved tier or an exhausted ephemeral range. + +**`accept() -> io::Result<(FipsStream, FipsAddr)>`** blocks until a flow +arrives. One `recvmsg`, one arrival: a `SOCK_SEQPACKET` message carries exactly +its own descriptor. Takes `&self`, so several threads may accept on one +listener concurrently. It does not map a would-block to `ETIMEDOUT`, so a +caller that put the listener in non-blocking mode sees a raw `WouldBlock`. +Failures: `EPIPE` when the message carried no descriptor, meaning the daemon +closed its half; `InvalidData` for an arrival that is not JSON or is missing a +field; and `io::Error::other` if the control buffer overflowed and a descriptor +was lost. + +Whatever the peer sent before `accept` returned is already queued on the +returned stream. **Refusing a flow is dropping the stream**, because there is +no other way to refuse one. An unparseable arrival is therefore reported only +after the descriptor it carried has been taken into ownership, so a parse +failure refuses the flow rather than leaking it. + +**`incoming()`** returns an `Incoming<'_>`, which borrows the listener for the +iterator's lifetime, so the listener cannot be moved or dropped mid-iteration. + +**`local_addr()`** returns the port actually held with the node's own key, not +an `io::Result`. + +**`set_nonblocking(bool)`** puts the listener in or out of non-blocking mode. +In that mode `accept` returns `WouldBlock` when no flow has arrived, and so +does every `incoming` item, which makes that iterator spin unless the caller +waits on the descriptor between items. It does not reach the flows the listener +yields: each arrives blocking. + +**`AsFd`** and **`AsRawFd`** return the listener's descriptor. **`poll` on it +reports readable exactly when `accept` would not block.** + +**There is no `set_read_timeout` here**, following `TcpListener`, which has +none either. A bounded accept is `set_nonblocking` plus a wait of the caller's +own on the descriptor. + +**Dropping** closes the descriptor and unbinds the port. Flows already accepted +from it are untouched; flows still pending on it go with it. + +### Incoming + +The iterator `incoming()` returns. Its item is `io::Result`: it +calls `accept` and discards the peer address, which is recoverable as +`stream.peer_addr()`. + +**It never returns `None`.** A listener has no last flow, and a failed accept +is yielded as an `Err` item rather than ending the iteration, so a `for` loop +over it never falls through. A reader who writes `let flow = flow?;` inside the +loop exits it on the first transient error, which is a different shape from +`TcpListener` habits. + +## Errors + +There is no bespoke error type. Every call returns `io::Result`, so the surface +matches `std::net` and a future C binding can return the number directly. + +**Match on `err.raw_os_error()` against `libc` constants, never on +`err.kind()`.** `EMFILE` and `EMSGSIZE` both carry +`ErrorKind::Uncategorized`, which is `#[non_exhaustive]` and cannot be named in +a match arm, so `kind()` cannot distinguish the node's flow ceiling or an +oversize datagram from anything else. + +### The errno table + +Compare against the `libc` constant, never against a literal. The client maps +each name onto `libc::` for the platform it was built for, so the number +differs between Linux and FreeBSD. + +| errno | `kind()` | What causes it | +| ----- | -------- | -------------- | +| `EADDRINUSE` | `AddrInUse` | The local port is held, or a flow already exists between those two ends on those two ports. | +| `EADDRNOTAVAIL` | `AddrNotAvailable` | A port in a reserved tier, local or remote, or the ephemeral range exhausted. Reserved rather than `EACCES`, because no program however privileged may hold port 256. | +| `EMFILE` | `Uncategorized` | The node is at `max_flows`, or the socket pair could not be made. Unmatchable by kind. | +| `EINVAL` | `InvalidInput` | An address that does not resolve locally, or a malformed command. | +| `EMSGSIZE` | `Uncategorized` | A `send` above `max_payload()`. Raised locally, before the syscall. Unmatchable by kind. | +| `EPIPE` | `BrokenPipe` | The daemon went away: it exited, the node shut down, or its own read failed. Never a peer finishing. | +| `ETIMEDOUT` | `TimedOut` | The five-second setup deadline, and nothing else on this surface. | +| `ECONNREFUSED` | `ConnectionRefused` | The node is shutting down, a debug command is disabled, nothing is accepting on the socket path, or the reply carried an errno name this client has no row for. | + +The last row is the catch-all, so `ECONNREFUSED` is the one code that does not +narrow the cause much. **The daemon sends errno names rather than numbers**, +because a number belongs to the platform the program was built for and the +daemon is not it. An unknown name is read as `ECONNREFUSED` for the same reason +a missing one is. The daemon also sends a human-readable message, and the +client discards it, so no caller can come to depend on prose. + +### The errors that carry no errno + +Three shapes have no `raw_os_error()` at all. + +**`ErrorKind::InvalidData`** and **`io::Error::other`** both mean this daemon is +not speaking the protocol: a reply or an arrival that is not JSON, an unknown +or missing status, a missing or malformed field, a port outside `u16`, a reply +carrying no descriptor, a descriptor arriving on a read that completed no line, +more than 64 KiB with no newline, or a truncated control message. None is worth +retrying, and none is a condition a correct daemon produces. + +**`ErrorKind::NotFound`** on a setup call means no daemon is running, or the +native API is disabled, which is the default. + +### What is worth retrying + +`EMFILE` and `EADDRNOTAVAIL` from an exhausted ephemeral range are load +conditions and may clear. A retry with a backoff is reasonable, and a program +that retries without one contributes to the exhaustion. `ETIMEDOUT` and `EPIPE` +mean the daemon is unhealthy or gone, so the useful retry is the whole setup +sequence and not the one call. `EADDRINUSE`, `EADDRNOTAVAIL` from a reserved +tier, `EINVAL` and `EMSGSIZE` are decisions about the arguments and produce the +same answer every time. + +## Where data disappears + +The API is datagram-unreliable. Nothing on this surface confirms delivery. +There is no acknowledgement, no retransmission, no ordering guarantee and no +flow control between the two ends. A program that needs confirmation gets it +from the peer, in the payload. + +Four places lose data with nothing reported to the client. + +**A full per-flow queue.** Inbound datagrams beyond `pending_per_flow` are +dropped with a trace log and no client-visible signal. A program that stops +reading a flow loses datagrams and is never told. + +**A listener that does not accept fast enough.** Whole arriving flows are +discarded, counted inside the daemon as `ListenerNotReading`. The bound on how +many flows a stalled listener holds is not the backlog: the send buffer on a +listener's pair is sized generously, because the approximation must err toward +accepting an arrival a program would have read. A program that binds a listener +and stops reading it accumulates flows on the order of `max_flows` rather than +of `backlog`. + +**An outbound datagram sent before a session exists.** The first `send` on a +new flow almost always takes this path, because `connect` contacts no peer and +leaves no FSP session behind it. The node holds the datagram and starts a +handshake. What holds it is bounded twice: at +`node.session.pending_packets_per_dest` datagrams for one destination, past +which a further datagram evicts the oldest one held, and at +`node.session.pending_max_destinations` destinations, past which a new +destination's datagram is dropped outright. The defaults are 16 and 256. +Neither eviction reaches the caller. + +**An outbound datagram after the MTU has fallen.** `max_payload()` is a +snapshot taken at setup. The daemon re-checks each outbound datagram against +the node's current limit and drops it silently if the transport MTU has since +fallen. + +### The drop causes, and what they mean + +An inbound datagram can be refused for eight reasons, which render as seven +texts: a pending flow's full queue and an established flow's are distinct to a +counter and alike to a client. The texts are what `DropReason::as_str` produces; +the counter names are what `fipsctl stats metrics` reports under `native`. + +| Text | Counter | Condition | +| ---- | ------- | --------- | +| `no listener or flow on that port` | `drop_no_port` | Nothing holds the destination port. | +| `listener backlog full` | `drop_backlog_full` | A listener holds the port but will not hold another pending flow. | +| `node flow ceiling reached` | `drop_too_many_flows` | The node is at `max_flows`. | +| `queue full` | `drop_pending_queue_full`, `drop_flow_queue_full` | A flow's queue is full, whether it is pending or established with a client that is not reading. Two counters, one text. | +| `arrival queue full` | `drop_arrival_queue_full` | The daemon's own queue to a listener's task is full. | +| `listener not reading arrivals` | `drop_listener_not_reading` | A listener's client is not reading its descriptor, so the arrival could not be written to it. | +| `listener closed` | `drop_listener_gone` | A listener's client closed its descriptor between the arrival being taken off the queue and being written. | + +**None of these reaches a client for real traffic.** There is no drop event and +no reply reports one. The texts are visible only through the debug `arrive` +command, which reports `dropped: ` as its outcome; the counters are +readable at any time through the control socket. + +`drop_oversize` is a ninth counter and is not in this table, because it is not a +dispatch refusal: it counts an **outbound** datagram the daemon discarded when +the transport MTU had fallen below it, which is the fourth case above. + +### No framing, in either direction + +There is no header and no length prefix. One send is one datagram and one +receive is one datagram. A program that adds its own length prefix on top of a +message-boundary-preserving transport is paying for something it already has. + +### Seeing what a node holds + +Neither type reports the node's state, and the client module exposes no +statistics. `fipsctl show native-flows` is the only way to see what a node +holds, and `fipsctl stats metrics` under `native` carries the per-cause drop +counters that no program is told about. See +[cli-fipsctl.md](cli-fipsctl.md) and +[control-socket.md](control-socket.md). + +## What a daemon restart costs + +**Every flow and every listener ends when the daemon does**, and descriptors do +not survive it. The daemon's halves close with the process, so a program reading +one gets `EPIPE` and a program writing one gets the same. There is no +reconnection and no resumption: a program that must survive a restart re-runs +its setup calls and gets new descriptors. + +**Datagrams already sent and not yet forwarded are lost.** At an orderly flow +close there is nothing in flight, because a flow's reader forwards what the +client wrote and only then notices the flow is over. At daemon exit every reader +stops at once with no such notice. The window is small and nothing bounds it. + +A program that needs to know its last datagram reached a peer needs an +acknowledgement from that peer. Neither this API nor the wire beneath it has one +to offer. + +## The line protocol + +The Rust client hides all of this. It is documented because a client in +another language has to implement it. For the obligations such a client +carries, see +[../how-to/write-a-native-api-client.md](../how-to/write-a-native-api-client.md). + +### Framing + +The encoding is the control socket's, so a client that speaks one speaks both. +One JSON object per line, terminated by `\n`. + +A request is: + +```json +{"command": "connect", "params": {"peer": "npub1...", "remote_port": 4242}} +``` + +Every command requires `params`. A request without the key is refused with +`command '' requires params`. + +A command line is capped at 8192 bytes. A longer one ends the connection with +`native API command too large`. **That is the only condition that ends the +connection instead of producing a reply.** The cap is applied whether or not +the chunk in hand holds the newline, so a well formed 9000-byte line ends the +connection exactly as a runaway one with no newline does. The connection is +dropped with nothing written back, so a client sees end of file and never a +message. + +Two kinds of line come back, and only two. A success: + +```json +{"status": "ok", "data": {}} +``` + +And a refusal, which carries the code a client acts on and prose it must not +match against: + +```json +{"status": "error", "data": {"errno": "EADDRNOTAVAIL"}, + "message": "port 256 is reserved for FIPS standard services"} +``` + +**There is no third kind.** No event is pushed on this connection and nothing +arrives on it unsolicited, so a reader that sends a command and takes the next +complete line as its answer is correct. A refusal is a normal reply line and +never drops the connection. A reply carrying no `errno` at all is read as +`ECONNREFUSED`, which covers a daemon older than the field. + +### The arrival message + +**An arriving flow is not a line.** It is one `SOCK_SEQPACKET` message on the +listener's own descriptor, with **no trailing newline**: the message boundary +is the framing, and a newline would offer a client a second framing to rely on. + +Its seven fields are `flow_id`, `peer` and `node` as npubs, `local_port`, +`remote_port`, `max_payload` and `held`. The `node` field is carried so an +accepted flow can answer `local_addr` without consulting the listener that +produced it. `flow_id` is the same identifier a `connect` reply carries and the +name the debug commands take. + +**`held` is the one field a client cannot infer.** It counts the datagrams the +daemon has already written onto the flow's descriptor before this message, +because the hand-off writes every held datagram first and the arrival last. It +says exactly how many `recv` calls a client may make on a newly accepted flow +without blocking. The Rust client reads neither `held` nor `flow_id`, because +neither names anything a caller can name. A client in another language may +ignore both on the same reasoning; it cannot ignore that they are there. + +### Passing the descriptor + +The daemon builds an `AF_UNIX` `SOCK_SEQPACKET` socket pair with +`SOCK_CLOEXEC`, keeps one half, and sends the other over `SCM_RIGHTS` in the +ancillary data of the same `sendmsg` that carries the reply line. It makes its +own half non-blocking and leaves the client's half blocking. `SOCK_SEQPACKET` +is what preserves message boundaries in both directions, which is why the +payload needs no framing. + +A refused `connect` leaves the port free: the socket pair is built before the +port is claimed, so a failure to build it needs no rollback. + +### The setup call + +One `AF_UNIX` `SOCK_STREAM` connection to the socket path, one command line +written, one reply line read. Replies come back in command order, and several +commands are permitted on one connection, so a client may pipeline. The shipped +Rust client does not: it opens a connection per setup call and drops it before +returning. + +**The connection owns nothing.** Closing it releases no flow and no listener, +and a descriptor kept across the close keeps working. What owns the flow is the +descriptor. + +## Command reference + +Five commands. Two are the interface; three are debug scaffolding, off by +default. + +| Command | Carries FD | What it does | +| ------- | ---------- | ------------ | +| `connect` | yes | Open a flow to a peer named by npub. Returns the flow's descriptor. | +| `listen` | yes | Hold a local port. Returns the listener's descriptor. | +| `stats` | no | Debug. Report what the daemon received on a flow. | +| `inject` | no | Debug. Write bytes into a flow from the daemon's side. | +| `arrive` | no | Debug. Dispatch a datagram as though a peer had sent it. | + +**There is no close command, no accept command, no reject command, and no +command to enumerate flows.** Closing a descriptor does the first three, and +`fipsctl show native-flows` does the fourth from the control socket. + +### connect + +| Parameter | Type | Meaning | +| --------- | ---- | ------- | +| `peer` | string | The far end, as an npub. | +| `remote_port` | u16 | The far end's port. Required. | +| `local_port` | u16 | Optional. Absent, `null` and `0` all mean an ephemeral port. | + +Reply data: `flow_id`, `local_port`, `remote_port`, `peer` (the daemon's own +re-encode of the key, not an echo of what the caller wrote), `node` (this +node's own npub), and `max_payload`. Carries the flow's descriptor. + +Refusals, in the order they are decided, which is what lets a client read an +`EADDRNOTAVAIL` as a tier refusal rather than an exhaustion: + +1. `EADDRNOTAVAIL`: the port is in a reserved tier, either port. Decided before + any node state is touched. +2. `EINVAL`: the npub does not decode. +3. `EMFILE`: the socket pair could not be built, or the node holds its maximum + flows. +4. `EADDRINUSE`: the named local port is held, or a flow to that peer between + those two ports already exists. +5. `EADDRNOTAVAIL`: no ephemeral port is free. +6. `ECONNREFUSED`: the node is shutting down. + +### listen + +One parameter, `local_port` (u16), where absent, `null` and `0` all mean an +ephemeral port. Reply data: `local_port`, `node`, `backlog`. **Carries the +listener's descriptor**, and a reply that carried none is a daemon that is not +this one. + +`backlog` reports the depth the daemon will hold for this listener, so an +operator can size a client's reader against it. Nothing on the Rust surface +takes a depth as a parameter or exposes the reported one. + +Refusals: `EADDRNOTAVAIL` for a reserved tier or an exhausted ephemeral range; +`EADDRINUSE` when the port is held, whether by another listener or by a +connected flow; `EMFILE` when the socket pair could not be built; +`ECONNREFUSED` when the node is shutting down. + +A flow announced on a listener but never taken is discarded after five seconds. +This is not an accept timeout and a client cannot reach it: the window it +bounds is a hand-off between two tasks inside the daemon, not a client's round +trip. + +### The debug commands + +`stats`, `inject` and `arrive` exist so the daemon's own checks can drive the +receive and dispatch paths without a peer. They are answered only where +`node.native_api.debug_commands` is set, which is off by default and which no +packaged node sets. A node with the key unset refuses all three by name, with +`ECONNREFUSED` and a message naming the key that would admit them, so a client +can tell "this node will not" from "this build cannot". + +**Do not build a client on them.** + +`stats` takes `flow_id` and reports `flow_id`, `local_port`, `rx_datagrams`, +`rx_bytes` and `closed`. The counters are the daemon's view of what the client +wrote into the descriptor, independent of whether any of it then reached a +peer. + +`inject` takes `flow_id`, `data` (a hex string) and an optional `repeat` +(default 1, maximum 64). It writes `repeat` separate datagrams of those bytes +onto the flow's descriptor from the daemon's side. The flow table it names into +is the node's, not the connection's, so a caller that can reach this command +can write into any flow on the node. + +`arrive` takes `peer` (npub), `src_port`, `dst_port` and `data` (hex), and +drives the same delivery decision the real receive path uses. It replies `ok` +whenever the node answers at all: + +```json +{"status": "ok", "data": {"outcome": "announced", "flow_id": 9}} +``` + +`outcome` is `delivered`, `announced`, `held`, or `dropped: `. +`flow_id` is non-null only for `announced` and `held`. The peer's key is decoded +from the npub the caller named, which makes it client-asserted rather than +authenticated. Port tiers are not checked here. + +## Compiled-in bounds + +Three bounds are not configurable: a command line of 8192 bytes, a per-arrival +send-buffer allowance of 4096 bytes on a listener's pair, and a `repeat` of 64 +on the debug `inject`. The daemon reads each descriptor with a 65535-byte +buffer, which is above anything the wire carries. + +`pending_per_flow` has a compiled ceiling of 64, checked when the configuration +loads rather than at the first arrival: the whole held batch is written onto a +socket pair whose client half has not been sent yet, so a larger value could +leave a listener's task with a write it cannot complete. Both it and `backlog` +have a floor of 1. A `backlog` of zero admits no flow at all. A +`pending_per_flow` of zero is worse, because the arrival is announced and the +datagram that caused it is then refused, so a peer's opening message vanishes +with no refusal a client or an operator can see. + +## See also + +- [../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md) + — enable the API and write a first program +- [../how-to/write-a-native-api-client.md](../how-to/write-a-native-api-client.md) + — the obligations a client in another language carries +- [../how-to/serve-many-peers-on-one-thread.md](../how-to/serve-many-peers-on-one-thread.md) + — one `poll` loop instead of a thread per peer +- [../design/fips-native-api.md](../design/fips-native-api.md) + — what this interface is for, what it is instead of, and what it is not +- [configuration.md](configuration.md#native-datagram-api-nodenative_api) + — every `node.native_api.*` key and its default +- [security.md](security.md#native-datagram-api) + — what `fips` group membership grants once the API is on +- [control-socket.md](control-socket.md) + — the `show_native_flows` response shape +- [cli-fipsctl.md](cli-fipsctl.md) + — `fipsctl show native-flows` diff --git a/docs/reference/security.md b/docs/reference/security.md index 12c64bdc..8f391cf3 100644 --- a/docs/reference/security.md +++ b/docs/reference/security.md @@ -189,11 +189,74 @@ mutation; rate-limited msg1s never reach the ACL. | `/etc/fips/peers.allow` | root:root | `0644` | Optional peer allowlist. | | `/etc/fips/peers.deny` | root:root | `0644` | Optional peer denylist. | | `/run/fips/control.sock` | root:fips | `0770` | Control socket (members of `fips` group can use `fipsctl`). | -| `/run/fips/` | root:fips | `0750` | Control socket parent directory. | +| `/run/fips/api.sock` | root:fips | `0770` | Native datagram API socket, when `node.native_api.enabled` is set (experimental; absent otherwise). | +| `/run/fips/` | root:fips | `0750` | Socket parent directory. | Adding a user to the `fips` group grants `fipsctl` access without requiring root. The daemon `chown`s the control socket and its parent -directory at bind time. +directory at bind time, and does the same for the native API socket when +that is enabled. + +## Native Datagram API + +**Experimental. Disabled by default** (`node.native_api.enabled`, default +`false`), and built on Linux and FreeBSD only. It is not a stable API +surface, not a reliability layer, and not the v2 external process API. No +compatibility promise is made about it. + +**Any user in the `fips` group can impersonate the node on the mesh.** The +API socket is created at mode `0770` owned by group `fips`, and that is the +entire authorization model. A process that can open it can: + +- send datagrams under this node's identity to any peer it names, which + peers authenticate as coming from this node; +- hold any port from 1024 upward and receive mesh traffic addressed to this + node on it, including traffic another local program expected; +- do both without authenticating, without a capability check, and without + any record beyond the daemon's own logs. + +Group membership is therefore equivalent to possession of the node's +identity for the purpose of sending on the mesh. **On a node with the native +API enabled, treat membership of the `fips` group exactly as you would treat +`/etc/fips/fips.key`.** Grant it to the accounts that are trusted to speak as +the node and to no others, and review it before enabling the API on a shared +machine. + +**The file descriptor carries the grant, not the connection.** A setup call +hands the client a socket descriptor and the connection it was made on is then +closed; the flow or the held port lives until that descriptor is closed. A +descriptor is an ordinary kernel object, so it survives `fork`, survives +`exec` unless the client asked for it close-on-exec when it received it, and +can be handed to another process over `SCM_RIGHTS`. A process holding one can +send as this node on that flow, or receive on that port, without ever opening +the API socket and without being in the `fips` group. +Nothing revokes a descriptor already handed out. Restarting the daemon closes +its own halves and ends every flow and listener at once, and that is the only +revocation there is. + +Two consequences follow for `fipsctl` access. First, the `fips` group is +already the control-socket group, so enabling the native API silently +upgrades every existing `fipsctl` user from "can read node state and manage +peers" to "can send as the node". Second, an operator who wants the two +audiences separated must not enable the API on a node whose `fips` group has +been handed out for monitoring. + +`node.native_api.debug_commands` (default `false`) is a second, independent +gate. It admits three commands (`inject`, `stats`, `arrive`) that exist for +the test harness: `arrive` makes the daemon dispatch a datagram as though a +peer had sent it, reaching any listener on this node under any peer identity +the caller names. Leave it off outside a test harness; a packaged node does +not enable it. + +The socket is local only. It is not reachable over the network, and nothing +about it changes the mesh's own authentication: a peer still verifies the +node's signature, which is precisely why a local caller that can send through +this socket is indistinguishable from the node itself. + +See [configuration.md](configuration.md#native-datagram-api-nodenative_api) +for the key list and +[../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md) +for the client. ## Threat-Resistance Matrix @@ -214,6 +277,7 @@ the FMP design document: | Sybil identities | Discretionary peering + handshake rate limiting + optional peer ACL | | Eclipse attack | Diverse peering across independent operators and transports | | Unauthorized peer admission | Optional `peers.allow` allowlist consulted before handshake | +| Local impersonation via the native datagram API | API disabled by default; when enabled, `fips` group membership is the only gate and must be treated as key access | See [../design/fips-mesh-layer.md](../design/fips-mesh-layer.md) for the unauthenticated-attack-surface analysis (only handshake msg1 is @@ -251,3 +315,6 @@ way to restrict inbound traffic on `fips0`. See — operator activation and drop-in recipes - [configuration.md](configuration.md) — full `node.rekey.*`, `node.rate_limit.*` parameter tables +- [../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md) + — enabling the experimental native datagram API, and what group + membership grants once it is on diff --git a/docs/tutorials/README.md b/docs/tutorials/README.md index 84879826..328a04bf 100644 --- a/docs/tutorials/README.md +++ b/docs/tutorials/README.md @@ -38,13 +38,19 @@ policy, and an understanding of both deployment modes — overlay on top of existing IP, and ground-up where the mesh is the network. -There is also a side trip you can take any time after tutorial 1: +There are also two side trips you can take: - [ipv6-adapter-walkthrough.md](ipv6-adapter-walkthrough.md) — trace one `ssh` from DNS query through session setup to the far-side TUN, using `fipstop` and `fipsctl` to watch each step. Optional, but if you like seeing how the pieces fit together, - this is the doc that shows you. + this is the doc that shows you. Take it any time after tutorial 1. + +- [native-api-walkthrough.md](native-api-walkthrough.md) — write a + program against the experimental native datagram API, addressing a + peer by public key and port with no IPv6 emulation and no TUN. Runs + two throwaway nodes on one machine, so it needs no mesh and no root, + and you can take it without doing the tutorials first. ## Advanced diff --git a/docs/tutorials/native-api-walkthrough.md b/docs/tutorials/native-api-walkthrough.md new file mode 100644 index 00000000..8baa5364 --- /dev/null +++ b/docs/tutorials/native-api-walkthrough.md @@ -0,0 +1,343 @@ +# Native Datagram API Walkthrough + +A side trip. You will run two FIPS nodes on one machine, write a listening +program and a connecting program against the native datagram API, and watch one +datagram cross between them. Then you will look at the flow from the outside +with `fipsctl` while it is still open. + +Nothing here touches the public mesh, and nothing needs root. Both nodes run +with no TUN device and no DNS, peered directly over loopback UDP, so the whole +session lives in one scratch directory you delete at the end. + +**This is not part of the numbered progression.** Take it any time. It assumes +you can build the daemon from source and can read Rust; it does not assume you +have worked through the tutorials. + +**The API is experimental.** Names, fields and the command set may change +without a deprecation cycle. It is Linux and FreeBSD only. + +## What you will end up with + +- Two nodes, each with its own identity, control socket and API socket. +- A listening program that holds port 4600 and echoes one datagram per flow. +- A connecting program that opens a flow to the other node's public key and + gets its datagram back. +- A reading of `fipsctl show native-flows` taken while the flow is open. + +## Step 1: Build the daemon and its tools + +From a checkout of the FIPS source: + +```sh +cargo build --release --bins +``` + +That gives you `target/release/fips` and `target/release/fipsctl`. Put them on +your path for the rest of this walkthrough: + +```sh +export PATH="$PWD/target/release:$PATH" +``` + +## Step 2: Make two identities + +`keygen -s` prints a keypair to stdout and writes nothing: + +```sh +fipsctl keygen -s +``` + +```text +nsec1... +npub1... +``` + +Run it twice and keep both pairs. Call them A and B. You need each node's +`nsec` for its own config, and each node's `npub` for the *other* node's peer +entry. + +```sh +mkdir -p ~/napi-lab/a ~/napi-lab/b +cd ~/napi-lab +``` + +## Step 3: Write the two configs + +Node A, at `~/napi-lab/a/fips.yaml`. Substitute A's `nsec` and B's `npub`: + +```yaml +node: + identity: + nsec: "" + control: + socket_path: "/home/YOU/napi-lab/a/control.sock" + native_api: + enabled: true + socket_path: "/home/YOU/napi-lab/a/api.sock" + +tun: + enabled: false + +dns: + enabled: false + +transports: + udp: + bind_addr: "127.0.0.1:2121" + mtu: 1472 + +peers: + - npub: "" + alias: "node-b" + addresses: + - transport: udp + addr: "127.0.0.1:2122" +``` + +Node B, at `~/napi-lab/b/fips.yaml`, is the mirror image: B's `nsec`, A's +`npub`, its own sockets under `b/`, `bind_addr` on `2122`, and its peer address +pointing at `2121`. + +> **Use absolute paths.** The daemon does not resolve a socket path relative to +> the config file. Putting an `nsec` in a config is fine for a throwaway lab +> node like this one; for anything you keep, use +> [../how-to/persistent-identity.md](../how-to/persistent-identity.md) instead. + +Disabling TUN and DNS is what lets both nodes run as your own user. A node with +a TUN device needs `CAP_NET_ADMIN`, and this walkthrough does not need one: +the native API is the path that does not go through the IPv6 adapter. + +## Step 4: Start both nodes + +In two terminals: + +```sh +fips --config ~/napi-lab/a/fips.yaml +``` + +```sh +fips --config ~/napi-lab/b/fips.yaml +``` + +Each should log that it bound its API socket: + +```text +Native API socket listening on /home/YOU/napi-lab/a/api.sock +``` + +In a third terminal, confirm the two found each other: + +```sh +fipsctl -s ~/napi-lab/a/control.sock show peers +``` + +Wait for B to appear with a session. The link forms over loopback UDP and +usually takes a second or two. **Wait for it before going on**: a `connect` on +a flow contacts no peer, so it will succeed whether or not the link is up, and +the datagram would simply be held and then dropped. + +## Step 5: Write the listening program + +Make a crate next to the lab directory: + +```sh +cargo new --bin napi-listen +cd napi-listen +``` + +Point it at your FIPS checkout in `Cargo.toml`: + +```toml +[dependencies] +fips = { path = "/path/to/your/fips/checkout" } +``` + +`src/main.rs`: + +```rust +//! Hold a port and echo one datagram per flow. + +use fips::native::client::{FipsListener, FipsStream}; +use std::env; +use std::error::Error; +use std::path::Path; +use std::thread; + +/// Return one datagram to where it came from, then release the flow. +fn serve(flow: FipsStream) { + // Sized at the flow's own limit, so no datagram it can carry is + // truncated on the way in and echoed short. + let mut buf = vec![0u8; flow.max_payload()]; + match flow.recv(&mut buf) { + Ok(len) => { + let _ = flow.send(&buf[..len]); + println!("returned {len} bytes to {}", flow.peer_addr()); + } + Err(error) => eprintln!("receiving: {error}"), + } + // Returning drops the flow, which closes its descriptor. That is what + // releases the flow at the daemon; there is no close call to make. +} + +fn main() -> Result<(), Box> { + let socket = env::args().nth(1).ok_or("usage: napi-listen ")?; + let listener = FipsListener::bind_at(Path::new(&socket), 4600)?; + println!("holding {}", listener.local_addr()); + + for arrival in listener.incoming() { + // Detached rather than joined: the accept loop must not wait on one + // peer, and the thread owns everything it touches. + match arrival { + Ok(flow) => drop(thread::spawn(move || serve(flow))), + Err(error) => eprintln!("accepting: {error}"), + } + } + Ok(()) +} +``` + +Run it against node B: + +```sh +cargo run -- ~/napi-lab/b/api.sock +``` + +```text +holding npub1...:4600 +``` + +**Note what `serve` does not do.** It does not loop reading until the flow +closes. The v1 wire carries no half-close, so nothing peer-driven would ever +end that loop; it would hold a thread and a flow slot per peer until the +process died. One exchange per flow is the program's own decision, and making +it is mandatory. See +[../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md#four-things-that-will-bite-you). + +**Note also what `incoming()` does not do.** It never returns `None`, and a +failed accept arrives as an `Err` item rather than ending the iteration. Writing +`let flow = arrival?;` here would exit the loop on the first transient error, +which is a different shape from `TcpListener` habits. + +## Step 6: Write the connecting program + +```sh +cd .. +cargo new --bin napi-connect +cd napi-connect +``` + +Same dependency line. `src/main.rs`: + +```rust +//! Open a flow to a peer, exchange one datagram, and exit. + +use fips::native::client::FipsStream; +use std::env; +use std::error::Error; +use std::io; +use std::path::Path; +use std::time::Duration; + +/// How long to wait for the peer's answer before giving up on it. +const REPLY: Duration = Duration::from_secs(10); + +fn main() -> Result<(), Box> { + let mut args = env::args().skip(1); + let (Some(socket), Some(peer)) = (args.next(), args.next()) else { + return Err("usage: napi-connect ".into()); + }; + + // One setup call, and it contacts no peer: the daemon registers the flow + // locally and hands back the descriptor it rides on. Success here says + // nothing about the peer existing, being reachable, or listening. + let flow = FipsStream::connect_at(Path::new(&socket), 0, (peer, 4600))?; + println!("{} -> {}", flow.local_addr(), flow.peer_addr()); + + // Before the first recv and not after it, because the peer may never + // answer at all and the deadline is what makes that a failure rather + // than a hang. + flow.set_read_timeout(Some(REPLY))?; + + flow.send(b"hello")?; + + let mut buf = vec![0u8; flow.max_payload()]; + match flow.recv(&mut buf) { + Ok(len) => println!("{}", String::from_utf8_lossy(&buf[..len])), + Err(error) if error.kind() == io::ErrorKind::WouldBlock => { + return Err(format!("no answer from {} in {REPLY:?}", flow.peer_addr()).into()); + } + Err(error) => return Err(error.into()), + } + Ok(()) +} +``` + +Run it against node A, naming node B's npub: + +```sh +cargo run -- ~/napi-lab/a/api.sock +``` + +```text +npub1...:49152 -> npub1...:4600 +hello +``` + +The listener's terminal reports the other half: + +```text +returned 5 bytes to npub1...:49152 +``` + +That datagram went from your connecting program, into node A over a Unix +socket, across loopback UDP inside an encrypted FSP session, into node B, and +out to your listening program on another Unix socket. No IPv6 address and no +TUN device was involved anywhere in it. + +## Step 7: Watch a flow from the outside + +The exchange above is over in milliseconds. To look at a live flow, make the +connector hold one open: add a `std::thread::sleep(Duration::from_secs(60));` +before the final `Ok(())` and run it again. + +While it sleeps: + +```sh +fipsctl -s ~/napi-lab/b/control.sock show native-flows +``` + +You get every flow node B holds, with its ports, its queue depth and its age, +plus every bound listener and its backlog. The counters are in: + +```sh +fipsctl -s ~/napi-lab/b/control.sock stats metrics +``` + +under `native`, where the `drop_*` fields separate a datagram refused for +having no listening port from one dropped because a client was not reading fast +enough. Those counters are the only way to see a drop: **nothing on the API +surface reports one to your program.** + +## Step 8: Clean up + +Stop both daemons with Ctrl-C, then: + +```sh +rm -rf ~/napi-lab +``` + +The identities were only ever in those config files, so removing the directory +removes them. Nothing was written outside it and nothing was published to any +relay. + +## Where to go next + +- [../how-to/use-the-native-datagram-api.md](../how-to/use-the-native-datagram-api.md) + — the same ground as a recipe, including enabling the API on a real node and + the security posture that grants +- [../reference/native-api.md](../reference/native-api.md) + — every type and method, the errno table, the ceilings, and what happens to + data that disappears +- [../how-to/write-a-native-api-client.md](../how-to/write-a-native-api-client.md) + — doing all of this from C, Python or Go, where there is no client library + and the obligations become yours diff --git a/examples/native-echo.rs b/examples/native-echo.rs new file mode 100644 index 00000000..7b8ac1ab --- /dev/null +++ b/examples/native-echo.rs @@ -0,0 +1,126 @@ +//! An echo server on the native datagram API, and the reference client for it. +//! +//! Run it beside a daemon whose native API socket is the first argument; it +//! holds the port given as the second and returns every datagram sent to that +//! port to whoever sent it. +//! +//! ```text +//! native-echo /run/fips/api.sock 4600 +//! ``` +//! +//! **One exchange, one flow.** A served flow takes a single datagram, sends it +//! back and closes. A datagram API carries no far-end close, so a server that +//! kept reading would hold the flow until its own process went away and would +//! be waiting for a signal that never comes; the accept loop takes the next +//! arrival instead. +//! +//! **It imports `fips::native::client` and nothing else from the crate.** That +//! is the point of the example as much as the echoing is: needing `serde_json`, +//! `libc`, or any knowledge of the line protocol here would mean the client +//! module had failed to hide something, and the fix would belong there. +//! +//! **Platform.** The client module is built on Linux and FreeBSD only, because +//! macOS has no `AF_UNIX` `SOCK_SEQPACKET` and Windows no `SCM_RIGHTS`. `main` +//! is gated to match rather than the file being Linux-only by accident: +//! `cargo clippy --all-targets` and `cargo nextest run` both compile example +//! targets, and both run on macOS, so an ungated file would break those runs +//! there instead of this program refusing to start. + +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +mod echo { + use fips::native::client::{FipsListener, FipsStream}; + use std::env; + use std::io::{self, Write}; + use std::path::Path; + use std::process::ExitCode; + use std::thread; + + /// Hold the port named on the command line and echo what arrives on it. + /// + /// Returns only on failure: a healthy server has no reason to stop, and the + /// caller kills it. + pub fn run() -> ExitCode { + let mut args = env::args().skip(1); + let (Some(socket), Some(port)) = (args.next(), args.next()) else { + eprintln!("usage: native-echo "); + return ExitCode::FAILURE; + }; + let port: u16 = match port.parse() { + Ok(port) => port, + Err(error) => { + eprintln!("native-echo: {port:?} is not a port: {error}"); + return ExitCode::FAILURE; + } + }; + + // "cannot hold" rather than "holding": a caller waits for the success + // line below by substring, and a failure line containing it would + // satisfy that wait and hide the reason. + let listener = match FipsListener::bind_at(Path::new(&socket), port) { + Ok(listener) => listener, + Err(error) => { + eprintln!("native-echo: cannot hold port {port} on {socket}: {error}"); + return ExitCode::FAILURE; + } + }; + + // The port the daemon actually held, which is what a caller asking for + // an ephemeral one needs. A caller waits on this line before it sends + // anything, so it is flushed rather than left to the line buffer. + println!("native-echo: holding port {}", listener.local_addr().port()); + let _ = io::stdout().flush(); + + for arrival in listener.incoming() { + match arrival { + // Detached rather than joined: the accept loop must not wait on + // one peer, and the thread owns everything it touches. + Ok(flow) => drop(thread::spawn(move || serve(flow))), + Err(error) => { + eprintln!("native-echo: accepting on port {port}: {error}"); + return ExitCode::FAILURE; + } + } + } + // `incoming` has no end: a listener has no last flow. Reaching here at + // all would mean the iterator broke its contract. + ExitCode::FAILURE + } + + /// Return one datagram to where it came from, then release the flow. + /// + /// Dropping the flow closes its descriptor, which is what tells the daemon + /// the flow is finished; there is no close command to send. + fn serve(flow: FipsStream) { + // Sized at the flow's own limit, so no datagram the flow can carry is + // truncated on the way in and echoed short. + let mut buf = vec![0u8; flow.max_payload()]; + let len = match flow.recv(&mut buf) { + Ok(len) => len, + Err(error) => { + eprintln!("native-echo: receiving from {}: {error}", flow.peer_addr()); + return; + } + }; + match flow.send(&buf[..len]) { + Ok(()) => println!("native-echo: returned {len} bytes to {}", flow.peer_addr()), + Err(error) => eprintln!("native-echo: returning to {}: {error}", flow.peer_addr()), + } + let _ = io::stdout().flush(); + } +} + +/// Serve until killed. +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +fn main() -> std::process::ExitCode { + echo::run() +} + +/// Refuse cleanly where the native API client is not built. +#[cfg(not(any(target_os = "linux", target_os = "freebsd")))] +fn main() -> std::process::ExitCode { + eprintln!( + "native-echo needs the native datagram API client, which is built on \ + Linux and FreeBSD only" + ); + std::process::ExitCode::FAILURE +} diff --git a/examples/native-surface.rs b/examples/native-surface.rs new file mode 100644 index 00000000..b3db6ce0 --- /dev/null +++ b/examples/native-surface.rs @@ -0,0 +1,661 @@ +//! Every public item of the native datagram API client, asserted against a +//! real daemon. +//! +//! ```text +//! native-surface walk /run/fips/api.sock npub1… +//! ``` +//! +//! **This is an assertion harness, not a program shape to copy.** It opens +//! flows nobody answers, asks for deadlines only so it can watch them expire, +//! and reaches for `poll(2)` on a descriptor the surface hands out. A program +//! that wanted to do something useful with this API would look like +//! `native-echo`, which is the example to read first. +//! +//! **It reaches past `fips::native::client` for `libc`, and that is a +//! departure.** `native-echo` states as a design property that needing `libc` +//! would mean the client module had failed to hide something. Here the +//! descriptor is the thing under test: `AsRawFd` and `AsFd` exist so a caller +//! can put a flow or a listener in its own event loop, `poll(2)` has no `std` +//! spelling, and asserting that the number really is a pollable descriptor is +//! the whole point of those items. Every `libc` use is inside the platform-gated +//! module below, because `libc` is a `cfg(unix)` dependency and the Windows leg +//! of CI compiles examples. +//! +//! **Every assertion goes through [`surface::step`], and there is no other way +//! to record one.** The count in the terminal line is that recorder's counter +//! rather than a number written into the format string, so a caller comparing +//! it against what it expected catches a block that stopped running as well as +//! a block that failed. +//! +//! **Platform.** The client module is built on Linux and FreeBSD only, so +//! `main` is gated to match, for the reason `native-echo` gives: example +//! targets are compiled by `cargo clippy --all-targets` on every platform in +//! the build matrix, and an ungated file would break those runs rather than +//! this program refusing to start. + +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +mod surface { + use fips::native::client::{FipsAddr, FipsListener, FipsStream, SOCKET, XOnlyPublicKey}; + use std::env; + use std::fmt; + use std::io::{self, Write}; + use std::os::fd::{AsFd, AsRawFd, RawFd}; + use std::path::Path; + use std::process::ExitCode; + use std::str::FromStr; + use std::sync::Mutex; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::thread; + use std::time::{Duration, Instant}; + + /// How long the whole run may take before the watchdog calls it wedged. + /// + /// Several of the defects these assertions exist to catch fail by blocking + /// for ever rather than by returning something wrong, and a container that + /// never exits reaches no accounting in the driver that started it. + const WATCHDOG: Duration = Duration::from_secs(30); + + /// The read deadline the expiry assertion arms. + /// + /// Long enough that scheduling noise cannot make the wait look absent, short + /// enough that waiting it out twice costs nothing. + const DEADLINE: Duration = Duration::from_millis(200); + + /// The largest datagram a flow carries on the harness node. + /// + /// A literal, because the number is the assertion: it is what the daemon + /// computes from `testing/native-api/node.yaml`'s 1472-byte UDP MTU, and the + /// harness's line-protocol checks assert the same 1362 from the other side. + /// A walk run against a node configured differently is expected to fail + /// here, which is why the port band and the socket path are checked too. + const PAYLOAD: usize = 1362; + + /// The lowest port the daemon's ephemeral allocator ever hands out. + /// + /// A floor rather than a value: the allocator is a forward-only cursor + /// shared with every check that ran before this one, so the exact port + /// depends on run order and only the range is a property of the surface. + const EPHEMERAL: u16 = 49152; + + /// The port the walk asks a listener to hold by name. + /// + /// From 4800-4809, a band no other check in `testing/native-api/` uses. + const HELD: u16 = 4800; + + /// The local port the walk asks a flow to be opened from by name. + const NAMED: u16 = 4801; + + /// How many assertions have held so far. + static PASSED: AtomicUsize = AtomicUsize::new(0); + + /// The assertion currently running, so a wedge can say which one wedged. + static IN_FLIGHT: Mutex<&'static str> = Mutex::new("start-up"); + + /// Run one assertion, counting it when it holds and ending the run when not. + /// + /// The only way to record an assertion, which is what makes [`PASSED`] a + /// measurement of what ran rather than a number kept by hand. + /// + /// It exits rather than returning an error because the walk is a sequence: + /// a flow that could not be opened has nothing to assert about, and a run + /// that carried on would bury the first failure under the noise of every + /// assertion downstream of it. + pub fn step(name: &'static str, body: impl FnOnce() -> Result) -> T { + // The guard is dropped at the end of this statement rather than held + // across `body`, or the watchdog could not read the name it needs. + *IN_FLIGHT + .lock() + .unwrap_or_else(|poison| poison.into_inner()) = name; + match body() { + Ok(value) => { + PASSED.fetch_add(1, Ordering::SeqCst); + value + } + Err(detail) => { + eprintln!("native-surface: FAILED at {name}: {detail}"); + let _ = io::stderr().flush(); + std::process::exit(1); + } + } + } + + /// Arm a thread that ends the run, by name, if it stops making progress. + /// + /// A deadline that never reached the descriptor and a non-blocking mode that + /// was never set both fail as a `recv` that never returns. Without this the + /// container would run until something outside it lost patience, and the + /// evidence of which assertion was in flight would be gone. + fn watchdog() { + drop(thread::spawn(|| { + thread::sleep(WATCHDOG); + let name = *IN_FLIGHT + .lock() + .unwrap_or_else(|poison| poison.into_inner()); + eprintln!( + "native-surface: FAILED at {name}: watchdog after {}s", + WATCHDOG.as_secs() + ); + let _ = io::stderr().flush(); + std::process::exit(1); + })); + } + + /// Compare what a call reported against what the surface promises. + fn same(what: &str, got: T, want: T) -> Result<(), String> { + if got == want { + return Ok(()); + } + Err(format!("{what} is {got:?}, expected {want:?}")) + } + + /// Assert a call was refused with a particular errno. + fn errno(what: &str, got: io::Result, want: i32) -> Result<(), String> { + match got { + Ok(value) => Err(format!( + "{what} succeeded with {value:?}, expected errno {want}" + )), + Err(error) if error.raw_os_error() == Some(want) => Ok(()), + Err(error) => Err(format!("{what} failed with {error}, expected errno {want}")), + } + } + + /// Assert a call refused rather than waiting. + /// + /// By kind rather than by errno: `WouldBlock` is what the surface documents, + /// and it is the one answer both an expired deadline and a non-blocking + /// descriptor give, which is why the two are told apart here by how long the + /// call took rather than by what it returned. + fn blocked(what: &str, got: io::Result) -> Result<(), String> { + match got { + Ok(value) => Err(format!( + "{what} succeeded with {value:?}, expected it to refuse to wait" + )), + Err(error) if error.kind() == io::ErrorKind::WouldBlock => Ok(()), + Err(error) => Err(format!( + "{what} failed with {error}, expected it to refuse to wait" + )), + } + } + + /// Whether `poll(2)` says a descriptor has something to read, right now. + /// + /// The one place this file reaches past the client module, and the reason it + /// is allowed to: what `AsRawFd` and `AsFd` promise is that the number names + /// a socket an event loop can wait on, and nothing in `std` asks that + /// question of a bare descriptor. + fn readable(fd: RawFd) -> Result { + let mut waiting = libc::pollfd { + fd, + events: libc::POLLIN, + revents: 0, + }; + // SAFETY: the pointer and count describe one `pollfd` this frame owns, + // and the descriptor belongs to a stream or listener still alive here. + let rc = unsafe { libc::poll(std::ptr::from_mut(&mut waiting), 1, 0) }; + if rc < 0 { + return Err(format!( + "poll on descriptor {fd}: {}", + io::Error::last_os_error() + )); + } + // Without this the mechanism failing and the assertion holding are the + // same value: the only poll assertion here asserts a NEGATIVE, and a + // descriptor poll(2) rejects outright comes back rc=1 with POLLNVAL and + // no POLLIN, which would read as a quiet "nothing to read" pass. + let broken = waiting.revents & (libc::POLLNVAL | libc::POLLERR); + if broken != 0 { + return Err(format!( + "poll rejected descriptor {fd}, revents {broken:#x}, so it names no open socket" + )); + } + Ok(waiting.revents & libc::POLLIN != 0) + } + + /// Assert what a freshly opened flow says about the address it was given. + /// + /// Through the accessors and `Display` rather than by comparing the whole + /// address, because those are themselves items under test. + fn opened(flow: &FipsStream, key: XOnlyPublicKey, port: u16) -> Result<(), String> { + let want = FipsAddr::new(key, port); + same("the peer key", flow.peer_addr().key(), key)?; + same("the peer port", flow.peer_addr().port(), port)?; + same( + "the peer address written out", + flow.peer_addr().to_string(), + want.to_string(), + ) + } + + /// Assert the whole surface against the daemon on `sock`. + /// + /// `peer` is an npub nothing answers on, which is what most of these + /// assertions need: a native `connect` is a local registration, so a flow to + /// a peer that does not exist is a real flow with a real descriptor, and + /// nothing arriving on it is what makes a deadline observable. + fn walk(sock: &Path, peer: &str, key: XOnlyPublicKey) { + // ── Setup entry points ──────────────────────────────────────────── + let held = step("bind_at holds the port it was told to hold", || { + let listener = + FipsListener::bind_at(sock, HELD).map_err(|e| format!("bind_at({HELD}): {e}"))?; + same("the port held", listener.local_addr().port(), HELD)?; + Ok(listener) + }); + + let ephemeral = step( + "bind with no socket path resolves SOCKET and takes an ephemeral port", + || { + let listener = FipsListener::bind(0).map_err(|e| format!("bind(0): {e}"))?; + let port = listener.local_addr().port(); + if port < EPHEMERAL { + return Err(format!( + "the port held is {port}, expected one from {EPHEMERAL} up" + )); + } + Ok(listener) + }, + ); + + let flow = step( + "connect_at opens a flow to the peer and port it was given", + || { + let port = 4809; + let flow = FipsStream::connect_at(sock, 0, (key, port)) + .map_err(|e| format!("connect_at(_, 0, (key, {port})): {e}"))?; + opened(&flow, key, port)?; + let local = flow.local_addr().port(); + if local < EPHEMERAL { + return Err(format!( + "the flow's local port is {local}, expected one from {EPHEMERAL} up" + )); + } + Ok(flow) + }, + ); + + step( + "the node's key is the same on a listener and a flow, and is not the peer's", + || { + same( + "the node key a flow reports", + flow.local_addr().key(), + ephemeral.local_addr().key(), + )?; + if flow.peer_addr().key() == flow.local_addr().key() { + return Err("a flow's peer key and node key are the same value".to_string()); + } + Ok(()) + }, + ); + + // ── Addressing: one connect per ToFipsAddr impl ─────────────────── + // + // Eight impls, eight calls, and the mapping is the audit: `grep -n + // 'impl.*ToFipsAddr for' src/native/client/mod.rs` returns eight lines. + // A ninth would have to appear here and in the harness's expected count + // before either could go green again, which is the intended friction. + let addr = FipsAddr::new(key, 4807); + + step("connect takes an address by value", || { + let flow = FipsStream::connect(addr).map_err(|e| format!("connect(FipsAddr): {e}"))?; + // connect delegates to connect_at with a local port of 0, so a + // defect that passed some other port instead shows up here as a + // local port below the ephemeral floor. + let local = flow.local_addr().port(); + if local < EPHEMERAL { + return Err(format!( + "the flow's local port is {local}, expected one from {EPHEMERAL} up" + )); + } + opened(&flow, key, addr.port()) + }); + + step("connect takes a key and a port as a tuple", || { + let port = 4802; + let flow = FipsStream::connect((key, port)) + .map_err(|e| format!("connect((XOnlyPublicKey, {port})): {e}"))?; + opened(&flow, key, port) + }); + + step( + "connect takes a serialized key and a port as a tuple", + || { + let port = 4803; + let flow = FipsStream::connect((key.serialize(), port)) + .map_err(|e| format!("connect(([u8; 32], {port})): {e}"))?; + opened(&flow, key, port) + }, + ); + + step("connect takes an npub slice and a port as a tuple", || { + let port = 4804; + let flow = FipsStream::connect((peer, port)) + .map_err(|e| format!("connect((&str, {port})): {e}"))?; + opened(&flow, key, port) + }); + + step("connect takes an owned npub and a port as a tuple", || { + let port = 4805; + let flow = FipsStream::connect((peer.to_string(), port)) + .map_err(|e| format!("connect((String, {port})): {e}"))?; + opened(&flow, key, port) + }); + + step( + "connect takes a whole address as one slice, parsed by FromStr", + || { + // The only route to `impl ToFipsAddr for str`: a `str` is + // unsized, so the argument is a `&str` and the blanket impl for + // `&T` is what dispatches to it. + let text = format!("{peer}:4806"); + let want = + FipsAddr::from_str(&text).map_err(|e| format!("parsing {text:?}: {e}"))?; + let flow = FipsStream::connect(text.as_str()) + .map_err(|e| format!("connect({text:?} as &str): {e}"))?; + opened(&flow, want.key(), want.port()) + }, + ); + + step("connect takes a whole address as one owned string", || { + let text = format!("{peer}:4808"); + let want = FipsAddr::from_str(&text).map_err(|e| format!("parsing {text:?}: {e}"))?; + let flow = + FipsStream::connect(text).map_err(|e| format!("connect(a String address): {e}"))?; + opened(&flow, want.key(), want.port()) + }); + + // The borrow is the assertion, not an accident: `&addr` is the only + // thing that reaches the blanket `impl ToFipsAddr for &T`, and passing + // `addr` by value as clippy suggests would exercise `impl for FipsAddr` + // a second time and leave the blanket impl untested with this `T`. + #[allow(clippy::needless_borrows_for_generic_args)] + step( + "connect_from names the local port, taking the address by reference", + || { + // The blanket impl again, with a different `T`, which is what + // makes this a separate exercise rather than a repeat. + let flow = FipsStream::connect_from(NAMED, &addr) + .map_err(|e| format!("connect_from({NAMED}, &FipsAddr): {e}"))?; + same("the local port asked for", flow.local_addr().port(), NAMED)?; + opened(&flow, key, addr.port()) + }, + ); + + // ── Deadlines ───────────────────────────────────────────────────── + step( + "read_timeout reads back the deadline set_read_timeout set", + || { + flow.set_read_timeout(Some(DEADLINE)) + .map_err(|e| format!("set_read_timeout(Some({DEADLINE:?})): {e}"))?; + let got = flow + .read_timeout() + .map_err(|e| format!("read_timeout: {e}"))?; + same("the read deadline", got, Some(DEADLINE)) + }, + ); + + step("recv gives up once the read deadline expires", || { + let mut buf = [0u8; 64]; + let started = Instant::now(); + let outcome = flow.recv(&mut buf); + let waited = started.elapsed(); + blocked("recv on a flow no peer answers", outcome)?; + // Both bounds matter: too soon means the deadline never reached the + // descriptor and the answer came from somewhere else, and too late + // means it reached a different option than the one that was set. + if waited < DEADLINE - Duration::from_millis(50) { + return Err(format!( + "recv gave up after {waited:?}, too soon to have waited the {DEADLINE:?} deadline" + )); + } + if waited > Duration::from_secs(5) { + return Err(format!( + "recv waited {waited:?}, far past the {DEADLINE:?} deadline" + )); + } + Ok(()) + }); + + step("setting the read deadline to None clears it", || { + flow.set_read_timeout(None) + .map_err(|e| format!("set_read_timeout(None): {e}"))?; + let got = flow + .read_timeout() + .map_err(|e| format!("read_timeout: {e}"))?; + same("the read deadline", got, None) + }); + + step("set_read_timeout refuses a zero duration", || { + errno( + "set_read_timeout(Some(0))", + flow.set_read_timeout(Some(Duration::ZERO)), + libc::EINVAL, + ) + }); + + step( + "write_timeout reads back the deadline set_write_timeout set", + || { + flow.set_write_timeout(Some(DEADLINE)) + .map_err(|e| format!("set_write_timeout(Some({DEADLINE:?})): {e}"))?; + let got = flow + .write_timeout() + .map_err(|e| format!("write_timeout: {e}"))?; + same("the write deadline", got, Some(DEADLINE)) + }, + ); + + step("setting the write deadline to None clears it", || { + flow.set_write_timeout(None) + .map_err(|e| format!("set_write_timeout(None): {e}"))?; + let got = flow + .write_timeout() + .map_err(|e| format!("write_timeout: {e}"))?; + same("the write deadline", got, None) + }); + + step("set_write_timeout refuses a zero duration", || { + errno( + "set_write_timeout(Some(0))", + flow.set_write_timeout(Some(Duration::ZERO)), + libc::EINVAL, + ) + }); + + // ── Non-blocking ────────────────────────────────────────────────── + step("a non-blocking flow refuses to wait in recv", || { + flow.set_nonblocking(true) + .map_err(|e| format!("set_nonblocking(true) on a flow: {e}"))?; + let mut buf = [0u8; 64]; + let started = Instant::now(); + let outcome = flow.recv(&mut buf); + let waited = started.elapsed(); + blocked("recv on a non-blocking flow", outcome)?; + // The read deadline was cleared two assertions ago, so a + // set_nonblocking that did nothing would park here for ever and the + // watchdog would name this step. This bound therefore carries only + // the narrow shape where set_nonblocking armed a short deadline + // instead of the descriptor's mode. It is deliberately loose: it is + // still an order of magnitude under the cleared-deadline case, and + // tightening it buys no discrimination while inviting a flake when + // the runner is loaded. + if waited > Duration::from_millis(250) { + return Err(format!( + "recv on a non-blocking flow took {waited:?}, which is a wait rather than a refusal" + )); + } + Ok(()) + }); + + step("a non-blocking listener refuses to wait in accept", || { + held.set_nonblocking(true) + .map_err(|e| format!("set_nonblocking(true) on a listener: {e}"))?; + blocked("accept on a non-blocking listener", held.accept()) + }); + + step( + "a failed accept is an incoming item rather than the end of the iteration", + || match held.incoming().next() { + None => Err("incoming ended, and a listener has no last flow".to_string()), + Some(item) => blocked("the first incoming item", item), + }, + ); + + // ── Descriptors ─────────────────────────────────────────────────── + step( + "the flow's borrowed descriptor is the number its raw one gives", + || { + same( + "the flow's descriptor", + flow.as_fd().as_raw_fd(), + flow.as_raw_fd(), + ) + }, + ); + + step( + "the listener's borrowed descriptor is the number its raw one gives", + || { + same( + "the listener's descriptor", + held.as_fd().as_raw_fd(), + held.as_raw_fd(), + ) + }, + ); + + step("a flow and a listener hold different descriptors", || { + let (one, other) = (flow.as_raw_fd(), held.as_raw_fd()); + if one == other { + return Err(format!("both report descriptor {one}")); + } + Ok(()) + }); + + step( + "the flow's descriptor can be duplicated, so it names an open file", + || { + flow.as_fd() + .try_clone_to_owned() + .map(drop) + .map_err(|e| format!("duplicating the flow's descriptor: {e}")) + }, + ); + + step( + "the listener's descriptor can be duplicated, so it names an open file", + || { + held.as_fd() + .try_clone_to_owned() + .map(drop) + .map_err(|e| format!("duplicating the listener's descriptor: {e}")) + }, + ); + + step( + "poll reports neither the flow nor the listener readable while nothing has arrived", + || { + if readable(flow.as_raw_fd())? { + return Err("the flow is readable and no peer has sent anything".to_string()); + } + if readable(held.as_raw_fd())? { + return Err("the listener is readable and no flow has arrived".to_string()); + } + Ok(()) + }, + ); + + // ── Limits ──────────────────────────────────────────────────────── + step( + "max_payload is what the daemon computed for this transport", + || { + same( + "the largest datagram this flow carries", + flow.max_payload(), + PAYLOAD, + ) + }, + ); + + step( + "a datagram of exactly max_payload bytes is accepted", + || { + let datagram = vec![0x5a; flow.max_payload()]; + flow.send(&datagram) + .map_err(|e| format!("send of {} bytes: {e}", datagram.len())) + }, + ); + + step( + "a datagram one byte past max_payload is refused with EMSGSIZE", + || { + let datagram = vec![0x5a; flow.max_payload() + 1]; + errno( + "send of one byte past the limit", + flow.send(&datagram), + libc::EMSGSIZE, + ) + }, + ); + } + + /// Walk the surface, and print how many assertions held. + pub fn run() -> ExitCode { + let mut args = env::args().skip(1); + let (Some(mode), Some(sock), Some(peer)) = (args.next(), args.next(), args.next()) else { + eprintln!("usage: native-surface walk "); + return ExitCode::FAILURE; + }; + if mode != "walk" { + eprintln!("native-surface: {mode:?} is not a mode; the modes are: walk"); + return ExitCode::FAILURE; + } + // The walk is given a path because `connect_at` and `bind_at` are items + // in their own right and need one. The forms that take no path resolve + // SOCKET, which is the same daemon only when the caller mounted it + // there; a mismatch would fail those calls with ENOENT and read as a + // defect in the surface rather than in the invocation. + if sock != SOCKET { + eprintln!( + "native-surface: the no-path calls resolve {SOCKET}, so the walk has to be given \ + that path, not {sock}" + ); + return ExitCode::FAILURE; + } + let node = match FipsAddr::from_str(&format!("{peer}:0")) { + Ok(node) => node, + Err(error) => { + eprintln!("native-surface: {peer:?} is not an npub: {error}"); + return ExitCode::FAILURE; + } + }; + let key: XOnlyPublicKey = node.key(); + + watchdog(); + walk(Path::new(&sock), &peer, key); + + // The count is the recorder's, not a literal: a block that stopped + // running still reaches this line, and only the number betrays it. + println!( + "native-surface: walk complete, {} assertions passed", + PASSED.load(Ordering::SeqCst) + ); + let _ = io::stdout().flush(); + ExitCode::SUCCESS + } +} + +/// Walk the surface once and report. +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +fn main() -> std::process::ExitCode { + surface::run() +} + +/// Refuse cleanly where the native API client is not built. +#[cfg(not(any(target_os = "linux", target_os = "freebsd")))] +fn main() -> std::process::ExitCode { + eprintln!( + "native-surface needs the native datagram API client, which is built on \ + Linux and FreeBSD only" + ); + std::process::ExitCode::FAILURE +} diff --git a/src/bin/fipsctl.rs b/src/bin/fipsctl.rs index 625b6a57..737dedee 100644 --- a/src/bin/fipsctl.rs +++ b/src/bin/fipsctl.rs @@ -177,6 +177,8 @@ enum ShowCommands { Routing, /// Identity cache entries (known node pubkeys) IdentityCache, + /// Native datagram API flows and listeners + NativeFlows, } #[derive(Subcommand, Debug)] @@ -200,6 +202,7 @@ impl ShowCommands { ShowCommands::Transports => "show_transports", ShowCommands::Routing => "show_routing", ShowCommands::IdentityCache => "show_identity_cache", + ShowCommands::NativeFlows => "show_native_flows", } } } @@ -1481,6 +1484,16 @@ mod tests { assert_eq!(AclCommands::Show.command_name(), "show_acl"); } + #[test] + fn test_cli_parses_show_native_flows_to_its_control_command() { + let cli = Cli::try_parse_from(["fipsctl", "show", "native-flows"]).unwrap(); + + let Commands::Show { what } = cli.command else { + panic!("expected a show subcommand"); + }; + assert_eq!(what.command_name(), "show_native_flows"); + } + #[test] fn test_cli_parses_acl_show() { let cli = Cli::try_parse_from(["fipsctl", "acl", "show"]).unwrap(); diff --git a/src/config/mod.rs b/src/config/mod.rs index 7c172abd..67491aa6 100644 --- a/src/config/mod.rs +++ b/src/config/mod.rs @@ -38,8 +38,8 @@ use zeroize::{Zeroize, Zeroizing}; pub use gateway::{ConntrackConfig, GatewayConfig, GatewayDnsConfig, PortForward, Proto}; pub use node::{ BloomConfig, BuffersConfig, CacheConfig, ControlConfig, LimitsConfig, LookupConfig, MmpConfig, - NodeConfig, NostrRendezvousConfig, NostrRendezvousPolicy, RateLimitConfig, RekeyConfig, - RendezvousConfig, RetryConfig, SessionConfig, SessionMmpConfig, TreeConfig, + NativeApiConfig, NodeConfig, NostrRendezvousConfig, NostrRendezvousPolicy, RateLimitConfig, + RekeyConfig, RendezvousConfig, RetryConfig, SessionConfig, SessionMmpConfig, TreeConfig, }; pub use peer::{ConnectPolicy, PeerAddress, PeerConfig, TransportSpec}; pub use transport::{ @@ -1099,6 +1099,37 @@ impl Config { } } + let native = &self.node.native_api; + // Both floors refuse a node that would start, answer every setup call + // and then drop every datagram a peer sent. A zero `backlog` makes the + // registry refuse every arrival; a zero `pending_per_flow` makes it + // announce the arrival and then refuse the datagram that caused it, so + // a peer's opening message is lost with no refusal anywhere. + if native.backlog < 1 { + return Err(ConfigError::Validation( + "node.native_api.backlog is 0 but must be at least 1: a listener with no \ + backlog admits no flow, so every arrival would be dropped" + .to_string(), + )); + } + if native.pending_per_flow < 1 { + return Err(ConfigError::Validation( + "node.native_api.pending_per_flow is 0 but must be at least 1: a flow that \ + can hold nothing loses its peer's opening datagram between the arrival \ + being announced and the client taking the flow" + .to_string(), + )); + } + if native.pending_per_flow > NativeApiConfig::MAX_PENDING_PER_FLOW { + return Err(ConfigError::Validation(format!( + "node.native_api.pending_per_flow is {} but must not exceed {}: the whole batch \ + is written onto a socket pair the client cannot read yet, and a larger one \ + would not fit the send buffer", + native.pending_per_flow, + NativeApiConfig::MAX_PENDING_PER_FLOW + ))); + } + // Reject loopback UDP bind combined with non-loopback peer addresses. // Linux pins the source IP to a loopback-bound socket, so packets // sent from such a socket to external peers are dropped at the @@ -2602,6 +2633,68 @@ node: .expect("positive established-bucket values must validate"); } + #[test] + fn test_validate_pending_per_flow_above_its_ceiling_rejected() { + let mut config = Config::default(); + config.node.native_api.pending_per_flow = NativeApiConfig::MAX_PENDING_PER_FLOW + 1; + + let err = config.validate().expect_err("validation should fail"); + let msg = err.to_string(); + assert!(msg.contains("pending_per_flow"), "got: {msg}"); + } + + #[test] + fn test_validate_pending_per_flow_at_its_ceiling_accepted() { + // The boundary is accepted, or the check would be refusing a value the + // send buffer takes and the ceiling would be a different number than + // the one it is documented as. + let mut config = Config::default(); + config.node.native_api.pending_per_flow = NativeApiConfig::MAX_PENDING_PER_FLOW; + + config + .validate() + .expect("the documented ceiling itself must validate"); + } + + #[test] + fn test_validate_native_api_backlog_zero_rejected() { + // Zero would start a node whose listeners admit no flow at all: the + // registry compares the pending depth against this number before it + // announces anything, so every arrival is dropped. + let mut config = Config::default(); + config.node.native_api.backlog = 0; + + let err = config.validate().expect_err("validation should fail"); + let msg = err.to_string(); + assert!(msg.contains("node.native_api.backlog"), "got: {msg}"); + } + + #[test] + fn test_validate_native_api_pending_per_flow_zero_rejected() { + // Zero is the worse of the two: the arrival is announced and the + // datagram that caused it is then refused, so a peer's opening message + // is lost with no refusal a client or an operator can see. + let mut config = Config::default(); + config.node.native_api.pending_per_flow = 0; + + let err = config.validate().expect_err("validation should fail"); + let msg = err.to_string(); + assert!(msg.contains("pending_per_flow"), "got: {msg}"); + } + + #[test] + fn test_validate_native_api_floors_accept_one() { + // The boundary itself validates, or the floors would be refusing a + // working node rather than a broken one. + let mut config = Config::default(); + config.node.native_api.backlog = 1; + config.node.native_api.pending_per_flow = 1; + + config + .validate() + .expect("a depth of one is a working node, not a refused one"); + } + #[test] fn test_validate_rekey_after_messages_zero_rejected() { let mut config = Config::default(); diff --git a/src/config/node.rs b/src/config/node.rs index 2847e217..fbc6b67a 100644 --- a/src/config/node.rs +++ b/src/config/node.rs @@ -934,6 +934,142 @@ impl ControlConfig { } } +/// Native datagram API socket (`node.native_api.*`). +/// +/// **Experimental, and built on Linux and FreeBSD only.** The API hands a +/// client a file descriptor over `SCM_RIGHTS`, which Windows has no equivalent +/// of, and does it over an `AF_UNIX` `SOCK_SEQPACKET` socket, which macOS does +/// not implement. No listener is built on either, and this section is ignored +/// there. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +pub struct NativeApiConfig { + /// Enable the native API socket (`node.native_api.enabled`). + /// + /// Disabled by default. Any process that can open the socket can send as + /// this node's identity, and can receive mesh traffic on a port it chooses, + /// so enabling it is an explicit operator decision rather than a default. + #[serde(default = "NativeApiConfig::default_enabled")] + pub enabled: bool, + + /// Unix socket path (`node.native_api.socket_path`). + #[serde(default = "NativeApiConfig::default_socket_path")] + pub socket_path: String, + + /// Datagrams held for one flow (`node.native_api.pending_per_flow`). + /// + /// Applies while a flow waits to be accepted and while an established + /// flow's client is slow to read. Mirrors + /// [`SessionConfig::pending_packets_per_dest`], which bounds the same shape + /// of problem on the session layer. + /// + /// Bounded above by [`NativeApiConfig::MAX_PENDING_PER_FLOW`] at config + /// load. The whole batch is written onto a socket pair no process can read + /// yet, so a value large enough to exceed the send buffer would leave the + /// listener's task with a write it cannot complete. + /// + /// Bounded below by 1 at the same place. Zero announces an arrival and then + /// refuses the datagram that caused it, losing a peer's opening message + /// with no refusal a client or an operator can see. + #[serde(default = "NativeApiConfig::default_pending_per_flow")] + pub pending_per_flow: usize, + + /// Flows awaiting accept on one listener (`node.native_api.backlog`). + /// + /// A client that announces interest and never answers cannot make the node + /// hold more than this, whatever a peer does. + /// + /// Bounded below by 1 at config load. Zero would admit no flow at all: the + /// registry compares a listener's pending depth against this before it + /// announces anything, so every arrival would be dropped. + #[serde(default = "NativeApiConfig::default_backlog")] + pub backlog: usize, + + /// Flows this node holds at once (`node.native_api.max_flows`). + #[serde(default = "NativeApiConfig::default_max_flows")] + pub max_flows: usize, + + /// Answer the debug commands (`node.native_api.debug_commands`). + /// + /// **Off by default, and not a supported interface.** The three commands + /// it admits (`inject`, `stats`, `arrive`) exist so the test harness can + /// drive the receive and dispatch paths without a wire. `inject` makes + /// the daemon write bytes the client chose into one of that client's own + /// flows, and `arrive` makes it dispatch a datagram as though a peer had + /// sent it, which reaches any listener this node holds. None of the three + /// belongs in a packaged node, so this key is what the test harness turns + /// on and nothing else does. + #[serde(default = "NativeApiConfig::default_debug_commands")] + pub debug_commands: bool, +} + +impl Default for NativeApiConfig { + fn default() -> Self { + Self { + enabled: Self::default_enabled(), + socket_path: Self::default_socket_path(), + pending_per_flow: Self::default_pending_per_flow(), + backlog: Self::default_backlog(), + max_flows: Self::default_max_flows(), + debug_commands: Self::default_debug_commands(), + } + } +} + +impl NativeApiConfig { + /// Largest `pending_per_flow` a node will start with. + /// + /// The held batch is at most this many datagrams of at most `max_payload` + /// bytes each, written without waiting onto a socket pair whose other half + /// is still on its way to the client. At 64 and a 1362-byte payload that is + /// about 87 KB, which an ordinary `AF_UNIX` send buffer takes. The bound is + /// checked at config load so a value that would wedge a listener's task is + /// refused at startup rather than at the first arrival. + pub const MAX_PENDING_PER_FLOW: usize = 64; + + fn default_enabled() -> bool { + false + } + + fn default_pending_per_flow() -> usize { + 16 + } + + fn default_backlog() -> usize { + 16 + } + + fn default_max_flows() -> usize { + 256 + } + + fn default_debug_commands() -> bool { + false + } + + /// Default native API socket path, resolved beside the control socket. + /// + /// On Windows the path is empty: the API is not built there, so no value + /// would be meaningful. + fn default_socket_path() -> String { + #[cfg(unix)] + { + super::resolve_default_socket("api.sock") + } + #[cfg(windows)] + { + String::new() + } + } + + /// Whether this section carries nothing but its defaults. + /// + /// Drives `skip_serializing_if` so a config file that never named the + /// section does not gain one when the config is serialized back out. + fn is_default(&self) -> bool { + *self == Self::default() + } +} + /// Internal buffers (`node.buffers.*`). #[derive(Debug, Clone, Serialize, Deserialize)] pub struct BuffersConfig { @@ -1157,6 +1293,11 @@ pub struct NodeConfig { #[serde(default)] pub control: ControlConfig, + /// Native datagram API (`node.native_api.*`). Experimental; the listener + /// is built on Linux and FreeBSD only. + #[serde(default, skip_serializing_if = "NativeApiConfig::is_default")] + pub native_api: NativeApiConfig, + /// Metrics Measurement Protocol — link layer (`node.mmp.*`). #[serde(default)] pub mmp: MmpConfig, @@ -1201,6 +1342,7 @@ impl Default for NodeConfig { session: SessionConfig::default(), buffers: BuffersConfig::default(), control: ControlConfig::default(), + native_api: NativeApiConfig::default(), mmp: MmpConfig::default(), session_mmp: SessionMmpConfig::default(), ecn: EcnConfig::default(), @@ -1404,6 +1546,23 @@ owd_window_size: 48 } } + #[test] + fn test_native_api_is_off_and_undebuggable_by_default() { + // Both gates default closed, and neither has any other guard in the + // library: the harness case that leaves `debug_commands` out of its + // YAML proves the serde path, not the value it lands on. + let config = NativeApiConfig::default(); + assert!(!config.enabled); + assert!(!config.debug_commands); + + // Enabling the API must not drag the debug commands in with it, which + // is the shape a real operator config takes. + let yaml = "enabled: true\n"; + let parsed: NativeApiConfig = serde_yaml::from_str(yaml).unwrap(); + assert!(parsed.enabled); + assert!(!parsed.debug_commands); + } + #[cfg(windows)] #[test] fn test_default_socket_path_windows() { diff --git a/src/control/mod.rs b/src/control/mod.rs index 57f748b8..d896f2ac 100644 --- a/src/control/mod.rs +++ b/src/control/mod.rs @@ -132,61 +132,6 @@ mod unix_impl { use std::path::{Path, PathBuf}; use tokio::net::UnixListener; - /// Ensure the socket's parent exists and report whether this call created - /// the leaf directory. - /// - /// `create_dir` gives us an atomic ownership decision: an `AlreadyExists` - /// result means another actor owns the existing directory, while success - /// means it is safe for this bind to apply FIPS ownership and mode. Missing - /// ancestors are created recursively, but only the requested leaf is later - /// treated as the socket's private directory. - fn ensure_socket_parent(parent: &Path) -> Result { - if parent.as_os_str().is_empty() { - return Ok(false); - } - - match std::fs::create_dir(parent) { - Ok(()) => Ok(true), - Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => { - if parent.is_dir() { - Ok(false) - } else { - Err(error) - } - } - Err(error) if error.kind() == std::io::ErrorKind::NotFound => { - let ancestor = parent.parent().ok_or(error)?; - ensure_socket_parent(ancestor)?; - ensure_socket_parent(parent) - } - Err(error) => Err(error), - } - } - - /// Apply access policy to a newly bound control socket. - /// - /// The socket is always group-owned. `managed_parent` is either a private - /// directory this bind created or a canonical FIPS runtime directory. A - /// shared or operator-owned existing parent is omitted so it retains its - /// ownership and mode. - fn set_control_socket_access( - socket_path: &Path, - managed_parent: Option<&Path>, - mut chown_to_fips_group: impl FnMut(&Path), - ) -> Result<(), std::io::Error> { - use std::os::unix::fs::PermissionsExt; - - std::fs::set_permissions(socket_path, std::fs::Permissions::from_mode(0o770))?; - chown_to_fips_group(socket_path); - - if let Some(parent) = managed_parent { - std::fs::set_permissions(parent, std::fs::Permissions::from_mode(0o750))?; - chown_to_fips_group(parent); - } - - Ok(()) - } - /// Control socket listener (Unix domain socket). /// /// Manages the Unix domain socket lifecycle: bind, accept, cleanup. @@ -202,36 +147,7 @@ mod unix_impl { /// and binds the Unix listener. pub fn bind(config: &ControlConfig) -> Result { let socket_path = PathBuf::from(&config.socket_path); - - // Creation is useful for diagnostics, but ownership is keyed to - // directory identity as well: systemd pre-creates /run/fips on - // every Linux service start and initially owns it as root:root. - let managed_parent = match socket_path.parent() { - Some(parent) => { - let created = ensure_socket_parent(parent)?; - if created { - debug!(path = %parent.display(), "Created private control socket directory"); - } - (created || crate::config::is_managed_socket_parent(parent)) - .then(|| parent.to_owned()) - } - None => None, - }; - - // Remove stale socket if it exists - if socket_path.exists() { - Self::remove_stale_socket(&socket_path)?; - } - - let listener = UnixListener::bind(&socket_path)?; - - // Make the socket and its managed private directory group-accessible - // so fips group members can use fipsctl/fipstop. - set_control_socket_access( - &socket_path, - managed_parent.as_deref(), - Self::chown_to_fips_group, - )?; + let listener = crate::utils::sockbind::bind(&socket_path, "control")?; info!(path = %socket_path.display(), "Control socket listening"); @@ -241,60 +157,6 @@ mod unix_impl { }) } - /// Remove a stale socket file. - /// - /// If the file exists but no one is listening, remove it so we can - /// bind. This handles unclean daemon exits. - fn remove_stale_socket(path: &Path) -> Result<(), std::io::Error> { - // Try connecting to see if someone is listening - match std::os::unix::net::UnixStream::connect(path) { - Ok(_) => { - // Someone is listening — don't remove it - Err(std::io::Error::new( - std::io::ErrorKind::AddrInUse, - format!("control socket already in use: {}", path.display()), - )) - } - Err(_) => { - // No one listening — remove the stale socket - debug!(path = %path.display(), "Removing stale control socket"); - std::fs::remove_file(path)?; - Ok(()) - } - } - } - - /// Set group ownership of a path to the 'fips' group (best-effort). - fn chown_to_fips_group(path: &Path) { - use std::ffi::CString; - use std::os::unix::ffi::OsStrExt; - - // Look up the 'fips' group - let group_name = CString::new("fips").unwrap(); - let grp = unsafe { libc::getgrnam(group_name.as_ptr()) }; - if grp.is_null() { - debug!( - "'fips' group not found, skipping chown for {}", - path.display() - ); - return; - } - let gid = unsafe { (*grp).gr_gid }; - - let c_path = match CString::new(path.as_os_str().as_bytes()) { - Ok(p) => p, - Err(_) => return, - }; - let ret = unsafe { libc::chown(c_path.as_ptr(), u32::MAX, gid) }; - if ret != 0 { - warn!( - path = %path.display(), - error = %std::io::Error::last_os_error(), - "Failed to chown control socket to 'fips' group" - ); - } - } - /// Run the accept loop, forwarding requests to the main event loop via mpsc. /// /// Each accepted connection is handled in a spawned task: @@ -334,17 +196,7 @@ mod unix_impl { /// Clean up the socket file. fn cleanup(&self) { - if self.socket_path.exists() { - if let Err(e) = std::fs::remove_file(&self.socket_path) { - warn!( - path = %self.socket_path.display(), - error = %e, - "Failed to remove control socket" - ); - } else { - debug!(path = %self.socket_path.display(), "Control socket removed"); - } - } + crate::utils::sockbind::cleanup(&self.socket_path, "control"); } } @@ -353,92 +205,6 @@ mod unix_impl { self.cleanup(); } } - - #[cfg(test)] - mod tests { - use super::{ensure_socket_parent, set_control_socket_access}; - use std::os::unix::fs::PermissionsExt; - - #[test] - fn parent_setup_distinguishes_existing_and_created_directories() { - let temp = tempfile::tempdir().unwrap(); - let existing = temp.path().join("existing"); - std::fs::create_dir(&existing).unwrap(); - assert!(!ensure_socket_parent(&existing).unwrap()); - - let nested = temp.path().join("missing").join("fips"); - assert!(ensure_socket_parent(&nested).unwrap()); - assert!(nested.is_dir()); - assert!(!ensure_socket_parent(&nested).unwrap()); - } - - #[test] - fn access_setup_leaves_an_existing_shared_parent_unchanged() { - let temp = tempfile::tempdir().unwrap(); - let parent = temp.path().join("shared"); - std::fs::create_dir(&parent).unwrap(); - std::fs::set_permissions(&parent, std::fs::Permissions::from_mode(0o711)).unwrap(); - let socket = parent.join("control.sock"); - std::fs::File::create(&socket).unwrap(); - - let mut chowned = Vec::new(); - set_control_socket_access(&socket, None, |path| chowned.push(path.to_path_buf())) - .unwrap(); - - assert_eq!(chowned, vec![socket.clone()]); - assert_eq!( - std::fs::metadata(&parent).unwrap().permissions().mode() & 0o777, - 0o711 - ); - assert_eq!( - std::fs::metadata(&socket).unwrap().permissions().mode() & 0o777, - 0o770 - ); - } - - #[test] - fn access_setup_secures_a_new_private_parent() { - let temp = tempfile::tempdir().unwrap(); - let parent = temp.path().join("fips"); - std::fs::create_dir(&parent).unwrap(); - let socket = parent.join("control.sock"); - std::fs::File::create(&socket).unwrap(); - - let mut chowned = Vec::new(); - set_control_socket_access(&socket, Some(&parent), |path| { - chowned.push(path.to_path_buf()) - }) - .unwrap(); - - assert_eq!(chowned, vec![socket, parent.clone()]); - assert_eq!( - std::fs::metadata(&parent).unwrap().permissions().mode() & 0o777, - 0o750 - ); - } - - #[test] - fn access_setup_secures_an_existing_managed_parent() { - let temp = tempfile::tempdir().unwrap(); - let parent = temp.path().join("managed"); - std::fs::create_dir(&parent).unwrap(); - std::fs::set_permissions(&parent, std::fs::Permissions::from_mode(0o700)).unwrap(); - let socket = parent.join("control.sock"); - std::fs::File::create(&socket).unwrap(); - - let mut chowned = Vec::new(); - set_control_socket_access(&socket, Some(&parent), |path| { - chowned.push(path.to_path_buf()) - }) - .unwrap(); - - assert_eq!(chowned, vec![socket, parent.clone()]); - assert_eq!( - std::fs::metadata(&parent).unwrap().permissions().mode() & 0o777, - 0o750 - ); - } - } } // ============================================================================ diff --git a/src/control/queries.rs b/src/control/queries.rs index 2e84b3e3..83fe847a 100644 --- a/src/control/queries.rs +++ b/src/control/queries.rs @@ -1621,6 +1621,124 @@ pub(crate) fn show_identity_cache_from_handle( }) } +/// How `show_native_flows` names a flow's lifecycle state. +/// +/// Two states and not three: the registry either holds a flow a client has +/// taken, or one it announced and is still waiting to be answered about. A +/// rejected or expired flow is gone from the registry entirely and has nothing +/// to report. +fn native_flow_state(established: bool) -> &'static str { + if established { + "established" + } else { + "pending_accept" + } +} + +/// `show_native_flows` — Native datagram API flows and listeners. +/// +/// Two representations of one peer, on purpose. `peer` is the npub, which is +/// the address a client names and the only form the native API itself reports. +/// `peer_addr` is the 16-byte node address, kept because this is an operator +/// surface and it is what `show_sessions` and `show_routing` key on. Both are +/// always present: the flow carries its peer's key, so nothing is resolved +/// here and nothing can be missing. +pub fn show_native_flows(node: &Node) -> Value { + let now = now_ms(); + + let flows: Vec = node + .native() + .flows() + .into_iter() + .map(|view| { + json!({ + "flow_id": view.flow, + "peer": crate::identity::encode_npub(&view.pubkey), + "peer_addr": hex::encode(view.key.peer.as_bytes()), + "local_port": view.key.local, + "remote_port": view.key.remote, + "state": native_flow_state(view.established), + "queued": view.queued, + "age_ms": now.saturating_sub(view.at), + }) + }) + .collect(); + + let listeners: Vec = node + .native() + .listeners() + .into_iter() + .map(|view| { + json!({ + "local_port": view.port, + "backlog": view.backlog, + }) + }) + .collect(); + + let native_stats = node.metrics().native.snapshot(); + + json!({ + "flows": flows, + "listeners": listeners, + "stats": serde_json::to_value(&native_stats).unwrap_or_default(), + }) +} + +/// Off-loop variant of [`show_native_flows`]: renders from the tick-published +/// [`NativeSnapshot`](super::snapshot::NativeSnapshot) plus the `native` +/// counter family from the `MetricsRegistry`. +/// `age_ms` is derived at render time from the captured `since_ms`, exactly as +/// [`show_native_flows`] computed it. Output is byte-identical to +/// [`show_native_flows`]. +/// +/// A flow's `queued` depth is as of the last publish rather than as of the +/// read, which is the same point-in-time property every other snapshot cell +/// has; the underlying channel is drained by the client's own task and has no +/// value a control task could read anyway. +pub(crate) fn show_native_flows_from_handle( + handle: &super::read_handle::ControlReadHandle, +) -> Value { + let native = handle.native(); + let now = now_ms(); + + let flows: Vec = native + .flows + .iter() + .map(|row| { + json!({ + "flow_id": row.flow, + "peer": crate::identity::encode_npub(&row.peer_key), + "peer_addr": hex::encode(row.peer.as_bytes()), + "local_port": row.local_port, + "remote_port": row.remote_port, + "state": native_flow_state(row.established), + "queued": row.queued, + "age_ms": now.saturating_sub(row.since_ms), + }) + }) + .collect(); + + let listeners: Vec = native + .listeners + .iter() + .map(|row| { + json!({ + "local_port": row.local_port, + "backlog": row.backlog, + }) + }) + .collect(); + + let native_stats = handle.metrics().native.snapshot(); + + json!({ + "flows": flows, + "listeners": listeners, + "stats": serde_json::to_value(&native_stats).unwrap_or_default(), + }) +} + /// `show_stats_list` — Enumerate available history metrics and their units. pub fn show_stats_list() -> Value { let metrics: Vec = ALL_METRICS @@ -2340,6 +2458,7 @@ pub(crate) fn show_metrics_from_handle(handle: &super::read_handle::ControlReadH "bloom": m.bloom.snapshot(), "congestion": m.congestion.snapshot(), "errors": m.errors.snapshot(), + "native": m.native.snapshot(), }) } @@ -2565,7 +2684,7 @@ mod tests { } } - // ---- 18 handler snapshot tests -------------------------------------- + // ---- 19 handler snapshot tests -------------------------------------- #[test] fn snapshot_show_status() { @@ -2645,6 +2764,12 @@ mod tests { assert_snapshot("show_identity_cache", &render(show_identity_cache(&node))); } + #[test] + fn snapshot_show_native_flows() { + let node = build_test_node(); + assert_snapshot("show_native_flows", &render(show_native_flows(&node))); + } + #[test] fn snapshot_show_stats_list() { // Static — no Node needed. @@ -2756,6 +2881,7 @@ mod tests { ("show_connections", None), ("show_transports", None), ("show_mmp", None), + ("show_native_flows", None), ]; for (cmd, params) in read_queries { let req = Request { @@ -2926,6 +3052,7 @@ mod tests { ("bloom", "accepted"), ("congestion", "ce_forwarded"), ("errors", "coords_required"), + ("native", "flows_opened"), ]; assert_eq!( obj.len(), @@ -3292,6 +3419,167 @@ mod tests { ); } + // ---- native datagram API coverage ------------------------------------ + + /// Freshness + fidelity: after a `record_stats_history()` tick (the native + /// publisher site) the off-loop `show_native_flows` render equals its + /// on-loop oracle byte-for-byte, and the query is served off-loop. + #[test] + fn native_snapshot_matches_on_loop_after_tick() { + use super::super::protocol::Request; + use super::super::read_handle::snapshot_dispatch; + + let mut node = build_test_node(); + + // Put something in the registry first. Asserting the seed cell is empty + // against an empty registry would observe the absence of the very thing + // it checks and could not fail; with a listener already bound, an empty + // cell says the handle reads a published snapshot rather than the live + // registry. + let (arrivals, _arrivals_rx) = tokio::sync::mpsc::channel(8); + node.native_registry_for_test() + .listen(Some(4242), arrivals) + .expect("port 4242 is free on a fresh node"); + + let handle = node.control_read_handle(); + assert!( + handle.native().flows.is_empty() && handle.native().listeners.is_empty(), + "the seed native snapshot is empty until the first tick publishes" + ); + + // Advance one tick (the publisher site). + node.record_stats_history(); + let handle = node.control_read_handle(); + assert_eq!( + handle.native().listeners.len(), + 1, + "the tick publishes the listener the registry already held" + ); + + let req = Request { + command: "show_native_flows".to_string(), + params: None, + }; + let resp = + snapshot_dispatch(&req, &handle).expect("show_native_flows must be served off-loop"); + assert_eq!( + resp.status, "ok", + "show_native_flows off-loop response not ok" + ); + + assert_eq!( + render(show_native_flows(&node)), + render(show_native_flows_from_handle(&handle)), + "off-loop show_native_flows must match on-loop output" + ); + } + + /// The publisher carries every field the oracle emits, for flows it can only + /// get wrong once there are some to get wrong. The registry is populated + /// directly (the rx_loop's own handler is what does this in the daemon) and + /// then published: an empty-registry parity test passes just as happily + /// against a publisher that drops every per-flow field. + /// + /// Both flow kinds appear, because they are rendered by different arms: a + /// flow the client opened, and one pending accept whose peer the node has + /// never had in any cache. + #[test] + fn native_snapshot_carries_a_populated_registry_faithfully() { + use crate::native::registry::Delivery; + + let mut node = build_test_node(); + + // A flow the client opened, naming its peer by key. + let known = Identity::from_secret_bytes(&[0x11; 32]).expect("valid secret key"); + let known_addr = *known.node_addr(); + + // And one a listener accepted, whose peer the node has never had in + // any cache: the key still reaches the report, because the flow + // carries it rather than the report resolving it. + let stranger = Identity::from_secret_bytes(&[0x5C; 32]).expect("valid secret key"); + let unknown_addr = *stranger.node_addr(); + + let (sink, _sink_rx) = tokio::sync::mpsc::channel(8); + node.native_registry_for_test() + .connect(known.pubkey(), 5000, Some(6000), sink, 1_000) + .expect("a fresh node holds no flows"); + let (arrivals, _arrivals_rx) = tokio::sync::mpsc::channel(8); + node.native_registry_for_test() + .listen(Some(4242), arrivals) + .expect("port 4242 is free on a fresh node"); + + let registry = node.native_registry_for_test(); + let announced = match registry.deliver(unknown_addr, stranger.pubkey(), 5001, 4242, 2_000) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + assert!(registry.hold(announced, b"held".to_vec())); + + node.record_stats_history(); + let handle = node.control_read_handle(); + + let on_loop = show_native_flows(&node); + let flows = on_loop["flows"].as_array().expect("flows is an array"); + assert_eq!(flows.len(), 2, "both flows are reported"); + + assert_eq!(flows[0]["flow_id"], json!(1)); + assert_eq!( + flows[0]["peer_addr"], + json!(hex::encode(known_addr.as_bytes())) + ); + assert_eq!(flows[0]["peer"], json!(known.npub())); + assert_eq!(flows[0]["local_port"], json!(6000)); + assert_eq!(flows[0]["remote_port"], json!(5000)); + assert_eq!(flows[0]["state"], json!("established")); + assert_eq!(flows[0]["queued"], json!(0)); + + assert_eq!(flows[1]["flow_id"], json!(announced)); + assert_eq!( + flows[1]["peer_addr"], + json!(hex::encode(unknown_addr.as_bytes())) + ); + assert_eq!( + flows[1]["peer"], + json!(stranger.npub()), + "a flow learned from the wire reports its peer's npub too: the key \ + was captured where the session authenticated it, so nothing has to \ + invert the node address" + ); + assert_eq!(flows[1]["local_port"], json!(4242)); + assert_eq!(flows[1]["remote_port"], json!(5001)); + assert_eq!(flows[1]["state"], json!("pending_accept")); + assert_eq!(flows[1]["queued"], json!(1), "the held datagram is queued"); + + let listeners = on_loop["listeners"] + .as_array() + .expect("listeners is an array"); + assert_eq!(listeners.len(), 1, "the bound listener is reported"); + assert_eq!(listeners[0]["local_port"], json!(4242)); + assert_eq!(listeners[0]["backlog"], json!(1)); + + // `age_ms` is derived from `since_ms` and is redacted by `render`, so the + // parity assertion below cannot see the publisher dropping the flow + // timestamp. Pin the captured absolute times against the ones the + // registry was given. + assert_eq!( + handle.native().flows[0].since_ms, + 1_000, + "the publisher carries the connect time the registry recorded" + ); + assert_eq!( + handle.native().flows[1].since_ms, + 2_000, + "the publisher carries the announce time the registry recorded" + ); + + // And the published snapshot renders the same thing. + assert_eq!( + render(on_loop), + render(show_native_flows_from_handle(&handle)), + "off-loop show_native_flows must match on-loop output for live flows" + ); + } + /// Structural sharing: a republish in which only /// one row changed re-allocates only that one `Arc` — every unchanged /// row is reused by pointer (`Arc::ptr_eq`). Exercises diff --git a/src/control/read_handle.rs b/src/control/read_handle.rs index 30a187bc..ad4d1f80 100644 --- a/src/control/read_handle.rs +++ b/src/control/read_handle.rs @@ -13,8 +13,11 @@ //! - `entities` — `ArcSwap`: peers / sessions / links / //! connections / transports, published from the tick with `Vec>` //! structural sharing. +//! - `native` — `ArcSwap`: native datagram API flows and +//! listeners, published from the tick because the registry never leaves the +//! rx_loop. //! -//! Publisher placement: all three snapshot cells are published from the +//! Publisher placement: all four snapshot cells are published from the //! periodic tick, which runs as one arm of the rx_loop's `select!`. Publishing //! therefore costs the rx_loop; what the handle removes is the read-side round //! trip out to the rx_loop and back, not the cost of publishing. The @@ -38,7 +41,7 @@ use crate::node::context::NodeContext; use crate::node::metrics::MetricsRegistry; use super::protocol::{Request, Response}; -use super::snapshot::{EntitySnapshot, RoutingSnapshot, StatsSnapshot}; +use super::snapshot::{EntitySnapshot, NativeSnapshot, RoutingSnapshot, StatsSnapshot}; /// Cloneable read-only view of node state for off-loop control serving. /// @@ -62,6 +65,9 @@ pub(crate) struct ControlReadHandle { /// connections / transports + mmp), published from the tick with /// `Vec>` structural sharing. entities: Arc>, + /// Native datagram API read view (flows / listeners), published + /// from the tick because the registry is the rx_loop's alone. + native: Arc>, } impl ControlReadHandle { @@ -75,6 +81,7 @@ impl ControlReadHandle { stats: Arc>, routing: Arc>, entities: Arc>, + native: Arc>, ) -> Self { Self { context, @@ -82,6 +89,7 @@ impl ControlReadHandle { stats, routing, entities, + native, } } @@ -112,6 +120,12 @@ impl ControlReadHandle { pub(crate) fn entities(&self) -> arc_swap::Guard> { self.entities.load() } + + /// Load the latest published native-API snapshot (freshest + /// available by construction; no staleness gate). + pub(crate) fn native(&self) -> arc_swap::Guard> { + self.native.load() + } } /// Attempt to serve a request entirely from the read handle, off the rx_loop. @@ -120,12 +134,13 @@ impl ControlReadHandle { /// snapshot cells, or `None` when it must take the mpsc → rx_loop path. /// /// The queries served here read any of the cells the handle bundles — -/// `context`, `metrics`, `stats`, `routing`, `entities` — plus host-OS facts -/// gathered in [`super::listening`] and [`super::firewall_state`], so they -/// render in the control task without touching `Node`. Taking a parameter is -/// not what decides it: `show_stats_history` is parameterized and is served -/// here. What falls back is a query needing live `Node` state the snapshot does -/// not carry, and every mutation. +/// `context`, `metrics`, `stats`, `routing`, `entities`, `native` — plus +/// host-OS facts gathered in [`super::listening`] and +/// [`super::firewall_state`], so they render in the control task without +/// touching `Node`. Taking a parameter is not what decides it: +/// `show_stats_history` is parameterized and is served here. What falls back is +/// a query needing live `Node` state the snapshot does not carry, and every +/// mutation. /// /// **It now also carries mutating commands**, namely the `profile_tick_*` /// family under the `profiling` feature. They are served here rather than on @@ -219,6 +234,10 @@ pub(crate) fn snapshot_dispatch(request: &Request, handle: &ControlReadHandle) - "show_connections" => Some(Response::ok(queries::show_connections_from_handle(handle))), "show_transports" => Some(Response::ok(queries::show_transports_from_handle(handle))), "show_mmp" => Some(Response::ok(queries::show_mmp_from_handle(handle))), + // Served from the tick-published `NativeSnapshot`. Peer npubs are + // resolved at publish time, and the `native` counter family comes from + // the `MetricsRegistry`, so this renders entirely off-loop. + "show_native_flows" => Some(Response::ok(queries::show_native_flows_from_handle(handle))), _ => None, } } diff --git a/src/control/snapshot.rs b/src/control/snapshot.rs index 042f79b3..4a7cd47b 100644 --- a/src/control/snapshot.rs +++ b/src/control/snapshot.rs @@ -25,6 +25,7 @@ use crate::node::NodeState; use crate::node::acl::PeerAclStatus; use crate::node::stats_history::StatsHistory; use crate::upper::tun::TunState; +use secp256k1::XOnlyPublicKey; /// Read-only snapshot of the stats-history rings plus the scalar gauges and /// counts `show_status` reports. Published from the tick. @@ -395,6 +396,76 @@ pub(crate) struct IdentityRow { pub last_seen_ms: u64, } +// ===================================================================== +// NativeSnapshot (native datagram API read view) +// ===================================================================== + +/// Read-only snapshot of the native datagram API registry that +/// `show_native_flows` renders. Published via `ArcSwap`. +/// +/// **Publisher placement.** The registry lives inside `Node` and is touched +/// only by the rx_loop, which is precisely why the native receive path takes no +/// lock; a control task cannot read it at all, and putting a lock on it to allow +/// that would give back the property the design was built for. So the +/// projection is published from the tick beside the other snapshot cells. +/// +/// **Cost.** A field read per flow. The peer's key is captured where its +/// session authenticated it and rides on the registry entry, so publishing a +/// row resolves nothing and consults no cache. +/// +/// Time-relative fields (`age_ms`) are derived at render time from the captured +/// absolute timestamps, so a rendered age stays fresh relative to the read. +#[derive(Clone, Default)] +pub(crate) struct NativeSnapshot { + /// One row per flow, established and pending, ordered by identifier. + pub flows: Vec, + /// One row per listener, ordered by port. + pub listeners: Vec, +} + +impl NativeSnapshot { + /// Build an empty snapshot for seeding the `ArcSwap` cell at construction, + /// before the first tick has published real state. + pub(crate) fn empty() -> Self { + Self::default() + } +} + +/// One native API flow in `show_native_flows`. +#[derive(Clone)] +pub(crate) struct NativeFlowRow { + /// The identifier the client names the flow by. + pub flow: u64, + /// The far end, by node address. Kept because this is an operator surface + /// and the node address is what `show_sessions` and `show_routing` key on, + /// which is what lets a reader correlate the three. + pub peer: NodeAddr, + /// The far end's address, as the x-only public key. Always present: a + /// connected flow decoded it from the npub its client named, and an + /// accepted one took it from the session that authenticated the peer. + pub peer_key: XOnlyPublicKey, + /// This node's port. + pub local_port: u16, + /// The far end's port. + pub remote_port: u16, + /// Whether a client has taken the flow, as opposed to still awaiting accept. + pub established: bool, + /// Datagrams the node is holding for it. + pub queued: usize, + /// Absolute open / accept / announce time (Unix ms); `age_ms` derived at + /// render time. + pub since_ms: u64, +} + +/// One native API listener in `show_native_flows`. +#[derive(Clone)] +pub(crate) struct NativeListenerRow { + /// The local port it holds. + pub local_port: u16, + /// Flows announced on it and not yet answered. + pub backlog: usize, +} + // ===================================================================== // EntitySnapshot (per-entity table read views) // ===================================================================== diff --git a/src/control/snapshots/show_native_flows.json b/src/control/snapshots/show_native_flows.json new file mode 100644 index 00000000..162a7c34 --- /dev/null +++ b/src/control/snapshots/show_native_flows.json @@ -0,0 +1,26 @@ +{ + "data": { + "flows": [], + "listeners": [], + "stats": { + "drop_arrival_queue_full": 0, + "drop_backlog_full": 0, + "drop_flow_queue_full": 0, + "drop_listener_gone": 0, + "drop_listener_not_reading": 0, + "drop_no_port": 0, + "drop_oversize": 0, + "drop_pending_queue_full": 0, + "drop_too_many_flows": 0, + "flows_accepted": 0, + "flows_closed": 0, + "flows_expired": 0, + "flows_opened": 0, + "received_bytes": 0, + "received_datagrams": 0, + "sent_bytes": 0, + "sent_datagrams": 0 + } + }, + "status": "ok" +} \ No newline at end of file diff --git a/src/gateway/control.rs b/src/gateway/control.rs index a77014dd..5f05ac26 100644 --- a/src/gateway/control.rs +++ b/src/gateway/control.rs @@ -7,7 +7,7 @@ use crate::control::protocol::{Request, Response}; use crate::gateway::pool::{MappingInfo, MappingState, PoolStatus}; -use std::path::{Path, PathBuf}; +use std::path::PathBuf; use std::time::Instant; use tokio::io::{AsyncBufReadExt, AsyncWriteExt, BufReader}; use tokio::net::UnixListener; @@ -54,35 +54,16 @@ pub struct GatewayControlSocket { } impl GatewayControlSocket { - /// Bind the gateway control socket. + /// Bind the gateway control socket under the shared FIPS access policy. /// - /// Creates parent directories if needed, removes stale socket files, - /// and sets `root:fips 0770` permissions. + /// The policy creates the parent directory, removes a stale socket file, + /// and applies mode `0770` plus the `fips` group. It also applies `0750` to + /// a parent it manages, which `/run/fips` is; systemd already creates that + /// directory at `0750` for the daemon this service requires, so under the + /// packaged deployment it is set to the value it already holds. pub fn bind() -> Result { let socket_path = PathBuf::from(GATEWAY_SOCKET_PATH); - - // Create parent directory if it doesn't exist - if let Some(parent) = socket_path.parent() - && !parent.exists() - { - std::fs::create_dir_all(parent)?; - debug!(path = %parent.display(), "Created gateway control socket directory"); - } - - // Remove stale socket if it exists - if socket_path.exists() { - Self::remove_stale_socket(&socket_path)?; - } - - let listener = UnixListener::bind(&socket_path)?; - - // Set permissions to 0770 and chown to fips group - use std::os::unix::fs::PermissionsExt; - std::fs::set_permissions(&socket_path, std::fs::Permissions::from_mode(0o770))?; - Self::chown_to_fips_group(&socket_path); - if let Some(parent) = socket_path.parent() { - Self::chown_to_fips_group(parent); - } + let listener = crate::utils::sockbind::bind(&socket_path, "gateway control")?; info!(path = %socket_path.display(), "Gateway control socket listening"); @@ -92,51 +73,6 @@ impl GatewayControlSocket { }) } - /// Remove a stale socket file from a previous unclean exit. - fn remove_stale_socket(path: &Path) -> Result<(), std::io::Error> { - match std::os::unix::net::UnixStream::connect(path) { - Ok(_) => Err(std::io::Error::new( - std::io::ErrorKind::AddrInUse, - format!("gateway control socket already in use: {}", path.display()), - )), - Err(_) => { - debug!(path = %path.display(), "Removing stale gateway control socket"); - std::fs::remove_file(path)?; - Ok(()) - } - } - } - - /// Set group ownership to the `fips` group (best-effort). - fn chown_to_fips_group(path: &Path) { - use std::ffi::CString; - use std::os::unix::ffi::OsStrExt; - - let group_name = CString::new("fips").unwrap(); - let grp = unsafe { libc::getgrnam(group_name.as_ptr()) }; - if grp.is_null() { - debug!( - "'fips' group not found, skipping chown for {}", - path.display() - ); - return; - } - let gid = unsafe { (*grp).gr_gid }; - - let c_path = match CString::new(path.as_os_str().as_bytes()) { - Ok(p) => p, - Err(_) => return, - }; - let ret = unsafe { libc::chown(c_path.as_ptr(), u32::MAX, gid) }; - if ret != 0 { - warn!( - path = %path.display(), - error = %std::io::Error::last_os_error(), - "Failed to chown gateway control socket to 'fips' group" - ); - } - } - /// Run the accept loop, reading the latest snapshot from the watch channel. pub async fn accept_loop(self, snapshot_rx: watch::Receiver>) { loop { @@ -218,17 +154,7 @@ impl GatewayControlSocket { /// Clean up the socket file. fn cleanup(&self) { - if self.socket_path.exists() { - if let Err(e) = std::fs::remove_file(&self.socket_path) { - warn!( - path = %self.socket_path.display(), - error = %e, - "Failed to remove gateway control socket" - ); - } else { - debug!(path = %self.socket_path.display(), "Gateway control socket removed"); - } - } + crate::utils::sockbind::cleanup(&self.socket_path, "gateway control"); } } diff --git a/src/lib.rs b/src/lib.rs index dcd35027..8afea9c3 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -21,6 +21,7 @@ pub mod identity; #[macro_use] pub(crate) mod instr; pub mod mdns; +pub mod native; pub mod node; pub mod noise; pub mod nostr; diff --git a/src/native/client/codec.rs b/src/native/client/codec.rs new file mode 100644 index 00000000..273510d3 --- /dev/null +++ b/src/native/client/codec.rs @@ -0,0 +1,408 @@ +//! The native API's line protocol, as a client sees it. +//! +//! Pure: no sockets, no descriptors and no node state, so every rule here is +//! testable without a running daemon. It is the client-side mirror of +//! [`crate::native::protocol`], which does the same job for the daemon: one +//! module that knows the encoding, and a shell that knows only I/O. +//! +//! The encoding is the control socket's, one JSON object per line. A client +//! sends `{"command": "...", "params": {...}}` and reads back one +//! [`Response`](crate::control::protocol::Response). **Replies only, in command +//! order**: nothing unsolicited arrives on this socket, so a line is always the +//! answer to the command just written. +//! +//! An arriving flow is not a line. It is one `SOCK_SEQPACKET` message on the +//! listener's own descriptor, carrying the same fields a `connect` reply +//! carries, which is why [`opened`] reads both. +//! +//! **Errors come back as `io::Error` carrying the errno the contract names**, +//! read from `data.errno`. The daemon's `message` is for an operator reading a +//! log and is deliberately dropped: a client that matched on it would be +//! relying on prose, and `io::Error::from_raw_os_error` is the only +//! constructor that leaves `raw_os_error` readable, which is what a C binding +//! would return from the corresponding call. + +use crate::identity::{decode_npub, encode_npub}; +use secp256k1::XOnlyPublicKey; +use serde_json::{Value, json}; +use std::io; + +/// An npub in text form, as an x-only public key. +/// +/// The client-side half of `fips_pton`: bech32 in one direction, and nothing +/// else. No daemon, no socket, no lookup. A string that is not an npub is +/// `EINVAL`, which is what `inet_pton`'s caller gets for a malformed address. +pub fn pton(text: &str) -> io::Result { + decode_npub(text).map_err(|_| io::Error::from_raw_os_error(libc::EINVAL)) +} + +/// An x-only public key as the npub that names it. +/// +/// The client-side half of `fips_ntop`, and the exact inverse of [`pton`]. +pub fn ntop(key: &XOnlyPublicKey) -> String { + encode_npub(key) +} + +/// The `connect` command line. +/// +/// `local` of 0 is the request for an ephemeral port, which is `bind(2)` with +/// port 0. It is spelled out rather than omitted: the daemon reads an absent +/// field the same way, but one spelling means one rule for `connect` and +/// `listen` both. +pub fn connect(peer: &XOnlyPublicKey, remote: u16, local: u16) -> Vec { + request( + "connect", + json!({"peer": ntop(peer), "remote_port": remote, "local_port": local}), + ) +} + +/// The `listen` command line, where 0 asks the daemon to pick the port. +pub fn listen(local: u16) -> Vec { + request("listen", json!({"local_port": local})) +} + +/// Encode one command as the newline-terminated line the daemon reads. +fn request(command: &str, params: Value) -> Vec { + let mut line = serde_json::to_vec(&json!({"command": command, "params": params})) + .expect("a JSON object built here always serializes"); + line.push(b'\n'); + line +} + +/// The `data` of one reply, or the error the daemon reported. +/// +/// A refusal becomes an `io::Error` here rather than at the call site, so every +/// setup path reports the same way and none of them has to know that a refusal +/// is spelled as a successful read of an error line. +pub fn reply(line: &[u8]) -> io::Result { + let value: Value = serde_json::from_slice(line).map_err(|error| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("the daemon sent a line that is not JSON: {error}"), + ) + })?; + + match value.get("status").and_then(Value::as_str) { + Some("ok") => Ok(value.get("data").cloned().unwrap_or(Value::Null)), + // A reply carrying no `errno` is `ECONNREFUSED`, which covers a daemon + // older than the code that added the field and any refusal raised + // outside the registry. + Some("error") => Err(io::Error::from_raw_os_error(code( + value + .get("data") + .and_then(|data| data.get("errno")) + .and_then(Value::as_str) + .unwrap_or("ECONNREFUSED"), + ))), + Some(other) => Err(io::Error::new( + io::ErrorKind::InvalidData, + format!("the daemon reported an unknown status '{other}'"), + )), + None => Err(missing("status")), + } +} + +/// The errno a name from the daemon stands for on this platform. +/// +/// The daemon sends names rather than numbers because the number belongs to the +/// platform the client was built for and the daemon is not it. An unknown name +/// is `ECONNREFUSED` for the same reason a missing one is: it is the daemon +/// refusing for a reason this client has no row for. +fn code(name: &str) -> i32 { + match name { + "EADDRINUSE" => libc::EADDRINUSE, + "EADDRNOTAVAIL" => libc::EADDRNOTAVAIL, + "EMFILE" => libc::EMFILE, + "EINVAL" => libc::EINVAL, + "EMSGSIZE" => libc::EMSGSIZE, + "EPIPE" => libc::EPIPE, + "ETIMEDOUT" => libc::ETIMEDOUT, + _ => libc::ECONNREFUSED, + } +} + +/// What the daemon said about a flow it opened. +/// +/// One shape for two producers. A `connect` reply and a listener's arrival +/// message carry the same fields, because the daemon re-encodes the peer's key +/// itself in both rather than echoing what a client wrote, so one reader serves +/// both and an accepted flow reports its peer exactly as a connected one does. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Opened { + /// The far end's public key. The address, not a name for one. + pub peer: XOnlyPublicKey, + /// This node's own public key, carried so an accepted flow can answer + /// `getsockname` without consulting the listener that produced it. + pub node: XOnlyPublicKey, + /// The local port the flow holds. + pub local: u16, + /// The far end's port. + pub remote: u16, + /// The largest datagram this flow carries. + pub max: usize, +} + +/// Read a flow description out of a `connect` reply's data or an arrival +/// message. +/// +/// The arrival's `flow_id` and `held` are not read. Neither addresses anything +/// a client can name: the flow is named by its descriptor, and the held +/// datagrams are already on that descriptor by the time the message carrying it +/// can be read. +pub fn opened(data: &Value) -> io::Result { + let max = data + .get("max_payload") + .and_then(Value::as_u64) + .ok_or_else(|| missing("max_payload"))?; + + Ok(Opened { + peer: key(data, "peer")?, + node: key(data, "node")?, + local: port(data, "local_port")?, + remote: port(data, "remote_port")?, + max: max as usize, + }) +} + +/// Read a flow description out of one arrival message. +/// +/// The message is a whole JSON object with **no trailing newline**: it is one +/// `SOCK_SEQPACKET` message, so the boundary is the framing and a newline would +/// offer a client a second framing to rely on. +pub fn arrival(message: &[u8]) -> io::Result { + let value: Value = serde_json::from_slice(message).map_err(|error| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("the daemon sent an arrival that is not JSON: {error}"), + ) + })?; + opened(&value) +} + +/// What the daemon said about a port it bound. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Bound { + /// This node's own public key, which is half of the listener's address. + pub node: XOnlyPublicKey, + /// The port actually held, which is what a client that asked for 0 needs. + pub local: u16, +} + +/// Read a listener description out of a `listen` reply's data. +/// +/// `backlog` is not read. It reports the depth the daemon holds so an operator +/// can size a client's reader, and nothing on this surface takes a depth. +pub fn bound(data: &Value) -> io::Result { + Ok(Bound { + node: key(data, "node")?, + local: port(data, "local_port")?, + }) +} + +/// Read one npub field as the key it encodes. +fn key(data: &Value, field: &'static str) -> io::Result { + let text = data + .get(field) + .and_then(Value::as_str) + .ok_or_else(|| missing(field))?; + pton(text).map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("the daemon's '{field}' is not an npub"), + ) + }) +} + +/// Read one port field, refusing a number that is not one. +fn port(data: &Value, field: &'static str) -> io::Result { + let value = data + .get(field) + .and_then(Value::as_u64) + .ok_or_else(|| missing(field))?; + u16::try_from(value).map_err(|_| { + io::Error::new( + io::ErrorKind::InvalidData, + format!("the daemon's '{field}' is outside the range a port has"), + ) + }) +} + +/// The error for a field this client needs and the daemon did not send. +fn missing(field: &str) -> io::Error { + io::Error::new( + io::ErrorKind::InvalidData, + format!("the daemon's line has no usable '{field}'"), + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// An npub the daemon could have written. Nothing here reaches a peer. + const PEER: &str = "npub1sjlh2c3x9w7kjsqg2ay080n2lff2uvt325vpan33ke34rn8l5jcqawh57m"; + + /// A second one, standing in for the node's own identity. + const NODE: &str = "npub1n9lpnv0592cc2ps6nm0ca3qls642vx7yjsv35rkxqzj2vgds52sqgpverl"; + + /// The `data` of a connect reply, and of an arrival message but for two + /// fields this client does not read. + fn flow() -> Value { + json!({ + "flow_id": 3, "local_port": 49152, "remote_port": 4242, + "peer": PEER, "node": NODE, "max_payload": 1362, + }) + } + + #[test] + fn an_npub_survives_a_round_trip_through_the_two_pure_conversions() { + let key = pton(PEER).unwrap(); + assert_eq!(ntop(&key), PEER); + } + + #[test] + fn a_string_that_is_not_an_npub_is_refused_as_einval() { + let error = pton("not-an-npub").unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EINVAL)); + // An nsec is bech32 and the right length, so only the prefix separates + // it from an address. Taking it would name a key nobody may address. + assert!(pton("nsec1vl029mgpspedva04g90vltkh6fvh240zqtv9k0t9af8935ke9laqsnlfe5").is_err()); + } + + #[test] + fn a_connect_command_names_the_peer_by_the_daemons_own_spelling() { + let key = pton(PEER).unwrap(); + let line = connect(&key, 4242, 0); + assert_eq!(line.last(), Some(&b'\n')); + let value: Value = serde_json::from_slice(line.trim_ascii_end()).unwrap(); + assert_eq!(value["command"], "connect"); + assert_eq!(value["params"]["peer"], PEER); + assert_eq!(value["params"]["remote_port"], 4242); + // Spelled rather than omitted: 0 is the ephemeral request, and one + // spelling covers connect and listen both. + assert_eq!(value["params"]["local_port"], 0); + } + + #[test] + fn a_listen_command_carries_the_port_the_caller_asked_for() { + let value: Value = serde_json::from_slice(listen(4242).trim_ascii_end()).unwrap(); + assert_eq!(value["command"], "listen"); + assert_eq!(value["params"]["local_port"], 4242); + } + + #[test] + fn an_ok_reply_yields_its_data() { + let line = br#"{"status":"ok","data":{"local_port":4242}}"#; + assert_eq!(reply(line).unwrap()["local_port"], 4242); + } + + #[test] + fn a_refusal_becomes_the_errno_the_daemon_named() { + let line = br#"{"status":"error","message":"port 4242 is already in use","data":{"errno":"EADDRINUSE"}}"#; + let error = reply(line).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EADDRINUSE)); + } + + #[test] + fn a_refusal_carrying_no_errno_falls_back_to_econnrefused() { + // A daemon older than the field, or a refusal raised outside the + // registry. Guessing a more specific code would tell a caller something + // the daemon did not say. + let line = br#"{"status":"error","message":"no such flow: 9"}"#; + assert_eq!( + reply(line).unwrap_err().raw_os_error(), + Some(libc::ECONNREFUSED) + ); + let unknown = br#"{"status":"error","data":{"errno":"EWHATEVER"}}"#; + assert_eq!( + reply(unknown).unwrap_err().raw_os_error(), + Some(libc::ECONNREFUSED) + ); + } + + #[test] + fn every_errno_the_daemon_can_name_maps_to_this_platforms_number() { + // The contract is the number a C binding would return, so a name this + // client did not translate would silently become ECONNREFUSED and + // collapse a row of the error table. + for (name, want) in [ + ("EADDRINUSE", libc::EADDRINUSE), + ("EADDRNOTAVAIL", libc::EADDRNOTAVAIL), + ("EMFILE", libc::EMFILE), + ("EINVAL", libc::EINVAL), + ("EMSGSIZE", libc::EMSGSIZE), + ("EPIPE", libc::EPIPE), + ("ETIMEDOUT", libc::ETIMEDOUT), + ] { + assert_eq!(code(name), want, "{name}"); + } + } + + #[test] + fn a_line_that_is_not_json_is_reported_as_such() { + let error = reply(b"not json at all").unwrap_err(); + assert_eq!(error.kind(), io::ErrorKind::InvalidData); + } + + #[test] + fn a_reply_with_no_status_is_refused_rather_than_guessed_at() { + let error = reply(br#"{"data":{}}"#).unwrap_err(); + assert!(error.to_string().contains("status"), "{error}"); + } + + #[test] + fn one_reader_describes_a_connect_reply_and_an_arrival_alike() { + let want = Opened { + peer: pton(PEER).unwrap(), + node: pton(NODE).unwrap(), + local: 49152, + remote: 4242, + max: 1362, + }; + assert_eq!(opened(&flow()).unwrap(), want); + + // The arrival message is the same object with two fields this client + // does not read. It must not need a second reader. + let mut arrival = flow(); + arrival["held"] = json!(2); + assert_eq!(opened(&arrival).unwrap(), want); + } + + #[test] + fn a_flow_whose_peer_is_not_an_npub_is_refused_rather_than_carried() { + // The peer is the address. A client that accepted a hex node address + // here would report something that cannot be sent to. + let mut data = flow(); + data["peer"] = json!("aabbccddeeff00112233445566778899"); + let error = opened(&data).unwrap_err(); + assert!(error.to_string().contains("not an npub"), "{error}"); + } + + #[test] + fn a_reply_missing_the_payload_limit_names_the_field() { + let mut data = flow(); + data.as_object_mut().unwrap().remove("max_payload"); + let error = opened(&data).unwrap_err(); + assert!(error.to_string().contains("max_payload"), "{error}"); + } + + #[test] + fn a_port_above_the_range_a_port_has_is_refused() { + let mut data = flow(); + data["local_port"] = json!(70000); + let error = opened(&data).unwrap_err(); + assert!(error.to_string().contains("local_port"), "{error}"); + } + + #[test] + fn a_listen_reply_reports_the_port_actually_held_and_the_nodes_own_key() { + let data = json!({"local_port": 49152, "node": NODE, "backlog": 16}); + assert_eq!( + bound(&data).unwrap(), + Bound { + node: pton(NODE).unwrap(), + local: 49152, + } + ); + } +} diff --git a/src/native/client/mod.rs b/src/native/client/mod.rs new file mode 100644 index 00000000..d46946a3 --- /dev/null +++ b/src/native/client/mod.rs @@ -0,0 +1,1402 @@ +//! A blocking client for the native datagram API, shaped like `std::net`. +//! +//! An external program links this crate and speaks the API through +//! [`FipsStream`] and [`FipsListener`] without knowing its line protocol. The +//! names, signatures and error type follow `TcpStream` and `TcpListener`, +//! because the descriptor the daemon hands back **is a real socket**: `send`, +//! `recv`, `poll`, `select`, `epoll`, `close` and `SO_RCVTIMEO` on it are the +//! genuine syscalls. Only setup is not, so setup is the part shaped to look +//! like Berkeley's. +//! +//! An address is an x-only secp256k1 public key and a port. The npub is that +//! key written down, and converting between the two is bech32 and nothing else: +//! no lookup, no resolution, no name service. The 16-byte node address on the +//! wire is a truncated hash of the key, it does not invert, and it appears +//! nowhere on this surface. [`ToFipsAddr`] is how one parameter takes every +//! spelling of the same address, exactly as `ToSocketAddrs` does. +//! +//! **A stream that outlives its connection is not representable, because the +//! connection is not an object.** [`FipsStream::connect`] opens an RPC +//! connection to the daemon socket, sends one command, receives the descriptor +//! and closes that connection. What it returns holds the descriptor and plain +//! copies of what the reply said, and no handle on anything else, which is what +//! a Berkeley setup call leaves behind. Both types are therefore `Send` and +//! `'static` with no `Arc` and no borrow. +//! +//! **The connection is read with `recvmsg` and never with a buffered reader.** +//! The daemon attaches a descriptor to the ancillary data of the same `sendmsg` +//! that carries the reply line, so a reader that consumed those bytes without a +//! control buffer would consume the descriptor into nothing. See [`Wire::fill`] +//! for the rule that decides which line a descriptor belongs to. + +mod codec; + +use super::{fdpass, seqpacket}; +use codec::Opened; +use std::collections::VecDeque; +use std::fmt; +use std::io; +use std::os::fd::{AsFd, AsRawFd, BorrowedFd, OwnedFd, RawFd}; +use std::os::unix::net::UnixStream; +use std::path::Path; +use std::str::FromStr; +use std::time::Duration; + +pub use secp256k1::XOnlyPublicKey; + +/// Where a packaged daemon puts its native API socket. +/// +/// The path-taking constructors exist for a program that is told where its +/// daemon is; everything else uses this, the way a Berkeley call needs no +/// argument to find the kernel. +pub const SOCKET: &str = "/run/fips/api.sock"; + +/// Bytes taken from the RPC connection per `recvmsg`. +/// +/// A reply is a few hundred bytes, so this holds several and the reader rarely +/// makes two syscalls for one line. The same size serves an arrival message, +/// which is smaller still. +const CHUNK: usize = 8192; + +/// Largest partial line the client will hold before giving up on the daemon. +/// +/// It bounds a daemon that stops sending newlines, which is the only way the +/// line buffer could grow without end. Well above any reply the daemon writes. +const MAX_LINE: usize = 65536; + +/// How long a setup command waits for its answer. +/// +/// Matches the control socket's per-connection timeout. **It is the only wait +/// on this surface**: a native `connect` is a local registration that contacts +/// no peer, so nothing else here can time out, and `ETIMEDOUT` has exactly this +/// one producer. +const SETUP: Duration = Duration::from_secs(5); + +/// One end of a flow: an x-only public key and a port. +/// +/// The Rust mirror of `struct sockaddr_fips`. [`Display`](fmt::Display) writes +/// `npub1…:4242` and [`FromStr`] reads it back; the colon is unambiguous +/// because bech32's character set does not contain one. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct FipsAddr { + key: XOnlyPublicKey, + port: u16, +} + +impl FipsAddr { + /// An address from a key and a port. + pub fn new(key: XOnlyPublicKey, port: u16) -> Self { + Self { key, port } + } + + /// The public key. This is the address; nothing else identifies an end. + pub fn key(&self) -> XOnlyPublicKey { + self.key + } + + /// The port. + pub fn port(&self) -> u16 { + self.port + } +} + +impl fmt::Display for FipsAddr { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "{}:{}", codec::ntop(&self.key), self.port) + } +} + +impl FromStr for FipsAddr { + type Err = io::Error; + + /// Read `npub1…:4242`, refusing anything else with `EINVAL`. + fn from_str(text: &str) -> io::Result { + let (npub, port) = text + .rsplit_once(':') + .ok_or_else(|| io::Error::from_raw_os_error(libc::EINVAL))?; + let port: u16 = port + .parse() + .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; + Ok(Self::new(codec::pton(npub)?, port)) + } +} + +/// Every spelling of one address, behind one parameter. +/// +/// Mirrors `std::net::ToSocketAddrs`. The implementations are the address in +/// binary ([`XOnlyPublicKey`] or 32 bytes), the address in text (an npub), and +/// the whole thing as one string. +pub trait ToFipsAddr { + /// The address this value names, or `EINVAL` if it names none. + fn to_fips_addr(&self) -> io::Result; +} + +impl ToFipsAddr for FipsAddr { + fn to_fips_addr(&self) -> io::Result { + Ok(*self) + } +} + +impl ToFipsAddr for (XOnlyPublicKey, u16) { + fn to_fips_addr(&self) -> io::Result { + Ok(FipsAddr::new(self.0, self.1)) + } +} + +impl ToFipsAddr for ([u8; 32], u16) { + fn to_fips_addr(&self) -> io::Result { + let key = XOnlyPublicKey::from_slice(&self.0) + .map_err(|_| io::Error::from_raw_os_error(libc::EINVAL))?; + Ok(FipsAddr::new(key, self.1)) + } +} + +impl ToFipsAddr for (&str, u16) { + fn to_fips_addr(&self) -> io::Result { + Ok(FipsAddr::new(codec::pton(self.0)?, self.1)) + } +} + +impl ToFipsAddr for (String, u16) { + fn to_fips_addr(&self) -> io::Result { + (self.0.as_str(), self.1).to_fips_addr() + } +} + +impl ToFipsAddr for str { + fn to_fips_addr(&self) -> io::Result { + self.parse() + } +} + +impl ToFipsAddr for String { + fn to_fips_addr(&self) -> io::Result { + self.as_str().to_fips_addr() + } +} + +impl ToFipsAddr for &T { + fn to_fips_addr(&self) -> io::Result { + (**self).to_fips_addr() + } +} + +/// The RPC connection's reader, and the write side beside it. +/// +/// Holds its own line buffer over `recvmsg` because a buffered reader is not +/// usable here: see [`Wire::fill`]. It lives only for the length of one setup +/// call, and dropping it is what closes the connection. +#[derive(Debug)] +struct Wire { + sock: UnixStream, + /// Bytes read that do not yet form a complete line. + partial: Vec, + /// Complete lines, oldest first, each with the descriptor it arrived with. + lines: VecDeque<(Vec, Option)>, +} + +impl Wire { + /// Open a connection to the daemon's socket, with the setup deadline on it. + /// + /// The deadline is `SO_RCVTIMEO`, so it bounds the wait for the answer + /// rather than the whole call, which is what the contract's one `ETIMEDOUT` + /// producer describes. + fn open(path: &Path) -> io::Result { + let sock = UnixStream::connect(path)?; + sock.set_read_timeout(Some(SETUP))?; + Ok(Self::new(sock)) + } + + /// A reader over an already connected socket. + fn new(sock: UnixStream) -> Self { + Self { + sock, + partial: Vec::new(), + lines: VecDeque::new(), + } + } + + /// Send one command and read its answer, with any descriptor it carried. + /// + /// The RPC socket carries **replies only, in command order**, so the next + /// complete line is the answer and there is nothing to queue. + fn call(&mut self, request: Vec) -> io::Result<(serde_json::Value, Option)> { + self.send(&request)?; + let (line, fd) = self.line()?; + Ok((codec::reply(&line)?, fd)) + } + + /// Write one command line. + /// + /// `libc::send` with `MSG_NOSIGNAL` rather than `write_all`, so a command + /// written to a daemon that has gone away is an `EPIPE` error whatever the + /// host program is. A Rust binary ignores `SIGPIPE` by default and would see + /// the error anyway, but a C program that loads this crate would be killed + /// instead. [`FipsStream::send`] makes the same choice for the same reason. + fn send(&mut self, line: &[u8]) -> io::Result<()> { + let mut rest = line; + while !rest.is_empty() { + // SAFETY: the socket is owned and open, and the pointer and length + // describe `rest`. + let sent = unsafe { + libc::send( + self.sock.as_raw_fd(), + rest.as_ptr().cast(), + rest.len(), + libc::MSG_NOSIGNAL, + ) + }; + if sent < 0 { + let error = io::Error::last_os_error(); + if error.kind() == io::ErrorKind::Interrupted { + continue; + } + return Err(error); + } + if sent == 0 { + // A blocking stream socket does not do this with bytes left to + // write. Reported rather than spun on. + return Err(io::Error::other("the native API socket took no bytes")); + } + rest = &rest[sent as usize..]; + } + Ok(()) + } + + /// Take the next complete line, reading until one is available. + fn line(&mut self) -> io::Result<(Vec, Option)> { + loop { + if let Some(line) = self.lines.pop_front() { + return Ok(line); + } + self.fill()?; + } + } + + /// One `recvmsg`, split into lines, with any descriptor placed by the + /// association rule. + /// + /// **A descriptor belongs to the last complete line of the read that + /// carried it, never to the next line the reader assembles.** Measured on + /// Linux 6.8.0-117 with a C program rather than reasoned about: a `recvmsg` + /// that returns ancillary data ends exactly at the end of the `sendmsg` that + /// carried it, but it may begin with any amount of data written before it. + /// So a line the daemon wrote earlier and a descriptor-bearing reply arrive + /// as one read, and a reader that attached the descriptor to the next line + /// it completed would hand a flow to the wrong one. + /// + /// Such a read never ends on a partial line, because the daemon writes + /// exactly one whole line per `sendmsg` and treats a short write as an error + /// rather than retrying. That is an invariant of these two programs rather + /// than of the socket type, which is why a read carrying a descriptor and no + /// complete line is reported instead of guessed at. + fn fill(&mut self) -> io::Result<()> { + let mut buf = [0u8; CHUNK]; + let chunk = fdpass::recv(self.sock.as_raw_fd(), &mut buf).map_err(expired)?; + if chunk.len == 0 { + // The daemon accepted the connection and then dropped it, which is + // the socket not accepting by a slower route. + return Err(io::Error::from_raw_os_error(libc::ECONNREFUSED)); + } + self.partial.extend_from_slice(&buf[..chunk.len]); + + let mut produced = 0usize; + while let Some(end) = self.partial.iter().position(|byte| *byte == b'\n') { + let mut line: Vec = self.partial.drain(..=end).collect(); + line.pop(); + self.lines.push_back((line, None)); + produced += 1; + } + + if self.partial.len() > MAX_LINE { + return Err(io::Error::new( + io::ErrorKind::InvalidData, + format!("{} bytes with no newline", self.partial.len()), + )); + } + + if let Some(fd) = chunk.fd { + if produced == 0 { + // Dropping the descriptor closes it. Holding it would mean + // guessing which later line it belongs to, which is the defect + // this rule exists to prevent. + return Err(io::Error::new( + io::ErrorKind::InvalidData, + "a descriptor arrived on a read that completed no line", + )); + } + self.lines + .back_mut() + .expect("a read that produced a line has one to attach to") + .1 = Some(fd); + } + Ok(()) + } +} + +/// Turn the setup deadline expiring into the errno the contract names. +/// +/// `SO_RCVTIMEO` reports a would-block, which says nothing to a caller about +/// which wait ended. `ETIMEDOUT` is the row this is, and it is the only row +/// that produces it. +fn expired(error: io::Error) -> io::Error { + match error.kind() { + io::ErrorKind::WouldBlock | io::ErrorKind::TimedOut => { + io::Error::from_raw_os_error(libc::ETIMEDOUT) + } + _ => error, + } +} + +/// One datagram flow, and the descriptor it is carried on. +/// +/// Five scalars and two keys, all fixed when the flow opened: a flow's key pair +/// and port pair cannot change while it is open, and `max_payload` is computed +/// once. There is nothing else to hold, which is what makes the type `Send` and +/// self-contained, so a server serving one thread per flow can move it there. +/// +/// The protocol has no close command. Dropping the stream closes its +/// descriptor, and that is what releases the flow and its local port at the +/// daemon. +#[derive(Debug)] +pub struct FipsStream { + fd: OwnedFd, + peer: FipsAddr, + local: FipsAddr, + max: usize, +} + +impl FipsStream { + /// Open a flow to `addr`, from a port the daemon picks. + /// + /// **This contacts no peer.** It is a local registration, and nothing about + /// it proves the peer exists, is reachable, or is listening. A + /// Berkeley-shaped surface invites the opposite reading, so it is said here + /// rather than left to a manual. + pub fn connect(addr: A) -> io::Result { + Self::connect_at(Path::new(SOCKET), 0, addr) + } + + /// Open a flow from a named local port, which a peer can be told in advance. + /// + /// `connect_from(0, addr)` means an ephemeral port, the same as + /// [`FipsStream::connect`]. There is no other spelling of it. + pub fn connect_from(local: u16, addr: A) -> io::Result { + Self::connect_at(Path::new(SOCKET), local, addr) + } + + /// The same call, for a program that is told where its daemon's socket is. + pub fn connect_at(sock: &Path, local: u16, addr: A) -> io::Result { + let addr = addr.to_fips_addr()?; + Self::open(Wire::open(sock)?, local, addr) + } + + /// Perform the `connect` exchange on an open connection, then close it. + /// + /// The connection is a local here on purpose: it is dropped before this + /// returns, so what the caller holds cannot outlive it. + fn open(mut wire: Wire, local: u16, addr: FipsAddr) -> io::Result { + let (data, fd) = wire.call(codec::connect(&addr.key, addr.port, local))?; + let fd = fd.ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + "a connect reply carried no descriptor", + ) + })?; + Ok(Self::wired(fd, codec::opened(&data)?)) + } + + /// A stream over a descriptor and the facts the daemon reported with it. + fn wired(fd: OwnedFd, opened: Opened) -> FipsStream { + FipsStream { + fd, + peer: FipsAddr::new(opened.peer, opened.remote), + local: FipsAddr::new(opened.node, opened.local), + max: opened.max, + } + } + + /// Send one datagram. + /// + /// No byte count: `SOCK_SEQPACKET` delivers a message whole or not at all, + /// so there is no short write and no count worth checking. A count would + /// give every `?`-using caller something it could ignore incorrectly, and + /// no test would catch it because the count is always the full length. + /// + /// Above [`FipsStream::max_payload`] this is `EMSGSIZE` before the syscall, + /// so a caller learns which datagram was too large instead of finding a gap + /// at the far end. The limit itself is allowed. + pub fn send(&self, buf: &[u8]) -> io::Result<()> { + if buf.len() > self.max { + return Err(io::Error::from_raw_os_error(libc::EMSGSIZE)); + } + loop { + // SAFETY: the descriptor is owned and open, and the pointer and + // length describe `buf`. + let sent = unsafe { + libc::send( + self.fd.as_raw_fd(), + buf.as_ptr().cast(), + buf.len(), + libc::MSG_NOSIGNAL, + ) + }; + if sent < 0 { + let error = io::Error::last_os_error(); + if error.kind() == io::ErrorKind::Interrupted { + continue; + } + return Err(error); + } + return Ok(()); + } + } + + /// Receive one datagram, returning how many bytes it held. + /// + /// **`Ok(0)` is an empty datagram**, which a peer may legitimately send. It + /// is not end of file, and this is the one place where mimicking Berkeley + /// exactly would be wrong: reading a zero-byte datagram as a close would let + /// a peer tear down a live flow by sending nothing. A closed daemon half is + /// `EPIPE`, discriminated by `POLLHUP`, measured for this socket pair in + /// [`seqpacket`](super::seqpacket). + /// + /// A datagram longer than `buf` is truncated and the remainder discarded, + /// which is `SOCK_SEQPACKET` behaviour. Size `buf` at + /// [`FipsStream::max_payload`] and it cannot happen. + pub fn recv(&self, buf: &mut [u8]) -> io::Result { + loop { + // SAFETY: the descriptor is owned and open, and the pointer and + // length describe `buf`. + let received = + unsafe { libc::recv(self.fd.as_raw_fd(), buf.as_mut_ptr().cast(), buf.len(), 0) }; + if received < 0 { + let error = io::Error::last_os_error(); + if error.kind() == io::ErrorKind::Interrupted { + continue; + } + return Err(error); + } + if received == 0 && seqpacket::peer_hung_up(self.fd.as_raw_fd()) { + return Err(io::Error::from_raw_os_error(libc::EPIPE)); + } + return Ok(received as usize); + } + } + + /// The far end, by public key and port. + /// + /// Not an `io::Result`, unlike `TcpStream::peer_addr`, which is a syscall + /// that can fail. This is a field read of what setup already reported, and a + /// `Result` that is structurally always `Ok` teaches a caller to `unwrap`. + pub fn peer_addr(&self) -> FipsAddr { + self.peer + } + + /// This end: the node's own public key and the port the flow holds. + pub fn local_addr(&self) -> FipsAddr { + self.local + } + + /// Set or clear the deadline on [`FipsStream::recv`]. + /// + /// Mirrors `TcpStream::set_read_timeout`. `None` clears it. When the + /// deadline expires, `recv` returns [`io::ErrorKind::WouldBlock`], which is + /// the same answer a non-blocking descriptor gives and is what the platform + /// reports; the two are told apart by which the caller asked for. + /// + /// **A zero duration is refused with `EINVAL`.** The kernel reads a zero + /// timeout as "wait for ever", which is the opposite of what a caller + /// passing zero means, so it is refused rather than silently inverted. + /// + /// This bounds one `recv`, not a conversation. Nothing peer-driven ever + /// ends a flow, so a program still decides its own termination. + pub fn set_read_timeout(&self, dur: Option) -> io::Result<()> { + seqpacket::set_timeout(self.fd.as_raw_fd(), libc::SO_RCVTIMEO, dur) + } + + /// Set or clear the deadline on [`FipsStream::send`]. + /// + /// Mirrors `TcpStream::set_write_timeout`. It matters less than the read + /// deadline and is not useless: `send` blocks when the daemon is not + /// draining this flow fast enough. Same zero-duration refusal. + pub fn set_write_timeout(&self, dur: Option) -> io::Result<()> { + seqpacket::set_timeout(self.fd.as_raw_fd(), libc::SO_SNDTIMEO, dur) + } + + /// The deadline on [`FipsStream::recv`], or `None` when there is none. + pub fn read_timeout(&self) -> io::Result> { + seqpacket::timeout(self.fd.as_raw_fd(), libc::SO_RCVTIMEO) + } + + /// The deadline on [`FipsStream::send`], or `None` when there is none. + pub fn write_timeout(&self) -> io::Result> { + seqpacket::timeout(self.fd.as_raw_fd(), libc::SO_SNDTIMEO) + } + + /// Set or clear non-blocking mode on this flow's descriptor. + /// + /// Mirrors `TcpStream::set_nonblocking`. In non-blocking mode + /// [`FipsStream::recv`] returns [`io::ErrorKind::WouldBlock`] instead of + /// waiting, and [`FipsStream::send`] does the same when the daemon is not + /// draining the flow fast enough. + /// + /// **This exists so a flow can be driven by a reactor.** `AsyncFd` and its + /// equivalents require a non-blocking descriptor, and without a safe call a + /// caller has to reach through [`AsRawFd`] and make the `fcntl` themselves, + /// or give the flow a thread of its own. + /// + /// A flow from [`FipsListener::accept`] is blocking however the listener was + /// set: they are separate sockets, and the daemon hands over a fresh one. + /// Set it on the flow if the flow is what you poll. + pub fn set_nonblocking(&self, nonblocking: bool) -> io::Result<()> { + seqpacket::set_nonblocking(self.fd.as_raw_fd(), nonblocking) + } + + /// The largest datagram this flow carries, as the daemon computed it. + /// + /// No `std::net` counterpart, because TCP has no such limit to report. It is + /// here because `EMSGSIZE` is reachable and a caller needs the threshold + /// before it sends. + pub fn max_payload(&self) -> usize { + self.max + } +} + +impl AsRawFd for FipsStream { + /// The flow's descriptor, for a caller with its own reactor or poll loop. + /// + /// The stream keeps ownership: the descriptor is closed when it is dropped, + /// which is what releases the flow at the daemon. + fn as_raw_fd(&self) -> RawFd { + self.fd.as_raw_fd() + } +} + +impl AsFd for FipsStream { + /// The flow's descriptor as a borrow, which is the form a reactor wants. + /// + /// Preferred over [`AsRawFd`]: the borrow cannot outlive the stream, so a + /// registered descriptor cannot be closed out from under the reactor and + /// then reused for something else by the next `open`. A `RawFd` carries no + /// such guarantee and is kept for callers whose interface demands an `int`. + fn as_fd(&self) -> BorrowedFd<'_> { + self.fd.as_fd() + } +} + +/// A held local port, and the descriptor arriving flows are delivered on. +/// +/// **The listener is a descriptor**, which is what makes it pollable: it joins +/// an existing `poll`, `select` or `epoll` loop with no new mechanism, and +/// [`FipsListener::accept`] is one `recvmsg` on it. Dropping the listener closes +/// that descriptor, which unbinds the port; flows already accepted from it are +/// untouched. +#[derive(Debug)] +pub struct FipsListener { + fd: OwnedFd, + local: FipsAddr, +} + +impl FipsListener { + /// Hold `port`, or an ephemeral one when `port` is 0. + pub fn bind(port: u16) -> io::Result { + Self::bind_at(Path::new(SOCKET), port) + } + + /// The same call, for a program that is told where its daemon's socket is. + pub fn bind_at(sock: &Path, port: u16) -> io::Result { + Self::hold(Wire::open(sock)?, port) + } + + /// Perform the `listen` exchange on an open connection, then close it. + fn hold(mut wire: Wire, port: u16) -> io::Result { + let (data, fd) = wire.call(codec::listen(port))?; + let fd = fd.ok_or_else(|| { + io::Error::new( + io::ErrorKind::InvalidData, + "a listen reply carried no descriptor", + ) + })?; + let bound = codec::bound(&data)?; + Ok(FipsListener { + fd, + local: FipsAddr::new(bound.node, bound.local), + }) + } + + /// Take the next arriving flow, blocking until one arrives. + /// + /// One `recvmsg`, one arrival: `SOCK_SEQPACKET` means the message carries + /// exactly its own descriptor, so the association rule the RPC socket needs + /// does not arise here. + /// + /// **Whatever the peer sent before this returned is already on the returned + /// stream's descriptor.** The daemon writes those datagrams before the + /// message that carries the descriptor, so the ordering is the guarantee and + /// a peer's opening datagram cannot be lost between the two. + /// + /// Refusing a flow is dropping the stream, which closes its descriptor. + /// There is no other way to refuse one, which is why an unreadable arrival + /// message is reported after the descriptor it carried has been taken: the + /// flow is then refused rather than leaked. + pub fn accept(&self) -> io::Result<(FipsStream, FipsAddr)> { + let mut buf = [0u8; CHUNK]; + let chunk = fdpass::recv(self.fd.as_raw_fd(), &mut buf)?; + let Some(fd) = chunk.fd else { + // Nothing to refuse and nothing to report on: either the daemon + // closed its half, or it wrote an arrival with no flow in it. + return Err(io::Error::from_raw_os_error(libc::EPIPE)); + }; + let stream = FipsStream::wired(fd, codec::arrival(&buf[..chunk.len])?); + let peer = stream.peer_addr(); + Ok((stream, peer)) + } + + /// An iterator over arriving flows, as `TcpListener::incoming` is. + /// + /// It never ends: a listener has no last flow, and a failed accept is an + /// item rather than the end of the iteration. + pub fn incoming(&self) -> Incoming<'_> { + Incoming { listener: self } + } + + /// Set or clear non-blocking mode on this listener's descriptor. + /// + /// Mirrors `TcpListener::set_nonblocking`. In non-blocking mode + /// [`FipsListener::accept`] returns [`io::ErrorKind::WouldBlock`] when no + /// flow has arrived, and so does every [`FipsListener::incoming`] item, + /// which makes that iterator spin unless the caller waits on the descriptor + /// between items. + /// + /// **This is what a reactor needs.** `AsyncFd` and its equivalents require a + /// non-blocking descriptor, and without a safe call a caller has to reach + /// through [`AsRawFd`] and make the `fcntl` themselves, or give the accept + /// loop a thread of its own. + /// + /// It does not reach the flows this listener yields: each arrives blocking. + pub fn set_nonblocking(&self, nonblocking: bool) -> io::Result<()> { + seqpacket::set_nonblocking(self.fd.as_raw_fd(), nonblocking) + } + + /// The port this listener holds, with the node's own public key. + /// + /// The port is the one actually held, so a caller that asked for 0 reads + /// what it got, which is `getsockname` after `bind(2)` with port 0. + pub fn local_addr(&self) -> FipsAddr { + self.local + } +} + +impl AsRawFd for FipsListener { + /// The listener's descriptor, for a caller with its own event loop. + /// + /// This is the point of the listener being a descriptor: `poll` on it + /// reports readable exactly when [`FipsListener::accept`] would not block. + fn as_raw_fd(&self) -> RawFd { + self.fd.as_raw_fd() + } +} + +impl AsFd for FipsListener { + /// The listener's descriptor as a borrow, which is the form a reactor wants. + /// + /// Preferred over [`AsRawFd`], for the reason given on the same impl for + /// [`FipsStream`]: the borrow is tied to the listener's lifetime, so the + /// descriptor cannot be closed while a reactor still holds it. + fn as_fd(&self) -> BorrowedFd<'_> { + self.fd.as_fd() + } +} + +/// Arriving flows, one per [`Iterator::next`]. +#[derive(Debug)] +pub struct Incoming<'a> { + listener: &'a FipsListener, +} + +impl Iterator for Incoming<'_> { + type Item = io::Result; + + fn next(&mut self) -> Option { + Some(self.listener.accept().map(|(stream, _peer)| stream)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::{Read, Write}; + use std::os::fd::AsFd; + use std::thread; + + /// An npub shaped like the ones a caller passes. No test reaches the peer. + const PEER: &str = "npub1sjlh2c3x9w7kjsqg2ay080n2lff2uvt325vpan33ke34rn8l5jcqawh57m"; + + /// A second one, standing in for the node's own identity. + const NODE: &str = "npub1n9lpnv0592cc2ps6nm0ca3qls642vx7yjsv35rkxqzj2vgds52sqgpverl"; + + /// The reply to a `connect`, which carries the flow's descriptor. + const CONNECT_REPLY: &[u8] = concat!( + r#"{"status":"ok","data":{"flow_id":3,"local_port":49152,"remote_port":4242,"#, + r#""peer":"npub1sjlh2c3x9w7kjsqg2ay080n2lff2uvt325vpan33ke34rn8l5jcqawh57m","#, + r#""node":"npub1n9lpnv0592cc2ps6nm0ca3qls642vx7yjsv35rkxqzj2vgds52sqgpverl","#, + r#""max_payload":1362}}"#, + "\n", + ) + .as_bytes(); + + /// The reply to a `listen`, which carries the listener's descriptor. + const LISTEN_REPLY: &[u8] = concat!( + r#"{"status":"ok","data":{"local_port":4242,"#, + r#""node":"npub1n9lpnv0592cc2ps6nm0ca3qls642vx7yjsv35rkxqzj2vgds52sqgpverl","#, + r#""backlog":16}}"#, + "\n", + ) + .as_bytes(); + + /// One arrival message, as the listener's task writes it: no newline, + /// because the `SOCK_SEQPACKET` boundary is the framing. + const ARRIVAL: &[u8] = concat!( + r#"{"flow_id":9,"#, + r#""peer":"npub1sjlh2c3x9w7kjsqg2ay080n2lff2uvt325vpan33ke34rn8l5jcqawh57m","#, + r#""node":"npub1n9lpnv0592cc2ps6nm0ca3qls642vx7yjsv35rkxqzj2vgds52sqgpverl","#, + r#""local_port":4242,"remote_port":5001,"max_payload":1362,"held":1}"#, + ) + .as_bytes(); + + /// A refusal, as the daemon writes one for a port that is already held. + const REFUSAL: &[u8] = concat!( + r#"{"status":"error","message":"port 4242 is already in use on this node","#, + r#""data":{"errno":"EADDRINUSE"}}"#, + "\n", + ) + .as_bytes(); + + /// Read one whole line from a socket, one byte at a time. + /// + /// A test daemon reads its command this way so it never consumes past the + /// newline, which a buffered reader would. + fn read_line(sock: &UnixStream) -> Vec { + let mut line = Vec::new(); + let mut byte = [0u8; 1]; + loop { + let read = (&*sock) + .read(&mut byte) + .expect("the client should be there"); + assert_eq!(read, 1, "the client closed mid-command"); + if byte[0] == b'\n' { + return line; + } + line.push(byte[0]); + } + } + + /// A stream whose far end is a socket the test drives, standing in for the + /// daemon's half of a real flow. + fn stream(max: usize) -> (FipsStream, UnixStream) { + let (ours, theirs) = seqpacket::pair().expect("a socket pair should be available"); + let stream = FipsStream::wired( + ours, + Opened { + peer: codec::pton(PEER).unwrap(), + node: codec::pton(NODE).unwrap(), + local: 4242, + remote: 5001, + max, + }, + ); + (stream, UnixStream::from(theirs)) + } + + #[test] + fn a_read_deadline_expires_rather_than_waiting_for_a_datagram_that_never_comes() { + let (flow, _daemon) = stream(1362); + flow.set_read_timeout(Some(Duration::from_millis(150))) + .expect("the descriptor is open"); + + // The far end is held open and silent, so without a deadline this + // parks for ever. Timing it is what shows the deadline did the work + // rather than something else returning early. + // + // The recv runs on its own thread and is collected through a channel + // with a bound of its own, because the defect this test exists to catch + // is precisely a deadline that never reached the descriptor: called + // inline, that parks for ever and wedges the whole `cargo test --lib` + // run with no diagnostic. A hang is not a red, so the bound is what + // makes this test able to fail. + let (report, collected) = std::sync::mpsc::channel(); + thread::spawn(move || { + let start = std::time::Instant::now(); + let mut buf = [0u8; 64]; + let outcome = flow.recv(&mut buf); + let _ = report.send((outcome, start.elapsed())); + }); + + let (outcome, waited) = collected + .recv_timeout(Duration::from_secs(5)) + .expect("the 150ms read deadline never bounded the recv"); + let error = outcome.expect_err("nothing was ever sent"); + + assert_eq!(error.kind(), io::ErrorKind::WouldBlock); + assert!( + waited >= Duration::from_millis(100), + "returned after {waited:?}, too soon to have waited out a 150ms deadline" + ); + } + + #[test] + fn a_zero_read_deadline_is_refused_because_the_kernel_would_read_it_as_no_deadline() { + let (flow, _daemon) = stream(1362); + let error = flow + .set_read_timeout(Some(Duration::ZERO)) + .expect_err("zero means the opposite to the kernel"); + assert_eq!(error.raw_os_error(), Some(libc::EINVAL)); + + // And it did not reach the socket: no deadline is set afterwards. + assert_eq!(flow.read_timeout().unwrap(), None); + } + + #[test] + fn a_deadline_reads_back_as_it_was_set_and_none_clears_it() { + let (flow, _daemon) = stream(1362); + assert_eq!(flow.read_timeout().unwrap(), None, "none is the default"); + assert_eq!(flow.write_timeout().unwrap(), None); + + flow.set_read_timeout(Some(Duration::from_millis(1500))) + .unwrap(); + assert_eq!( + flow.read_timeout().unwrap(), + Some(Duration::from_millis(1500)) + ); + + // The two directions are separate options; setting one must not set + // the other, which a copied constant would. + assert_eq!(flow.write_timeout().unwrap(), None, "send is untouched"); + + flow.set_write_timeout(Some(Duration::from_secs(2))) + .unwrap(); + assert_eq!(flow.write_timeout().unwrap(), Some(Duration::from_secs(2))); + assert_eq!( + flow.read_timeout().unwrap(), + Some(Duration::from_millis(1500)), + "recv keeps its own" + ); + + flow.set_read_timeout(None).unwrap(); + assert_eq!(flow.read_timeout().unwrap(), None, "cleared"); + assert_eq!( + flow.write_timeout().unwrap(), + Some(Duration::from_secs(2)), + "clearing one leaves the other" + ); + } + + #[test] + fn a_nonblocking_flow_reports_would_block_rather_than_waiting_for_a_datagram() { + let (flow, _daemon) = stream(1362); + + // Blocking is the default, so the flag has somewhere to move from. + let raw = flow.as_raw_fd(); + let before = unsafe { libc::fcntl(raw, libc::F_GETFL) }; + assert_eq!(before & libc::O_NONBLOCK, 0, "a flow starts out blocking"); + + flow.set_nonblocking(true).expect("the descriptor is open"); + + // Nothing has been sent, so a blocking recv would park here for ever. + let mut buf = [0u8; 64]; + let error = flow.recv(&mut buf).expect_err("no datagram is waiting"); + assert_eq!(error.kind(), io::ErrorKind::WouldBlock); + } + + #[test] + fn clearing_nonblocking_returns_a_flow_to_blocking_and_leaves_its_other_flags_alone() { + let (flow, _daemon) = stream(1362); + let raw = flow.as_raw_fd(); + + // A second flag to watch, and it has to be one F_SETFL actually + // honours. The access mode looks like the obvious choice and is not: + // F_SETFL ignores it, so the kernel preserves O_RDWR however carelessly + // the flag word is written, and asserting on it gives a test that + // cannot fail. O_ASYNC is settable on a socket, so it does discriminate. + let start = unsafe { libc::fcntl(raw, libc::F_GETFL) }; + assert!(start >= 0); + assert_eq!( + unsafe { libc::fcntl(raw, libc::F_SETFL, start | libc::O_ASYNC) }, + 0, + "O_ASYNC should be settable on a seqpacket socket" + ); + let original = unsafe { libc::fcntl(raw, libc::F_GETFL) }; + assert_ne!(original & libc::O_ASYNC, 0, "the flag under test is set"); + + flow.set_nonblocking(true).unwrap(); + assert_ne!( + unsafe { libc::fcntl(raw, libc::F_GETFL) } & libc::O_NONBLOCK, + 0 + ); + + flow.set_nonblocking(false).unwrap(); + let after = unsafe { libc::fcntl(raw, libc::F_GETFL) }; + assert_eq!(after & libc::O_NONBLOCK, 0, "blocking mode is restored"); + assert_ne!( + after & libc::O_ASYNC, + 0, + "a flag the caller set survived the round trip, which a bare \ + F_SETFL of O_NONBLOCK alone would have cleared" + ); + assert_eq!( + after, original, + "the whole flag word is unchanged but for O_NONBLOCK" + ); + } + + #[test] + fn a_nonblocking_listener_reports_would_block_rather_than_waiting_for_a_flow() { + let (ours, _theirs) = seqpacket::pair().expect("a socket pair should be available"); + let listener = FipsListener { + fd: ours, + local: FipsAddr::new(codec::pton(NODE).unwrap(), 4242), + }; + + listener + .set_nonblocking(true) + .expect("the descriptor is open"); + + let error = listener.accept().expect_err("no flow has arrived"); + assert_eq!(error.kind(), io::ErrorKind::WouldBlock); + } + + #[test] + fn the_borrowed_descriptor_is_the_same_one_the_raw_accessor_reports() { + let (flow, _daemon) = stream(1362); + assert_eq!(flow.as_fd().as_raw_fd(), flow.as_raw_fd()); + + let (ours, _theirs) = seqpacket::pair().unwrap(); + let listener = FipsListener { + fd: ours, + local: FipsAddr::new(codec::pton(NODE).unwrap(), 4242), + }; + assert_eq!(listener.as_fd().as_raw_fd(), listener.as_raw_fd()); + } + + #[test] + fn a_descriptor_lands_on_the_last_complete_line_of_the_read_that_carried_it() { + let (daemon, client) = UnixStream::pair().unwrap(); + let (passed, held) = seqpacket::pair().unwrap(); + let held = UnixStream::from(held); + + // Both written before the client reads, so one recvmsg carries the + // plain line, the descriptor-bearing line and the descriptor. This is + // the measured rule: the ancillary data belongs to the sendmsg the read + // ended on, not to the first line the reader completes. + (&daemon).write_all(REFUSAL).unwrap(); + fdpass::send_once(daemon.as_raw_fd(), CONNECT_REPLY, Some(passed.as_fd())).unwrap(); + drop(passed); + + let mut wire = Wire::new(client); + wire.fill().unwrap(); + assert_eq!( + wire.lines.len(), + 2, + "the plain line and the descriptor-bearing one should arrive as one \ + recvmsg; if they did not, the measured association rule this \ + module rests on does not hold on this platform" + ); + + let (first, first_fd) = wire.line().unwrap(); + assert!( + first.starts_with(br#"{"status":"error""#), + "first line was {first:?}" + ); + assert!( + first_fd.is_none(), + "the earlier line must not be given the descriptor" + ); + + let (second, second_fd) = wire.line().unwrap(); + assert!(second.starts_with(br#"{"status":"ok""#)); + let received = second_fd.expect("the reply must carry the descriptor"); + + // A live socket rather than merely a number: the far half sees it. + let mut received = UnixStream::from(received); + received.write_all(b"alive").unwrap(); + let mut got = [0u8; 5]; + (&held).read_exact(&mut got).unwrap(); + assert_eq!(&got, b"alive"); + } + + #[test] + fn a_descriptor_on_a_read_that_completed_no_line_is_reported_rather_than_guessed_at() { + let (daemon, client) = UnixStream::pair().unwrap(); + let (passed, _held) = seqpacket::pair().unwrap(); + + // A descriptor with no line to attach it to. Holding it would mean + // guessing which later line owns it, and the guess loses a flow. + fdpass::send_once(daemon.as_raw_fd(), b"{\"status\":", Some(passed.as_fd())).unwrap(); + + let mut wire = Wire::new(client); + let error = wire.fill().unwrap_err(); + assert!( + error.to_string().contains("completed no line"), + "got {error}" + ); + } + + #[test] + fn connect_returns_a_stream_and_closes_the_connection_that_made_it() { + let (daemon, client) = UnixStream::pair().unwrap(); + let (passed, _held) = seqpacket::pair().unwrap(); + + let worker = thread::spawn(move || { + let command = read_line(&daemon); + fdpass::send_once(daemon.as_raw_fd(), CONNECT_REPLY, Some(passed.as_fd())).unwrap(); + // The connection must be gone once connect has returned: a stream + // that outlived its connection is the defect this shape removes, + // and a reply is the last thing this socket is for. + // A deadline, or the defect this asserts against fails as a hang + // rather than as a named test: a client that kept its connection + // leaves this read blocked for ever and the suite never finishes. + daemon + .set_read_timeout(Some(Duration::from_secs(5))) + .unwrap(); + let mut rest = [0u8; 16]; + let read = (&daemon) + .read(&mut rest) + .expect("the RPC connection outlived the setup call that opened it"); + (command, read) + }); + + let addr = (PEER, 4242).to_fips_addr().unwrap(); + let flow = FipsStream::open(Wire::new(client), 0, addr).unwrap(); + + assert_eq!(flow.peer_addr().to_string(), format!("{PEER}:4242")); + assert_eq!(flow.local_addr().to_string(), format!("{NODE}:49152")); + assert_eq!(flow.max_payload(), 1362); + + let (command, read) = worker.join().unwrap(); + let command: serde_json::Value = serde_json::from_slice(&command).unwrap(); + assert_eq!(command["command"], "connect"); + assert_eq!(command["params"]["peer"], PEER); + assert_eq!(command["params"]["local_port"], 0); + assert_eq!(read, 0, "the RPC connection was still open"); + } + + #[test] + fn bind_keeps_the_listener_descriptor_and_reports_the_port_actually_held() { + let (daemon, client) = UnixStream::pair().unwrap(); + let (passed, held) = seqpacket::pair().unwrap(); + let held = UnixStream::from(held); + + let worker = thread::spawn(move || { + let command = read_line(&daemon); + fdpass::send_once(daemon.as_raw_fd(), LISTEN_REPLY, Some(passed.as_fd())).unwrap(); + drop(passed); + // As for connect: setup is over, so the connection is over. A + // listener that held one would take its flows down with it. + // A deadline, or the defect this asserts against fails as a hang + // rather than as a named test: a client that kept its connection + // leaves this read blocked for ever and the suite never finishes. + daemon + .set_read_timeout(Some(Duration::from_secs(5))) + .unwrap(); + let mut rest = [0u8; 16]; + let read = (&daemon) + .read(&mut rest) + .expect("the RPC connection outlived the setup call that opened it"); + (command, read) + }); + + // Asking for 0 is asking the daemon to pick, so the reply's port is the + // only place the answer exists. + let listener = FipsListener::hold(Wire::new(client), 0).unwrap(); + assert_eq!(listener.local_addr().to_string(), format!("{NODE}:4242")); + + let (command, read) = worker.join().unwrap(); + let command: serde_json::Value = serde_json::from_slice(&command).unwrap(); + assert_eq!(command["command"], "listen"); + assert_eq!(command["params"]["local_port"], 0); + assert_eq!(read, 0, "the RPC connection was still open"); + + // The descriptor is a live socket, not merely a number the reply named. + let mut buf = [0u8; 5]; + (&held).write_all(b"alive").unwrap(); + // SAFETY: the listener's descriptor is open and owned by it. + let got = unsafe { libc::recv(listener.as_raw_fd(), buf.as_mut_ptr().cast(), 5, 0) }; + assert_eq!(got, 5, "{}", io::Error::last_os_error()); + } + + #[test] + fn a_refusal_becomes_the_error_a_bind_would_have_returned() { + let (daemon, client) = UnixStream::pair().unwrap(); + + let worker = thread::spawn(move || { + read_line(&daemon); + (&daemon).write_all(REFUSAL).unwrap(); + daemon + }); + + let error = FipsListener::hold(Wire::new(client), 4242).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EADDRINUSE)); + + let _daemon = worker.join().unwrap(); + } + + #[test] + fn a_daemon_that_never_answers_is_the_one_producer_of_etimedout() { + let (_daemon, client) = UnixStream::pair().unwrap(); + // The real deadline is five seconds and its subject is the same wait; + // shortening it here keeps the suite quick without changing the path. + client + .set_read_timeout(Some(Duration::from_millis(50))) + .unwrap(); + + let error = FipsListener::hold(Wire::new(client), 4242).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::ETIMEDOUT)); + } + + #[test] + fn a_daemon_that_goes_away_ends_a_blocking_setup_call_rather_than_looping() { + // Gone before the command: the write is what fails, and a write to a + // socket whose far end has closed is `EPIPE`. It is an error here + // rather than a signal because the command is sent with `MSG_NOSIGNAL`. + let (daemon, client) = UnixStream::pair().unwrap(); + drop(daemon); + let error = FipsListener::hold(Wire::new(client), 4242).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EPIPE)); + + // Gone after the command: the read ends at once instead of waiting out + // the setup deadline, and a connection the daemon accepted and then + // dropped is the socket not accepting by a slower route. + let (daemon, client) = UnixStream::pair().unwrap(); + let worker = thread::spawn(move || { + read_line(&daemon); + drop(daemon); + }); + let error = FipsListener::hold(Wire::new(client), 4242).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::ECONNREFUSED)); + worker.join().unwrap(); + } + + #[test] + fn accept_takes_the_flow_the_arrival_carried_and_names_its_peer_by_npub() { + let (ours, theirs) = seqpacket::pair().unwrap(); + let (passed, held) = seqpacket::pair().unwrap(); + let held = UnixStream::from(held); + + let listener = FipsListener { + fd: ours, + local: FipsAddr::new(codec::pton(NODE).unwrap(), 4242), + }; + + // The daemon writes the peer's opening datagram onto the flow's own + // half first, and the arrival that carries that flow's descriptor + // second. The order is the guarantee: whatever arrived before the + // client could read the arrival is already there when it holds it. + (&held).write_all(b"opening").unwrap(); + fdpass::send_once(theirs.as_raw_fd(), ARRIVAL, Some(passed.as_fd())).unwrap(); + drop(passed); + + let (flow, peer) = listener.accept().unwrap(); + assert_eq!(peer.to_string(), format!("{PEER}:5001")); + assert_eq!(flow.peer_addr(), peer); + // From the arrival's own `node`, not from the listener: an accepted + // stream answers `getsockname` without consulting what produced it. + assert_eq!(flow.local_addr().to_string(), format!("{NODE}:4242")); + assert_eq!(flow.max_payload(), 1362); + + let mut buf = [0u8; 64]; + assert_eq!(flow.recv(&mut buf).unwrap(), 7); + assert_eq!(&buf[..7], b"opening"); + + // The descriptor is the flow's own half and nothing else's. + flow.send(b"back").unwrap(); + let mut got = [0u8; 64]; + assert_eq!((&held).read(&mut got).unwrap(), 4); + } + + #[test] + fn a_listener_descriptor_is_pollable_and_reports_a_waiting_arrival() { + // The point of the listener being a descriptor. Without this, `accept` + // could only be discovered by blocking in it. + let (ours, theirs) = seqpacket::pair().unwrap(); + let (passed, _held) = seqpacket::pair().unwrap(); + let listener = FipsListener { + fd: ours, + local: FipsAddr::new(codec::pton(NODE).unwrap(), 4242), + }; + + let mut poll = libc::pollfd { + fd: listener.as_raw_fd(), + events: libc::POLLIN, + revents: 0, + }; + // SAFETY: `poll` names one live descriptor and the call cannot block. + assert_eq!(unsafe { libc::poll(&mut poll, 1, 0) }, 0, "readable early"); + + fdpass::send_once(theirs.as_raw_fd(), ARRIVAL, Some(passed.as_fd())).unwrap(); + + poll.revents = 0; + // SAFETY: as above. + assert_eq!(unsafe { libc::poll(&mut poll, 1, 0) }, 1); + assert_ne!(poll.revents & libc::POLLIN, 0); + listener.accept().unwrap(); + } + + #[test] + fn an_arrival_with_no_descriptor_is_reported_rather_than_taken_as_a_flow() { + let (ours, theirs) = seqpacket::pair().unwrap(); + let listener = FipsListener { + fd: ours, + local: FipsAddr::new(codec::pton(NODE).unwrap(), 4242), + }; + fdpass::send_once(theirs.as_raw_fd(), ARRIVAL, None).unwrap(); + + assert_eq!( + listener.accept().unwrap_err().raw_os_error(), + Some(libc::EPIPE) + ); + } + + #[test] + fn incoming_yields_the_same_flows_accept_would() { + let (ours, theirs) = seqpacket::pair().unwrap(); + let (passed, _held) = seqpacket::pair().unwrap(); + let listener = FipsListener { + fd: ours, + local: FipsAddr::new(codec::pton(NODE).unwrap(), 4242), + }; + fdpass::send_once(theirs.as_raw_fd(), ARRIVAL, Some(passed.as_fd())).unwrap(); + + let flow = listener.incoming().next().unwrap().unwrap(); + assert_eq!(flow.peer_addr().to_string(), format!("{PEER}:5001")); + } + + #[test] + fn datagrams_cross_a_flow_whole_in_both_directions() { + let (flow, far) = stream(1362); + + flow.send(b"out").unwrap(); + flow.send(b"again").unwrap(); + let mut got = [0u8; 64]; + assert_eq!((&far).read(&mut got).unwrap(), 3); + assert_eq!(&got[..3], b"out"); + assert_eq!((&far).read(&mut got).unwrap(), 5); + assert_eq!(&got[..5], b"again"); + + (&far).write_all(b"back").unwrap(); + let mut buf = [0u8; 64]; + assert_eq!(flow.recv(&mut buf).unwrap(), 4); + assert_eq!(&buf[..4], b"back"); + } + + #[test] + fn an_empty_datagram_is_not_reported_as_a_closed_flow() { + let (flow, far) = stream(1362); + + // Both produce a zero-byte read, and a client that read the first as a + // close would tear down a live flow because a peer sent nothing. + // + // `write_all(b"")` is a no-op in Rust and never reaches the socket, so + // the zero-length datagram is sent with `send` directly. The first + // version of this test used `write_all` and blocked for ever waiting + // for a datagram nothing had sent. + // SAFETY: the descriptor is open and owned by `far`. + let sent = unsafe { libc::send(far.as_raw_fd(), std::ptr::null(), 0, 0) }; + assert_eq!(sent, 0, "{}", io::Error::last_os_error()); + + let mut buf = [0u8; 64]; + assert_eq!(flow.recv(&mut buf).unwrap(), 0); + + drop(far); + let error = flow.recv(&mut buf).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EPIPE)); + } + + #[test] + fn a_datagram_above_the_flows_limit_is_refused_before_it_is_sent() { + let (flow, far) = stream(8); + let error = flow.send(&[0u8; 9]).unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EMSGSIZE)); + + // Nothing reached the socket: the far end has no datagram waiting. + far.set_read_timeout(Some(Duration::from_millis(50))) + .unwrap(); + let mut buf = [0u8; 64]; + assert!( + (&far).read(&mut buf).is_err(), + "a refused datagram was sent" + ); + + // The limit itself is allowed, so the check is not off by one. + flow.send(&[0u8; 8]).unwrap(); + assert_eq!((&far).read(&mut buf).unwrap(), 8); + } + + #[test] + fn a_flow_can_move_to_another_thread() { + // The point of a stream holding no reference to a connection: a server + // can hand one to a worker thread. + let (flow, far) = stream(1362); + let worker = thread::spawn(move || { + flow.send(b"from the worker").unwrap(); + flow + }); + let flow = worker.join().unwrap(); + assert_eq!(flow.local_addr().port(), 4242); + + let mut got = [0u8; 64]; + assert_eq!((&far).read(&mut got).unwrap(), 15); + } + + #[test] + fn one_parameter_takes_every_spelling_of_the_same_address() { + let want = FipsAddr::new(codec::pton(PEER).unwrap(), 4242); + let key = want.key(); + + assert_eq!((PEER, 4242u16).to_fips_addr().unwrap(), want); + assert_eq!((PEER.to_string(), 4242u16).to_fips_addr().unwrap(), want); + assert_eq!((key, 4242u16).to_fips_addr().unwrap(), want); + assert_eq!((key.serialize(), 4242u16).to_fips_addr().unwrap(), want); + assert_eq!(format!("{PEER}:4242").to_fips_addr().unwrap(), want); + assert_eq!(want.to_fips_addr().unwrap(), want); + + // Through a generic parameter, which is how the setup calls take one: + // this is what exercises the blanket reference implementation, and + // without it a caller could not pass `&addr` at all. + fn resolve(addr: A) -> FipsAddr { + addr.to_fips_addr().expect("the address should resolve") + } + let text: &str = &format!("{PEER}:4242"); + let by_ref: &FipsAddr = &want; + assert_eq!(resolve(text), want); + assert_eq!(resolve(by_ref), want); + } + + #[test] + fn an_address_that_does_not_parse_is_einval_with_no_daemon_involved() { + // Local and pure, like `inet_pton`: nothing here opens a socket. + for bad in [ + "not-an-npub", + "npub1abc:4242", + &format!("{PEER}:70000"), + PEER, + ] { + let error = bad.to_fips_addr().unwrap_err(); + assert_eq!(error.raw_os_error(), Some(libc::EINVAL), "{bad}"); + } + assert_eq!( + ([0u8; 32], 4242u16) + .to_fips_addr() + .unwrap_err() + .raw_os_error(), + Some(libc::EINVAL) + ); + } + + #[test] + fn an_address_written_out_reads_back_as_itself() { + let addr = FipsAddr::new(codec::pton(PEER).unwrap(), 4242); + assert_eq!(addr.to_string(), format!("{PEER}:4242")); + assert_eq!(addr.to_string().parse::().unwrap(), addr); + } +} diff --git a/src/native/fdpass.rs b/src/native/fdpass.rs new file mode 100644 index 00000000..97400ed0 --- /dev/null +++ b/src/native/fdpass.rs @@ -0,0 +1,312 @@ +//! Write a reply on the client connection, optionally carrying a descriptor. +//! +//! Passing a file descriptor between processes is `sendmsg` with an `SCM_RIGHTS` +//! control message, which neither std nor tokio exposes. The descriptor travels +//! in the ancillary data of the same `sendmsg` that carries the reply line, so a +//! client reads one message and gets both, with no window in which it holds one +//! without the other. +//! +//! Every write goes through `try_io`, including the ones with no descriptor, so +//! the connection's `UnixStream` is only ever borrowed shared. That is what lets +//! the reader keep the stream inside a `BufReader` while replies are written +//! through `get_ref`. +//! +//! The receiving half lives here too, in `recv`. It is blocking and uses no +//! tokio, because a client process is what runs it, but it is the same concern +//! read backwards and it needs the same control message sizing. + +use std::io; +use std::mem; +use std::os::fd::{AsRawFd, BorrowedFd, FromRawFd, OwnedFd, RawFd}; +use tokio::io::Interest; +use tokio::net::UnixStream; + +/// Space for the control message, sized at runtime and checked against this. +/// +/// `CMSG_SPACE(4)` is 24 bytes on the platforms this builds for. The array is +/// `u64` so it carries the alignment `cmsghdr` requires, and is larger than +/// needed so a platform with a wider header is caught by the assertion rather +/// than by memory corruption. +type CmsgBuf = [u64; 8]; + +/// Send `line` on `stream`, with `fd` in the ancillary data when given. +/// +/// The whole reply goes in one datagram-shaped `sendmsg`. A short write is +/// treated as an error rather than retried: the replies are a few hundred bytes +/// into an empty socket buffer, so a partial one means something is wrong that a +/// retry loop would hide. +pub async fn reply(stream: &UnixStream, line: &[u8], fd: Option>) -> io::Result<()> { + loop { + stream.writable().await?; + match stream.try_io(Interest::WRITABLE, || { + send_once(stream.as_raw_fd(), line, fd) + }) { + Ok(written) if written == line.len() => return Ok(()), + Ok(written) => { + return Err(io::Error::other(format!( + "native API reply truncated: wrote {written} of {}", + line.len() + ))); + } + Err(error) if error.kind() == io::ErrorKind::WouldBlock => continue, + Err(error) => return Err(error), + } + } +} + +/// One `sendmsg` that reports a full send buffer instead of waiting for one. +/// +/// The listener's task writes onto socket pairs whose client half it has not +/// handed over yet, so no process can read either one and a task that waited +/// would stop serving that listener for good. [`Seqpacket::send`] is the wrong +/// tool for exactly that reason: it treats a would-block as a reason to await +/// readiness rather than as an answer. This is the raw syscall on a descriptor +/// the reactor has already made non-blocking, so a full buffer comes back as +/// `WouldBlock` and the caller decides. +/// +/// [`Seqpacket::send`]: super::seqpacket::Seqpacket::send +pub(super) fn try_send(sock: RawFd, line: &[u8], fd: Option>) -> io::Result { + send_once(sock, line, fd) +} + +/// One `sendmsg`, with or without an `SCM_RIGHTS` control message. +/// +/// Visible within [`super`] so the client's tests can play daemon through the +/// real ancillary framing rather than an imitation of it. +pub(super) fn send_once(sock: i32, line: &[u8], fd: Option>) -> io::Result { + let mut iov = libc::iovec { + iov_base: line.as_ptr() as *mut libc::c_void, + iov_len: line.len(), + }; + // SAFETY: msghdr is a plain C struct with no invalid bit patterns; every + // field this call reads is set below. + let mut msg: libc::msghdr = unsafe { mem::zeroed() }; + msg.msg_iov = &mut iov; + msg.msg_iovlen = 1; + + let mut control: CmsgBuf = [0; 8]; + if let Some(fd) = fd { + // SAFETY: CMSG_SPACE is a pure size computation over its argument. + let space = unsafe { libc::CMSG_SPACE(mem::size_of::() as u32) } as usize; + if space > mem::size_of::() { + return Err(io::Error::other( + "control message buffer too small for a descriptor", + )); + } + + msg.msg_control = control.as_mut_ptr().cast(); + msg.msg_controllen = space as _; + + // SAFETY: msg_control points at `control`, which is aligned for + // cmsghdr and at least `space` bytes long, so the header the kernel + // macro returns lies inside it. + unsafe { + let header = libc::CMSG_FIRSTHDR(&msg); + if header.is_null() { + return Err(io::Error::other("control message header unavailable")); + } + (*header).cmsg_level = libc::SOL_SOCKET; + (*header).cmsg_type = libc::SCM_RIGHTS; + (*header).cmsg_len = libc::CMSG_LEN(mem::size_of::() as u32) as _; + let raw = fd.as_raw_fd(); + std::ptr::copy_nonoverlapping( + std::ptr::addr_of!(raw).cast::(), + libc::CMSG_DATA(header), + mem::size_of::(), + ); + } + } + + // SAFETY: `sock` is the connection's open descriptor, and `msg` describes + // buffers that outlive this call. + let sent = unsafe { libc::sendmsg(sock, &msg, libc::MSG_NOSIGNAL) }; + if sent < 0 { + return Err(io::Error::last_os_error()); + } + Ok(sent as usize) +} + +/// What one `recvmsg` on a client's RPC connection produced. +/// +/// The bytes and the descriptor are reported together because they arrive +/// together. Handing them back separately would reopen the window this module +/// exists to close. +/// +/// Visible within [`super`] only, like [`send_once`]: the client module is the +/// one caller, and the API this crate publishes is `FipsAddr`, `FipsStream` and +/// `FipsListener` rather than the framing underneath them. +#[derive(Debug)] +pub(super) struct Chunk { + /// How many bytes landed in the caller's buffer. Zero is end of file. + pub(super) len: usize, + /// The descriptor the message carried, where it carried one. + pub(super) fd: Option, +} + +/// Receive one message into `buf`, keeping any descriptor that came with it. +/// +/// Blocking and `libc`-only, with no tokio: this is the half a client process +/// runs. It lives beside [`reply`] because the two share the control message +/// sizing and the same safety argument. +/// +/// **Every read on the RPC connection must come through here.** A plain `read` +/// consumes a descriptor-bearing message's bytes with no ancillary buffer, and +/// the kernel closes the descriptor rather than queueing it, so the reply looks +/// right and the flow is silently gone. +/// +/// `MSG_CMSG_CLOEXEC` keeps a received descriptor out of a child the client +/// forks later. `EINTR` is retried, because a signal delivered during the wait +/// says nothing about the connection. +pub(super) fn recv(sock: RawFd, buf: &mut [u8]) -> io::Result { + loop { + let mut iov = libc::iovec { + iov_base: buf.as_mut_ptr().cast(), + iov_len: buf.len(), + }; + let mut control: CmsgBuf = [0; 8]; + // SAFETY: as in send_once; every field this call reads is set below. + let mut msg: libc::msghdr = unsafe { mem::zeroed() }; + msg.msg_iov = &mut iov; + msg.msg_iovlen = 1; + msg.msg_control = control.as_mut_ptr().cast(); + msg.msg_controllen = mem::size_of::() as _; + + // SAFETY: `sock` is the caller's open socket, and `msg` describes + // buffers that outlive the call. + let received = unsafe { libc::recvmsg(sock, &mut msg, libc::MSG_CMSG_CLOEXEC) }; + if received < 0 { + let error = io::Error::last_os_error(); + if error.kind() == io::ErrorKind::Interrupted { + continue; + } + return Err(error); + } + + // SAFETY: recvmsg succeeded, so it filled `msg_control` within + // `msg_controllen`, and `control` is still alive. + let mut fds = unsafe { take_fds(&msg) }; + + // Whatever did arrive is taken before the truncation check, so nothing + // leaks on that path: dropping an `OwnedFd` closes it. The connection + // cannot continue either way, because a descriptor the kernel dropped + // is one no later read can recover. + if (msg.msg_flags & libc::MSG_CTRUNC) != 0 { + return Err(io::Error::other( + "native API control message truncated: a descriptor was lost", + )); + } + + // More than one descriptor is not something this protocol sends. The + // extras are dropped, and so closed, rather than leaked. + let fd = if fds.is_empty() { + None + } else { + Some(fds.swap_remove(0)) + }; + + return Ok(Chunk { + len: received as usize, + fd, + }); + } +} + +/// Collect every descriptor an `SCM_RIGHTS` control message carried. +/// +/// The whole control buffer is walked rather than only its first header: a +/// reader that took `CMSG_FIRSTHDR` alone would leak any descriptor behind it. +/// +/// # Safety +/// +/// `msg` must be a `msghdr` that a successful `recvmsg` filled in, whose +/// `msg_control` buffer is still live and unmodified since. +unsafe fn take_fds(msg: &libc::msghdr) -> Vec { + let mut fds = Vec::new(); + // SAFETY: the caller guarantees `msg` came from a successful recvmsg, so + // every header these macros return lies inside its control buffer, and + // every descriptor named there is one this process now owns. + unsafe { + let mut header = libc::CMSG_FIRSTHDR(msg); + while !header.is_null() { + if (*header).cmsg_level == libc::SOL_SOCKET && (*header).cmsg_type == libc::SCM_RIGHTS { + let payload = (*header).cmsg_len as usize - libc::CMSG_LEN(0) as usize; + for index in 0..payload / mem::size_of::() { + let mut raw: libc::c_int = 0; + std::ptr::copy_nonoverlapping( + libc::CMSG_DATA(header).add(index * mem::size_of::()), + std::ptr::addr_of_mut!(raw).cast::(), + mem::size_of::(), + ); + fds.push(OwnedFd::from_raw_fd(raw)); + } + } + header = libc::CMSG_NXTHDR(msg, header); + } + } + fds +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::native::seqpacket::{Received, Seqpacket, pair}; + use std::io::Write; + use std::os::fd::AsFd; + use std::os::unix::net::UnixStream as StdUnixStream; + + /// Receive one line plus an optional descriptor, the way a client does. + /// + /// Goes through [`recv`] rather than repeating its `recvmsg`, so these + /// tests exercise the receiving half a client actually runs. + fn recv_with_fd(sock: &StdUnixStream) -> (Vec, Option) { + let mut buf = [0u8; 4096]; + let chunk = recv(sock.as_raw_fd(), &mut buf).expect("recvmsg should succeed"); + (buf[..chunk.len].to_vec(), chunk.fd) + } + + #[tokio::test] + async fn a_reply_without_a_descriptor_arrives_whole() { + let (ours, theirs) = StdUnixStream::pair().unwrap(); + ours.set_nonblocking(true).unwrap(); + let ours = UnixStream::from_std(ours).unwrap(); + + reply(&ours, b"{\"status\":\"ok\"}\n", None).await.unwrap(); + + let (line, fd) = recv_with_fd(&theirs); + assert_eq!(line, b"{\"status\":\"ok\"}\n"); + assert!(fd.is_none()); + } + + #[tokio::test] + async fn a_descriptor_arrives_with_its_reply_and_still_works() { + let (ours, theirs) = StdUnixStream::pair().unwrap(); + ours.set_nonblocking(true).unwrap(); + let ours = UnixStream::from_std(ours).unwrap(); + + let (daemon_half, client_half) = pair().unwrap(); + let daemon_half = Seqpacket::new(daemon_half).unwrap(); + + reply(&ours, b"{\"status\":\"ok\"}\n", Some(client_half.as_fd())) + .await + .unwrap(); + // The daemon closes its copy once it is sent; the receiver holds the + // only remaining reference to the client half. + drop(client_half); + + let (line, fd) = recv_with_fd(&theirs); + assert_eq!(line, b"{\"status\":\"ok\"}\n"); + let fd = fd.expect("a descriptor should have arrived"); + + // The passed descriptor is a live half of the flow, not merely a + // number: writing on it must reach the daemon's side. + let mut received = StdUnixStream::from(fd); + received.write_all(b"through the passed fd").unwrap(); + + let mut buf = [0u8; 64]; + assert_eq!( + daemon_half.recv(&mut buf).await.unwrap(), + Received::Datagram(21) + ); + assert_eq!(&buf[..21], b"through the passed fd"); + } +} diff --git a/src/native/link.rs b/src/native/link.rs new file mode 100644 index 00000000..ce06bc65 --- /dev/null +++ b/src/native/link.rs @@ -0,0 +1,393 @@ +//! What a native API client task asks the node to do. +//! +//! The registry lives inside `Node`, so a client task cannot touch it directly. +//! It sends one of these instead and waits on the `oneshot` it carried. That is +//! the same shape the control socket uses, and it is why the receive path needs +//! no lock: only the `rx_loop` ever holds the registry. +//! +//! Every variant that can fail carries its reply channel. A dropped reply means +//! the node is shutting down, which the client task reports as such rather than +//! waiting. + +use super::registry::{Arrival, Datagram, Delivery, DropCause, FlowKey, Registry, RegistryError}; +use crate::identity::NodeAddr; +use secp256k1::XOnlyPublicKey; +use tokio::sync::{mpsc, oneshot}; +use tracing::trace; + +/// A request from a client task to the node's registry. +#[derive(Debug)] +pub enum NativeMessage { + /// Bind a listener to a local port, or to an ephemeral one. + Listen { + /// The port to hold, or `None` for an ephemeral one. + port: Option, + /// Where the node announces new peers on that port. + arrivals: mpsc::Sender, + /// The port actually held, or why none could be. + reply: oneshot::Sender>, + }, + + /// Open a flow to a peer. + Connect { + /// The far end, by the x-only public key that is its address. + peer: XOnlyPublicKey, + /// The far end's port. + remote: u16, + /// The local port, or `None` for an ephemeral one. + local: Option, + /// Where the node delivers this flow's datagrams. + sink: mpsc::Sender, + /// What the flow holds, or why it could not be opened. + reply: oneshot::Sender>, + }, + + /// Take a flow a listener announced. + Accept { + /// Which announced flow. + flow: u64, + /// Where the node delivers its datagrams from now on. + sink: mpsc::Sender, + /// The flow's key, whatever arrived before it was accepted, and the + /// payload limit. + reply: oneshot::Sender>, + }, + + /// Give up a flow whose descriptor closed, or a listener whose port is + /// being unbound. + /// + /// Carries no reply: the sender is a task that is ending and has nothing + /// left to do with the answer. It is sent from the task rather than from a + /// `Drop`, so it can be awaited and cannot be silently lost to a full + /// channel. + Release { + /// Flows to forget. + flows: Vec, + /// Listener ports to free, along with anything pending on them. + listeners: Vec, + }, + + /// Undo a flow the listener's task promoted but could not hand over. + /// + /// Separate from [`NativeMessage::Release`] because the two count + /// differently: a release is a flow a client finished with, and this is one + /// no client ever held. Sending one message for both halves of the undo + /// keeps the registry entry and the counter from disagreeing. + Discard { + /// The flow to forget. + key: FlowKey, + /// Why it could not be handed over. + reason: DropReason, + }, + + /// **Debug.** Deliver a datagram as though it had arrived from the mesh. + /// + /// This drives the same [`Registry::deliver`](super::registry::Registry::deliver) + /// the FSP receive path will call, so the dispatch rule is exercised before + /// the wire exists and the wire, when it lands, changes the caller rather + /// than the rule. + Arrive { + /// The peer it appears to come from, by wire address. + peer: NodeAddr, + /// That peer's key, decoded from the npub the caller named. Client + /// asserted rather than authenticated, which is one of the reasons the + /// command is gated. + pubkey: XOnlyPublicKey, + /// Its source port. + src: u16, + /// Its destination port on this node. + dst: u16, + /// The payload. + data: Datagram, + /// What the registry decided to do with it. + reply: oneshot::Sender, + }, +} + +/// One datagram a client wrote to its descriptor, on its way to the mesh. +/// +/// Travels on its own channel rather than through [`NativeMessage`], so a burst +/// of client traffic cannot delay a registration and the data arm can drain in +/// batches the way the TUN arm does. +#[derive(Debug)] +pub struct Outbound { + /// The flow it belongs to, which carries both ports and the destination. + pub key: FlowKey, + /// The destination's address. Always known: a connected flow decoded it + /// from the npub its client named, and an accepted one took it from the + /// session that authenticated the peer. + pub peer: XOnlyPublicKey, + /// The payload, with no port header: the send path adds that. + pub payload: Datagram, +} + +/// A flow the client opened. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Opened { + /// The local port it holds. + pub local: u16, + /// The identifier the client names it by. + pub flow: u64, + /// The largest payload it may send, in bytes. + pub max: u16, +} + +/// A flow the client accepted from a listener. +#[derive(Debug)] +pub struct Accepted { + /// Which flow, in both directions. + pub key: FlowKey, + /// The peer's address. + pub peer: XOnlyPublicKey, + /// Whatever arrived before the client answered. + pub held: Vec, + /// The largest payload it may send, in bytes. + pub max: u16, +} + +/// Why a datagram was not delivered, across every path that can refuse one. +/// +/// [`DropCause`] covers the refusals the registry decides. Delivery can also +/// fail after the registry has agreed, when a bounded channel to a client is +/// full, and those three cases are the remaining variants. Keeping one type +/// over the whole set is what lets a counter match be exhaustive: a match over +/// `DropCause` alone would silently miss the case a slow client actually causes. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DropReason { + /// No listener and no flow holds the destination port. + NoPort, + /// A listener holds the port but will not hold another pending flow. + BacklogFull, + /// The node is at its flow ceiling. + TooManyFlows, + /// A flow awaiting accept will hold no more datagrams. + PendingQueueFull, + /// An established flow's client is not draining its descriptor. + FlowQueueFull, + /// A listener's client is not reading the arrivals it asked for. The flow + /// is unregistered as well, so nothing is left pending that nobody knows of. + ArrivalQueueFull, + /// A listener's client is not reading its descriptor, so the arrival could + /// not be written to it. Distinct from `ArrivalQueueFull`: that one is the + /// rx_loop refusing before anything was opened, and this one is a flow the + /// daemon had already wired and has to take apart again. + ListenerNotReading, + /// A listener's client closed its descriptor between the arrival being + /// taken off the queue and being written to it. The same cleanup as + /// `ListenerNotReading` and a different counter: this one is a race a + /// healthy client can lose, and that one is a client falling behind. + ListenerGone, +} + +impl From for DropReason { + fn from(cause: DropCause) -> Self { + match cause { + DropCause::NoPort => DropReason::NoPort, + DropCause::BacklogFull => DropReason::BacklogFull, + DropCause::TooManyFlows => DropReason::TooManyFlows, + DropCause::QueueFull => DropReason::PendingQueueFull, + } + } +} + +impl DropReason { + /// The client-facing text for this reason. + /// + /// `PendingQueueFull` and `FlowQueueFull` deliberately render alike. They + /// are distinct to a counter and indistinguishable to a client, which is + /// what keeps the strings the debug `arrive` command already answers with + /// unchanged. + pub fn as_str(self) -> &'static str { + match self { + DropReason::NoPort => "no listener or flow on that port", + DropReason::BacklogFull => "listener backlog full", + DropReason::TooManyFlows => "node flow ceiling reached", + DropReason::PendingQueueFull | DropReason::FlowQueueFull => "queue full", + DropReason::ArrivalQueueFull => "arrival queue full", + DropReason::ListenerNotReading => "listener not reading arrivals", + DropReason::ListenerGone => "listener closed", + } + } +} + +/// What became of a delivered datagram. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Outcome { + /// An established flow received it. + Delivered, + /// A listener was told a new peer arrived, and the datagram was held for + /// whoever accepts it. + Announced(u64), + /// A flow already announced held it while it waits to be accepted. + Held(u64), + /// Nothing took it. + Dropped(DropReason), +} + +/// What one served request did, so the shell can count it. +/// +/// The core decides and answers the client on the request's own channel; this +/// exists only so the node can bump a counter without the core having to know +/// what a counter is. Only outcomes something counts are named; everything else +/// is [`Served::Untracked`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Served { + /// A client opened a flow to a peer it named. + Opened, + /// A listener's task took a pending flow on a client's behalf. + Accepted, + /// A flow the daemon wired and could not hand over was taken apart again. + Discarded(DropReason), + /// This many established flows were given back when their descriptors + /// closed or their listener unbound. + Released(usize), + /// A datagram of this many bytes was dispatched by the debug arrival + /// command, which runs the same rule the wire does and is counted the same + /// way. The length rides along because `deliver` consumes the datagram. + Delivered(Outcome, usize), + /// Nothing counted: a listen, or a request the registry refused. + Untracked, +} + +/// Apply one request to `registry` and answer it. +/// +/// A free function over the registry rather than a method on the node: the node +/// supplies only `now`, and everything else here is a decision plus the sends +/// that decision names. That is what lets this be driven in a test without +/// building a node, and it is the same code the running daemon executes. +pub fn serve(registry: &mut Registry, message: NativeMessage, now: u64, max: u16) -> Served { + match message { + NativeMessage::Listen { + port, + arrivals, + reply, + } => { + let _ = reply.send(registry.listen(port, arrivals)); + Served::Untracked + } + + NativeMessage::Discard { key, reason } => { + registry.release(&key); + Served::Discarded(reason) + } + + NativeMessage::Connect { + peer, + remote, + local, + sink, + reply, + } => { + let opened = registry + .connect(peer, remote, local, sink, now) + .map(|(local, flow)| Opened { local, flow, max }); + let served = if opened.is_ok() { + Served::Opened + } else { + Served::Untracked + }; + let _ = reply.send(opened); + served + } + + NativeMessage::Accept { flow, sink, reply } => { + let accepted = registry + .accept(flow, sink, now) + .map(|(key, peer, held)| Accepted { + key, + peer, + held, + max, + }); + let served = if accepted.is_ok() { + Served::Accepted + } else { + Served::Untracked + }; + let _ = reply.send(accepted); + served + } + + NativeMessage::Release { flows, listeners } => { + let mut closed = 0; + for key in &flows { + if registry.release(key) { + closed += 1; + } + } + for port in &listeners { + registry.release_listener(*port); + } + Served::Released(closed) + } + + NativeMessage::Arrive { + peer, + pubkey, + src, + dst, + data, + reply, + } => { + let bytes = data.len(); + let outcome = deliver(registry, peer, pubkey, src, dst, data, now); + let _ = reply.send(outcome); + Served::Delivered(outcome, bytes) + } + } +} + +/// Deliver one inbound datagram to whatever owns its destination port. +/// +/// The FSP receive path calls this once the wire is connected; until then the +/// debug arrival command is its only caller. Either way the decision is the +/// registry's and this only performs it. +pub fn deliver( + registry: &mut Registry, + peer: NodeAddr, + pubkey: XOnlyPublicKey, + src: u16, + dst: u16, + data: Datagram, + now: u64, +) -> Outcome { + match registry.deliver(peer, pubkey, src, dst, now) { + Delivery::Flow(sink) => { + // Never block the receive path on a slow client. A full queue costs + // that client a datagram, not the node its tick. + match sink.try_send(data) { + Ok(()) => Outcome::Delivered, + Err(_) => { + trace!(dst, "Native API flow queue full, dropping datagram"); + Outcome::Dropped(DropReason::FlowQueueFull) + } + } + } + + Delivery::Arrived(arrivals, arrival) => { + let flow = arrival.flow; + if arrivals.try_send(arrival).is_err() { + // The client is not reading its own announcements. Undo the + // registration rather than leaving a pending flow nobody will + // ever be told about. + let _ = registry.reject(flow); + return Outcome::Dropped(DropReason::ArrivalQueueFull); + } + if registry.hold(flow, data) { + Outcome::Announced(flow) + } else { + Outcome::Dropped(DropReason::PendingQueueFull) + } + } + + Delivery::Pending(flow) => { + if registry.hold(flow, data) { + Outcome::Held(flow) + } else { + Outcome::Dropped(DropReason::PendingQueueFull) + } + } + + Delivery::Drop(cause) => Outcome::Dropped(cause.into()), + } +} diff --git a/src/native/mod.rs b/src/native/mod.rs new file mode 100644 index 00000000..87ec2f16 --- /dev/null +++ b/src/native/mod.rs @@ -0,0 +1,2148 @@ +//! Native datagram API socket (experimental). +//! +//! A client connects to this socket, sends line-delimited JSON commands, and +//! receives line-delimited JSON replies. Both replies that succeed carry a file +//! descriptor in their ancillary data: a flow's descriptor for `connect`, a +//! listener's for `listen`. The client reads and writes datagrams on a flow's +//! descriptor with no framing of any kind, and reads one message per arriving +//! flow on a listener's. +//! +//! The API addresses a peer by public key and a service by FSP port, so a +//! datagram travels from key to key with no IPv6 emulation and no TUN device. +//! Three representations of a peer exist and only two of them are a client's: +//! the x-only public key is the address, the npub is that key written down, and +//! the 16-byte node address is a truncated hash that travels on the wire and +//! appears in no field this API hands a client. +//! +//! **The RPC connection owns nothing.** It carries setup calls and their +//! replies and then has no further part in anything it opened. A flow lives +//! until its own descriptor reaches end of file and a listener until its own +//! does, whichever task holds them, which is what makes a descriptor this API +//! hands back behave like one a syscall would have. +//! +//! **The wire is connected.** A datagram a client writes leaves this node over +//! FSP, and one arriving on a held port reaches its flow. `max_payload` is the +//! real limit: the transport MTU less the FIPS encapsulation and the four-byte +//! port header. +//! +//! **Platform support.** The listener is built on Linux and FreeBSD only, and +//! two separate things bound that. Windows has no `SCM_RIGHTS` and so no way to +//! hand a descriptor to another process at all. macOS has `SCM_RIGHTS` but does +//! not implement `SOCK_SEQPACKET` for `AF_UNIX`, so only the descriptor's +//! socket type is missing there; see [`seqpacket`] for what a macOS port would +//! have to settle. The gate is explicit rather than `cfg(unix)` so macOS fails +//! to build here instead of failing at `socketpair` on a running node. +//! +//! - `protocol.rs` — the command types and the pure decisions over them. No +//! I/O, no node state. +//! - `registry.rs` — which local ports are held and where an inbound datagram +//! goes. Lives inside `Node`; decisions only. +//! - `link.rs` — what a client task asks the node's registry to do. +//! - `seqpacket.rs` — the socket pair behind a flow and behind a listener, the +//! daemon half under the reactor. +//! - `fdpass.rs` — replies, arrival messages, and the `SCM_RIGHTS` hand-off of +//! a descriptor, in both directions. +//! - `client/` — a blocking, std-only client an external program links this +//! crate for, so it speaks the API without knowing the line protocol. + +// The registry, its request types and the command types are plain decisions +// over maps and carry no socket. They build everywhere, which is what lets the +// `rx_loop` arm that serves them avoid a `cfg` — `tokio::select!` does not +// accept one. Only the listener and the descriptor machinery are gated. +pub mod link; +pub mod protocol; +pub mod registry; + +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +pub mod client; +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +pub mod fdpass; +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +pub mod seqpacket; + +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +pub use unix_impl::NativeApi; + +#[cfg(any(target_os = "linux", target_os = "freebsd"))] +mod unix_impl { + use super::fdpass; + use super::link::{Accepted, DropReason, NativeMessage, Outbound, Outcome}; + use super::protocol::{self, Command, Connect, Inject, Listen}; + use super::registry::{Arrival, Datagram, FlowKey, Limits, RegistryError}; + use super::seqpacket::{Received, Seqpacket, pair, set_sndbuf}; + use crate::config::NativeApiConfig; + use crate::control::protocol::{Request, Response}; + use crate::identity::{NodeAddr, decode_npub, encode_npub}; + use secp256k1::XOnlyPublicKey; + use std::collections::HashMap; + use std::os::fd::{AsFd, OwnedFd}; + use std::path::PathBuf; + use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; + use std::sync::{Arc, Mutex}; + use tokio::io::BufReader; + use tokio::net::{UnixListener, UnixStream}; + use tokio::sync::{mpsc, oneshot}; + use tracing::{debug, info, warn}; + + /// Largest command line a client may send, in bytes. + /// + /// A command carries an npub, two ports and nothing else, so this is + /// generous by two orders of magnitude. It exists to bound a client that + /// never sends a newline, not to constrain a real command. + pub(super) const MAX_COMMAND: usize = 8192; + + /// Receive buffer for a flow, in bytes. + /// + /// A datagram longer than this is truncated by the kernel, which is + /// `SOCK_SEQPACKET` behaviour. Sized well above anything the wire will + /// carry, because the real limit arrives with the wire and this buffer + /// should not be the thing that imposes one first. + const MAX_DATAGRAM: usize = 65535; + + /// Send-buffer bytes allowed per unread arrival on a listener's pair. + /// + /// `SO_SNDBUF` on an `AF_UNIX` socket accounts bytes plus per-message + /// overhead rather than messages, so this is a byte ceiling standing in for + /// a message count. It is set generously on purpose: the approximation must + /// err toward accepting an arrival a client would have read rather than + /// toward dropping it. An arrival message is a few hundred bytes, so this + /// leaves an order of magnitude of slack, and the number that would settle + /// it is the real per-message accounting measured on the target kernel. + /// + /// **The slack is what a wedged listener holds.** At a `backlog` of 16 this + /// asks for 64 KB, which Linux doubles, and `AF_UNIX` charges each message + /// its `skb->truesize` rather than its length, so the buffer takes arrivals + /// by the hundred rather than by the `backlog`. A client that binds a + /// listener and then stops reading therefore holds flows on the order of + /// `node.native_api.max_flows`, not on the order of `backlog`. Sizing this + /// down would trade that for dropping arrivals a client would have read, + /// which is the worse failure, so the node-wide ceiling is what bounds it. + const ARRIVAL_ALLOWANCE: usize = 4096; + + /// The native API listener. + /// + /// Binding is separate from serving so the caller can bind synchronously + /// during startup and see the failure there, in a deterministic order, + /// rather than at whatever later moment a spawned bind happened to run. + pub struct NativeApi { + listener: UnixListener, + socket_path: PathBuf, + limits: Limits, + /// Every flow this node holds a daemon-side half for. + flows: Arc, + /// Whether this node answers the debug commands at all. + debug: bool, + } + + impl NativeApi { + /// Bind the native API socket under the shared FIPS access policy. + /// + /// The socket is mode `0o770` and group `fips`, which is the whole of + /// the authorization model: any process that can open it can send as + /// this node's identity. That is why the feature is off by default. + pub fn bind(config: &NativeApiConfig) -> Result { + let socket_path = PathBuf::from(&config.socket_path); + let listener = crate::utils::sockbind::bind(&socket_path, "native API")?; + + info!(path = %socket_path.display(), "Native API socket listening"); + + Ok(Self { + listener, + socket_path, + limits: Limits { + per_flow: config.pending_per_flow, + backlog: config.backlog, + max_flows: config.max_flows, + }, + flows: Arc::new(Flows::new()), + debug: config.debug_commands, + }) + } + + /// Accept connections until the task is cancelled. + /// + /// Each connection is served by its own task and ends with its last + /// command. `npub` is this node's own address, reported in every setup + /// reply so a client can answer `getsockname` without asking again. + pub async fn accept_loop( + self, + node: mpsc::Sender, + outbound: mpsc::Sender, + npub: String, + ) { + let npub: Arc = Arc::from(npub); + loop { + let (stream, _addr) = match self.listener.accept().await { + Ok(conn) => conn, + Err(error) => { + warn!(error = %error, "Native API accept failed"); + continue; + } + }; + + let node = node.clone(); + let outbound = outbound.clone(); + let flows = Arc::clone(&self.flows); + let limits = self.limits; + let npub = Arc::clone(&npub); + let debug = self.debug; + tokio::spawn(async move { + if let Err(error) = + serve(stream, node, outbound, flows, limits, npub, debug).await + { + debug!(error = %error, "Native API connection ended"); + } + }); + } + } + + /// The path this socket is bound to. + pub fn socket_path(&self) -> &PathBuf { + &self.socket_path + } + } + + impl Drop for NativeApi { + fn drop(&mut self) { + crate::utils::sockbind::cleanup(&self.socket_path, "native API"); + } + } + + /// What a flow's reader task has observed, published for `stats`. + #[derive(Debug, Default)] + struct Counts { + datagrams: AtomicU64, + bytes: AtomicU64, + closed: AtomicBool, + } + + /// One flow the node holds the daemon-side half of. + struct Flow { + /// The local port, reported by `stats`. + local: u16, + /// The daemon's half of the client's socket pair, which `inject` + /// writes into. + sock: Arc, + counts: Arc, + } + + /// What `stats` reports about one flow. + struct FlowStats { + local: u16, + datagrams: u64, + bytes: u64, + closed: bool, + } + + /// Every flow the node holds a daemon-side half for. + /// + /// Node-scoped rather than connection-scoped, because a flow outlives the + /// RPC connection that opened it and a flow a listener accepted was never + /// opened on an RPC connection at all. A plain mutex is enough: every + /// operation is one map lookup, taken from a client or listener task and + /// never from the `rx_loop`. + struct Flows { + held: Mutex>, + } + + impl Flows { + /// An empty table. + fn new() -> Self { + Self { + held: Mutex::new(HashMap::new()), + } + } + + /// The table, recovering rather than propagating a poisoned lock. + /// + /// A client task that panicked mid-lookup left the map intact, and + /// refusing every later flow because of it would turn one client's bug + /// into the node's. + fn table(&self) -> std::sync::MutexGuard<'_, HashMap> { + self.held + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + } + + /// Record a flow the node now holds. + fn record(&self, id: u64, flow: Flow) { + self.table().insert(id, flow); + } + + /// Forget a flow whose descriptor has closed. + fn forget(&self, id: u64) { + self.table().remove(&id); + } + + /// What the debug `stats` command reports, or `None` for a flow this + /// node does not hold. + fn stats(&self, id: u64) -> Option { + let table = self.table(); + let flow = table.get(&id)?; + Some(FlowStats { + local: flow.local, + datagrams: flow.counts.datagrams.load(Ordering::Relaxed), + bytes: flow.counts.bytes.load(Ordering::Relaxed), + closed: flow.counts.closed.load(Ordering::Relaxed), + }) + } + + /// The daemon's half of a flow, cloned out so the caller can write to + /// it without holding the lock across an await. + fn sock(&self, id: u64) -> Option> { + self.table().get(&id).map(|flow| Arc::clone(&flow.sock)) + } + } + + /// A flow's socket pair and inbound queue, before its key is known. + struct Wiring { + sock: Arc, + inbound: mpsc::Receiver, + } + + /// State for one client connection. + /// + /// It holds no flow and no listener: both outlive it, and the connection is + /// a setup channel with nothing to give back when it ends. + pub(super) struct Connection { + /// Where registry requests go. + node: mpsc::Sender, + /// Where datagrams a client wrote go. + outbound: mpsc::Sender, + /// The node's flow table, shared with every listener task. + flows: Arc, + limits: Limits, + /// This node's own npub, reported in every setup reply. + npub: Arc, + /// Whether the debug commands are answered, from + /// `node.native_api.debug_commands`. Held per connection because it is + /// fixed for the node's lifetime and reading it here costs no lookup. + debug: bool, + } + + /// The node is gone, so nothing further can be registered. + fn shutting_down() -> Response { + deny("ECONNREFUSED", "node is shutting down".to_string()) + } + + /// Render a registry refusal for the client. + fn refused(error: RegistryError) -> Response { + deny(protocol::errno(&error), error.to_string()) + } + + /// An error reply carrying the code a C binding would return from the + /// corresponding call. + /// + /// The code rides in `data` because [`Response`] has no field for one and + /// the control socket shares the type. A client reads `data.errno` and never + /// the message: the message is prose for an operator, and a reply carrying + /// no code at all is `ECONNREFUSED`, which is what a daemon older than this + /// contract sends. + /// + /// **Every refusal this API makes goes through here**, the debug commands + /// included. They have no Berkeley call to take a code from, so they use the + /// nearest one for the argument that was wrong; the alternative is a bare + /// message, which reaches the client as `ECONNREFUSED` and says the node + /// declined rather than that the flow does not exist. + fn deny(errno: &'static str, message: String) -> Response { + Response { + status: "error".to_string(), + data: Some(serde_json::json!({ "errno": errno })), + message: Some(message), + } + } + + impl Connection { + /// A connection holding nothing. + fn new( + node: mpsc::Sender, + outbound: mpsc::Sender, + flows: Arc, + limits: Limits, + npub: Arc, + debug: bool, + ) -> Self { + Self { + node, + outbound, + flows, + limits, + npub, + debug, + } + } + + /// Send one request to the node and wait for its answer. + async fn ask( + &self, + build: impl FnOnce(oneshot::Sender) -> NativeMessage, + ) -> Option { + let (tx, rx) = oneshot::channel(); + self.node.send(build(tx)).await.ok()?; + rx.await.ok() + } + + /// Build the descriptor pair and the inbound queue for one flow. + /// + /// Returns the client's half plus everything the node keeps. The socket + /// pair is created before the flow is registered, so a failure here + /// leaves the registry untouched and needs no rollback. + fn wire(&self) -> std::io::Result<(Wiring, OwnedFd, mpsc::Sender)> { + wire_flow(self.limits.per_flow) + } + + /// Decide the reply to one command, and any descriptor that goes with it. + /// + /// Every path returns a `Response` rather than an error, because a bad + /// command from one client is not a reason to drop its connection or to + /// disturb any other. + pub(super) async fn answer(&mut self, line: &[u8]) -> (Response, Option) { + let request: Request = match serde_json::from_slice(line) { + Ok(request) => request, + Err(error) => { + return (deny("EINVAL", format!("invalid request: {error}")), None); + } + }; + + let command = match protocol::parse(&request) { + Ok(command) => command, + Err(error) => { + return ( + deny(protocol::command_errno(&error), error.to_string()), + None, + ); + } + }; + + // Refused by name rather than reported as unknown: a client driving + // the harness needs to tell "this node will not do that" from "this + // build has no such command", and hiding the difference would send + // whoever hits it looking for a typo. + if let Some(name) = command.debug_name() + && !self.debug + { + return ( + deny( + "ECONNREFUSED", + format!( + "'{name}' is a debug command and is disabled; \ + set node.native_api.debug_commands to enable it" + ), + ), + None, + ); + } + + match command { + Command::Connect(connect) => self.connect(connect).await, + Command::Listen(listen) => self.listen(listen).await, + Command::Stats(flow) => (self.stats(flow), None), + Command::Inject(inject) => (self.inject(inject).await, None), + Command::Arrive(arrive) => (self.arrive(arrive).await, None), + } + } + + /// Open a flow to a named peer. + async fn connect(&mut self, connect: Connect) -> (Response, Option) { + let peer = match decode_npub(&connect.peer) { + Ok(key) => key, + Err(error) => { + // The address did not parse. The client library refuses this + // before it sends, so reaching it means a caller wrote the + // line itself. + return (deny("EINVAL", format!("invalid peer: {error}")), None); + } + }; + + // The pair exists before the port is claimed, so a failure to build + // it cannot leave a port held by a flow that does not exist. + let (wiring, fd, sink) = match self.wire() { + Ok(parts) => parts, + Err(error) => { + // No descriptor to give, which is what `EMFILE` says about a + // socket call whatever the underlying cause was. + return ( + deny("EMFILE", format!("could not open a flow: {error}")), + None, + ); + } + }; + + let answer = self + .ask(|reply| NativeMessage::Connect { + peer, + remote: connect.remote, + local: connect.local, + sink, + reply, + }) + .await; + + let opened = match answer { + Some(Ok(opened)) => opened, + Some(Err(error)) => return (refused(error), None), + None => return (shutting_down(), None), + }; + + let key = FlowKey { + peer: NodeAddr::from_pubkey(&peer), + remote: connect.remote, + local: opened.local, + }; + start( + opened.flow, + key, + peer, + wiring, + &self.outbound, + &self.node, + &self.flows, + ); + + ( + Response::ok(serde_json::json!({ + "flow_id": opened.flow, + "local_port": opened.local, + "remote_port": connect.remote, + // The daemon's own re-encode of the key it decoded, not an + // echo of the client's string, so connect and an arrival + // name one peer the same way and one reader serves both. + "peer": encode_npub(&peer), + "node": self.npub.as_ref(), + "max_payload": opened.max, + })), + Some(fd), + ) + } + + /// Hold a local port and receive the flows that arrive on it. + /// + /// The reply carries the listener's own descriptor, which is what makes + /// a listener pollable and makes accepting a `recvmsg` rather than a + /// round trip on this connection. + async fn listen(&mut self, listen: Listen) -> (Response, Option) { + // The reply reports what the registry enforces, which is the raw + // configured depth: a reply claiming one while the registry refused + // every arrival is worse than a refusal at startup. The channel + // capacity is that number floored at one, because `mpsc::channel(0)` + // panics and a `Limits` built outside the config loader, as a test's + // is, is not bound by the loader's floor. + let backlog = self.limits.backlog; + let queue = backlog.max(1); + let (ours, theirs) = match pair() { + Ok(pair) => pair, + Err(error) => { + return ( + deny("EMFILE", format!("could not open a listener: {error}")), + None, + ); + } + }; + + // The buffer is the only bound on arrivals a client has stopped + // reading, so a failure to size it is worth a warning rather than a + // refusal: the listener still works, with the system default. + if let Err(error) = set_sndbuf(&ours, queue * ARRIVAL_ALLOWANCE) { + warn!(error = %error, "Could not size a native API listener's send buffer"); + } + + let sock = match Seqpacket::new(ours) { + Ok(sock) => sock, + Err(error) => { + return ( + deny("EMFILE", format!("could not open a listener: {error}")), + None, + ); + } + }; + + let (arrivals, incoming) = mpsc::channel::(queue); + let answer = self + .ask(|reply| NativeMessage::Listen { + port: listen.local, + arrivals, + reply, + }) + .await; + + let port = match answer { + Some(Ok(port)) => port, + Some(Err(error)) => return (refused(error), None), + None => return (shutting_down(), None), + }; + + tokio::spawn(watch( + port, + sock, + incoming, + self.node.clone(), + self.outbound.clone(), + Arc::clone(&self.flows), + self.limits, + Arc::clone(&self.npub), + )); + + ( + Response::ok(serde_json::json!({ + "local_port": port, + "node": self.npub.as_ref(), + "backlog": backlog, + })), + Some(theirs), + ) + } + + /// Report what a flow's reader task has seen. + fn stats(&self, flow: u64) -> Response { + match self.flows.stats(flow) { + Some(stats) => Response::ok(serde_json::json!({ + "flow_id": flow, + "local_port": stats.local, + "rx_datagrams": stats.datagrams, + "rx_bytes": stats.bytes, + "closed": stats.closed, + })), + None => deny("ENOENT", format!("no such flow: {flow}")), + } + } + + /// Write bytes into a flow from the daemon's side. + /// + /// Addressed node-wide, because a flow is the node's and not the + /// connection's. Any process that can open the socket can therefore + /// write into any flow on the node, its own or another client's. That + /// is why the command is behind `node.native_api.debug_commands`, which + /// no packaged node has: a caller that can reach it can already deliver + /// to any listener on the node under any identity it names. + async fn inject(&self, inject: Inject) -> Response { + let Some(sock) = self.flows.sock(inject.flow) else { + return deny("ENOENT", format!("no such flow: {}", inject.flow)); + }; + + for index in 0..inject.repeat { + if let Err(error) = sock.send(&inject.data).await { + return deny( + "EIO", + format!("wrote {index} of {} datagrams: {error}", inject.repeat), + ); + } + } + + Response::ok(serde_json::json!({ + "flow_id": inject.flow, + "datagrams": inject.repeat, + "bytes": inject.data.len() as u64 * u64::from(inject.repeat), + })) + } + + /// Deliver a datagram as though it had arrived from the mesh. + /// + /// Drives the same registry decision the FSP receive path does, so the + /// dispatch rule is exercised without a peer. The peer's key is decoded + /// from the npub the caller named, which makes it client-asserted + /// rather than authenticated; this is the one path where it is, and it + /// is one of the reasons the command is gated. + async fn arrive(&self, arrive: super::protocol::Arrive) -> Response { + let pubkey = match decode_npub(&arrive.peer) { + Ok(key) => key, + Err(error) => return deny("EINVAL", format!("invalid peer: {error}")), + }; + + let answer = self + .ask(|reply| NativeMessage::Arrive { + peer: NodeAddr::from_pubkey(&pubkey), + pubkey, + src: arrive.src, + dst: arrive.dst, + data: arrive.data, + reply, + }) + .await; + + match answer { + Some(outcome) => Response::ok(serde_json::json!({ + "outcome": match outcome { + Outcome::Delivered => "delivered".to_string(), + Outcome::Announced(_) => "announced".to_string(), + Outcome::Held(_) => "held".to_string(), + Outcome::Dropped(why) => format!("dropped: {}", why.as_str()), + }, + "flow_id": match outcome { + Outcome::Announced(flow) | Outcome::Held(flow) => Some(flow), + _ => None, + }, + })), + None => shutting_down(), + } + } + } + + #[cfg(test)] + impl Connection { + /// Build a connection wired to a channel a test serves. + pub(super) fn for_test( + node: mpsc::Sender, + outbound: mpsc::Sender, + limits: Limits, + debug: bool, + ) -> Self { + Self::new( + node, + outbound, + Arc::new(Flows::new()), + limits, + Arc::from(super::tests::NODE), + debug, + ) + } + + /// Wait until a flow's reader task has counted `want` datagrams. + /// + /// The reader runs on the same runtime, so this yields rather than + /// asserting into a race a test would lose intermittently. + pub(super) async fn settle(&self, flow: u64, want: u64) { + for _ in 0..1000 { + match self.flows.stats(flow) { + Some(stats) if stats.datagrams >= want => return, + _ => tokio::task::yield_now().await, + } + } + } + + /// Wait until a flow's reader task has observed the client's close. + pub(super) async fn settle_closed(&self, flow: u64) { + for _ in 0..1000 { + match self.flows.stats(flow) { + Some(stats) if stats.closed => return, + _ => tokio::task::yield_now().await, + } + } + } + } + + /// Build one flow's socket pair and inbound queue. + /// + /// A free function rather than a method because the listener's task needs + /// it too, and a listener has no connection to reach it through. + fn wire_flow(per_flow: usize) -> std::io::Result<(Wiring, OwnedFd, mpsc::Sender)> { + let (ours, theirs) = pair()?; + let sock = Arc::new(Seqpacket::new(ours)?); + let (sink, inbound) = mpsc::channel::(per_flow.max(1)); + Ok((Wiring { sock, inbound }, theirs, sink)) + } + + /// Start a flow's tasks now that its key is final, and record it node-side. + /// + /// The reader cannot start any earlier: it stamps every datagram it + /// forwards with the flow's key and the peer's address, and the local port + /// is not known until the registry has answered. + fn start( + id: u64, + key: FlowKey, + peer: XOnlyPublicKey, + wiring: Wiring, + outbound: &mpsc::Sender, + node: &mpsc::Sender, + flows: &Arc, + ) { + let counts = Arc::new(Counts::default()); + tokio::spawn(drain( + id, + key, + peer, + Arc::clone(&wiring.sock), + Arc::clone(&counts), + outbound.clone(), + node.clone(), + Arc::clone(flows), + )); + tokio::spawn(feed(Arc::clone(&wiring.sock), wiring.inbound)); + flows.record( + id, + Flow { + local: key.local, + sock: wiring.sock, + counts, + }, + ); + } + + /// Serve one listener until its client closes the descriptor. + /// + /// Two arms, and the second is not optional: end of file is observable only + /// by reading, so a listener that never called `recv` on its own half could + /// not notice its client had gone and would hold the port for the node's + /// lifetime. A client has no message to send here, so anything the read arm + /// produces other than end of file is discarded rather than answered: a + /// listener that replied would be a second protocol on a socket that has + /// none. + #[allow( + clippy::too_many_arguments, + reason = "one hand-off of plumbing to a task, with no state to group" + )] + async fn watch( + port: u16, + sock: Seqpacket, + mut incoming: mpsc::Receiver, + node: mpsc::Sender, + outbound: mpsc::Sender, + flows: Arc, + limits: Limits, + npub: Arc, + ) { + /// What the listener's select produced. + enum Next { + /// A peer arrived on this listener's port. + Arrival(Arrival), + /// The client closed the listener, or the node went away. + Done, + /// The client wrote something a listener has no use for. + Ignored, + } + + let mut buf = vec![0u8; MAX_DATAGRAM]; + loop { + // Both arms are cancel-safe. `mpsc::Receiver::recv` is documented + // so, and `Seqpacket::recv` holds nothing across a cancellation: it + // awaits readiness and then performs one `recvmsg`, and + // `SOCK_SEQPACKET` has no partial message to lose. This is the one + // property this task's shape depends on. + let next = tokio::select! { + arrival = incoming.recv() => match arrival { + Some(arrival) => Next::Arrival(arrival), + None => Next::Done, + }, + received = sock.recv(&mut buf) => match received { + Ok(Received::Eof) | Err(_) => Next::Done, + Ok(Received::Datagram(_)) => Next::Ignored, + }, + }; + + match next { + Next::Done => break, + Next::Ignored => continue, + Next::Arrival(arrival) => { + hand_over(arrival, &sock, &node, &outbound, &flows, limits, &npub).await; + } + } + } + + debug!(port, "Native API listener closed by its client"); + let _ = node + .send(NativeMessage::Release { + flows: Vec::new(), + listeners: vec![port], + }) + .await; + } + + /// Wire one arriving flow and hand its descriptor to the listener's client. + /// + /// Promotion comes first because taking the flow and taking what it held + /// are one registry operation. Then the held datagrams go onto the flow's + /// half and the arrival message onto the listener's, in that order, so + /// whatever arrived before the client could read the arrival is already on + /// the descriptor by the time the client holds it. + /// + /// Both writes are try-sends. They go onto a socket pair whose client half + /// has not been sent yet, so no process can read either one and a task that + /// parked on one would stop serving this listener entirely. + async fn hand_over( + arrival: Arrival, + listener: &Seqpacket, + node: &mpsc::Sender, + outbound: &mpsc::Sender, + flows: &Arc, + limits: Limits, + npub: &str, + ) { + // A failure here leaves the flow pending, which the registry's own + // deadline reclaims and counts. Nothing has been promoted yet, so there + // is nothing to undo. + let (wiring, theirs, sink) = match wire_flow(limits.per_flow) { + Ok(parts) => parts, + Err(error) => { + warn!(error = %error, "Could not open a socket pair for an arriving flow"); + return; + } + }; + + let (tx, rx) = oneshot::channel(); + if node + .send(NativeMessage::Accept { + flow: arrival.flow, + sink, + reply: tx, + }) + .await + .is_err() + { + return; + } + let accepted: Accepted = match rx.await { + Ok(Ok(accepted)) => accepted, + // The flow expired between the announcement and this hop, or the + // node is shutting down. Either way there is nothing to hand over. + Ok(Err(error)) => { + debug!(error = %error, "An arriving flow was gone before it could be wired"); + return; + } + Err(_) => return, + }; + + // The byte counts are dropped rather than checked: `SOCK_SEQPACKET` + // writes a message whole or fails, so a short write is not a state + // this can be in. + let failed = perform(arrival.flow, &accepted, npub, |step| match step { + Handoff::Held(datagram) => { + fdpass::try_send(wiring.sock.raw(), datagram, None).map(drop) + } + Handoff::Arrival(message) => { + fdpass::try_send(listener.raw(), message, Some(theirs.as_fd())).map(drop) + } + }) + .err(); + + if let Some(reason) = failed { + // Three parts, and all of them are required: closing the pair, + // giving the registry entry and its port claim back, and counting + // the drop. Missing any one leaks a descriptor, a port, or both, + // once per unread arrival, at a rate a remote peer sets. + drop(wiring); + drop(theirs); + let _ = node + .send(NativeMessage::Discard { + key: accepted.key, + reason, + }) + .await; + return; + } + + start( + arrival.flow, + accepted.key, + accepted.peer, + wiring, + outbound, + node, + flows, + ); + // Dropping our copy leaves the client holding the only reference to its + // half, so its close tears the flow down. + drop(theirs); + } + + /// What a failed hand-off write says about the client, for the counter. + /// + /// Both take the same three-part cleanup, so this decides only what an + /// operator is told: `EPIPE` is a client that closed its listener between + /// the arrival being taken off the queue and this write, which a healthy + /// client can lose, and anything else is a full send buffer, which is a + /// client that stopped reading. Reporting the two alike would leave a + /// normal close looking like a fault. + pub(super) fn why(error: &std::io::Error) -> DropReason { + if error.kind() == std::io::ErrorKind::BrokenPipe { + DropReason::ListenerGone + } else { + DropReason::ListenerNotReading + } + } + + /// Perform one arriving flow's hand-off, stopping at the first failed write. + /// + /// Builds the plan and consumes it here rather than taking one from the + /// caller, so no caller has a sequence it could reorder. That is the whole + /// reason this is a function: with the plan built in [`handoff`] and + /// performed inline in [`hand_over`], reversing the consumption was + /// invisible to every test, because both writes are synchronous and a + /// client reading afterwards sees the same bytes either way. Reversed, a + /// successful arrival write followed by a failed held write hands the + /// client a live descriptor for a flow the caller then tears down. + /// + /// `write` is a parameter for the same reason: the real writes need two + /// live socket pairs and a descriptor to pass, and a test that supplies + /// them can observe the result but not the order. + /// + /// Returns the reason for the failed write, ready for the caller's counter. + pub(super) fn perform( + flow: u64, + accepted: &Accepted, + npub: &str, + mut write: W, + ) -> Result<(), DropReason> + where + W: FnMut(&Handoff<'_>) -> std::io::Result<()>, + { + for step in handoff(flow, accepted, npub) { + write(&step).map_err(|error| why(&error))?; + } + Ok(()) + } + + /// One write of a hand-off, in the order the writes must happen. + pub(super) enum Handoff<'a> { + /// A datagram the node held for this flow before its client existed. + Held(&'a Datagram), + /// The arrival message, which carries the flow's descriptor with it. + Arrival(Vec), + } + + /// The writes that hand one arriving flow to a listener's client, in order. + /// + /// **The order is the guarantee.** Every held datagram is on the flow's + /// descriptor before the arrival message that carries that descriptor, so + /// whatever arrived before the client could read the arrival is already + /// there when the client holds it. Losing that drops a peer's opening + /// message, which this code has done once already. + /// + /// Built as a sequence rather than performed inline because the order is + /// then a value a test can read. Performed inline it is a race no test can + /// observe: both writes are synchronous and nothing can interleave between + /// them, so reversing them is invisible to any client that reads afterwards. + /// The sequence is consumed only by [`perform`], which is why no caller has + /// one to reorder. + /// + /// The arrival reports the whole held batch, because a held write that fails + /// ends the hand-off before the arrival is ever attempted. + pub(super) fn handoff<'a>(flow: u64, accepted: &'a Accepted, npub: &str) -> Vec> { + let mut steps: Vec> = accepted.held.iter().map(Handoff::Held).collect(); + steps.push(Handoff::Arrival(announce( + flow, + accepted, + npub, + accepted.held.len(), + ))); + steps + } + + /// The message a listener's client reads for one arriving flow. + /// + /// One `SOCK_SEQPACKET` message per arrival and **no trailing newline**: + /// the message boundary is the framing, and a newline would offer a client + /// a second one to rely on. The peer is named by npub, which is the address + /// its session authenticated; `node` is this node's own npub, carried so an + /// accepted stream can answer `getsockname` without consulting the listener + /// that produced it. + fn announce(id: u64, accepted: &Accepted, npub: &str, held: usize) -> Vec { + serde_json::to_vec(&serde_json::json!({ + "flow_id": id, + "peer": encode_npub(&accepted.peer), + "node": npub, + "local_port": accepted.key.local, + "remote_port": accepted.key.remote, + "max_payload": accepted.max, + "held": held, + })) + .unwrap_or_default() + } + + /// Drain one flow's daemon half, forwarding what the client writes. + /// + /// Counting continues alongside the forwarding: `stats` is how a check + /// observes that a datagram reached the daemon, independently of whether it + /// then reached a peer. + #[allow( + clippy::too_many_arguments, + reason = "one hand-off of plumbing to a task, with no state to group" + )] + async fn drain( + id: u64, + key: FlowKey, + peer: XOnlyPublicKey, + sock: Arc, + counts: Arc, + outbound: mpsc::Sender, + node: mpsc::Sender, + flows: Arc, + ) { + let mut buf = vec![0u8; MAX_DATAGRAM]; + loop { + match sock.recv(&mut buf).await { + Ok(Received::Datagram(len)) => { + counts.datagrams.fetch_add(1, Ordering::Relaxed); + counts.bytes.fetch_add(len as u64, Ordering::Relaxed); + let sent = outbound + .send(Outbound { + key, + peer, + payload: buf[..len].to_vec(), + }) + .await; + if sent.is_err() { + debug!(local = key.local, "Node is gone, ending this flow"); + // Still through `free`: the registry send fails too when + // the node is gone, but the node's own record of this + // flow is in this process and would otherwise outlive + // the task that owns it. + free(&node, &flows, id, key).await; + return; + } + } + Ok(Received::Eof) => { + counts.closed.store(true, Ordering::Relaxed); + debug!(local = key.local, "Native API flow closed by its client"); + free(&node, &flows, id, key).await; + return; + } + Err(error) => { + debug!(local = key.local, error = %error, "Native API flow read failed"); + free(&node, &flows, id, key).await; + return; + } + } + } + } + + /// Give one flow's registry entry back when its client is done with it. + /// + /// A flow lives until its own descriptor reaches end of file, whether it + /// was connected or accepted, so this is the only site that releases one. A + /// program serving a flow per exchange depends on it: without the per-flow + /// release it walks into the node's flow ceiling. + /// + /// The node's record goes at the same time, so `stats` on a closed flow + /// reports what a client would find with any other name: no such flow. + async fn free(node: &mpsc::Sender, flows: &Arc, id: u64, key: FlowKey) { + let _ = node + .send(NativeMessage::Release { + flows: vec![key], + listeners: Vec::new(), + }) + .await; + flows.forget(id); + } + + /// Write datagrams the node delivered onto the client's descriptor. + async fn feed(sock: Arc, mut inbound: mpsc::Receiver) { + while let Some(datagram) = inbound.recv().await { + if let Err(error) = sock.send(&datagram).await { + debug!(error = %error, "Native API flow write failed"); + return; + } + } + } + + /// Serve one client connection until it closes or misbehaves. + /// + /// The connection carries replies only, in command order. There is no event + /// on it and so no select over a partially-read command, which was not + /// cancellation-safe: an arrival becoming ready while the reader waited for + /// the rest of a line dropped the accumulated bytes, and the client's + /// half-command vanished with no reply and no error. + async fn serve( + stream: UnixStream, + node: mpsc::Sender, + outbound: mpsc::Sender, + flows: Arc, + limits: Limits, + npub: Arc, + debug: bool, + ) -> Result<(), std::io::Error> { + let mut connection = Connection::new(node, outbound, flows, limits, npub, debug); + let mut reader = BufReader::new(stream); + let mut line = Vec::new(); + + while read_command(&mut reader, &mut line).await? { + let (response, fd) = connection.answer(&line).await; + let mut json = serde_json::to_vec(&response)?; + json.push(b'\n'); + fdpass::reply(reader.get_ref(), &json, fd.as_ref().map(AsFd::as_fd)).await?; + // Dropping our copy leaves the client holding the only reference to + // its half, so its close tears the flow or the listener down. + drop(fd); + } + + Ok(()) + } + + /// Read one newline-terminated command into `line`, refusing an oversized + /// one. + /// + /// Returns `false` at end of file. The buffer is filled and consumed a + /// chunk at a time so a client that never sends a newline is cut off at + /// [`MAX_COMMAND`] rather than growing the buffer without bound. + pub(super) async fn read_command( + reader: &mut R, + line: &mut Vec, + ) -> Result + where + R: tokio::io::AsyncBufRead + Unpin, + { + use tokio::io::AsyncBufReadExt; + + line.clear(); + loop { + let available = reader.fill_buf().await?; + if available.is_empty() { + return Ok(!line.is_empty()); + } + + // The newline may or may not be in this chunk, and the cap has to + // hold either way. Checking it only on the no-newline branch makes + // enforcement depend on where the reader happened to split the + // input, which is a guard that works only intermittently. + let (take, complete) = match available.iter().position(|byte| *byte == b'\n') { + Some(end) => (end, true), + None => (available.len(), false), + }; + + if line.len() + take > MAX_COMMAND { + return Err(std::io::Error::new( + std::io::ErrorKind::InvalidData, + "native API command too large", + )); + } + + line.extend_from_slice(&available[..take]); + reader.consume(if complete { take + 1 } else { take }); + + if complete { + return Ok(true); + } + } + } +} + +#[cfg(all(test, any(target_os = "linux", target_os = "freebsd")))] +mod tests { + use super::link::{self, NativeMessage, Outbound}; + use super::registry::{Limits, Registry}; + use super::unix_impl::Connection; + use std::io::{Read, Write}; + use std::os::fd::{AsRawFd, OwnedFd}; + use std::os::unix::net::UnixStream as StdUnixStream; + use tokio::sync::mpsc; + + /// An npub the tests can decode. Any valid one will do; the tests never + /// reach the peer it names. + const PEER: &str = "npub1sjlh2c3x9w7kjsqg2ay080n2lff2uvt325vpan33ke34rn8l5jcqawh57m"; + + /// The npub a test node reports as its own, standing in for the identity a + /// running daemon would report. A real one, so a reader cannot mistake the + /// field for something the daemon fabricates. + pub(super) const NODE: &str = "npub10xlxvlhemja6c4dqv22uapctqupfhlxm9h8z3k2e72q4k9hcz7vqpkge6d"; + + fn limits() -> Limits { + Limits { + per_flow: 4, + backlog: 2, + max_flows: 8, + } + } + + /// The payload limit the fake node reports, standing in for one derived + /// from a transport MTU. + const MAX_PAYLOAD: u16 = 1362; + + /// A connection wired to a task serving a real registry. + /// + /// The task runs the same `link::serve` the daemon's `rx_loop` runs, so + /// these tests exercise the whole request and reply path rather than a + /// stand-in for it. + fn connect() -> (Connection, mpsc::Receiver) { + wire_connection(true) + } + + /// The same connection with the debug commands off, as a packaged node has + /// them. + fn connect_without_debug() -> (Connection, mpsc::Receiver) { + wire_connection(false) + } + + /// Build a connection over a real registry, with the debug gate as given. + fn wire_connection(debug: bool) -> (Connection, mpsc::Receiver) { + let (tx, mut rx) = mpsc::channel::(16); + tokio::spawn(async move { + let mut registry = Registry::new(limits()); + let mut now = 0u64; + while let Some(message) = rx.recv().await { + link::serve(&mut registry, message, now, MAX_PAYLOAD); + now += 1; + } + }); + let (out_tx, out_rx) = mpsc::channel::(64); + (Connection::for_test(tx, out_tx, limits(), debug), out_rx) + } + + /// Send one command and read the reply as JSON. + async fn ask(connection: &mut Connection, line: &str) -> serde_json::Value { + let (response, fd) = connection.answer(line.as_bytes()).await; + assert!(fd.is_none(), "this command should carry no descriptor"); + serde_json::to_value(response).unwrap() + } + + /// Send one command that opens a flow, returning the reply and descriptor. + async fn open(connection: &mut Connection, line: &str) -> (serde_json::Value, StdUnixStream) { + let (response, fd) = connection.answer(line.as_bytes()).await; + let value = serde_json::to_value(response).unwrap(); + let fd = fd.expect("this command should carry a descriptor"); + (value, StdUnixStream::from(fd)) + } + + /// Bind a listener and keep its descriptor, the way a client does. + async fn listen(connection: &mut Connection, port: u16) -> (serde_json::Value, StdUnixStream) { + open( + connection, + &format!(r#"{{"command":"listen","params":{{"local_port":{port}}}}}"#), + ) + .await + } + + /// Read one arrival message and the flow descriptor that rides with it. + /// + /// Goes through the same `recvmsg` a client process runs, so what these + /// tests assert about the ancillary framing is what a client would see. + fn accept(listener: &StdUnixStream) -> (serde_json::Value, StdUnixStream) { + let mut buf = [0u8; 4096]; + let chunk = super::fdpass::recv(listener.as_raw_fd(), &mut buf) + .expect("an arrival should be readable on the listener"); + let fd: OwnedFd = chunk.fd.expect("an arrival carries the flow's descriptor"); + let value: serde_json::Value = + serde_json::from_slice(&buf[..chunk.len]).expect("the arrival is one JSON object"); + (value, StdUnixStream::from(fd)) + } + + /// Deliver a datagram as though a peer had sent it. + fn arrival(src: u16, dst: u16, data: &str) -> String { + arrival_from(PEER, src, dst, data) + } + + /// The same, naming the peer, so a test can name one that will not decode. + fn arrival_from(peer: &str, src: u16, dst: u16, data: &str) -> String { + format!( + r#"{{"command":"arrive","params":{{"peer":"{peer}","src_port":{src},"dst_port":{dst},"data":"{data}"}}}}"# + ) + } + + #[tokio::test] + async fn closing_one_flow_frees_its_port_while_the_connection_stays_open() { + let (mut connection, _outbound) = connect(); + let line = format!( + r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242,"local_port":4243}}}}"# + ); + let (value, client) = open(&mut connection, &line).await; + let flow = value["data"]["flow_id"].as_u64().unwrap(); + + drop(client); + connection.settle_closed(flow).await; + + // `settle_closed` observes the reader's flag, which it sets before it + // sends the release, so the registry may not have processed it yet. + // Retry rather than sleep: without the reclaim every attempt fails and + // the loop runs out, which is the failure this test exists to produce. + let mut last = serde_json::Value::Null; + for _ in 0..1000 { + // Not `ask`: a connect that succeeds carries a descriptor, and that + // is the outcome being waited for. + let (response, fd) = connection.answer(line.as_bytes()).await; + last = serde_json::to_value(response).unwrap(); + if last["status"] == "ok" { + assert!(fd.is_some(), "a reopened flow still gets a descriptor"); + return; + } + tokio::task::yield_now().await; + } + panic!("the closed flow never gave its port back: {last}"); + } + + #[tokio::test] + async fn a_listener_holds_its_port_against_a_second_binder() { + let (mut connection, _outbound) = connect(); + let (value, _listener) = listen(&mut connection, 4242).await; + assert_eq!(value["status"], "ok"); + assert_eq!(value["data"]["local_port"], 4242); + assert_eq!(value["data"]["node"], NODE); + assert_eq!(value["data"]["backlog"], limits().backlog); + + let value = ask( + &mut connection, + r#"{"command":"listen","params":{"local_port":4242}}"#, + ) + .await; + assert_eq!(value["status"], "error"); + assert!( + value["message"] + .as_str() + .unwrap() + .contains("already in use"), + "message was {}", + value["message"] + ); + } + + #[tokio::test] + async fn a_listener_that_names_no_port_is_told_the_one_it_was_given() { + let (mut connection, _outbound) = connect(); + let (value, _listener) = open( + &mut connection, + r#"{"command":"listen","params":{"local_port":0}}"#, + ) + .await; + assert_eq!(value["status"], "ok"); + let port = value["data"]["local_port"].as_u64().unwrap(); + assert!( + port >= u64::from(super::protocol::PORT_EPHEMERAL_MIN), + "an allocated listener port comes from the ephemeral range, got {port}" + ); + + // The reported port is the one actually held. A reply that echoed the + // zero it was asked for would leave a client with no port to name. + let value = ask( + &mut connection, + &format!(r#"{{"command":"listen","params":{{"local_port":{port}}}}}"#), + ) + .await; + assert_eq!(value["status"], "error"); + } + + #[tokio::test] + async fn a_reserved_port_is_refused_before_it_reaches_the_registry() { + let (mut connection, _outbound) = connect(); + let value = ask( + &mut connection, + r#"{"command":"listen","params":{"local_port":256}}"#, + ) + .await; + assert_eq!(value["status"], "error"); + assert!( + value["message"] + .as_str() + .unwrap() + .contains("standard services") + ); + } + + #[tokio::test] + async fn malformed_json_and_unknown_commands_are_refused() { + let (mut connection, _outbound) = connect(); + let value = ask(&mut connection, "{not json").await; + assert_eq!(value["status"], "error"); + assert!( + value["message"] + .as_str() + .unwrap() + .contains("invalid request") + ); + + let value = ask(&mut connection, r#"{"command":"teleport"}"#).await; + assert_eq!(value["status"], "error"); + assert!(value["message"].as_str().unwrap().contains("teleport")); + } + + #[tokio::test] + async fn accept_and_reject_are_no_longer_commands() { + // They were removed with the round trip they served. A client that + // still sends one must be told the command does not exist, rather than + // having it quietly ignored or, worse, half-served. + let (mut connection, _outbound) = connect(); + for command in ["accept", "reject"] { + let value = ask( + &mut connection, + &format!(r#"{{"command":"{command}","params":{{"flow_id":1}}}}"#), + ) + .await; + assert_eq!(value["status"], "error", "{command} was answered"); + assert!( + value["message"].as_str().unwrap().contains(command), + "the refusal should name {command}: {}", + value["message"] + ); + } + } + + #[tokio::test] + async fn a_refusal_a_client_acts_on_carries_the_errno_for_it() { + // A client turns `data.errno` into the error its language's `bind` or + // `connect` would have returned, and it must never have to match the + // message to do it. Every row here is one the client library maps. + let (mut connection, _outbound) = connect(); + + let (_value, _listener) = listen(&mut connection, 4242).await; + let value = ask( + &mut connection, + r#"{"command":"listen","params":{"local_port":4242}}"#, + ) + .await; + assert_eq!(value["data"]["errno"], "EADDRINUSE", "{value}"); + + let value = ask( + &mut connection, + r#"{"command":"listen","params":{"local_port":256}}"#, + ) + .await; + assert_eq!(value["data"]["errno"], "EADDRNOTAVAIL", "{value}"); + + let value = ask( + &mut connection, + r#"{"command":"connect","params":{"peer":"not-an-npub","remote_port":4242}}"#, + ) + .await; + assert_eq!(value["data"]["errno"], "EINVAL", "{value}"); + + let value = ask(&mut connection, r#"{"command":"teleport"}"#).await; + assert_eq!(value["data"]["errno"], "EINVAL", "{value}"); + + // The node's flow ceiling, which is the one row a registry refusal + // other than a taken port produces. The descriptors are kept so the + // flows stay open while the ceiling is tested. + let mut open_flows = Vec::new(); + for index in 0..limits().max_flows { + let (value, client) = open( + &mut connection, + &format!( + r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242,"local_port":{}}}}}"#, + 5000 + index + ), + ) + .await; + assert_eq!(value["status"], "ok", "flow {index} was refused: {value}"); + open_flows.push(client); + } + let value = ask( + &mut connection, + &format!( + r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242,"local_port":6000}}}}"# + ), + ) + .await; + assert_eq!(value["data"]["errno"], "EMFILE", "{value}"); + } + + #[tokio::test] + async fn a_connect_yields_a_descriptor_and_holds_a_port() { + let (mut connection, _outbound) = connect(); + let (value, mut client) = open( + &mut connection, + &format!( + r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242,"local_port":5000}}}}"# + ), + ) + .await; + assert_eq!(value["status"], "ok"); + assert_eq!(value["data"]["local_port"], 5000); + assert_eq!( + value["data"]["peer"], PEER, + "the reply names the peer by npub, re-encoded from the key it decoded" + ); + assert_eq!( + value["data"]["node"], NODE, + "and names this node, so a client can answer getsockname" + ); + let flow = value["data"]["flow_id"].as_u64().unwrap(); + + client.write_all(b"hello").unwrap(); + connection.settle(flow, 1).await; + + let value = ask( + &mut connection, + &format!(r#"{{"command":"stats","params":{{"flow_id":{flow}}}}}"#), + ) + .await; + assert_eq!(value["data"]["rx_datagrams"], 1); + assert_eq!(value["data"]["rx_bytes"], 5); + + // The port the flow holds is now unavailable to a listener. + let value = ask( + &mut connection, + r#"{"command":"listen","params":{"local_port":5000}}"#, + ) + .await; + assert_eq!(value["status"], "error"); + } + + #[tokio::test] + async fn what_a_client_writes_reaches_the_node_with_its_flow_key_and_its_peer() { + let (mut connection, mut outbound) = connect(); + let (value, mut client) = open( + &mut connection, + &format!( + r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242,"local_port":5000}}}}"# + ), + ) + .await; + assert_eq!(value["data"]["max_payload"], MAX_PAYLOAD); + + client.write_all(b"over the wire").unwrap(); + + let sent = tokio::time::timeout(std::time::Duration::from_secs(5), outbound.recv()) + .await + .expect("the datagram should reach the node") + .expect("the outbound channel should stay open"); + assert_eq!(sent.payload, b"over the wire"); + assert_eq!(sent.key.local, 5000); + assert_eq!(sent.key.remote, 4242); + assert_eq!( + crate::identity::encode_npub(&sent.peer), + PEER, + "the flow carries the peer's address, so the node needs no lookup \ + to start a session" + ); + } + + #[tokio::test] + async fn a_bad_peer_is_refused_without_holding_a_port() { + let (mut connection, _outbound) = connect(); + let value = ask( + &mut connection, + r#"{"command":"connect","params":{"peer":"not-an-npub","remote_port":4242,"local_port":5000}}"#, + ) + .await; + assert_eq!(value["status"], "error"); + assert!(value["message"].as_str().unwrap().contains("invalid peer")); + + // The port must still be free: a refusal that leaked a reservation + // would take a port out of service for the node's lifetime. + let (value, _listener) = listen(&mut connection, 5000).await; + assert_eq!(value["status"], "ok"); + } + + #[tokio::test] + async fn an_arrival_reaches_the_listener_descriptor_with_its_flow_and_what_it_held() { + let (mut connection, _outbound) = connect(); + let (_value, listener) = listen(&mut connection, 4242).await; + + let value = ask(&mut connection, &arrival(5000, 4242, "00ff10")).await; + assert_eq!(value["status"], "ok"); + assert_eq!(value["data"]["outcome"], "announced"); + let flow = value["data"]["flow_id"].as_u64().unwrap(); + + let (message, mut client) = accept(&listener); + assert_eq!(message["flow_id"], flow); + assert_eq!( + message["peer"], PEER, + "the peer is named by npub, which is its address, and never by hex" + ); + assert!( + message.get("peer_addr").is_none(), + "no client-facing field carries the node address" + ); + assert_eq!(message["node"], NODE); + assert_eq!(message["local_port"], 4242); + assert_eq!(message["remote_port"], 5000); + assert_eq!(message["max_payload"], MAX_PAYLOAD); + assert_eq!(message["held"], 1); + + // The datagram that arrived before the client could read the arrival is + // already on the descriptor, rather than lost to the hand-off. + let mut buf = [0u8; 64]; + assert_eq!(client.read(&mut buf).unwrap(), 3); + assert_eq!(&buf[..3], &[0x00, 0xff, 0x10]); + } + + #[test] + fn a_listener_that_closed_and_one_that_stopped_reading_are_counted_apart() { + // Both take the same cleanup, so the counter is the only place the + // difference survives, and an operator reads it to tell a client that + // closed normally from one that is wedged. + use super::unix_impl::why; + use std::io::{Error, ErrorKind}; + + assert_eq!( + why(&Error::from(ErrorKind::BrokenPipe)), + link::DropReason::ListenerGone + ); + assert_eq!( + why(&Error::from(ErrorKind::WouldBlock)), + link::DropReason::ListenerNotReading + ); + } + + /// One accepted flow holding `held`, for the hand-off tests. + fn arrived(held: Vec>) -> link::Accepted { + let peer = crate::identity::decode_npub(PEER).unwrap(); + link::Accepted { + key: super::registry::FlowKey { + peer: crate::identity::NodeAddr::from_pubkey(&peer), + remote: 5000, + local: 4242, + }, + peer, + held, + max: MAX_PAYLOAD, + } + } + + /// What one hand-off write was, for an order assertion. + fn wrote(step: &super::unix_impl::Handoff<'_>) -> String { + match step { + super::unix_impl::Handoff::Held(datagram) => { + String::from_utf8_lossy(datagram).into_owned() + } + super::unix_impl::Handoff::Arrival(_) => "arrival".to_string(), + } + } + + #[test] + fn performing_a_hand_off_writes_every_held_datagram_before_the_arrival_carrying_the_descriptor() + { + // Over the writes the daemon actually performs, not over the plan it + // built. Asserting the plan alone leaves the consumption free to run in + // any order, and reversed it hands the client a live descriptor for a + // flow whose held datagrams never reached it. + use super::unix_impl::perform; + + let accepted = arrived(vec![b"first".to_vec(), b"second".to_vec()]); + let mut written: Vec = Vec::new(); + let done = perform(9, &accepted, NODE, |step| { + written.push(wrote(step)); + Ok(()) + }); + + assert!(done.is_ok(), "every write succeeded"); + assert_eq!( + written, + ["first", "second", "arrival"], + "the descriptor must be the last thing written, not the first" + ); + } + + #[test] + fn a_hand_off_whose_held_write_fails_never_writes_the_arrival_and_names_the_reason() { + // The arrival is what hands the descriptor over, so attempting it after + // a lost held datagram would give the client a flow missing its peer's + // opening message rather than no flow at all. + use super::unix_impl::perform; + use std::io::{Error, ErrorKind}; + + let accepted = arrived(vec![b"first".to_vec(), b"second".to_vec()]); + let mut written: Vec = Vec::new(); + let reason = perform(9, &accepted, NODE, |step| { + written.push(wrote(step)); + Err(Error::from(ErrorKind::WouldBlock)) + }) + .expect_err("the first write failed"); + + assert_eq!(reason, link::DropReason::ListenerNotReading); + assert_eq!(written, ["first"], "the hand-off stopped at the failure"); + } + + #[test] + fn a_hand_off_whose_arrival_write_fails_reports_the_listener_gone() { + // A client that closed its listener between the arrival leaving the + // queue and this write is a race a healthy client can lose, and it must + // not be counted as a client falling behind. + use super::unix_impl::{Handoff, perform}; + use std::io::{Error, ErrorKind}; + + let accepted = arrived(vec![b"first".to_vec()]); + let reason = perform(9, &accepted, NODE, |step| match step { + Handoff::Held(_) => Ok(()), + Handoff::Arrival(_) => Err(Error::from(ErrorKind::BrokenPipe)), + }) + .expect_err("the arrival write failed"); + + assert_eq!(reason, link::DropReason::ListenerGone); + } + + #[test] + fn the_hand_off_writes_what_a_flow_held_before_the_arrival_that_carries_it() { + // The plan, which is the half of the ordering guarantee this test owns: + // what the batch contains and what the arrival says about it. That the + // writes then happen in the planned order is the neighbouring test's, + // and both are needed, because a correct plan consumed backwards is + // invisible to a client that reads afterwards. + use super::unix_impl::{Handoff, handoff}; + + let accepted = arrived(vec![b"first".to_vec(), b"second".to_vec()]); + let steps = handoff(9, &accepted, NODE); + let order: Vec<&str> = steps + .iter() + .map(|step| match step { + Handoff::Held(_) => "held", + Handoff::Arrival(_) => "arrival", + }) + .collect(); + assert_eq!( + order, + ["held", "held", "arrival"], + "the descriptor must be the last thing written, not the first" + ); + + let held: Vec<&[u8]> = steps + .iter() + .filter_map(|step| match step { + Handoff::Held(datagram) => Some(datagram.as_slice()), + Handoff::Arrival(_) => None, + }) + .collect(); + assert_eq!(held, [b"first".as_slice(), b"second".as_slice()]); + + let Some(Handoff::Arrival(message)) = steps.last() else { + panic!("the hand-off ends with the arrival"); + }; + let value: serde_json::Value = serde_json::from_slice(message).unwrap(); + assert_eq!(value["held"], 2, "the arrival counts the whole batch"); + assert_eq!(value["flow_id"], 9); + assert_eq!(value["peer"], PEER); + } + + #[tokio::test] + async fn every_datagram_from_one_peer_reaches_the_one_flow_it_belongs_to() { + // The first datagram from a new peer announces an arrival and the rest + // join the flow it created, rather than announcing the same peer again. + // Which of the three the daemon holds and which it delivers depends on + // when the listener's task wins its hop, so the test asserts what does + // not depend on that: three datagrams, one flow, in order. + let (mut connection, _outbound) = connect(); + let (_value, listener) = listen(&mut connection, 4242).await; + let arrive = arrival(5000, 4242, "aa"); + + for _ in 0..3 { + let value = ask(&mut connection, &arrive).await; + let outcome = value["data"]["outcome"].as_str().unwrap(); + assert!( + ["announced", "held", "delivered"].contains(&outcome), + "a datagram from a peer with a listener on its port was {outcome}" + ); + } + + let (message, mut client) = accept(&listener); + assert_eq!(message["local_port"], 4242); + + let mut buf = [0u8; 64]; + for _ in 0..3 { + assert_eq!(client.read(&mut buf).unwrap(), 1); + assert_eq!(buf[0], 0xaa); + } + } + + #[tokio::test] + async fn closing_a_listener_descriptor_unbinds_its_port() { + let (mut connection, _outbound) = connect(); + let (_value, listener) = listen(&mut connection, 4242).await; + drop(listener); + + // The unbind is what `close(listen_fd)` means in Berkeley, and the + // daemon can only observe it by reading its own half. Without the read + // arm the port is held for the node's lifetime and every attempt below + // fails. + for _ in 0..1000 { + let (response, fd) = connection + .answer(br#"{"command":"listen","params":{"local_port":4242}}"#) + .await; + if serde_json::to_value(response).unwrap()["status"] == "ok" { + assert!(fd.is_some(), "a rebound listener still gets a descriptor"); + return; + } + tokio::task::yield_now().await; + } + panic!("the closed listener never gave its port back"); + } + + #[tokio::test] + async fn closing_the_rpc_connection_leaves_its_flows_and_listeners_alone() { + // The property the redesign turns on: the RPC connection is a setup + // channel and owns nothing. Before, dropping it released every flow and + // listener it had opened, which is what made a stream that outlived its + // client object unusable. + // The registry is shared rather than owned by the serving task, + // because the count has to be taken while the flow and the listener are + // still open. A task that reported its count on the way out could only + // be asked after everything under test had ended. + let registry = std::sync::Arc::new(std::sync::Mutex::new(Registry::new(limits()))); + let (tx, mut rx) = mpsc::channel::(16); + let serving = std::sync::Arc::clone(®istry); + tokio::spawn(async move { + while let Some(message) = rx.recv().await { + let mut held = serving.lock().unwrap(); + link::serve(&mut held, message, 0, MAX_PAYLOAD); + } + }); + + let (out_tx, _out_rx) = mpsc::channel::(64); + let mut connection = Connection::for_test(tx.clone(), out_tx, limits(), true); + let (_value, _listener) = listen(&mut connection, 4242).await; + let (_value, _client) = open( + &mut connection, + &format!( + r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242,"local_port":5000}}}}"# + ), + ) + .await; + + drop(connection); + // Both descriptors are still open, so neither the flow nor the listener + // has any reason to end. Give any release the connection might have + // sent a chance to be served before the count is taken. + for _ in 0..100 { + tokio::task::yield_now().await; + } + + assert_eq!( + registry.lock().unwrap().port_count(), + 2, + "the flow's port and the listener's port both survive the connection" + ); + } + + #[tokio::test] + async fn discarding_a_wired_flow_gives_its_port_back_and_counts_the_drop() { + // The registry half of the three-part cleanup a listener's task runs + // when it cannot hand a flow over. Driven directly, because the write + // failure that triggers it needs a client that has stopped reading. + let mut registry = Registry::new(limits()); + let (arrivals, _arrivals_rx) = mpsc::channel(4); + registry.listen(Some(4242), arrivals).unwrap(); + + let peer = crate::identity::decode_npub(PEER).unwrap(); + let addr = crate::identity::NodeAddr::from_pubkey(&peer); + let flow = match registry.deliver(addr, peer, 5000, 4242, 0) { + super::registry::Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + let (sink, _sink_rx) = mpsc::channel(4); + let (key, _peer, _held) = registry.accept(flow, sink, 0).unwrap(); + assert_eq!(registry.flow_count(), 1); + + let served = link::serve( + &mut registry, + NativeMessage::Discard { + key, + reason: super::link::DropReason::ListenerNotReading, + }, + 0, + MAX_PAYLOAD, + ); + assert_eq!( + served, + link::Served::Discarded(super::link::DropReason::ListenerNotReading), + "the drop is reported so the node counts it" + ); + assert_eq!(registry.flow_count(), 0, "the flow is gone"); + assert_eq!( + registry.port_count(), + 1, + "only the listener's port is left; the flow shared it and claimed none" + ); + } + + #[tokio::test] + async fn a_datagram_for_an_unheld_port_is_dropped() { + let (mut connection, _outbound) = connect(); + let value = ask(&mut connection, &arrival(5000, 9999, "aa")).await; + assert_eq!(value["status"], "ok"); + assert!( + value["data"]["outcome"] + .as_str() + .unwrap() + .starts_with("dropped"), + "outcome was {}", + value["data"]["outcome"] + ); + } + + #[tokio::test] + async fn inject_writes_separate_datagrams_to_the_client() { + let (mut connection, _outbound) = connect(); + let (value, mut client) = open( + &mut connection, + &format!(r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242}}}}"#), + ) + .await; + let flow = value["data"]["flow_id"].as_u64().unwrap(); + + let value = ask( + &mut connection, + &format!( + r#"{{"command":"inject","params":{{"flow_id":{flow},"data":"00ff10","repeat":3}}}}"# + ), + ) + .await; + assert_eq!(value["status"], "ok"); + + // Three reads of three bytes, not one read of nine. + let mut buf = [0u8; 64]; + for _ in 0..3 { + assert_eq!(client.read(&mut buf).unwrap(), 3); + assert_eq!(&buf[..3], &[0x00, 0xff, 0x10]); + } + } + + #[tokio::test] + async fn stats_and_inject_refuse_a_flow_the_node_does_not_hold() { + let (mut connection, _outbound) = connect(); + let value = ask( + &mut connection, + r#"{"command":"stats","params":{"flow_id":99}}"#, + ) + .await; + assert_eq!(value["status"], "error"); + assert!(value["message"].as_str().unwrap().contains("no such flow")); + // The errno, not only the prose: a refusal with no code reaches a + // client as ECONNREFUSED, which reads as the node declining rather than + // as the flow not existing. + assert_eq!(value["data"]["errno"], "ENOENT"); + + let value = ask( + &mut connection, + r#"{"command":"inject","params":{"flow_id":99,"data":"00"}}"#, + ) + .await; + assert_eq!(value["status"], "error"); + assert_eq!(value["data"]["errno"], "ENOENT"); + } + + #[tokio::test] + async fn a_debug_command_naming_a_peer_it_cannot_decode_refuses_with_einval() { + // The debug commands answer under the same errno contract as the two + // setup calls. Without a code the reply reaches a client as + // ECONNREFUSED, which says nothing about the argument that was wrong. + let (mut connection, _outbound) = wire_connection(true); + let value = ask( + &mut connection, + &arrival_from("not-an-npub", 4242, 5000, "00"), + ) + .await; + + assert_eq!(value["status"], "error"); + assert!(value["message"].as_str().unwrap().contains("invalid peer")); + assert_eq!(value["data"]["errno"], "EINVAL"); + } + + /// The three debug command lines, aimed at a flow that really exists. + /// + /// Naming a live flow and a live port is what makes the gate tests + /// discriminating: with the gate open every one of these succeeds, so a + /// refusal can only have come from the gate. + fn debug_commands(flow: u64, local: u16) -> [String; 3] { + [ + format!(r#"{{"command":"inject","params":{{"flow_id":{flow},"data":"00ff10"}}}}"#), + format!(r#"{{"command":"stats","params":{{"flow_id":{flow}}}}}"#), + arrival(4242, local, "00"), + ] + } + + /// Open a flow and return its identifier, local port and client half. + async fn open_flow(connection: &mut Connection) -> (u64, u16, StdUnixStream) { + let (value, client) = open( + connection, + &format!(r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242}}}}"#), + ) + .await; + let flow = value["data"]["flow_id"].as_u64().unwrap(); + let local = value["data"]["local_port"].as_u64().unwrap() as u16; + (flow, local, client) + } + + #[tokio::test] + async fn every_debug_command_is_refused_when_the_gate_is_closed() { + let (mut connection, _outbound) = connect_without_debug(); + let (flow, local, _client) = open_flow(&mut connection).await; + + for line in debug_commands(flow, local) { + let value = ask(&mut connection, &line).await; + assert_eq!(value["status"], "error", "{line} was not refused"); + let message = value["message"].as_str().unwrap(); + // The key by name, so an operator reading the refusal knows what to + // change, and so this cannot pass on some unrelated error. + assert!( + message.contains("node.native_api.debug_commands"), + "{line} was refused for the wrong reason: {message}" + ); + } + + // The gate refuses the command, not the connection: a client that tries + // one on a node that will not serve it keeps its flows. + let (value, _listener) = listen(&mut connection, 4243).await; + assert_eq!(value["status"], "ok"); + } + + #[tokio::test] + async fn every_debug_command_is_answered_when_the_gate_is_open() { + let (mut connection, _outbound) = connect(); + let (flow, local, _client) = open_flow(&mut connection).await; + + for line in debug_commands(flow, local) { + let value = ask(&mut connection, &line).await; + assert_eq!(value["status"], "ok", "{line} was refused: {value}"); + } + + // The arrival reaches the flow rather than being dropped for want of a + // port. The harness asserts the same outcome, and would be asserting + // nothing if a dropped arrival also answered "ok". + let value = ask(&mut connection, &debug_commands(flow, local)[2]).await; + assert_eq!(value["data"]["outcome"], "delivered"); + } + + #[tokio::test] + async fn an_unreasonable_repeat_is_refused() { + let (mut connection, _outbound) = connect(); + for repeat in ["0", "99999"] { + let value = ask( + &mut connection, + &format!( + r#"{{"command":"inject","params":{{"flow_id":1,"data":"00","repeat":{repeat}}}}}"# + ), + ) + .await; + assert_eq!(value["status"], "error"); + assert!(value["message"].as_str().unwrap().contains("repeat")); + } + } + + #[tokio::test] + async fn closing_the_descriptor_is_seen_by_the_daemon() { + let (mut connection, _outbound) = connect(); + let (value, client) = open( + &mut connection, + &format!(r#"{{"command":"connect","params":{{"peer":"{PEER}","remote_port":4242}}}}"#), + ) + .await; + let flow = value["data"]["flow_id"].as_u64().unwrap(); + + drop(client); + connection.settle_closed(flow).await; + + // The node forgets a flow whose descriptor closed, so `stats` answers + // for it the way it answers for any other name it does not hold. + let value = ask( + &mut connection, + &format!(r#"{{"command":"stats","params":{{"flow_id":{flow}}}}}"#), + ) + .await; + assert_eq!(value["status"], "error"); + } + + #[tokio::test] + async fn the_command_size_cap_holds_on_both_read_paths() { + use super::unix_impl::{MAX_COMMAND, read_command}; + use tokio::io::BufReader; + + // With a newline in the same chunk, and with none at all: both branches + // must refuse, or the guard only works for inputs that split the way it + // expects. + for input in [ + format!("{}\n", "x".repeat(MAX_COMMAND + 1)), + "x".repeat(MAX_COMMAND + 1), + ] { + let mut reader = BufReader::new(input.as_bytes()); + let mut line = Vec::new(); + let error = read_command(&mut reader, &mut line).await.unwrap_err(); + assert_eq!(error.kind(), std::io::ErrorKind::InvalidData); + } + + // And the healthy path: a command at exactly the cap is accepted. + let ok = format!("{}\n", "x".repeat(MAX_COMMAND)); + let mut reader = BufReader::new(ok.as_bytes()); + let mut line = Vec::new(); + assert!(read_command(&mut reader, &mut line).await.unwrap()); + assert_eq!(line.len(), MAX_COMMAND); + } + + #[tokio::test] + async fn two_commands_on_one_connection_are_read_in_turn() { + use super::unix_impl::read_command; + use tokio::io::BufReader; + + let input = "{\"command\":\"listen\"}\n{\"command\":\"stats\"}\n"; + let mut reader = BufReader::new(input.as_bytes()); + let mut line = Vec::new(); + + assert!(read_command(&mut reader, &mut line).await.unwrap()); + assert_eq!(line, br#"{"command":"listen"}"#); + assert!(read_command(&mut reader, &mut line).await.unwrap()); + assert_eq!(line, br#"{"command":"stats"}"#); + assert!(!read_command(&mut reader, &mut line).await.unwrap()); + } +} diff --git a/src/native/protocol.rs b/src/native/protocol.rs new file mode 100644 index 00000000..f6f51cd4 --- /dev/null +++ b/src/native/protocol.rs @@ -0,0 +1,566 @@ +//! Native datagram API command types and the pure decisions taken over them. +//! +//! No I/O and no node state. The shell in [`super`] reads a line, calls +//! [`parse`], and acts on the typed command. Every rule that can be decided +//! from the request alone is decided here, so it is testable without a socket +//! and without a running node. +//! +//! The encoding is the control socket's: one JSON object per line, shaped +//! `{"command": "...", "params": {...}}`, answered by a +//! [`Response`](crate::control::protocol::Response). The two sockets are +//! separate listeners with separate lifetimes, but there is no reason for a +//! client to learn two envelopes. + +use super::registry::RegistryError; +use crate::control::protocol::Request; +use serde::Deserialize; +use thiserror::Error; + +/// Highest port reserved for protocol use, per the FSP port registry. +pub const PORT_PROTOCOL_MAX: u16 = 255; + +/// Highest port reserved for FIPS standard services. Port 256 in this range is +/// the IPv6 shim ([`FSP_PORT_IPV6_SHIM`](crate::proto::fsp::FSP_PORT_IPV6_SHIM)). +pub const PORT_STANDARD_MAX: u16 = 1023; + +/// Lowest port the daemon hands out when a client names no local port. +/// +/// Below this the application range is available for an explicit bind, which is +/// what lets a service hold a port a peer can be told about in advance. +pub const PORT_EPHEMERAL_MIN: u16 = 49152; + +/// A parsed native API command. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum Command { + /// Open a flow to a named peer. + Connect(Connect), + /// Receive flows from any peer on a local port. + Listen(Listen), + /// **Debug, gated.** Make the daemon write bytes the client chose into one + /// of that client's flows, so the receive direction is exercisable without + /// a peer. + Inject(Inject), + /// **Debug, gated.** Report what the daemon has received on a flow. + Stats(u64), + /// **Debug, gated.** Deliver a datagram as though it had arrived from the + /// mesh, which reaches any listener this node holds. + Arrive(Arrive), +} + +impl Command { + /// The name a client used, when the command is one of the debug three. + /// + /// The three are not a supported interface: they exist for the test + /// harness, and a node answers them only where + /// `node.native_api.debug_commands` is on. Whether that key is on is node + /// configuration and so is not decidable here; classifying the command is, + /// which is the half this module owns. + pub fn debug_name(&self) -> Option<&'static str> { + match self { + Command::Inject(_) => Some("inject"), + Command::Stats(_) => Some("stats"), + Command::Arrive(_) => Some("arrive"), + Command::Connect(_) | Command::Listen(_) => None, + } + } +} + +/// Parameters of a `connect` command, after validation. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Connect { + /// The far end, as the npub the client supplied. Decoding it to a key is + /// the shell's job: this module holds no identity types. + pub peer: String, + /// The far end's port. + pub remote: u16, + /// The local port, or `None` to let the daemon allocate an ephemeral one. + pub local: Option, +} + +/// Parameters of a `listen` command, after validation. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Listen { + /// The local port to receive on, or `None` to let the daemon allocate an + /// ephemeral one. Port 0 and an absent field mean the same thing, which is + /// what `bind(2)` with port 0 means and what `connect` already accepted. + pub local: Option, +} + +/// Parameters of the debug `arrive` command, after validation. +/// +/// This drives the node's inbound dispatch without a wire, so the rule that +/// decides between an established flow, a listener and a drop is exercised +/// before FSP is involved. The wire is a second caller of the same rule rather +/// than a new one, which is why the command outlived the work that first +/// needed it; it stays behind `node.native_api.debug_commands` because a caller that +/// reaches it can deliver to any listener on this node under any peer identity +/// it names. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Arrive { + /// The peer it appears to come from, as an npub. + pub peer: String, + /// Its source port. + pub src: u16, + /// Its destination port on this node. + pub dst: u16, + /// The payload, decoded from hex. + pub data: Vec, +} + +/// Parameters of the debug `inject` command, after validation. +/// +/// This command exists so the receive direction is testable without a peer +/// sending anything. It can only write into a flow the same connection opened, +/// so it grants that client nothing it does not already have, but it makes the +/// daemon emit bytes the client chose on a path a reader cannot distinguish +/// from the mesh. It is therefore gated on +/// `node.native_api.debug_commands` and off in a packaged node. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Inject { + /// Which flow to write into. + pub flow: u64, + /// The bytes to write, decoded from the hex the client sent. Hex rather + /// than text so a check can assert byte fidelity, including bytes that are + /// not valid UTF-8. + pub data: Vec, + /// How many separate datagrams to write. Sending one payload several times + /// is how a check observes that message boundaries survive. + pub repeat: u32, +} + +/// Why a client's command was refused. +/// +/// Each variant is a distinct refusal a client can act on, rather than one +/// string it would have to parse. +#[derive(Debug, Clone, PartialEq, Eq, Error)] +pub enum CommandError { + /// The command name is not one this API serves. + #[error("unknown command '{0}'")] + Unknown(String), + + /// The command needs a `params` object and none was supplied. + #[error("command '{0}' requires params")] + NoParams(&'static str), + + /// The `params` object did not match the command's shape. + #[error("invalid params for '{command}': {reason}")] + BadParams { + /// The command whose params failed to parse. + command: &'static str, + /// What serde objected to. + reason: String, + }, + + /// The port is in the range the protocol itself reserves. + #[error("port {0} is reserved for protocol use")] + PortProtocol(u16), + + /// The port is in the range reserved for FIPS standard services, which is + /// where the IPv6 shim lives. + #[error("port {0} is reserved for FIPS standard services")] + PortStandard(u16), + + /// A hex field did not decode. + #[error("invalid hex in '{field}': {reason}")] + BadHex { + /// The field that failed to decode. + field: &'static str, + /// What the decoder objected to. + reason: String, + }, + + /// A repeat count outside what the command will do in one call. + #[error("repeat must be between 1 and {max}, got {got}")] + BadRepeat { + /// The largest accepted value. + max: u32, + /// What the client asked for. + got: u32, + }, +} + +/// Largest number of datagrams one `inject` writes. +/// +/// Bounded because the command runs inline on the connection task: a client +/// asking for a million would hold that task for the duration. +pub const MAX_INJECT_REPEAT: u32 = 64; + +/// Raw `connect` parameters, before validation. +#[derive(Debug, Deserialize)] +struct ConnectParams { + peer: String, + remote_port: u16, + #[serde(default)] + local_port: Option, +} + +/// Raw `listen` parameters, before validation. +#[derive(Debug, Deserialize)] +struct ListenParams { + #[serde(default)] + local_port: Option, +} + +/// Raw `stats` parameters, before validation. +#[derive(Debug, Deserialize)] +struct FlowParams { + flow_id: u64, +} + +/// Raw `arrive` parameters, before validation. +#[derive(Debug, Deserialize)] +struct ArriveParams { + peer: String, + src_port: u16, + dst_port: u16, + data: String, +} + +/// Raw `inject` parameters, before validation. +#[derive(Debug, Deserialize)] +struct InjectParams { + flow_id: u64, + data: String, + #[serde(default = "InjectParams::one")] + repeat: u32, +} + +impl InjectParams { + /// One datagram, when the client names no repeat count. + fn one() -> u32 { + 1 + } +} + +/// Turn a request into a typed command, refusing anything a client may not ask +/// for. +/// +/// Port policy is applied here rather than at the registry, so a refusal costs +/// no node state and reads the same whether or not a node is running. +pub fn parse(request: &Request) -> Result { + match request.command.as_str() { + "connect" => { + let params: ConnectParams = take(request, "connect")?; + check_bindable(params.remote_port)?; + let local = match params.local_port { + // Port 0 is the UDP convention for "any port", and a client + // that reaches for it means the same thing as omitting the + // field. Accepting both spellings costs one branch and saves a + // refusal a client would find surprising. + None | Some(0) => None, + Some(port) => { + check_bindable(port)?; + Some(port) + } + }; + Ok(Command::Connect(Connect { + peer: params.peer, + remote: params.remote_port, + local, + })) + } + "listen" => { + let params: ListenParams = take(request, "listen")?; + // The same rule `connect` applies to its local port, so one + // spelling of "any port" covers both commands. + let local = match params.local_port { + None | Some(0) => None, + Some(port) => { + check_bindable(port)?; + Some(port) + } + }; + Ok(Command::Listen(Listen { local })) + } + "stats" => Ok(Command::Stats( + take::(request, "stats")?.flow_id, + )), + "arrive" => { + let params: ArriveParams = take(request, "arrive")?; + let data = hex::decode(¶ms.data).map_err(|error| CommandError::BadHex { + field: "data", + reason: error.to_string(), + })?; + Ok(Command::Arrive(Arrive { + peer: params.peer, + src: params.src_port, + dst: params.dst_port, + data, + })) + } + "inject" => { + let params: InjectParams = take(request, "inject")?; + if params.repeat == 0 || params.repeat > MAX_INJECT_REPEAT { + return Err(CommandError::BadRepeat { + max: MAX_INJECT_REPEAT, + got: params.repeat, + }); + } + let data = hex::decode(¶ms.data).map_err(|error| CommandError::BadHex { + field: "data", + reason: error.to_string(), + })?; + Ok(Command::Inject(Inject { + flow: params.flow_id, + data, + repeat: params.repeat, + })) + } + other => Err(CommandError::Unknown(other.to_string())), + } +} + +/// Deserialize a command's `params` object, naming the command in any error. +fn take Deserialize<'de>>( + request: &Request, + command: &'static str, +) -> Result { + let params = request + .params + .as_ref() + .ok_or(CommandError::NoParams(command))?; + serde_json::from_value(params.clone()).map_err(|error| CommandError::BadParams { + command, + reason: error.to_string(), + }) +} + +/// Decide whether a client may name `port`. +/// +/// The tiers come from the FSP port registry, which describes what already +/// ships on the v1 wire: 0-255 is protocol use, and 256-1023 is reserved for +/// FIPS standard services, of which 256 is the IPv6 shim. Only the application +/// range above them is a client's to use. +pub fn check_bindable(port: u16) -> Result<(), CommandError> { + if port <= PORT_PROTOCOL_MAX { + return Err(CommandError::PortProtocol(port)); + } + if port <= PORT_STANDARD_MAX { + return Err(CommandError::PortStandard(port)); + } + Ok(()) +} + +/// The `errno` name a refused setup command reports to its client. +/// +/// A client must never match on the message: that is for an operator reading a +/// log, and the codes are the contract. The match is exhaustive over +/// [`RegistryError`] and over the [`CommandError`] its `Port` variant wraps, so +/// a variant added without a code is a compile error rather than a silent +/// `ECONNREFUSED`. +/// +/// The names rather than the numbers, because the number belongs to the +/// platform the client is built for and this crate is not it. The client turns +/// the name into `io::Error::from_raw_os_error(libc::EADDRINUSE)` and so on. +pub fn errno(error: &RegistryError) -> &'static str { + match error { + // One port, one owner, and one key, one flow. Both are the address a + // caller asked for and cannot have. + RegistryError::PortTaken(_) | RegistryError::FlowTaken { .. } => "EADDRINUSE", + RegistryError::NoPort => "EADDRNOTAVAIL", + RegistryError::TooManyFlows(_) => "EMFILE", + // Neither reaches a setup reply today: a full backlog is decided on the + // receive path and counted rather than answered, and a missing pending + // flow is a hand-off the daemon lost to its own deadline. If either ever + // does reach a client, it is the daemon refusing for a reason of its + // own, which is what the catch-all names. + RegistryError::BacklogFull(_) | RegistryError::NoPending(_) => "ECONNREFUSED", + RegistryError::Port(command) => command_errno(command), + } +} + +/// The `errno` name for a command this API refused before any node state was +/// touched. +/// +/// A reserved port is `EADDRNOTAVAIL` rather than `EACCES`: `bind(2)` gives +/// `EACCES` below 1024 because that is a privilege question, and this is not +/// one. No client, however privileged, may hold port 256, because the IPv6 shim +/// has it. Everything else here is a malformed command, which is `EINVAL`. +pub fn command_errno(error: &CommandError) -> &'static str { + match error { + CommandError::PortProtocol(_) | CommandError::PortStandard(_) => "EADDRNOTAVAIL", + CommandError::Unknown(_) + | CommandError::NoParams(_) + | CommandError::BadParams { .. } + | CommandError::BadHex { .. } + | CommandError::BadRepeat { .. } => "EINVAL", + } +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + + /// Build a request the way the line reader would, from JSON text. + fn request(text: &str) -> Request { + serde_json::from_str(text).expect("test request should parse") + } + + #[test] + fn connect_carries_the_peer_and_both_ports() { + let parsed = parse(&request( + r#"{"command":"connect","params":{"peer":"npub1abc","remote_port":4242,"local_port":5000}}"#, + )) + .unwrap(); + assert_eq!( + parsed, + Command::Connect(Connect { + peer: "npub1abc".to_string(), + remote: 4242, + local: Some(5000), + }) + ); + } + + #[test] + fn an_absent_local_port_asks_for_an_ephemeral_one() { + let parsed = parse(&request( + r#"{"command":"connect","params":{"peer":"npub1abc","remote_port":4242}}"#, + )) + .unwrap(); + let Command::Connect(connect) = parsed else { + panic!("expected a connect"); + }; + assert_eq!(connect.local, None); + } + + #[test] + fn a_zero_local_port_means_the_same_as_an_absent_one() { + let parsed = parse(&request( + r#"{"command":"connect","params":{"peer":"npub1abc","remote_port":4242,"local_port":0}}"#, + )) + .unwrap(); + let Command::Connect(connect) = parsed else { + panic!("expected a connect"); + }; + assert_eq!(connect.local, None); + } + + #[test] + fn the_protocol_port_range_is_refused() { + let error = parse(&request( + r#"{"command":"listen","params":{"local_port":200}}"#, + )) + .unwrap_err(); + assert_eq!(error, CommandError::PortProtocol(200)); + } + + #[test] + fn the_standard_service_port_range_is_refused() { + // 256 is the IPv6 shim; refusing the whole tier keeps a client from + // binding it or anything reserved beside it. + let error = parse(&request( + r#"{"command":"listen","params":{"local_port":256}}"#, + )) + .unwrap_err(); + assert_eq!(error, CommandError::PortStandard(256)); + assert_eq!(check_bindable(1023), Err(CommandError::PortStandard(1023))); + assert!(check_bindable(1024).is_ok()); + } + + #[test] + fn a_remote_port_in_a_reserved_tier_is_refused_too() { + // The far end's shim is no more addressable than our own: a client that + // could name port 256 remotely would be injecting into the peer's IPv6 + // plane. + let error = parse(&request( + r#"{"command":"connect","params":{"peer":"npub1abc","remote_port":256}}"#, + )) + .unwrap_err(); + assert_eq!(error, CommandError::PortStandard(256)); + } + + #[test] + fn every_refusal_carries_the_errno_its_contract_names() { + // The rows of the table the client turns into `io::Error`. A client + // must be able to tell "that port is taken" from "that port is not + // yours to take" without matching English prose, which is the whole + // reason the code is on the wire. + assert_eq!(errno(&RegistryError::PortTaken(4242)), "EADDRINUSE"); + assert_eq!( + errno(&RegistryError::FlowTaken { + local: 4242, + remote: 5000 + }), + "EADDRINUSE" + ); + assert_eq!(errno(&RegistryError::NoPort), "EADDRNOTAVAIL"); + assert_eq!(errno(&RegistryError::TooManyFlows(256)), "EMFILE"); + assert_eq!( + errno(&RegistryError::Port(CommandError::PortStandard(256))), + "EADDRNOTAVAIL" + ); + assert_eq!( + errno(&RegistryError::Port(CommandError::PortProtocol(200))), + "EADDRNOTAVAIL" + ); + + // And the refusals that never reach the registry. + assert_eq!( + command_errno(&CommandError::Unknown("teleport".to_string())), + "EINVAL" + ); + assert_eq!(command_errno(&CommandError::NoParams("stats")), "EINVAL"); + } + + #[test] + fn an_unknown_command_names_itself_in_the_refusal() { + let error = parse(&request(r#"{"command":"teleport"}"#)).unwrap_err(); + assert_eq!(error, CommandError::Unknown("teleport".to_string())); + } + + #[test] + fn a_command_that_needs_params_refuses_without_them() { + let error = parse(&request(r#"{"command":"stats"}"#)).unwrap_err(); + assert_eq!(error, CommandError::NoParams("stats")); + } + + #[test] + fn params_of_the_wrong_shape_name_the_command() { + let request = Request { + command: "listen".to_string(), + params: Some(json!({"local_port": "not a number"})), + }; + let error = parse(&request).unwrap_err(); + let CommandError::BadParams { command, .. } = error else { + panic!("expected a params error"); + }; + assert_eq!(command, "listen"); + } + + #[test] + fn accept_and_reject_are_not_commands_this_api_serves() { + // They went with the round trip they existed for: a flow is taken by + // reading the listener's descriptor and refused by closing the one that + // arrives with it. A client still sending either must be told so. + for command in ["accept", "reject"] { + let error = parse(&request(&format!( + r#"{{"command":"{command}","params":{{"flow_id":7}}}}"# + ))) + .unwrap_err(); + assert_eq!(error, CommandError::Unknown(command.to_string())); + } + } + + #[test] + fn a_listen_may_name_no_port_and_gets_an_ephemeral_one() { + for line in [ + r#"{"command":"listen","params":{"local_port":0}}"#, + r#"{"command":"listen","params":{}}"#, + ] { + assert_eq!( + parse(&request(line)).unwrap(), + Command::Listen(Listen { local: None }), + "{line} should ask for an ephemeral port" + ); + } + assert_eq!( + parse(&request( + r#"{"command":"listen","params":{"local_port":4242}}"# + )) + .unwrap(), + Command::Listen(Listen { local: Some(4242) }) + ); + } +} diff --git a/src/native/registry.rs b/src/native/registry.rs new file mode 100644 index 00000000..d3b99fac --- /dev/null +++ b/src/native/registry.rs @@ -0,0 +1,1117 @@ +//! Which local ports the native API holds, and where an inbound datagram goes. +//! +//! The registry is the node's, not a client connection's: two clients must +//! conflict with each other over a port, and an inbound datagram has to find its +//! flow without knowing which connection opened it. It lives inside `Node` and +//! is reached from the client tasks through the `rx_loop`, so there is no lock +//! on the receive path. +//! +//! Everything here is a decision over maps. No I/O, no clock read, no sockets: +//! the caller passes the time in and performs whatever the decision names. That +//! is what lets the delivery rule be tested without a wire, and it is why the +//! same [`Registry::deliver`] serves both the debug arrival command and, once it +//! exists, the real FSP receive path. + +use crate::identity::NodeAddr; +use secp256k1::XOnlyPublicKey; +use std::collections::HashMap; +use thiserror::Error; +use tokio::sync::mpsc::Sender; + +use super::protocol::{PORT_EPHEMERAL_MIN, check_bindable}; + +/// How long a flow stays pending between the announcement and the listener +/// task wiring it, in milliseconds. +/// +/// Compiled in rather than configured: the window it bounds is a task hop +/// inside the daemon, not a client's behaviour, so an operator has nothing to +/// tune it against. Five seconds is generous for a hop and short enough that a +/// wedged listener task cannot accumulate pending flows. It is reasoned rather +/// than measured, and the announce-to-wired latency under load is what would +/// settle it. +pub const PENDING_DEADLINE_MS: u64 = 5_000; + +/// What identifies one flow on the wire, in both directions. +/// +/// The local port alone is not enough: two peers may both send to a listener's +/// port, and each is a separate flow. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct FlowKey { + /// The far end. + pub peer: NodeAddr, + /// The far end's port. + pub remote: u16, + /// This node's port. + pub local: u16, +} + +/// A datagram handed to an established flow. +pub type Datagram = Vec; + +/// What a listener is told when a new peer arrives on its port. +#[derive(Debug, Clone)] +pub struct Arrival { + /// The identifier the listener's task names the flow by. + pub flow: u64, + /// Which flow arrived. + pub key: FlowKey, + /// The peer's address, as the x-only public key its session authenticated. + /// Carried here rather than resolved later because the node address in + /// `key` is a truncated hash and does not invert. + pub pubkey: XOnlyPublicKey, +} + +/// Bounds on what the registry will hold. +#[derive(Debug, Clone, Copy)] +pub struct Limits { + /// Datagrams held for one flow, whether pending accept or established. + pub per_flow: usize, + /// Flows awaiting accept on one listener. + pub backlog: usize, + /// Flows this node holds at once, across every client. + pub max_flows: usize, +} + +/// Why a registry request was refused. +#[derive(Debug, Clone, PartialEq, Eq, Error)] +pub enum RegistryError { + /// Another listener or flow already holds this local port. + #[error("port {0} is already in use on this node")] + PortTaken(u16), + + /// A flow to that peer between those two ports already exists. + /// + /// Distinct from [`RegistryError::PortTaken`], and reachable when that one + /// is not: a flow accepted from a listener never owned the listener's port, + /// so closing the listener leaves the port free while the flow's key is + /// still held. The key is what an inbound datagram is demultiplexed on, so + /// a second flow carrying it would take the first one's traffic. + #[error("a flow to that peer between ports {local} and {remote} already exists")] + FlowTaken { + /// This node's port. + local: u16, + /// The far end's port. + remote: u16, + }, + + /// Every port in the ephemeral range is held. + #[error("no ephemeral port is free")] + NoPort, + + /// The node is already holding as many flows as it will. + #[error("this node holds its maximum of {0} flows")] + TooManyFlows(usize), + + /// The listener already has as many unaccepted flows as it will hold. + #[error("listener on port {0} has a full backlog")] + BacklogFull(u16), + + /// No pending flow carries this identifier. + #[error("no pending flow {0}")] + NoPending(u64), + + /// The port is one a client may not name. + #[error("{0}")] + Port(#[from] super::protocol::CommandError), +} + +/// Where an inbound datagram goes. +#[derive(Debug)] +pub enum Delivery<'a> { + /// An established flow owns it. The caller sends on this. + Flow(&'a Sender), + /// A listener owns the port and this peer is new to it. The caller raises + /// the arrival on that listener, then holds the datagram against the flow. + Arrived(&'a Sender, Arrival), + /// This flow was already announced and is waiting to be accepted. The + /// caller holds the datagram against it with [`Registry::hold`], and counts + /// a [`DropCause::QueueFull`] if it will not fit. + Pending(u64), + /// Nothing owns the port, or a bound is full. Count it and drop it. + Drop(DropCause), +} + +/// Why a datagram was not delivered. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DropCause { + /// No listener and no flow holds the destination port. + NoPort, + /// A listener holds the port but will not hold another pending flow. + BacklogFull, + /// The node is at its flow ceiling. + TooManyFlows, + /// The flow is pending accept and its queue is full. + QueueFull, +} + +/// What holds a local port. +#[derive(Debug)] +enum Owner { + /// A listener, which may own many flows on this port. + Listener, + /// One connected flow. + Flow(FlowKey), +} + +/// A listener bound to a local port. +struct Listener { + /// Where arrivals are announced. + arrivals: Sender, + /// Flows announced on this listener and not yet taken by its task. + pending: Vec, +} + +/// A flow the client has accepted or opened. +struct Flow { + /// The peer's address, so every report of this flow names the peer the + /// same way a connected flow's client named it. + pubkey: XOnlyPublicKey, + /// Where its datagrams go. + sink: Sender, + /// When the client took it, in milliseconds. A pending flow already + /// records its announcement time; an established one needs its own, or an + /// observer cannot tell a flow opened an hour ago from one opened just now. + at: u64, +} + +/// A flow announced to a listener and awaiting the client's answer. +struct Pending { + key: FlowKey, + /// The peer's address, captured where the session authenticated it. + pubkey: XOnlyPublicKey, + /// The port whose listener announced it. + listener: u16, + /// Datagrams held until the client accepts, so the first message from a new + /// peer is not lost to the accept round trip. + held: Vec, + /// When the arrival was announced, in milliseconds. + at: u64, +} + +/// One flow as an observer sees it. +/// +/// A copy rather than a borrow of the registry's own entry: the caller projects +/// it into a control snapshot the control socket reads off the rx_loop, and the +/// registry itself never leaves the loop that owns it. +#[derive(Debug, Clone, Copy)] +pub struct FlowView { + /// The identifier the client names it by. + pub flow: u64, + /// The peer and both ports. + pub key: FlowKey, + /// The peer's address. Always present, for an accepted flow as much as for + /// a connected one, because it is captured rather than resolved. + pub pubkey: XOnlyPublicKey, + /// Whether a client has taken it, as opposed to still awaiting accept. + pub established: bool, + /// Datagrams the node is holding for it: what a pending flow has kept back + /// for whoever accepts it, or what an established flow's client has not yet + /// read off its descriptor. + pub queued: usize, + /// When it was opened, accepted, or announced, in milliseconds. + pub at: u64, +} + +/// One listener as an observer sees it. +#[derive(Debug, Clone, Copy)] +pub struct ListenerView { + /// The local port it holds. + pub port: u16, + /// Flows announced on it and not yet answered. + pub backlog: usize, +} + +/// The node's native API registry. +pub struct Registry { + ports: HashMap, + listeners: HashMap, + flows: HashMap, + /// Client-facing identifiers for established flows. A client names a flow + /// by number rather than by the three-part key, and the numbers come from + /// the same counter as pending flows so the two can never collide. + by_id: HashMap, + pending: HashMap, + limits: Limits, + next_flow: u64, + next_ephemeral: u16, +} + +impl Registry { + /// An empty registry under `limits`. + pub fn new(limits: Limits) -> Self { + Self { + ports: HashMap::new(), + listeners: HashMap::new(), + flows: HashMap::new(), + by_id: HashMap::new(), + pending: HashMap::new(), + limits, + next_flow: 1, + next_ephemeral: PORT_EPHEMERAL_MIN, + } + } + + /// How many flows the node holds, established and pending. + pub fn flow_count(&self) -> usize { + self.flows.len() + self.pending.len() + } + + /// How many local ports are held. + pub fn port_count(&self) -> usize { + self.ports.len() + } + + /// Every flow the node holds, established and pending, ordered by + /// identifier so a reader sees a stable list across reads. + /// + /// An established flow's queue depth is what its bounded channel is + /// carrying, which is the same thing a pending flow's held datagrams are: + /// what the node is keeping because the client has not taken it yet. + pub fn flows(&self) -> Vec { + let established = self.by_id.iter().filter_map(|(id, key)| { + let flow = self.flows.get(key)?; + Some(FlowView { + flow: *id, + key: *key, + pubkey: flow.pubkey, + established: true, + queued: flow + .sink + .max_capacity() + .saturating_sub(flow.sink.capacity()), + at: flow.at, + }) + }); + let pending = self.pending.iter().map(|(id, entry)| FlowView { + flow: *id, + key: entry.key, + pubkey: entry.pubkey, + established: false, + queued: entry.held.len(), + at: entry.at, + }); + let mut views: Vec = established.chain(pending).collect(); + views.sort_by_key(|view| view.flow); + views + } + + /// Every listener the node holds, ordered by port. + pub fn listeners(&self) -> Vec { + let mut views: Vec = self + .listeners + .iter() + .map(|(port, listener)| ListenerView { + port: *port, + backlog: listener.pending.len(), + }) + .collect(); + views.sort_by_key(|view| view.port); + views + } + + /// Bind a listener to `port`, or to an ephemeral one when it is `None`, + /// announcing arrivals on `arrivals`. + /// + /// Returns the port actually held, which is what a client that asked for + /// an ephemeral one needs and is what `getsockname` reports after a + /// `bind(2)` with port 0. A named port still goes through + /// [`check_bindable`], so the reserved tiers are refused either way. + /// + /// A port that only accepted flows still hold is bindable, and the new + /// listener hears from new peers alone: an established key outranks the + /// port in [`Registry::deliver`], so the flows that outlived the previous + /// listener keep their own traffic. + pub fn listen( + &mut self, + port: Option, + arrivals: Sender, + ) -> Result { + let port = match port { + Some(port) => { + check_bindable(port)?; + if self.ports.contains_key(&port) { + return Err(RegistryError::PortTaken(port)); + } + port + } + None => self.take_ephemeral()?, + }; + self.ports.insert(port, Owner::Listener); + self.listeners.insert( + port, + Listener { + arrivals, + pending: Vec::new(), + }, + ); + Ok(port) + } + + /// Open a flow to `peer` on `remote`, from `local` or an ephemeral port. + /// + /// The peer is named by its x-only public key, which is the address, and + /// the node address the wire keys on is derived from it here rather than + /// passed alongside: two arguments naming one peer could disagree. + /// + /// Returns the local port the flow holds and the identifier the client uses + /// to name it. `now` is kept so the flow can report its own age. + pub fn connect( + &mut self, + peer: XOnlyPublicKey, + remote: u16, + local: Option, + sink: Sender, + now: u64, + ) -> Result<(u16, u64), RegistryError> { + check_bindable(remote)?; + if self.flow_count() >= self.limits.max_flows { + return Err(RegistryError::TooManyFlows(self.limits.max_flows)); + } + + let local = match local { + Some(port) => { + check_bindable(port)?; + if self.ports.contains_key(&port) { + return Err(RegistryError::PortTaken(port)); + } + port + } + None => self.take_ephemeral()?, + }; + + let key = FlowKey { + peer: NodeAddr::from_pubkey(&peer), + remote, + local, + }; + // Checked before the port is claimed, so a refusal leaves nothing to + // undo. Inserting over a live entry instead would drop that flow's sink + // without telling its client, and hand the older flow's release the + // power to free a port the newer one now owns. Only established flows + // are consulted: a pending flow sits on a live listener's port, and + // that listener refuses this call above. + if self.flows.contains_key(&key) { + return Err(RegistryError::FlowTaken { local, remote }); + } + let id = self.next_flow; + self.next_flow += 1; + self.ports.insert(local, Owner::Flow(key)); + self.flows.insert( + key, + Flow { + pubkey: peer, + sink, + at: now, + }, + ); + self.by_id.insert(id, key); + Ok((local, id)) + } + + /// The key of an established flow, by the identifier a client holds. + pub fn key_of(&self, flow: u64) -> Option { + self.by_id.get(&flow).copied() + } + + /// Claim a free port from the ephemeral range. + /// + /// Sweeps forward from where the last claim left off and wraps once, so a + /// port is not reused immediately after it is released. + fn take_ephemeral(&mut self) -> Result { + let span = u32::from(u16::MAX - PORT_EPHEMERAL_MIN) + 1; + for _ in 0..span { + let port = self.next_ephemeral; + self.next_ephemeral = if port == u16::MAX { + PORT_EPHEMERAL_MIN + } else { + port + 1 + }; + if !self.ports.contains_key(&port) { + return Ok(port); + } + } + Err(RegistryError::NoPort) + } + + /// Decide where an inbound datagram goes. + /// + /// Three steps, in order: an established flow that matches the whole key, a + /// listener on the destination port, then nothing. The order is what makes a + /// reply from a known peer reach its own flow rather than announcing itself + /// as a new arrival every time, and it is what keeps a flow that outlived + /// its listener receiving after another client has rebound the port. + /// + /// `peer` is what the wire supplied and is what every lookup keys on. + /// `pubkey` is what the session authenticated, and is used by the announce + /// arm alone: it is the address a new flow carries for the rest of its life. + pub fn deliver( + &mut self, + peer: NodeAddr, + pubkey: XOnlyPublicKey, + src: u16, + dst: u16, + now: u64, + ) -> Delivery<'_> { + let key = FlowKey { + peer, + remote: src, + local: dst, + }; + + // A flow already announced and not yet answered holds its datagrams + // rather than announcing again, so a peer that keeps sending does not + // fill the backlog with duplicates of itself. + if let Some(id) = self.pending_for(&key) { + return Delivery::Pending(id); + } + + if self.flows.contains_key(&key) { + let sink = &self.flows.get(&key).expect("checked above").sink; + return Delivery::Flow(sink); + } + + if !self.listeners.contains_key(&dst) { + return Delivery::Drop(DropCause::NoPort); + } + if self.flow_count() >= self.limits.max_flows { + return Delivery::Drop(DropCause::TooManyFlows); + } + { + let listener = self.listeners.get(&dst).expect("checked above"); + if listener.pending.len() >= self.limits.backlog { + return Delivery::Drop(DropCause::BacklogFull); + } + } + + let id = self.next_flow; + self.next_flow += 1; + self.pending.insert( + id, + Pending { + key, + pubkey, + listener: dst, + held: Vec::new(), + at: now, + }, + ); + let listener = self.listeners.get_mut(&dst).expect("checked above"); + listener.pending.push(id); + let arrival = Arrival { + flow: id, + key, + pubkey, + }; + Delivery::Arrived(&listener.arrivals, arrival) + } + + /// Hold a datagram for a pending flow, reporting whether it fitted. + pub fn hold(&mut self, flow: u64, datagram: Datagram) -> bool { + let Some(entry) = self.pending.get_mut(&flow) else { + return false; + }; + if entry.held.len() >= self.limits.per_flow { + return false; + } + entry.held.push(datagram); + true + } + + /// The pending flow carrying `key`, if one was announced. + fn pending_for(&self, key: &FlowKey) -> Option { + self.pending + .iter() + .find(|(_, entry)| entry.key == *key) + .map(|(id, _)| *id) + } + + /// Accept a pending flow, returning its key, its peer and whatever it held. + /// + /// The held datagrams go back to the caller so the first message from a new + /// peer reaches the client, rather than being lost to the hop between the + /// announcement and the listener task wiring the flow. + pub fn accept( + &mut self, + flow: u64, + sink: Sender, + now: u64, + ) -> Result<(FlowKey, XOnlyPublicKey, Vec), RegistryError> { + let entry = self + .pending + .remove(&flow) + .ok_or(RegistryError::NoPending(flow))?; + if let Some(listener) = self.listeners.get_mut(&entry.listener) { + listener.pending.retain(|id| *id != flow); + } + self.flows.insert( + entry.key, + Flow { + pubkey: entry.pubkey, + sink, + at: now, + }, + ); + self.by_id.insert(flow, entry.key); + Ok((entry.key, entry.pubkey, entry.held)) + } + + /// Discard a pending flow and anything it held. + /// + /// The undo for an announcement the listener's arrival channel refused: + /// the entry is registered by then, so dropping the datagram alone would + /// leave a pending flow nobody will ever be told about. + pub fn reject(&mut self, flow: u64) -> Result { + let entry = self + .pending + .remove(&flow) + .ok_or(RegistryError::NoPending(flow))?; + if let Some(listener) = self.listeners.get_mut(&entry.listener) { + listener.pending.retain(|id| *id != flow); + } + Ok(entry.key) + } + + /// Release an established flow and the port it held, if it held one. + /// + /// A flow accepted from a listener shares that listener's port and does not + /// own it, so only a flow the registry recorded as the port's owner + /// releases it. + /// Returns whether a flow was actually removed, so a caller counting + /// closures counts real ones rather than requests to close. + pub fn release(&mut self, key: &FlowKey) -> bool { + let removed = self.flows.remove(key).is_some(); + self.by_id.retain(|_, held| held != key); + if let Some(Owner::Flow(owned)) = self.ports.get(&key.local) + && owned == key + { + self.ports.remove(&key.local); + } + removed + } + + /// Release a listener, its port, and every flow still pending on it. + /// + /// **Flows already accepted on that port survive**, which is what + /// `close(listen_fd)` means: an accepted flow is keyed on its whole + /// [`FlowKey`] and never owned the listener's port, so it keeps receiving + /// its own peer's datagrams while the port becomes bindable again. What + /// stops that from crossing two clients' traffic is [`Registry::deliver`] + /// matching the key before it looks at the port, and [`Registry::connect`] + /// refusing a key a live flow already holds. + pub fn release_listener(&mut self, port: u16) { + if matches!(self.ports.get(&port), Some(Owner::Listener)) { + self.ports.remove(&port); + } + if let Some(listener) = self.listeners.remove(&port) { + for flow in listener.pending { + self.pending.remove(&flow); + } + } + } + + /// Discard pending flows the listener's task never wired. + /// + /// Returns what was discarded, so the caller can count it. The window this + /// closes is a task hop rather than a client round trip, so a non-zero + /// count means the daemon could not complete an arrival and is worth + /// alerting on. + pub fn expire(&mut self, now: u64) -> Vec { + let deadline = PENDING_DEADLINE_MS; + let expired: Vec = self + .pending + .iter() + .filter(|(_, entry)| now.saturating_sub(entry.at) >= deadline) + .map(|(id, _)| *id) + .collect(); + for flow in &expired { + if let Some(entry) = self.pending.remove(flow) + && let Some(listener) = self.listeners.get_mut(&entry.listener) + { + listener.pending.retain(|id| id != flow); + } + } + expired + } +} + +#[cfg(test)] +mod tests { + use super::*; + use tokio::sync::mpsc; + + fn limits() -> Limits { + Limits { + per_flow: 4, + backlog: 2, + max_flows: 8, + } + } + + /// A distinct peer key per byte. The address is derived from it rather + /// than invented, so a test cannot pair a key with an address the wire + /// would never have carried together. + fn key(byte: u8) -> XOnlyPublicKey { + crate::identity::Identity::from_secret_bytes(&[byte; 32]) + .expect("a non-zero secret is a valid key") + .pubkey() + } + + fn peer(byte: u8) -> NodeAddr { + NodeAddr::from_pubkey(&key(byte)) + } + + fn sink() -> (Sender, mpsc::Receiver) { + mpsc::channel(16) + } + + fn arrivals() -> (Sender, mpsc::Receiver) { + mpsc::channel(16) + } + + #[test] + fn a_port_has_one_owner() { + let mut registry = Registry::new(limits()); + let (tx, _rx) = arrivals(); + registry.listen(Some(4242), tx).unwrap(); + + let (tx, _rx) = arrivals(); + assert_eq!( + registry.listen(Some(4242), tx).unwrap_err(), + RegistryError::PortTaken(4242) + ); + + // A connected flow may not take a listener's port either: the owner is + // one or the other, never both. + let (tx, _rx) = sink(); + assert_eq!( + registry + .connect(key(1), 5000, Some(4242), tx, 0) + .unwrap_err(), + RegistryError::PortTaken(4242) + ); + } + + #[test] + fn a_reserved_port_is_refused_by_the_registry_too() { + // The command layer refuses these as well. Checking here means a caller + // that reaches the registry by another route cannot bind the shim's + // port. + let mut registry = Registry::new(limits()); + let (tx, _rx) = arrivals(); + assert!(matches!( + registry.listen(Some(256), tx).unwrap_err(), + RegistryError::Port(_) + )); + } + + #[test] + fn releasing_a_flow_frees_its_port() { + let mut registry = Registry::new(limits()); + let (tx, _rx) = sink(); + let (port, _id) = registry.connect(key(1), 5000, Some(4242), tx, 0).unwrap(); + assert_eq!(port, 4242); + assert_eq!(registry.port_count(), 1); + + registry.release(&FlowKey { + peer: peer(1), + remote: 5000, + local: 4242, + }); + assert_eq!(registry.port_count(), 0); + + // And the port can be taken again. + let (tx, _rx) = arrivals(); + assert!(registry.listen(Some(4242), tx).is_ok()); + } + + #[test] + fn ephemeral_ports_come_from_the_ephemeral_range_and_do_not_repeat() { + let mut registry = Registry::new(limits()); + let mut seen = Vec::new(); + for _ in 0..4 { + let (tx, _rx) = sink(); + let (port, _id) = registry.connect(key(1), 5000, None, tx, 0).unwrap(); + assert!(port >= PORT_EPHEMERAL_MIN, "got {port}"); + assert!(!seen.contains(&port), "port {port} handed out twice"); + seen.push(port); + } + } + + #[test] + fn an_established_flow_wins_over_a_listener_on_the_same_port() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + + // Announce a peer, accept it, then send again: the second datagram must + // reach the flow rather than announce the same peer twice. + let delivery = registry.deliver(peer(7), key(7), 5000, 4242, 0); + let flow = match delivery { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + let (tx, _rx) = sink(); + registry.accept(flow, tx, 0).unwrap(); + + assert!(matches!( + registry.deliver(peer(7), key(7), 5000, 4242, 0), + Delivery::Flow(_) + )); + } + + /// The listener is one object and the flows it produced are others, so + /// closing it must not disturb them and must not let another client take + /// one over. + fn accepted_flow_on_a_closed_listener() -> (Registry, mpsc::Receiver) { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + + let flow = match registry.deliver(peer(7), key(7), 5000, 4242, 0) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + let (tx, rx) = sink(); + registry.accept(flow, tx, 0).unwrap(); + + registry.release_listener(4242); + (registry, rx) + } + + #[test] + fn a_flow_accepted_from_a_listener_outlives_it_and_keeps_receiving_after_a_rebind() { + let (mut registry, _rx) = accepted_flow_on_a_closed_listener(); + + // The port is bindable again, because the flow never owned it. + let (arrival_tx, _arrival_rx) = arrivals(); + registry + .listen(Some(4242), arrival_tx) + .expect("a closed listener's port is free even while its flows live"); + + // The peer that was accepted still reaches its own flow rather than the + // new listener, which is what keeps two clients' traffic apart. + assert!( + matches!( + registry.deliver(peer(7), key(7), 5000, 4242, 0), + Delivery::Flow(_) + ), + "the surviving flow's key must outrank the new listener's port" + ); + + // And a peer that is new to the port reaches the new listener. + assert!(matches!( + registry.deliver(peer(8), key(8), 5001, 4242, 0), + Delivery::Arrived(_, _) + )); + } + + #[test] + fn a_connect_may_not_take_over_the_key_of_a_flow_that_outlived_its_listener() { + let (mut registry, mut rx) = accepted_flow_on_a_closed_listener(); + assert_eq!( + registry.port_count(), + 0, + "the closed listener gave its port back" + ); + + let (tx, _rx) = sink(); + assert_eq!( + registry + .connect(key(7), 5000, Some(4242), tx, 0) + .unwrap_err(), + RegistryError::FlowTaken { + local: 4242, + remote: 5000 + }, + "a second flow on one key would take the first one's datagrams" + ); + + // The refusal claimed nothing, and the flow it protected still works. + assert_eq!(registry.port_count(), 0); + match registry.deliver(peer(7), key(7), 5000, 4242, 0) { + Delivery::Flow(sink) => sink.try_send(b"still mine".to_vec()).unwrap(), + other => panic!("expected the surviving flow, got {other:?}"), + } + assert_eq!(rx.try_recv().unwrap(), b"still mine".to_vec()); + } + + #[test] + fn a_different_peer_on_the_same_listener_is_a_separate_flow() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + + let first = match registry.deliver(peer(1), key(1), 5000, 4242, 0) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + let second = match registry.deliver(peer(2), key(2), 5000, 4242, 0) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + assert_ne!(first, second); + } + + #[test] + fn an_unowned_port_drops() { + let mut registry = Registry::new(limits()); + assert!(matches!( + registry.deliver(peer(1), key(1), 5000, 9999, 0), + Delivery::Drop(DropCause::NoPort) + )); + } + + #[test] + fn a_full_backlog_drops_rather_than_growing() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + + // The backlog is two, so a third distinct peer is refused. Without this + // an unaccepting client would let any peer grow the node's memory. + for byte in 1..=2 { + assert!(matches!( + registry.deliver(peer(byte), key(byte), 5000, 4242, 0), + Delivery::Arrived(_, _) + )); + } + assert!(matches!( + registry.deliver(peer(3), key(3), 5000, 4242, 0), + Delivery::Drop(DropCause::BacklogFull) + )); + } + + #[test] + fn accepting_returns_what_the_flow_held() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + + let flow = match registry.deliver(peer(1), key(1), 5000, 4242, 0) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + assert!(registry.hold(flow, b"first".to_vec())); + + let (tx, _rx) = sink(); + let (flow_key, pubkey, held) = registry.accept(flow, tx, 0).unwrap(); + assert_eq!(flow_key.peer, peer(1)); + assert_eq!( + pubkey, + key(1), + "an accepted flow carries the peer's key, not only its wire address" + ); + assert_eq!(held, vec![b"first".to_vec()]); + } + + #[test] + fn a_pending_flow_holds_no_more_than_its_share() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + let flow = match registry.deliver(peer(1), key(1), 5000, 4242, 0) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + + for _ in 0..limits().per_flow { + assert!(registry.hold(flow, b"x".to_vec())); + } + assert!(!registry.hold(flow, b"x".to_vec()), "the cap should hold"); + } + + #[test] + fn rejecting_frees_the_backlog_slot() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + + let flow = match registry.deliver(peer(1), key(1), 5000, 4242, 0) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + registry.reject(flow).unwrap(); + assert_eq!(registry.flow_count(), 0); + assert_eq!( + registry.reject(flow).unwrap_err(), + RegistryError::NoPending(flow) + ); + } + + #[test] + fn a_pending_flow_expires_and_a_fresh_one_does_not() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + let flow = match registry.deliver(peer(1), key(1), 5000, 4242, 1_000) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + + // One millisecond short of the deadline, nothing expires. The healthy + // path matters as much as the guard: an expiry that fired early would + // discard flows a client was about to accept. + assert!(registry.expire(1_000 + PENDING_DEADLINE_MS - 1).is_empty()); + assert_eq!(registry.expire(1_000 + PENDING_DEADLINE_MS), vec![flow]); + assert_eq!(registry.flow_count(), 0); + } + + #[test] + fn releasing_a_listener_takes_its_pending_flows_with_it() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + registry.deliver(peer(1), key(1), 5000, 4242, 0); + assert_eq!(registry.flow_count(), 1); + + registry.release_listener(4242); + assert_eq!(registry.flow_count(), 0); + assert_eq!(registry.port_count(), 0); + } + + #[test] + fn the_node_flow_ceiling_holds() { + let mut registry = Registry::new(limits()); + for _ in 0..limits().max_flows { + let (tx, _rx) = sink(); + registry.connect(key(1), 5000, None, tx, 0).unwrap(); + } + let (tx, _rx) = sink(); + assert_eq!( + registry.connect(key(1), 5000, None, tx, 0).unwrap_err(), + RegistryError::TooManyFlows(limits().max_flows) + ); + } + + #[test] + fn the_flow_view_reports_both_kinds_of_flow_with_their_queue_depth_and_age() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + + let (tx, _rx) = sink(); + let (_port, opened) = registry + .connect(key(1), 5000, Some(6000), tx, 1_000) + .unwrap(); + + let announced = match registry.deliver(peer(2), key(2), 5001, 4242, 2_000) { + Delivery::Arrived(_, arrival) => arrival.flow, + other => panic!("expected an arrival, got {other:?}"), + }; + assert!(registry.hold(announced, b"held".to_vec())); + + let views = registry.flows(); + assert_eq!(views.len(), 2); + + let established = views + .iter() + .find(|v| v.flow == opened) + .expect("opened flow"); + assert!(established.established); + assert_eq!(established.key.peer, peer(1)); + assert_eq!(established.key.local, 6000); + assert_eq!(established.key.remote, 5000); + assert_eq!(established.queued, 0, "nothing has been delivered to it"); + assert_eq!(established.at, 1_000); + + let pending = views + .iter() + .find(|v| v.flow == announced) + .expect("pending flow"); + assert!(!pending.established); + assert_eq!(pending.key.peer, peer(2)); + assert_eq!(pending.queued, 1, "the held datagram is queued against it"); + assert_eq!(pending.at, 2_000); + } + + #[test] + fn an_established_flow_reports_the_datagrams_its_client_has_not_read() { + let mut registry = Registry::new(limits()); + let (tx, _rx) = sink(); + registry.connect(key(1), 5000, Some(6000), tx, 0).unwrap(); + + let flow = FlowKey { + peer: peer(1), + remote: 5000, + local: 6000, + }; + match registry.deliver(peer(1), key(1), 5000, 6000, 0) { + Delivery::Flow(sink) => sink.try_send(b"unread".to_vec()).unwrap(), + other => panic!("expected the established flow, got {other:?}"), + } + + assert_eq!(registry.flows()[0].queued, 1); + assert!(registry.release(&flow)); + } + + #[test] + fn the_listener_view_reports_its_port_and_backlog_depth() { + let mut registry = Registry::new(limits()); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4242), arrival_tx).unwrap(); + let (arrival_tx, _arrival_rx) = arrivals(); + registry.listen(Some(4243), arrival_tx).unwrap(); + + assert_eq!( + registry + .listeners() + .iter() + .map(|v| (v.port, v.backlog)) + .collect::>(), + vec![(4242, 0), (4243, 0)], + "ports come out ordered and idle listeners have no backlog" + ); + + registry.deliver(peer(1), key(1), 5000, 4242, 0); + registry.deliver(peer(2), key(2), 5000, 4242, 0); + assert_eq!(registry.listeners()[0].backlog, 2); + assert_eq!(registry.listeners()[1].backlog, 0); + } + + #[test] + fn a_listener_that_names_no_port_is_given_an_ephemeral_one_and_told_which() { + let mut registry = Registry::new(limits()); + let (tx, _rx) = arrivals(); + let port = registry.listen(None, tx).unwrap(); + assert!( + port >= PORT_EPHEMERAL_MIN, + "an allocated listener port comes from the ephemeral range, got {port}" + ); + + // The reported port is the one actually held, which is the whole point + // of reporting it: a listener that answered 0 would leave its client + // unable to name the port to a peer. + let (tx, _rx) = arrivals(); + assert_eq!( + registry.listen(Some(port), tx).unwrap_err(), + RegistryError::PortTaken(port) + ); + } + + #[test] + fn every_flow_reports_the_peers_key_whether_it_was_opened_or_accepted() { + let mut registry = Registry::new(limits()); + let (tx, _rx) = arrivals(); + registry.listen(Some(4242), tx).unwrap(); + + let (tx, _rx) = sink(); + let (_port, opened) = registry.connect(key(1), 5000, Some(6000), tx, 0).unwrap(); + let announced = match registry.deliver(peer(2), key(2), 5001, 4242, 0) { + Delivery::Arrived(_, arrival) => { + assert_eq!(arrival.pubkey, key(2), "the announcement names the peer"); + arrival.flow + } + other => panic!("expected an arrival, got {other:?}"), + }; + + let views = registry.flows(); + let of = |id: u64| views.iter().find(|v| v.flow == id).expect("flow").pubkey; + assert_eq!(of(opened), key(1)); + assert_eq!( + of(announced), + key(2), + "a flow learned from the wire reports the same kind of address as one \ + the client opened" + ); + } +} diff --git a/src/native/seqpacket.rs b/src/native/seqpacket.rs new file mode 100644 index 00000000..fe54cb60 --- /dev/null +++ b/src/native/seqpacket.rs @@ -0,0 +1,401 @@ +//! A connected `SOCK_SEQPACKET` pair, one half driven by tokio readiness. +//! +//! `SOCK_SEQPACKET` is what a datagram API wants from a local socket: it keeps +//! message boundaries, it is flow controlled, and it reports end of file when +//! the peer closes. `SOCK_STREAM` loses the boundaries and `SOCK_DGRAM` gives a +//! weaker close signal. +//! +//! Tokio ships no type for it — `UnixStream` is `SOCK_STREAM` and +//! `UnixDatagram` is `SOCK_DGRAM` — so the daemon's half is driven through +//! [`AsyncFd`], which is tokio's supported way to put an arbitrary file +//! descriptor under the reactor. +//! +//! **Not available on macOS**, which does not implement `SOCK_SEQPACKET` for +//! `AF_UNIX`. Everything else this module needs does work there: `SCM_RIGHTS` +//! is supported, and `AsyncFd` is backed by kqueue. So a macOS port is a change +//! of socket type to `SOCK_DGRAM`, which macOS does support and which also +//! keeps message boundaries, plus one thing that has to be **measured on a Mac +//! rather than assumed**: whether a closed peer on a connected `SOCK_DGRAM` +//! pair is distinguishable from an empty datagram. The `POLLHUP` result below +//! was measured on Linux and BSD poll semantics for datagram sockets differ. + +use std::io; +use std::os::fd::{AsRawFd, FromRawFd, OwnedFd, RawFd}; +use std::time::Duration; +use tokio::io::unix::AsyncFd; + +/// What one receive attempt produced. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Received { + /// A datagram of this many bytes. Zero is a legitimate value: a client may + /// send an empty datagram, and that is not the same as closing. + Datagram(usize), + /// The peer closed its half of the pair. + Eof, +} + +/// Create a connected `SOCK_SEQPACKET` pair. +/// +/// Both descriptors are close-on-exec so neither leaks into a child process. +/// Neither is set non-blocking here: the two halves are independent sockets, so +/// [`Seqpacket::new`] can make the daemon's half non-blocking for the reactor +/// while the half handed to the client stays blocking, which is what a client +/// calling `recv` in a loop expects. +pub fn pair() -> io::Result<(OwnedFd, OwnedFd)> { + let mut fds = [0 as libc::c_int; 2]; + // SAFETY: `fds` is a two-element array of the type socketpair writes, and + // the return value is checked before either descriptor is read. + let rc = unsafe { + libc::socketpair( + libc::AF_UNIX, + libc::SOCK_SEQPACKET | libc::SOCK_CLOEXEC, + 0, + fds.as_mut_ptr(), + ) + }; + if rc != 0 { + return Err(io::Error::last_os_error()); + } + // SAFETY: socketpair reported success, so both entries are open + // descriptors this process now owns. + Ok(unsafe { (OwnedFd::from_raw_fd(fds[0]), OwnedFd::from_raw_fd(fds[1])) }) +} + +/// Size the send buffer of `fd`, in bytes. +/// +/// Used on the daemon's half of a listener's pair, where the buffer is the only +/// bound on arrivals a client has stopped reading. `SO_SNDBUF` on an `AF_UNIX` +/// socket accounts bytes plus per-message overhead rather than messages, so a +/// caller converting a message count to bytes is approximating and should +/// approximate generously. Linux doubles what it is given and clamps to its own +/// minimum, which is why nothing here reads the value back and asserts on it. +pub fn set_sndbuf(fd: &OwnedFd, bytes: usize) -> io::Result<()> { + let size = libc::c_int::try_from(bytes) + .map_err(|_| io::Error::new(io::ErrorKind::InvalidInput, "send buffer size too large"))?; + // SAFETY: the descriptor is open for the call, and the pointer and length + // describe one `c_int` that outlives it. + let rc = unsafe { + libc::setsockopt( + fd.as_raw_fd(), + libc::SOL_SOCKET, + libc::SO_SNDBUF, + std::ptr::addr_of!(size).cast(), + std::mem::size_of::() as libc::socklen_t, + ) + }; + if rc != 0 { + return Err(io::Error::last_os_error()); + } + Ok(()) +} + +/// The daemon's half of a flow's or a listener's socket pair, registered with +/// the reactor. +pub struct Seqpacket { + inner: AsyncFd, +} + +impl Seqpacket { + /// Put `fd` under the reactor, making it non-blocking first. + /// + /// `AsyncFd` requires a non-blocking descriptor: a blocking one would stall + /// the whole runtime thread inside a syscall the reactor believed would + /// return at once. + pub fn new(fd: OwnedFd) -> io::Result { + set_nonblocking(fd.as_raw_fd(), true)?; + Ok(Self { + inner: AsyncFd::new(fd)?, + }) + } + + /// The raw descriptor, for a caller that must perform its own syscall. + /// + /// The one such caller is the listener's task, which needs a send that + /// reports a full buffer rather than waiting for one; see + /// [`try_send`](super::fdpass::try_send). Everything else goes through + /// [`Seqpacket::send`] and [`Seqpacket::recv`]. + pub(super) fn raw(&self) -> RawFd { + self.inner.get_ref().as_raw_fd() + } + + /// Receive one datagram, or report that the peer closed. + /// + /// A datagram longer than `buf` is truncated and the remainder discarded, + /// which is `SOCK_SEQPACKET` behaviour. Callers size `buf` at the largest + /// payload the API accepts, so a truncation means the client exceeded it. + pub async fn recv(&self, buf: &mut [u8]) -> io::Result { + loop { + let mut guard = self.inner.readable().await?; + let attempt = guard.try_io(|inner| recv_once(inner.get_ref().as_raw_fd(), buf)); + match attempt { + Ok(result) => return result, + // The reactor said readable and the syscall disagreed. Clear + // the readiness and wait again rather than reporting an error. + Err(_would_block) => continue, + } + } + } + + /// Send one datagram. + /// + /// `SOCK_SEQPACKET` delivers it whole or not at all, so a short write is + /// not a case the caller has to handle. + pub async fn send(&self, buf: &[u8]) -> io::Result { + loop { + let mut guard = self.inner.writable().await?; + let attempt = guard.try_io(|inner| { + // SAFETY: the descriptor is owned and open, and the pointer and + // length describe `buf`. + let n = unsafe { + libc::send( + inner.get_ref().as_raw_fd(), + buf.as_ptr().cast(), + buf.len(), + libc::MSG_NOSIGNAL, + ) + }; + if n < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(n as usize) + } + }); + match attempt { + Ok(result) => return result, + Err(_would_block) => continue, + } + } + } +} + +/// One receive, distinguishing an empty datagram from end of file. +/// +/// Both produce a zero-byte read, so something else has to tell them apart. +/// `MSG_EOR` is the technique the manual pages suggest and it **does not work +/// here**: measured on Linux 6.8, `recvmsg` on an `AF_UNIX` `SOCK_SEQPACKET` +/// socket returns `msg_flags == 0` for a normal message, for an empty message +/// and at end of file alike, so the flag carries no information. +/// +/// `POLLHUP` does discriminate, measured the same way. After a zero-byte read, +/// a queued empty datagram leaves the socket with no events pending, while a +/// closed peer leaves `POLLHUP` set and latched. So a zero-byte read is end of +/// file only when the peer has hung up. +/// +/// This matters because reading an empty datagram as a close would let a client +/// tear down its own flow by sending nothing, and the defect would present as a +/// spurious disconnect. +fn recv_once(fd: RawFd, buf: &mut [u8]) -> io::Result { + // SAFETY: the descriptor is open and the pointer and length describe `buf`. + let n = unsafe { libc::recv(fd, buf.as_mut_ptr().cast(), buf.len(), 0) }; + if n < 0 { + return Err(io::Error::last_os_error()); + } + if n == 0 && peer_hung_up(fd) { + return Ok(Received::Eof); + } + Ok(Received::Datagram(n as usize)) +} + +/// Whether the peer has closed its half of the pair. +/// +/// `events` is left empty on purpose: `POLLHUP` is reported in `revents` +/// whether or not it was requested, which was confirmed by measurement rather +/// than assumed. The poll does not block, and runs only on the zero-byte path. +/// +/// Visible within [`super`] because the client half needs the same +/// discrimination on the same socket pair: the rule belongs to the pair, not to +/// the end that reads it. +pub(super) fn peer_hung_up(fd: RawFd) -> bool { + let mut poll = libc::pollfd { + fd, + events: 0, + revents: 0, + }; + // SAFETY: `poll` points at one live pollfd and the call cannot block. + let rc = unsafe { libc::poll(&mut poll, 1, 0) }; + rc > 0 && (poll.revents & libc::POLLHUP) != 0 +} + +/// Set or clear a receive or send timeout on a socket. +/// +/// `option` is `SO_RCVTIMEO` or `SO_SNDTIMEO`. `None` clears the timeout, which +/// the kernel spells as a zero `timeval`. +/// +/// A zero duration is refused with `EINVAL` rather than passed through, because +/// the kernel reads a zero `timeval` as "no timeout" and a caller asking for +/// zero means the opposite. `std::net` makes the same refusal for the same +/// reason, and silently inverting the request would be worse than failing it. +pub(super) fn set_timeout(fd: RawFd, option: libc::c_int, dur: Option) -> io::Result<()> { + let timeout = match dur { + Some(d) if d == Duration::ZERO => { + return Err(io::Error::from_raw_os_error(libc::EINVAL)); + } + // Saturating rather than wrapping: a duration past `time_t` becomes the + // longest wait the kernel can express, which is the caller's intent. + Some(d) => libc::timeval { + tv_sec: d.as_secs().min(libc::time_t::MAX as u64) as libc::time_t, + tv_usec: libc::suseconds_t::from(d.subsec_micros()), + }, + None => libc::timeval { + tv_sec: 0, + tv_usec: 0, + }, + }; + // SAFETY: the descriptor is open, and the pointer and length describe a + // `timeval` this frame owns. + let rc = unsafe { + libc::setsockopt( + fd, + libc::SOL_SOCKET, + option, + std::ptr::from_ref(&timeout).cast(), + size_of::() as libc::socklen_t, + ) + }; + if rc < 0 { + return Err(io::Error::last_os_error()); + } + Ok(()) +} + +/// Read back a receive or send timeout, or `None` when none is set. +pub(super) fn timeout(fd: RawFd, option: libc::c_int) -> io::Result> { + let mut timeout = libc::timeval { + tv_sec: 0, + tv_usec: 0, + }; + let mut len = size_of::() as libc::socklen_t; + // SAFETY: the descriptor is open, and both pointers describe values this + // frame owns and keeps alive across the call. + let rc = unsafe { + libc::getsockopt( + fd, + libc::SOL_SOCKET, + option, + std::ptr::from_mut(&mut timeout).cast(), + &mut len, + ) + }; + if rc < 0 { + return Err(io::Error::last_os_error()); + } + if timeout.tv_sec == 0 && timeout.tv_usec == 0 { + return Ok(None); + } + Ok(Some( + Duration::from_secs(timeout.tv_sec as u64) + Duration::from_micros(timeout.tv_usec as u64), + )) +} + +/// Set or clear a descriptor's non-blocking mode, preserving its other flags. +/// +/// Read-modify-write rather than a bare `F_SETFL`, because the flag word also +/// carries the access mode and `O_APPEND`, and writing `O_NONBLOCK` alone would +/// drop them. +pub(super) fn set_nonblocking(fd: RawFd, nonblocking: bool) -> io::Result<()> { + // SAFETY: the descriptor is open for the duration of both calls. + let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) }; + if flags < 0 { + return Err(io::Error::last_os_error()); + } + let wanted = if nonblocking { + flags | libc::O_NONBLOCK + } else { + flags & !libc::O_NONBLOCK + }; + // SAFETY: as above; `wanted` is the value just read back, one bit changed. + if unsafe { libc::fcntl(fd, libc::F_SETFL, wanted) } < 0 { + return Err(io::Error::last_os_error()); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::{Read, Write}; + use std::os::unix::net::UnixStream as StdUnixStream; + + /// Turn the client half into something a blocking test can drive, the way a + /// client process would after receiving it. + fn client(fd: OwnedFd) -> StdUnixStream { + StdUnixStream::from(fd) + } + + #[tokio::test] + async fn message_boundaries_survive_in_both_directions() { + let (daemon, theirs) = pair().unwrap(); + let daemon = Seqpacket::new(daemon).unwrap(); + let mut theirs = client(theirs); + + // Three writes must arrive as three datagrams, not one run of bytes. + // This is the property SOCK_STREAM would lose. + theirs.write_all(b"one").unwrap(); + theirs.write_all(b"two").unwrap(); + theirs.write_all(b"three").unwrap(); + + let mut buf = [0u8; 64]; + assert_eq!(daemon.recv(&mut buf).await.unwrap(), Received::Datagram(3)); + assert_eq!(&buf[..3], b"one"); + assert_eq!(daemon.recv(&mut buf).await.unwrap(), Received::Datagram(3)); + assert_eq!(&buf[..3], b"two"); + assert_eq!(daemon.recv(&mut buf).await.unwrap(), Received::Datagram(5)); + assert_eq!(&buf[..5], b"three"); + + daemon.send(b"alpha").await.unwrap(); + daemon.send(b"beta").await.unwrap(); + let mut got = [0u8; 64]; + assert_eq!(theirs.read(&mut got).unwrap(), 5); + assert_eq!(&got[..5], b"alpha"); + assert_eq!(theirs.read(&mut got).unwrap(), 4); + assert_eq!(&got[..4], b"beta"); + } + + #[tokio::test] + async fn closing_the_client_half_reports_end_of_file() { + let (daemon, theirs) = pair().unwrap(); + let daemon = Seqpacket::new(daemon).unwrap(); + drop(client(theirs)); + + let mut buf = [0u8; 64]; + assert_eq!(daemon.recv(&mut buf).await.unwrap(), Received::Eof); + } + + #[tokio::test] + async fn an_empty_datagram_is_not_end_of_file() { + // The whole reason recv_once uses recvmsg. If an empty datagram read as + // a close, a client could tear down its own flow by sending nothing, + // and the bug would look like a spurious disconnect. + let (daemon, theirs) = pair().unwrap(); + let daemon = Seqpacket::new(daemon).unwrap(); + let mut theirs = client(theirs); + + // `write_all(b"")` is a no-op in Rust and never reaches the socket, so + // the zero-length datagram has to be sent with `send` directly. The + // first version of this test used `write_all` and asserted a behaviour + // it had not exercised. + // SAFETY: the descriptor is open and owned by `theirs`. + let sent = unsafe { libc::send(theirs.as_raw_fd(), std::ptr::null(), 0, 0) }; + assert_eq!(sent, 0, "{}", io::Error::last_os_error()); + theirs.write_all(b"after").unwrap(); + + let mut buf = [0u8; 64]; + assert_eq!(daemon.recv(&mut buf).await.unwrap(), Received::Datagram(0)); + assert_eq!(daemon.recv(&mut buf).await.unwrap(), Received::Datagram(5)); + } + + #[tokio::test] + async fn the_client_half_is_left_blocking() { + // The daemon's half is made non-blocking for the reactor. If that flag + // reached the client's half, a client doing an ordinary blocking recv + // would get EAGAIN instead of waiting, which is a trap worth a test + // rather than a comment. + let (daemon, theirs) = pair().unwrap(); + let _daemon = Seqpacket::new(daemon).unwrap(); + + // SAFETY: the descriptor is open and owned by `theirs`. + let flags = unsafe { libc::fcntl(theirs.as_raw_fd(), libc::F_GETFL) }; + assert!(flags >= 0); + assert_eq!(flags & libc::O_NONBLOCK, 0); + } +} diff --git a/src/node/dataplane/rx_loop.rs b/src/node/dataplane/rx_loop.rs index 642ca75e..43723ccf 100644 --- a/src/node/dataplane/rx_loop.rs +++ b/src/node/dataplane/rx_loop.rs @@ -134,6 +134,51 @@ impl Node { // Drop unused sender to avoid keeping channel open if control is disabled drop(control_tx); + // Native datagram API socket. Experimental, off by default, and built + // on Linux and FreeBSD only: Windows has no way to pass a descriptor + // between processes, and macOS has no AF_UNIX SOCK_SEQPACKET. Bound + // synchronously so a bad path or a socket already in use is reported + // here, before the node starts serving, rather than at whatever later + // moment a spawned bind happened to run. + // Two channels, as the TUN plane has: registrations go one way and are + // rare, datagrams go the other and arrive in bursts. Keeping them apart + // lets the data arm drain in batches without a registration waiting + // behind a burst of traffic. + let (native_out_tx, mut native_outbound_rx) = + tokio::sync::mpsc::channel::(1024); + let _native_out_guard = native_out_tx.clone(); + let (mut native_rx, _native_guard) = { + let (tx, rx) = tokio::sync::mpsc::channel::(64); + // The `cfg` sits on the binding rather than on the arm that drains + // the receiver, because `tokio::select!` does not accept one. Where + // there is no listener the channel exists and nothing ever sends, + // and the guard keeps it open so the arm never sees a closed + // receiver. + #[cfg(any(target_os = "linux", target_os = "freebsd"))] + let guard = { + let mut guard = Some(tx.clone()); + if self.config().node.native_api.enabled { + match crate::native::NativeApi::bind(&self.config().node.native_api) { + Ok(socket) => { + // The accept loop owns a sender, so the dummy guard + // is dropped: the channel closes when the last + // client task ends, not while one is still serving. + guard = None; + let npub = self.npub(); + tokio::spawn(socket.accept_loop(tx, native_out_tx.clone(), npub)); + } + Err(e) => { + warn!(error = %e, "Failed to bind native API socket"); + } + } + } + guard + }; + #[cfg(not(any(target_os = "linux", target_os = "freebsd")))] + let guard = Some(tx.clone()); + (rx, guard) + }; + // Decrypt-worker fallback receiver. The worker pushes each // authenticated FMP plaintext here so rx_loop can finish the // per-peer side-effects (stats, MMP, ECN, link dispatch). @@ -311,6 +356,30 @@ impl Node { ); self.register_identity(identity.node_addr, identity.pubkey); } + // Native API datagrams a client wrote to its descriptor. Drained + // in a burst like the TUN arm, for the same reason: one wake-up + // should clear what a client handed over, not one datagram. + Some(out) = native_outbound_rx.recv() => { + self.handle_native_outbound(out.key, out.peer, out.payload).await; + let mut drained = 0; + while drained < 256 { + match native_outbound_rx.try_recv() { + Ok(next) => { + self.handle_native_outbound(next.key, next.peer, next.payload).await; + drained += 1; + } + Err(_) => break, + } + } + } + // Native API registry requests. Placed after the hot inbound + // path so a burst of client registrations cannot delay packet + // processing. No `cfg` here: `tokio::select!` does not accept + // one, so the channel exists on every platform and only the + // listener that feeds it is gated. + Some(message) = native_rx.recv() => { + self.handle_native(message); + } Some((request, response_tx)) = control_rx.recv() => { // Only mutating COMMAND requests (`connect` / `disconnect`) // reach the rx_loop now. Every pure-read `show_*` query is @@ -348,6 +417,10 @@ impl Node { instr_step!(instr_on, crate::instr::Domain::Tick, crate::instr::Step::WholeTick, { instr_step!(instr_on, crate::instr::Domain::Tick, crate::instr::Step::CheckTimeouts, self.check_timeouts().await); + // Discard flows the rx_loop announced and a listener's + // own task never took. Cheap: it walks only the pending + // map, which the backlog bounds. + self.native_expire(); let now_ms = Self::now_ms(); instr_step!(instr_on, crate::instr::Domain::Tick, crate::instr::Step::ReloadPeerAcl, self.reload_peer_acl().await); diff --git a/src/node/handlers/mod.rs b/src/node/handlers/mod.rs index 4abfe046..c0e9b64e 100644 --- a/src/node/handlers/mod.rs +++ b/src/node/handlers/mod.rs @@ -3,6 +3,8 @@ pub(in crate::node) mod handshake; pub(crate) mod lookup; mod mmp; +mod native; +pub(in crate::node) use native::PendingNative; pub(crate) mod probe; mod rekey; pub(in crate::node) mod session; diff --git a/src/node/handlers/native.rs b/src/node/handlers/native.rs new file mode 100644 index 00000000..93a0ebd9 --- /dev/null +++ b/src/node/handlers/native.rs @@ -0,0 +1,228 @@ +//! Serve the native API registry from inside the `rx_loop`. +//! +//! The registry lives in `Node`, so every client request arrives here and is +//! answered on the `oneshot` it carried. Nothing else touches the registry, +//! which is why the receive path takes no lock. +//! +//! The work itself is in [`crate::native::link`], as a free function over the +//! registry. This file supplies the clock and nothing else, so the logic the +//! daemon runs is the same logic a test drives without building a node. + +use crate::identity::NodeAddr; +use crate::native::link::{self, NativeMessage, Outcome, Served}; +use crate::native::registry::FlowKey; +use crate::node::Node; +use crate::proto::fsp::wire::FSP_PORT_HEADER_SIZE; +use crate::upper::icmp::FIPS_OVERHEAD; +use secp256k1::XOnlyPublicKey; +use tracing::debug; + +/// One native datagram held while its destination's session establishes. +/// +/// Carries its ports, which is exactly what the TUN pending queue lacks and +/// why native datagrams need a queue of their own. +#[derive(Debug, Clone)] +pub(in crate::node) struct PendingNative { + /// The flow this datagram belongs to. + pub key: FlowKey, + /// The payload, without any port header: `send_session_data` adds that. + pub payload: Vec, +} + +impl Node { + /// Answer one native API request, and count what it did. + /// + /// The counting is here rather than in [`link::serve`] because the core + /// decides and the shell observes. `serve` reports its outcome and knows + /// nothing about counters. + pub(in crate::node) fn handle_native(&mut self, message: NativeMessage) { + let max = self.native_max_payload(); + let served = link::serve(&mut self.native, message, Self::now_ms(), max); + let native = &self.metrics.native; + match served { + Served::Opened => native.flows_opened.inc(), + Served::Accepted => native.flows_accepted.inc(), + Served::Discarded(reason) => native.record_drop(reason), + Served::Released(closed) => native.flows_closed.add(closed as u64), + Served::Delivered(outcome, bytes) => self.count_native_delivery(outcome, bytes), + Served::Untracked => {} + } + } + + /// Count one dispatched inbound datagram. + /// + /// Shared by the wire path and the debug arrival command, which run the same + /// dispatch, so a counter cannot disagree between them. + fn count_native_delivery(&self, outcome: Outcome, bytes: usize) { + let native = &self.metrics.native; + match outcome { + Outcome::Delivered | Outcome::Announced(_) | Outcome::Held(_) => { + native.record_received(bytes); + } + Outcome::Dropped(reason) => native.record_drop(reason), + } + } + + /// Deliver one inbound datagram to whatever owns its destination port. + /// + /// `pubkey` is the peer's address, read from the session entry that + /// authenticated it. It is captured here, at the one call site the wire + /// has, rather than resolved when a report is rendered: the node address + /// the wire carries is a truncated hash and does not invert. + pub(in crate::node) fn native_deliver( + &mut self, + peer: crate::identity::NodeAddr, + pubkey: XOnlyPublicKey, + src: u16, + dst: u16, + data: Vec, + ) -> Outcome { + let bytes = data.len(); + let outcome = link::deliver( + &mut self.native, + peer, + pubkey, + src, + dst, + data, + Self::now_ms(), + ); + self.count_native_delivery(outcome, bytes); + outcome + } + + /// The largest payload a native flow may send, in bytes. + /// + /// Wire size is `FIPS_OVERHEAD` plus the four-byte port header plus the + /// payload, so the payload is whatever is left of the transport MTU. This + /// is 40 bytes more than the same application data gets through the IPv6 + /// shim, which spends them on a header the application never sees. + pub(in crate::node) fn native_max_payload(&self) -> u16 { + self.transport_mtu() + .saturating_sub(FIPS_OVERHEAD) + .saturating_sub(FSP_PORT_HEADER_SIZE as u16) + } + + /// Send one datagram a native API client wrote to its descriptor. + /// + /// Mirrors `handle_tun_outbound` with three differences: the destination + /// comes from the flow rather than from a parsed IPv6 header, an error is + /// reported to the client rather than as an ICMPv6 message, and the flow's + /// own ports are used instead of the shim's. + pub(in crate::node) async fn handle_native_outbound( + &mut self, + key: FlowKey, + peer: XOnlyPublicKey, + payload: Vec, + ) { + let max = self.native_max_payload() as usize; + if payload.len() > max { + debug!( + len = payload.len(), + max, "Native API datagram exceeds the path payload limit, dropping" + ); + self.metrics.native.drop_oversize.inc(); + return; + } + + if let Some(entry) = self.sessions.get(&key.peer) { + if entry.is_established() { + let bytes = payload.len(); + if let Err(error) = self + .send_session_data(&key.peer, key.local, key.remote, &payload) + .await + { + debug!( + peer = %self.peer_display_name(&key.peer), + error = %error, + "Failed to send a native datagram" + ); + } else { + self.metrics.native.record_sent(bytes); + } + return; + } + self.queue_native(key, payload); + return; + } + + // No session. Start one and hold the datagram; if there is no route, + // ask discovery for one and hold it anyway, exactly as the TUN path + // does. + // + // Every flow carries its peer's address, whether the client named it or + // the session authenticated it, so there is no cache lookup here and no + // datagram dropped for want of a key. `initiate_session` wants a full + // key, and the parity it is lifted with does not matter: the handshake + // hashes the DH output x-only and forces even parity in the XK + // premessage, so both parities derive the same material. + let pubkey = peer.public_key(secp256k1::Parity::Even); + if let Err(error) = self.initiate_session(key.peer, pubkey).await { + debug!( + peer = %self.peer_display_name(&key.peer), + error = %error, + "Failed to initiate a session for a native flow, trying discovery" + ); + self.maybe_initiate_lookup(&key.peer).await; + } + self.queue_native(key, payload); + } + + /// Hold a native datagram until its destination's session establishes. + /// + /// Bounded the same way the TUN queue is, and by the same configured + /// numbers, so one client cannot make the node hold without limit. + fn queue_native(&mut self, key: FlowKey, payload: Vec) { + let max_dests = self.config().node.session.pending_max_destinations; + if !self.pending_native.contains_key(&key.peer) && self.pending_native.len() >= max_dests { + return; + } + let per_dest = self.config().node.session.pending_packets_per_dest; + let queue = self.pending_native.entry(key.peer).or_default(); + crate::proto::fsp::push_bounded_pending(queue, PendingNative { key, payload }, per_dest); + } + + /// Send the native datagrams held for a destination whose session is up. + /// + /// Called beside `flush_pending_packets` at every site that flushes the TUN + /// queue. A separate queue means a separate flush, and forgetting one would + /// leave datagrams held until the session went away. + pub(in crate::node) async fn flush_pending_native(&mut self, dest: &NodeAddr) { + let Some(held) = self.pending_native.remove(dest) else { + return; + }; + for entry in held { + let bytes = entry.payload.len(); + if let Err(error) = self + .send_session_data(dest, entry.key.local, entry.key.remote, &entry.payload) + .await + { + debug!( + peer = %self.peer_display_name(dest), + error = %error, + "Failed to send a queued native datagram" + ); + break; + } + self.metrics.native.record_sent(bytes); + } + } + + /// Discard pending flows a listener's task never wired. + /// + /// Called from the maintenance tick. The window it closes is the hop + /// between the rx_loop announcing an arrival and the listener's task taking + /// it, so a non-zero count is the daemon failing to complete an arrival + /// rather than a client failing to answer for one. It should be zero on a + /// healthy node. + pub(in crate::node) fn native_expire(&mut self) { + let expired = self.native.expire(Self::now_ms()); + self.metrics.native.flows_expired.add(expired.len() as u64); + if !expired.is_empty() { + debug!( + count = expired.len(), + "Native API pending flows expired without an answer" + ); + } + } +} diff --git a/src/node/handlers/session.rs b/src/node/handlers/session.rs index 6e68a2de..58bc204f 100644 --- a/src/node/handlers/session.rs +++ b/src/node/handlers/session.rs @@ -389,6 +389,7 @@ impl Node { debug!(len = rest.len(), "DataPacket too short for port header"); return; } + let src_port = u16::from_le_bytes([rest[0], rest[1]]); let dst_port = u16::from_le_bytes([rest[2], rest[3]]); let service_payload = &rest[FSP_PORT_HEADER_SIZE..]; @@ -431,11 +432,43 @@ impl Node { } } _ => { - debug!( - src = %self.peer_display_name(src_addr), - dst_port, - "Unknown FSP service port, dropping DataPacket" - ); + // Every other port belongs to the native datagram API, + // which decides between an established flow, a listener + // and a drop. The same decision serves the debug + // arrival command, so the rule was exercised before + // this call site existed. + // The peer's key comes from the session entry rather + // than from the wire: the handler above refuses + // anything but an Established session, and an + // Established entry holds the key its handshake + // attested. Nothing has to invert the node address. + let payload = service_payload.to_vec(); + // The entry was removed for the trial-decrypt cascade + // and re-inserted above, and this arm is reached only + // for an Established session, so the lookup finds one. + // It is written as a lookup rather than an unwrap + // because a future path that skipped the re-insert + // would otherwise panic on a peer's datagram. Skipping + // is scoped to the native dispatch alone: the idle + // timer and the pending-outbound flush at the end of + // this function are this message's bookkeeping and are + // owed whether or not it had anywhere to go. + if let Some(peer_key) = self + .sessions + .get(src_addr) + .map(|session| session.remote_pubkey().x_only_public_key().0) + { + let outcome = self + .native_deliver(*src_addr, peer_key, src_port, dst_port, payload); + if let crate::native::link::Outcome::Dropped(why) = outcome { + debug!( + src = %self.peer_display_name(src_addr), + dst_port, + why = why.as_str(), + "Unknown FSP service port, dropping DataPacket" + ); + } + } } } } @@ -470,6 +503,7 @@ impl Node { // Flush any pending outbound packets (e.g., simultaneous initiation // where responder also had queued outbound packets) self.flush_pending_packets(src_addr).await; + self.flush_pending_native(src_addr).await; } /// Handle an incoming SessionSetup (Noise XK msg1). @@ -969,6 +1003,7 @@ impl Node { // Flush any queued outbound packets for this destination self.flush_pending_packets(src_addr).await; + self.flush_pending_native(src_addr).await; info!(src = %self.peer_display_name(src_addr), "Session established (initiator, XK)"); } @@ -1188,6 +1223,7 @@ impl Node { // Flush any pending packets self.flush_pending_packets(src_addr).await; + self.flush_pending_native(src_addr).await; info!(src = %self.peer_display_name(src_addr), "Session established (responder, XK)"); } diff --git a/src/node/metrics.rs b/src/node/metrics.rs index 963f5b83..30180f9a 100644 --- a/src/node/metrics.rs +++ b/src/node/metrics.rs @@ -16,7 +16,7 @@ use std::sync::atomic::{AtomicU64, Ordering}; use crate::node::reject::{BloomReject, DiscoveryReject, ForwardingReject, TreeReject}; use crate::node::stats::{ BloomStatsSnapshot, CongestionStatsSnapshot, ErrorSignalStatsSnapshot, ForwardingStatsSnapshot, - LookupStatsSnapshot, TreeStatsSnapshot, + LookupStatsSnapshot, NativeStatsSnapshot, TreeStatsSnapshot, }; /// An atomic counter. @@ -495,11 +495,104 @@ impl ErrorMetrics { } } +/// Native datagram API metric counters. +/// +/// Every one of these is bumped on the rx_loop, in the native request handler, +/// the outbound handler and the inbound dispatch arm. No client task touches a +/// counter, so there is one path per counter rather than two. +/// +/// `flows_accepted` counts flows promoted out of a listener's backlog, which +/// happens before the daemon writes the arrival that carries one to its client. +/// A hand-off that then fails is counted under `drop_listener_not_reading` or +/// `drop_listener_gone` and is not taken back off `flows_accepted`, so what a +/// client actually received is the first less the other two. +#[derive(Default)] +pub struct NativeMetrics { + pub flows_opened: Counter, + pub flows_accepted: Counter, + pub flows_closed: Counter, + pub flows_expired: Counter, + pub sent_datagrams: Counter, + pub sent_bytes: Counter, + pub received_datagrams: Counter, + pub received_bytes: Counter, + pub drop_no_port: Counter, + pub drop_backlog_full: Counter, + pub drop_too_many_flows: Counter, + pub drop_pending_queue_full: Counter, + pub drop_flow_queue_full: Counter, + pub drop_arrival_queue_full: Counter, + pub drop_listener_not_reading: Counter, + pub drop_listener_gone: Counter, + pub drop_oversize: Counter, +} + +impl NativeMetrics { + /// Route a typed drop to its counter. + /// + /// Exhaustive over [`DropReason`](crate::native::link::DropReason) with no + /// wildcard arm, which is what makes a variant added later a compile error + /// here rather than a datagram that vanishes uncounted. `DropReason` rather + /// than `DropCause` because delivery can also fail after the registry has + /// agreed, and those cases are the ones a slow client causes. + #[inline] + pub fn record_drop(&self, reason: crate::native::link::DropReason) { + use crate::native::link::DropReason; + match reason { + DropReason::NoPort => self.drop_no_port.inc(), + DropReason::BacklogFull => self.drop_backlog_full.inc(), + DropReason::TooManyFlows => self.drop_too_many_flows.inc(), + DropReason::PendingQueueFull => self.drop_pending_queue_full.inc(), + DropReason::FlowQueueFull => self.drop_flow_queue_full.inc(), + DropReason::ArrivalQueueFull => self.drop_arrival_queue_full.inc(), + DropReason::ListenerNotReading => self.drop_listener_not_reading.inc(), + DropReason::ListenerGone => self.drop_listener_gone.inc(), + } + } + + /// Count one datagram leaving a client for the mesh. + #[inline] + pub fn record_sent(&self, bytes: usize) { + self.sent_datagrams.inc(); + self.sent_bytes.add(bytes as u64); + } + + /// Count one datagram reaching a client from the mesh. + #[inline] + pub fn record_received(&self, bytes: usize) { + self.received_datagrams.inc(); + self.received_bytes.add(bytes as u64); + } + + /// Sample every counter into a serializable snapshot. + pub fn snapshot(&self) -> NativeStatsSnapshot { + NativeStatsSnapshot { + flows_opened: self.flows_opened.get(), + flows_accepted: self.flows_accepted.get(), + flows_closed: self.flows_closed.get(), + flows_expired: self.flows_expired.get(), + sent_datagrams: self.sent_datagrams.get(), + sent_bytes: self.sent_bytes.get(), + received_datagrams: self.received_datagrams.get(), + received_bytes: self.received_bytes.get(), + drop_no_port: self.drop_no_port.get(), + drop_backlog_full: self.drop_backlog_full.get(), + drop_too_many_flows: self.drop_too_many_flows.get(), + drop_pending_queue_full: self.drop_pending_queue_full.get(), + drop_flow_queue_full: self.drop_flow_queue_full.get(), + drop_arrival_queue_full: self.drop_arrival_queue_full.get(), + drop_listener_not_reading: self.drop_listener_not_reading.get(), + drop_listener_gone: self.drop_listener_gone.get(), + drop_oversize: self.drop_oversize.get(), + } + } +} + /// Atomic counter registry shared across the node via `Arc`. /// /// Sole storage for the forwarding, discovery, tree, bloom, congestion, -/// and error counter families; these were migrated off `NodeStats`, which -/// now holds only the session, handshake, mmp, and transport families. +/// error, and native counter families; these were migrated off `NodeStats`, +/// which now holds only the session, handshake, mmp, and transport families. #[derive(Default)] pub struct MetricsRegistry { pub forwarding: ForwardingMetrics, @@ -508,6 +601,7 @@ pub struct MetricsRegistry { pub bloom: BloomMetrics, pub congestion: CongestionMetrics, pub errors: ErrorMetrics, + pub native: NativeMetrics, } impl MetricsRegistry { @@ -520,6 +614,54 @@ impl MetricsRegistry { mod tests { use super::*; + #[test] + fn native_record_drop_routes_every_reason_to_its_own_counter() { + use crate::native::link::DropReason; + let m = NativeMetrics::default(); + for reason in [ + DropReason::NoPort, + DropReason::BacklogFull, + DropReason::TooManyFlows, + DropReason::PendingQueueFull, + DropReason::FlowQueueFull, + DropReason::ArrivalQueueFull, + DropReason::ListenerNotReading, + DropReason::ListenerGone, + ] { + m.record_drop(reason); + } + let snap = m.snapshot(); + // Each reason lands in one counter and no reason lands in two, which is + // what a shared counter would hide. + assert_eq!(snap.drop_no_port, 1); + assert_eq!(snap.drop_backlog_full, 1); + assert_eq!(snap.drop_too_many_flows, 1); + assert_eq!(snap.drop_pending_queue_full, 1); + assert_eq!(snap.drop_flow_queue_full, 1); + assert_eq!(snap.drop_arrival_queue_full, 1); + assert_eq!(snap.drop_listener_not_reading, 1); + assert_eq!(snap.drop_listener_gone, 1); + // Nothing bumped the send-side refusal, which no DropReason reaches. + assert_eq!(snap.drop_oversize, 0); + } + + #[test] + fn native_pending_and_flow_queue_drops_read_alike_to_a_client_and_apart_to_a_counter() { + use crate::native::link::DropReason; + // The two render identically on purpose, so the debug arrival command + // answers the same strings it always has. A counter must still tell + // them apart, because only one of them means a slow client. + assert_eq!( + DropReason::PendingQueueFull.as_str(), + DropReason::FlowQueueFull.as_str() + ); + let m = NativeMetrics::default(); + m.record_drop(DropReason::FlowQueueFull); + let snap = m.snapshot(); + assert_eq!(snap.drop_flow_queue_full, 1); + assert_eq!(snap.drop_pending_queue_full, 0); + } + #[test] fn forwarding_received_tracks_packets_and_bytes() { let m = ForwardingMetrics::default(); diff --git a/src/node/mod.rs b/src/node/mod.rs index 27e6630b..851ae047 100644 --- a/src/node/mod.rs +++ b/src/node/mod.rs @@ -457,6 +457,18 @@ pub struct Node { /// Keyed by destination NodeAddr, bounded per-dest and total. pending_tun_packets: HashMap>>, + /// Native API registry: which local ports are held, and where an inbound + /// datagram goes. Reached only from the `rx_loop`, so it takes no lock. + native: crate::native::registry::Registry, + + /// Native datagrams held per destination while its session establishes. + /// + /// Deliberately **not** `pending_tun_packets`: that queue is drained + /// through `send_ipv6_packet`, which compresses its bytes as an IPv6 + /// header, and it records neither a port nor a kind, so nothing could tell + /// a native datagram from an IPv6 packet once it was in there. + pending_native: HashMap>, + // === Discovery === /// Discovery-subsystem state: recent-request dedup cache, in-flight /// lookups, originator-side backoff, and transit-side forward limiter. @@ -505,6 +517,12 @@ pub struct Node { /// see [`Self::publish_entities_snapshot`] for the rationale. entities_snapshot: std::sync::Arc>, + /// Read-side snapshot of the native datagram API registry (flows + /// / listeners) that the `show_native_flows` query renders off the rx_loop. + /// Published from the tick; see [`Self::publish_native_snapshot`] for why + /// the registry itself cannot be shared instead. + native_snapshot: std::sync::Arc>, + // === TUN Interface === /// TUN device state. tun_state: TunState, @@ -760,6 +778,12 @@ impl Node { sessions: HashMap::new(), identity_cache: HashMap::new(), pending_tun_packets: HashMap::new(), + pending_native: HashMap::new(), + native: crate::native::registry::Registry::new(crate::native::registry::Limits { + per_flow: config.node.native_api.pending_per_flow, + backlog: config.node.native_api.backlog, + max_flows: config.node.native_api.max_flows, + }), next_link_id: 1, next_transport_id: 1, stats: stats::NodeStats::new(), @@ -774,6 +798,9 @@ impl Node { entities_snapshot: std::sync::Arc::new(arc_swap::ArcSwap::from_pointee( crate::control::snapshot::EntitySnapshot::empty(), )), + native_snapshot: std::sync::Arc::new(arc_swap::ArcSwap::from_pointee( + crate::control::snapshot::NativeSnapshot::empty(), + )), tun_state, tun_name: None, index_allocator: IndexAllocator::new(), @@ -908,6 +935,12 @@ impl Node { sessions: HashMap::new(), identity_cache: HashMap::new(), pending_tun_packets: HashMap::new(), + pending_native: HashMap::new(), + native: crate::native::registry::Registry::new(crate::native::registry::Limits { + per_flow: config.node.native_api.pending_per_flow, + backlog: config.node.native_api.backlog, + max_flows: config.node.native_api.max_flows, + }), next_link_id: 1, next_transport_id: 1, stats: stats::NodeStats::new(), @@ -922,6 +955,9 @@ impl Node { entities_snapshot: std::sync::Arc::new(arc_swap::ArcSwap::from_pointee( crate::control::snapshot::EntitySnapshot::empty(), )), + native_snapshot: std::sync::Arc::new(arc_swap::ArcSwap::from_pointee( + crate::control::snapshot::NativeSnapshot::empty(), + )), tun_state, tun_name: None, index_allocator: IndexAllocator::new(), @@ -1513,6 +1549,7 @@ impl Node { self.stats_snapshot.clone(), self.routing_snapshot.clone(), self.entities_snapshot.clone(), + self.native_snapshot.clone(), ) } @@ -1661,6 +1698,72 @@ impl Node { // Publish the per-entity read view from the same tick, with // `Vec>` structural sharing against the previous snapshot. self.publish_entities_snapshot(); + + // Publish the native datagram API read view from the same tick. + self.publish_native_snapshot(); + } + + /// Project the native datagram API registry into a + /// [`NativeSnapshot`](crate::control::snapshot::NativeSnapshot) and publish + /// it via `ArcSwap`, so `show_native_flows` renders off the rx_loop. + /// + /// **Publisher placement.** The registry is reached only from the rx_loop, + /// which is what lets the native receive path take no lock; sharing it with + /// the control task would mean locking it and giving that property back. + /// The tick is therefore the publisher, as it is for the other three cells. + /// The peer's npub needs nothing from `&Node`: the key rides on the + /// registry entry, captured where the session authenticated it. + fn publish_native_snapshot(&self) { + use crate::control::snapshot as snap; + + let flows: Vec = self + .native + .flows() + .into_iter() + .map(|view| snap::NativeFlowRow { + flow: view.flow, + peer: view.key.peer, + peer_key: view.pubkey, + local_port: view.key.local, + remote_port: view.key.remote, + established: view.established, + queued: view.queued, + since_ms: view.at, + }) + .collect(); + + let listeners: Vec = self + .native + .listeners() + .into_iter() + .map(|view| snap::NativeListenerRow { + local_port: view.port, + backlog: view.backlog, + }) + .collect(); + + self.native_snapshot + .store(std::sync::Arc::new(snap::NativeSnapshot { + flows, + listeners, + })); + } + + /// Borrow the native datagram API registry, read-only. + /// + /// For the on-loop `show_native_flows` oracle. Every mutation still goes + /// through the rx_loop's own handler. + pub(crate) fn native(&self) -> &crate::native::registry::Registry { + &self.native + } + + /// Test-only: reach the native registry mutably, so a test can populate it + /// the way the rx_loop's handler does without standing up a client task and + /// a socket pair. A parity test over an empty registry proves nothing about + /// a publisher that drops fields, which is why this exists. + #[cfg(test)] + pub(crate) fn native_registry_for_test(&mut self) -> &mut crate::native::registry::Registry { + &mut self.native } /// Resolve the npub of the spanning-tree root for `show_tree`'s `root_npub`. diff --git a/src/node/stats.rs b/src/node/stats.rs index f72639ea..e9699085 100644 --- a/src/node/stats.rs +++ b/src/node/stats.rs @@ -431,6 +431,38 @@ pub struct CongestionStatsSnapshot { pub kernel_drop_events: u64, } +/// Native datagram API counters. +/// +/// Eight of the `drop_*` fields are one per +/// [`DropReason`](crate::native::link::DropReason) variant, so a variant added +/// later has nowhere to hide. `drop_oversize` is the ninth and has no reason +/// because it is refused on the send side, before the registry sees the +/// datagram. +/// +/// Present on every platform even though the API builds only on Linux and +/// FreeBSD, so `show_metrics` carries one schema everywhere and a consumer does +/// not branch on the host. +#[derive(Clone, Debug, Default, Serialize)] +pub struct NativeStatsSnapshot { + pub flows_opened: u64, + pub flows_accepted: u64, + pub flows_closed: u64, + pub flows_expired: u64, + pub sent_datagrams: u64, + pub sent_bytes: u64, + pub received_datagrams: u64, + pub received_bytes: u64, + pub drop_no_port: u64, + pub drop_backlog_full: u64, + pub drop_too_many_flows: u64, + pub drop_pending_queue_full: u64, + pub drop_flow_queue_full: u64, + pub drop_arrival_queue_full: u64, + pub drop_listener_not_reading: u64, + pub drop_listener_gone: u64, + pub drop_oversize: u64, +} + #[cfg(test)] mod tests { use super::*; diff --git a/src/proto/fsp/core.rs b/src/proto/fsp/core.rs index b22c7349..dc061cb0 100644 --- a/src/proto/fsp/core.rs +++ b/src/proto/fsp/core.rs @@ -437,9 +437,9 @@ pub(crate) fn should_apply_path_mtu(existing: Option, candidate: u16) -> bo /// oldest entry first when the queue is at `per_dest` capacity. Pure transform /// over a passed-in queue (the max-destinations cap is a shell-side map-level /// check). -pub(crate) fn push_bounded_pending( - queue: &mut alloc::collections::VecDeque>, - packet: Vec, +pub(crate) fn push_bounded_pending( + queue: &mut alloc::collections::VecDeque, + packet: T, per_dest: usize, ) { if queue.len() >= per_dest { diff --git a/src/utils/mod.rs b/src/utils/mod.rs index 1fd66c36..9c18a16b 100644 --- a/src/utils/mod.rs +++ b/src/utils/mod.rs @@ -1,6 +1,8 @@ //! Utility modules. //! //! Shared infrastructure that doesn't belong to a specific protocol layer: -//! session index allocation and other cross-cutting concerns. +//! session index allocation, Unix socket binding, and other cross-cutting +//! concerns. pub mod index; +pub mod sockbind; diff --git a/src/utils/sockbind.rs b/src/utils/sockbind.rs new file mode 100644 index 00000000..9a32245c --- /dev/null +++ b/src/utils/sockbind.rs @@ -0,0 +1,269 @@ +//! Bind a Unix domain socket under the FIPS access policy. +//! +//! Every FIPS Unix socket needs the same four things before it can listen: a +//! parent directory that exists and whose ownership is known, any stale socket +//! file removed, a listener bound, and group access applied so members of the +//! `fips` group can reach it. +//! +//! The policy lives in one place so a change to it reaches every socket rather +//! than whichever copy an author happened to be looking at. All three sockets +//! use it: the control socket, the gateway control socket and the native +//! datagram API socket. The gateway previously kept its own copy, which had +//! already drifted in that it never set the parent's mode at all. + +#[cfg(unix)] +use std::path::{Path, PathBuf}; +#[cfg(unix)] +use tokio::net::UnixListener; +#[cfg(unix)] +use tracing::{debug, warn}; + +/// Bind a Unix listener at `path` under the FIPS access policy. +/// +/// `what` names the socket for diagnostics: it is the noun in the "already in +/// use" error a caller sees when another process is listening there, and a +/// structured field on the directory and stale-socket log lines. The caller +/// emits its own "listening" line, so each socket keeps its own wording. +/// +/// Creates missing ancestors, removes a stale socket file, binds, then applies +/// mode `0o770` to the socket and `0o750` to the parent when this bind owns the +/// parent. An `AddrInUse` error means a live listener already holds the path. +#[cfg(unix)] +pub fn bind(path: &Path, what: &str) -> Result { + // Creation is useful for diagnostics, but ownership is keyed to directory + // identity as well: systemd pre-creates /run/fips on every Linux service + // start and initially owns it as root:root. + let managed_parent = match path.parent() { + Some(parent) => { + let created = ensure_socket_parent(parent)?; + if created { + debug!(path = %parent.display(), socket = what, "Created private socket directory"); + } + (created || crate::config::is_managed_socket_parent(parent)).then(|| parent.to_owned()) + } + None => None, + }; + + if path.exists() { + remove_stale_socket(path, what)?; + } + + let listener = UnixListener::bind(path)?; + + set_socket_access(path, managed_parent.as_deref(), chown_to_fips_group)?; + + Ok(listener) +} + +/// Ensure the socket's parent exists and report whether this call created the +/// leaf directory. +/// +/// `create_dir` gives us an atomic ownership decision: an `AlreadyExists` +/// result means another actor owns the existing directory, while success means +/// it is safe for this bind to apply FIPS ownership and mode. Missing ancestors +/// are created recursively, but only the requested leaf is later treated as the +/// socket's private directory. +#[cfg(unix)] +fn ensure_socket_parent(parent: &Path) -> Result { + if parent.as_os_str().is_empty() { + return Ok(false); + } + + match std::fs::create_dir(parent) { + Ok(()) => Ok(true), + Err(error) if error.kind() == std::io::ErrorKind::AlreadyExists => { + if parent.is_dir() { + Ok(false) + } else { + Err(error) + } + } + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + let ancestor = parent.parent().ok_or(error)?; + ensure_socket_parent(ancestor)?; + ensure_socket_parent(parent) + } + Err(error) => Err(error), + } +} + +/// Apply access policy to a newly bound socket. +/// +/// The socket is always group-owned. `managed_parent` is either a private +/// directory this bind created or a canonical FIPS runtime directory. A shared +/// or operator-owned existing parent is omitted so it retains its ownership and +/// mode. +/// +/// `chown_to_fips_group` is a parameter so the policy can be tested without +/// requiring the `fips` group to exist on the machine running the tests. +#[cfg(unix)] +fn set_socket_access( + socket_path: &Path, + managed_parent: Option<&Path>, + mut chown_to_fips_group: impl FnMut(&Path), +) -> Result<(), std::io::Error> { + use std::os::unix::fs::PermissionsExt; + + std::fs::set_permissions(socket_path, std::fs::Permissions::from_mode(0o770))?; + chown_to_fips_group(socket_path); + + if let Some(parent) = managed_parent { + std::fs::set_permissions(parent, std::fs::Permissions::from_mode(0o750))?; + chown_to_fips_group(parent); + } + + Ok(()) +} + +/// Remove a stale socket file. +/// +/// If the file exists but no one is listening, remove it so we can bind. This +/// handles unclean daemon exits. A live listener yields `AddrInUse` instead, so +/// two daemons cannot silently take the same path. +#[cfg(unix)] +fn remove_stale_socket(path: &Path, what: &str) -> Result<(), std::io::Error> { + match std::os::unix::net::UnixStream::connect(path) { + Ok(_) => Err(std::io::Error::new( + std::io::ErrorKind::AddrInUse, + format!("{what} socket already in use: {}", path.display()), + )), + Err(_) => { + debug!(path = %path.display(), socket = what, "Removing stale socket"); + std::fs::remove_file(path)?; + Ok(()) + } + } +} + +/// Set group ownership of a path to the `fips` group (best-effort). +/// +/// A missing group is not an error: a source build on a developer machine has +/// no `fips` group, and the socket is still usable by its owner. +#[cfg(unix)] +fn chown_to_fips_group(path: &Path) { + use std::ffi::CString; + use std::os::unix::ffi::OsStrExt; + + let group_name = CString::new("fips").unwrap(); + let grp = unsafe { libc::getgrnam(group_name.as_ptr()) }; + if grp.is_null() { + debug!( + "'fips' group not found, skipping chown for {}", + path.display() + ); + return; + } + let gid = unsafe { (*grp).gr_gid }; + + let c_path = match CString::new(path.as_os_str().as_bytes()) { + Ok(p) => p, + Err(_) => return, + }; + let ret = unsafe { libc::chown(c_path.as_ptr(), u32::MAX, gid) }; + if ret != 0 { + warn!( + path = %path.display(), + error = %std::io::Error::last_os_error(), + "Failed to chown socket to 'fips' group" + ); + } +} + +/// Remove a socket file at teardown, ignoring a path that is already gone. +#[cfg(unix)] +pub fn cleanup(path: &PathBuf, what: &str) { + if !path.exists() { + return; + } + match std::fs::remove_file(path) { + Ok(()) => debug!(path = %path.display(), socket = what, "Socket file removed"), + Err(error) => { + warn!(path = %path.display(), socket = what, error = %error, "Failed to remove socket file") + } + } +} + +#[cfg(all(test, unix))] +mod tests { + use super::{ensure_socket_parent, set_socket_access}; + use std::os::unix::fs::PermissionsExt; + + #[test] + fn parent_setup_distinguishes_existing_and_created_directories() { + let temp = tempfile::tempdir().unwrap(); + let existing = temp.path().join("existing"); + std::fs::create_dir(&existing).unwrap(); + assert!(!ensure_socket_parent(&existing).unwrap()); + + let nested = temp.path().join("missing").join("fips"); + assert!(ensure_socket_parent(&nested).unwrap()); + assert!(nested.is_dir()); + assert!(!ensure_socket_parent(&nested).unwrap()); + } + + #[test] + fn access_setup_leaves_an_existing_shared_parent_unchanged() { + let temp = tempfile::tempdir().unwrap(); + let parent = temp.path().join("shared"); + std::fs::create_dir(&parent).unwrap(); + std::fs::set_permissions(&parent, std::fs::Permissions::from_mode(0o711)).unwrap(); + let socket = parent.join("control.sock"); + std::fs::File::create(&socket).unwrap(); + + let mut chowned = Vec::new(); + set_socket_access(&socket, None, |path| chowned.push(path.to_path_buf())).unwrap(); + + assert_eq!(chowned, vec![socket.clone()]); + assert_eq!( + std::fs::metadata(&parent).unwrap().permissions().mode() & 0o777, + 0o711 + ); + assert_eq!( + std::fs::metadata(&socket).unwrap().permissions().mode() & 0o777, + 0o770 + ); + } + + #[test] + fn access_setup_secures_a_new_private_parent() { + let temp = tempfile::tempdir().unwrap(); + let parent = temp.path().join("fips"); + std::fs::create_dir(&parent).unwrap(); + let socket = parent.join("control.sock"); + std::fs::File::create(&socket).unwrap(); + + let mut chowned = Vec::new(); + set_socket_access(&socket, Some(&parent), |path| { + chowned.push(path.to_path_buf()) + }) + .unwrap(); + + assert_eq!(chowned, vec![socket, parent.clone()]); + assert_eq!( + std::fs::metadata(&parent).unwrap().permissions().mode() & 0o777, + 0o750 + ); + } + + #[test] + fn access_setup_secures_an_existing_managed_parent() { + let temp = tempfile::tempdir().unwrap(); + let parent = temp.path().join("managed"); + std::fs::create_dir(&parent).unwrap(); + std::fs::set_permissions(&parent, std::fs::Permissions::from_mode(0o700)).unwrap(); + let socket = parent.join("control.sock"); + std::fs::File::create(&socket).unwrap(); + + let mut chowned = Vec::new(); + set_socket_access(&socket, Some(&parent), |path| { + chowned.push(path.to_path_buf()) + }) + .unwrap(); + + assert_eq!(chowned, vec![socket, parent.clone()]); + assert_eq!( + std::fs::metadata(&parent).unwrap().permissions().mode() & 0o777, + 0o750 + ); + } +} diff --git a/testing/ci-local.sh b/testing/ci-local.sh index 1dc06645..7a4ad345 100755 --- a/testing/ci-local.sh +++ b/testing/ci-local.sh @@ -198,6 +198,7 @@ NAT_SUITES=(cone symmetric lan) NOSTR_RELAY_SUITES=(nostr-publish-consume) STUN_FAULTS_SUITES=(stun-faults) DNS_RESOLVER_SUITES=(dns-resolver) +NATIVE_API_SUITES=(native-api) DEB_INSTALL_SUITES=(deb-install) TOR_SUITES=(tor-socks5 tor-directory) @@ -249,6 +250,9 @@ list_suites() { echo " Sidecar:" for s in "${SIDECAR_SUITES[@]}"; do echo " $s"; done echo "" + echo " Native API:" + for s in "${NATIVE_API_SUITES[@]}"; do echo " $s"; done + echo "" echo " DNS resolver:" for s in "${DNS_RESOLVER_SUITES[@]}"; do echo " $s"; done echo "" @@ -497,8 +501,12 @@ run_build() { return 1 fi - info "cargo build --release" - if cargo build --release 2>&1; then + info "cargo build --release --bins --examples" + # --bins --examples rather than the bare default: the native datagram API's + # echo server is a cargo example, and the native-api harness's image needs + # it built here rather than separately. Naming --bins keeps the daemon and + # its tools in the build, which --examples alone would drop. + if cargo build --release --bins --examples 2>&1; then record "build" 0 else record "build" 1 @@ -606,7 +614,13 @@ install_binaries() { cp target/release/fipsctl "$dest/fipsctl" [[ -f target/release/fipstop ]] && cp target/release/fipstop "$dest/fipstop" || true [[ -f target/release/fips-gateway ]] && cp target/release/fips-gateway "$dest/fips-gateway" || true - chmod +x "$dest/fips" "$dest/fipsctl" + # Not optional: the native-api harness runs native-echo as one end of its + # echo check and native-surface as its surface walk, so a missing one must + # fail the image build rather than fail a check later with the container + # exiting on a missing entrypoint. + cp target/release/examples/native-echo "$dest/native-echo" + cp target/release/examples/native-surface "$dest/native-surface" + chmod +x "$dest/fips" "$dest/fipsctl" "$dest/native-echo" "$dest/native-surface" [[ -f "$dest/fipstop" ]] && chmod +x "$dest/fipstop" || true [[ -f "$dest/fips-gateway" ]] && chmod +x "$dest/fips-gateway" || true } @@ -960,6 +974,20 @@ run_stun_faults() { record "stun-faults" $rc } +# Run the native datagram API harness. +# +# Reads FIPS_TEST_IMAGE, so it exercises this run's binaries rather than a +# separately built one. Its two-node check creates and removes its own docker +# network, so nothing here has to. +run_native_api() { + info "[native-api] Running native datagram API test" + if FIPS_TEST_IMAGE="$CI_IMAGE_TEST" bash testing/native-api/test.sh 2>&1; then + record "native-api" 0 + else + record "native-api" 1 + fi +} + # Run dns-resolver harness (multi-distro + e2e scenarios) run_dns_resolver() { info "[dns-resolver] Running multi-distro test (slow — builds per-distro images)" @@ -1017,7 +1045,7 @@ run_integration() { local _f for _f in "$SCRIPT_DIR"/docker/*; do case "$(basename "$_f")" in - fips|fipsctl|fipstop|fips-gateway) continue ;; + fips|fipsctl|fipstop|fips-gateway|native-echo|native-surface) continue ;; esac cp -a "$_f" "$CI_BUILD_CONTEXT/" || { record "docker-build" 1; return; } done @@ -1140,6 +1168,9 @@ run_integration() { # Sidecar run_sidecar + # Native datagram API (light — one single-node run plus a two-node pair) + run_native_api + # DNS resolver multi-distro suite (heavy — per-distro systemd images) run_dns_resolver @@ -1194,6 +1225,8 @@ run_suite() { run_sidecar ;; dns-resolver) run_dns_resolver ;; + native-api) + run_native_api ;; deb-install) run_deb_install ;; tor-socks5) diff --git a/testing/docker/.gitignore b/testing/docker/.gitignore index da850b8b..5d4f87b8 100644 --- a/testing/docker/.gitignore +++ b/testing/docker/.gitignore @@ -2,3 +2,5 @@ fips fips-gateway fipsctl fipstop +native-echo +native-surface diff --git a/testing/docker/Dockerfile b/testing/docker/Dockerfile index 72471335..cec3b8d2 100644 --- a/testing/docker/Dockerfile +++ b/testing/docker/Dockerfile @@ -36,8 +36,15 @@ RUN printf '%s\n' \ 'no-resolv' \ >> /etc/dnsmasq.conf -COPY fips fipsctl fipstop fips-gateway /usr/local/bin/ -RUN chmod +x /usr/local/bin/fips /usr/local/bin/fipsctl /usr/local/bin/fipstop /usr/local/bin/fips-gateway +# native-echo and native-surface are the native datagram API's example +# programs, built as cargo examples rather than bins. The native-api harness +# runs native-echo as one end of a two-node check and native-surface as the +# assertion walk over the client's whole public surface, so the image carries +# both beside the daemon they talk to. +COPY fips fipsctl fipstop fips-gateway native-echo native-surface /usr/local/bin/ +RUN chmod +x /usr/local/bin/fips /usr/local/bin/fipsctl /usr/local/bin/fipstop \ + /usr/local/bin/fips-gateway /usr/local/bin/native-echo \ + /usr/local/bin/native-surface # Mirror systemd's RuntimeDirectory=fips so the daemon's resolver picks # /run/fips/control.sock — matches production layout and the chaos sim diff --git a/testing/interop/build-images.sh b/testing/interop/build-images.sh index 9107f4c0..9abab1be 100755 --- a/testing/interop/build-images.sh +++ b/testing/interop/build-images.sh @@ -183,6 +183,20 @@ build_one() { chmod +x "$ctx/$bin" done + # The shared Dockerfile COPYs the native datagram API's example programs, + # which are cargo examples that older refs do not carry and no interop + # check runs: these images exercise the wire between daemon versions. The + # cargo invocation above names only the four bins for that reason. Stage a + # stub for each so the image still builds, and so anything that does reach + # for one says why it is not there. + for example in native-echo native-surface; do + printf '%s\n' \ + '#!/bin/sh' \ + "echo \"$example is not built into interop images\" >&2" \ + 'exit 1' > "$ctx/$example" + chmod +x "$ctx/$example" + done + docker build \ --label "fips.interop.slot=$slot" \ --label "fips.interop.ref=$ref" \ diff --git a/testing/native-api/README.md b/testing/native-api/README.md new file mode 100644 index 00000000..b75b30ec --- /dev/null +++ b/testing/native-api/README.md @@ -0,0 +1,182 @@ +# Native Datagram API Harness + +Checks for the experimental native datagram API: a client process opens a flow +to a remote pubkey over a Unix socket, receives a file descriptor, and sends and +receives datagrams on it with no IPv6 emulation and no TUN device. + +Design of record: `design/native-api/v1-datagram-experiment.md` in the project +workspace, which is a separate tree from this repository. The feature is off by +default and Unix only. + +## Shape + +The client runs in **its own container**, reaching the daemon through a +bind-mounted `/run/fips`. That is the real deployment shape — a separate process +with its own filesystem opening the socket — rather than a test speaking to the +daemon from inside the daemon's container. It also makes the access policy +observable: the host sees the socket file and reads its mode directly. + +The step scripts are Python rather than Rust so a check changes without +rebuilding the daemon, which is what keeps the outside-in loop fast. That buys +speed at the cost of covering nothing of the Rust surface a caller links +against, so two compiled programs run here as well, both built on +`fips::native::client`: `examples/native-echo.rs`, which arrived with A5 and +serves the echo check, and `examples/native-surface.rs`, which walks the whole +public surface against a live daemon. + +The table covers this directory and the two example programs the driver runs. + +| File | What it is | +| ---- | ---------- | +| `test.sh` | The driver. Holds the scenarios and the pass/fail accounting. | +| `client.py` | A thin RPC client. Runs a script of steps over one connection and checks the replies. | +| `control.py` | A thin control-socket client, used to read `show_native_flows` back while a flow is open. | +| `node.yaml` | One node with the API enabled, no TUN, no DNS, no peers. Turns the debug commands on. | +| `node-api-off.yaml` | The same node with the API disabled, for the default-off check. | +| `node-debug-off.yaml` | The API enabled and the debug commands left at their default, for the gate check. | +| `../../examples/native-echo.rs` | The echo server for `check_echo_round_trip`. A program shape to copy. | +| `../../examples/native-surface.rs` | The surface walk for `check_surface_walk`. An assertion harness, not a shape to copy. | + +## Running + +```bash +cargo build --release --bins --examples # the driver refuses a stale binary +./testing/native-api/test.sh +``` + +`FIPS_TEST_IMAGE` is used when set, which is how `ci-local.sh` passes its +per-run image. There is deliberately no `fips-test:latest` to fall back on, so a +consumer that stops reading the variable fails loudly. Without it the driver +builds a minimal image from the locally compiled binary. + +The driver **refuses to run against a stale binary**. Three binaries are built +or read from this tree, the daemon and the two examples, and each is probed +against what it is actually built from: `src/` plus `Cargo.toml` for all three, +this directory's `*.py` because the harness client is bind-mounted live rather +than built in, and, for an example, its own `.rs` and no other. A guard rooted +only at `src/` would let a stale example pass a check written about new code. +A stale binary is the worst outcome available here: the checks would run and +report a verdict about code that is not the working tree's. + +An example is probed against its own source rather than all of `examples/` +because cargo does not relink `target/release/fips` when only an example +changes. Probing the daemon against every example would leave it permanently +older than a just-edited one, and the rebuild the refusal prescribes would not +clear the condition. + +All three binaries must come from one profile directory. `resolve_image` refuses +a profile that holds only the daemon, which is what a bare +`cargo build --release` leaves behind. + +## Increments + +The API is built outside-in, and this harness grows with it. Each increment's +checks must pass before the next one starts. + +| # | What it covers | State | +| - | -------------- | ----- | +| A1 | The socket, its access mode, the line framing, the command validation, the reserved-port refusals, and that the API is off by default | present | +| A2 | Descriptor passing over `SCM_RIGHTS`, message boundaries, `poll` readability, close reaching the daemon, flow isolation | present | +| A3 | Port ownership across clients, listening and accepting, the dispatch order, and reclaim when a descriptor closes | present | +| A4 | The end-to-end path between two nodes, and that a queued datagram is not IPv6-compressed | present | +| A5 | Counters, `show_native_flows` read back over the control socket, the Rust client module and echo example, and the debug-command gate | present | +| A6 | Every public item of `fips::native::client` walked against a live daemon: the five setup entry points, all eight `ToFipsAddr` spellings, the deadlines, non-blocking mode, the descriptor traits, and the payload limit | present | + +**The `"stub": true` marker is gone.** It meant "this flow reaches no peer", +and after A4 every flow does. `max_payload` is now the real limit — the +transport MTU less the FIPS encapsulation and the port header, 1362 bytes on a +1472-byte transport — and the end-to-end check asserts that number rather than +accepting whatever is reported. + +The tightening it existed for happened three times. A1's `connect` checks failed +the moment A2 began returning a descriptor, because `client.py` treats an +unannounced descriptor as a defect rather than ignoring it. A1's `accept` and +`reject` checks failed when A3 gave those commands a real registry, since a flow +no listener announced became a refusal. And the remaining `stub` assertions +failed at A4 when the field disappeared. Checks that had quietly kept passing +would have been worth nothing. + +**The `accept` and `reject` commands are gone**, and so is the `incoming` event. +A listener now returns its own descriptor, so it is pollable, accepting is one +`recvmsg` on it that carries the arriving flow's descriptor, and refusing a flow +is closing that descriptor. The command socket carries replies only, in command +order. A step names a listener descriptor with `keep_listener` and takes flows +off it with an `accept` step; every descriptor a reply carries must be named, or +the run fails rather than dropping a flow silently. + +**The backlog is no longer the bound a client sees.** It bounds arrivals the +daemon has announced and not yet wired, and the daemon drains that queue itself, +so a listener that never accepts is bounded by its send buffer and by +`node.native_api.max_flows` instead. `check_backlog_is_not_the_clients_bound` +asserts the change; the drop paths behind the new bound are covered by the +daemon's own tests, because neither is a number a shell check can produce. + +**Flow identifiers are assigned by the node and keep counting up for its +lifetime.** A check must capture one with `keep_flow` rather than assume a +literal, or it holds only for the first flow the daemon ever made. + +## The surface walk + +`check_surface_walk` runs `examples/native-surface.rs` against the shared +single node, last among the single-node checks. Its subject is the Rust surface +rather than the wire: until it existed, `FipsStream::connect`, `connect_from`, +`connect_at`, `FipsListener::bind` and `bind_at` had no coverage of any kind, +and every other public item was exercised only against the hand-written stand-in +daemon in the crate's unit tests. That stand-in has already hidden a real defect +once, by being kinder than the daemon, which is why the walk talks to the real +one. + +It runs last because `check_ephemeral_allocation` asserts 49152, 49153 and 49154 +as the first three ports the node ever hands out and the allocator is a +forward-only cursor. The walk therefore asserts only that its own ephemeral +ports are `>= 49152`, and takes its named ports from the otherwise unused +4800-4809 band. + +**The check asserts three things, not one:** that the container exited 0, that +its completion line is there, and that the count in that line equals +`SURFACE_ASSERTIONS` in `test.sh`. The third is the anti-silence measure. The +binary prints the recorder's own counter rather than a literal, so an assertion +block that stopped running — a `#[cfg]` gate that no longer matches, an early +return — still exits 0 and still prints the line, and only the count betrays it. +The number is deliberately brittle: adding an assertion must force an edit in +`test.sh`, so the two cannot drift apart quietly. + +**A hang has to become a red, and has to name itself.** The walk's own subjects +fail by blocking forever: a read deadline never applied to the descriptor, a +`set_nonblocking` that did nothing. The binary arms a 30-second watchdog that +prints the assertion it was in and exits 1, and `run_surface_at` bounds the +container at 60 seconds as a backstop for a wedge before that thread is armed. + +`timeout 60 docker run` is **not** that backstop, which a break-check measured +rather than a reading of the manual. `timeout` signals the docker client, the +client proxies SIGTERM to the container, and the walk is PID 1 there with no +handler for it, so the kernel discards the signal: the container was still up +five minutes after the bound passed and `docker run` never returned. The helper +runs the container detached, polls its state, and removes it by force, since +`docker rm -f` is a SIGKILL and PID 1 cannot discard that. + +## The two-node check + +`check_end_to_end` is the only check that runs more than one node. It derives +two identities with `testing/lib/derive_keys.py`, brings both up on their own +docker network peered by npub, and sends a datagram from a client on one to a +client on the other. + +Three orderings are waited on explicitly rather than assumed, each because +assuming it produced an intermittent failure: + +- **The link forms** before either client runs, watched for by the spanning tree + adopting a parent. Not by a peer-promotion log line: on this path — a + configured peer, dialled outbound — that line is never emitted. +- **The listener has bound its port** before the sender starts, watched for in + the listener's own output. Launching it first is not the same as it having + registered. +- **The client runs unbuffered** (`python3 -u`). Without it the marker above + never reaches the log file, so the wait cannot see it and every run fails at + the gate meant to make the check reliable. + +The payload is deliberately not a valid IPv6 packet, and it is sent before any +session exists so it goes through the native pending queue. If a native datagram +were ever routed through the TUN pending queue it would be handed to the IPv6 +compressor, which would refuse it, and this check would fail. The trap is +asserted rather than trusted. diff --git a/testing/native-api/client.py b/testing/native-api/client.py new file mode 100755 index 00000000..b377b9a6 --- /dev/null +++ b/testing/native-api/client.py @@ -0,0 +1,441 @@ +#!/usr/bin/env python3 +"""Native datagram API client for the increment checks. + +Speaks the line-delimited JSON command protocol on the daemon's native API +socket. A run takes a script: a list of steps sent over ONE connection. The +connection owns nothing — a flow lives until its own descriptor is closed, and a +listener until its own is — so the single connection is a convenience for the +checks rather than a lifetime the daemon respects. Descriptors are what keep +things alive, and this tool holds them until the step that closes them or until +it exits. + +Kinds of step: + + RPC step: {"command": str, "params": {...}?, "expect": {"dotted.key": val}?, + "keep_fd": name?, "keep_listener": name?, "keep_flow": name?} + Sends a command and checks the reply. `keep_fd` stores a flow + descriptor under that name, `keep_listener` a listener descriptor; + every reply that carries one must name it, because a descriptor + nothing named is a flow or a port silently dropped. `keep_flow` + stores the reply's data.flow_id. + + A parameter or expectation whose value is the string "@name" is + replaced by the flow identifier stored under `name`. Identifiers + are assigned by the node and keep counting up for its lifetime, so + a check that asserted a literal 1 would hold only for the first + flow the daemon ever made. + + Accept step: {"accept": listener, "keep_fd": name, "expect": {...}?, + "keep_flow": name?} + One recvmsg on a stored listener descriptor. There is no accept + command: an arriving flow is one SOCK_SEQPACKET message on the + listener itself, carrying the flow's descriptor as ancillary data + and the arrival object as its payload. Expectations are checked + against that object, whose peer is an npub and never a hex address. + + Sleep step: {"sleep": seconds} + Holds every descriptor open for a while, which is what a check that + reads the daemon's own view of a live flow needs. + + Flow step: {"fd": name, ...} operating on a stored descriptor: + "write": hex, "repeat": n? send n datagrams of those bytes + "read": n, "expect_bytes": hex?, "sizes": [..]? + read n datagrams and check them + "readable": bool check poll readability now + "close": true close the descriptor + `readable` and `close` work on a listener descriptor too: a + listener is pollable, and closing it unbinds its port. + +Reading is per-datagram: the descriptor is SOCK_SEQPACKET, so one recv is one +datagram. A check that reads three and gets one concatenated blob is a real +failure, not a quirk of the tool. + +Usage: + client.py --socket PATH --script '' + client.py --socket PATH --script-file steps.json + +Exit 0 when every expectation holds, 1 otherwise, 2 on a connection failure. +""" + +from __future__ import annotations + +import argparse +import array +import json +import os +import select +import socket +import sys +import time +from typing import Any + +# A flow's descriptor and a listener's are both AF_UNIX SOCK_SEQPACKET, so the +# wrap happens to be the same for both. Naming the roles anyway is the point: +# the next descriptor kind that is not one of these must not be wrapped +# correctly by accident. +FLOW = "flow" +LISTENER = "listener" + + +def recvfds(sock: socket.socket, bufsize: int, maxfds: int) -> tuple[bytes, list[int]]: + """One recvmsg, returning its payload and whatever descriptors it carried. + + Written out rather than calling `socket.recv_fds`, which takes a `flags` + argument and never forwards it to `recvmsg`: MSG_CMSG_CLOEXEC passed to that + helper does nothing, and a descriptor the harness kept would then survive + into any child process it forked. Measured on CPython 3.12 by reading + FD_CLOEXEC back with `fcntl.F_GETFD` after each of the two calls. + """ + fds = array.array("i") + data, ancillary, _flags, _addr = sock.recvmsg( + bufsize, socket.CMSG_LEN(maxfds * fds.itemsize), socket.MSG_CMSG_CLOEXEC + ) + for level, kind, payload in ancillary: + if level == socket.SOL_SOCKET and kind == socket.SCM_RIGHTS: + # Truncated to whole descriptors: the kernel may cut the array + # short, and a partial one names nothing. + fds.frombytes(payload[: len(payload) - (len(payload) % fds.itemsize)]) + return data, list(fds) + + +class Protocol(Exception): + """The daemon broke the local protocol, so the run cannot continue.""" + + +class Client: + """One connection to the native API socket, plus the descriptors it holds.""" + + def __init__(self, path: str, timeout: float) -> None: + """Connect to the socket at `path`, failing after `timeout` seconds.""" + self.sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + self.sock.settimeout(timeout) + self.sock.connect(path) + self.timeout = timeout + self.buf = b"" + # Complete lines, oldest first, each with the descriptor it arrived + # with. See `fill` for the rule that decides which line that is. + self.lines: list[list[Any]] = [] + self.fds: dict[str, tuple[socket.socket, str]] = {} + self.flows: dict[str, int] = {} + + def call(self, command: str, params: dict | None) -> tuple[dict, int | None]: + """Send one command; return the decoded reply and any descriptor. + + The socket carries replies only, in command order, so the next complete + line is this command's answer and there is nothing to separate out. + """ + request: dict[str, Any] = {"command": command} + if params is not None: + request["params"] = params + self.sock.sendall(json.dumps(request).encode() + b"\n") + line, fd = self.line() + return json.loads(line), fd + + def line(self) -> tuple[bytes, int | None]: + """Take the next complete line, reading until one is available.""" + while not self.lines: + self.fill() + line, fd = self.lines.pop(0) + return line, fd + + def fill(self) -> None: + """One recvmsg, split into lines, with any descriptor placed by the rule. + + A DESCRIPTOR BELONGS TO THE LAST COMPLETE LINE OF THE READ THAT CARRIED + IT, never to the next line the reader assembles. A recvmsg returning + ancillary data ends exactly at the end of the sendmsg that carried it, + but it may begin with any amount of data written before it, so a reader + that attached the descriptor to the first line it completed would hand a + flow to the wrong reply. Both reply kinds carry a descriptor now, so + this is reachable rather than theoretical. + + A read that carries a descriptor and completes no line is reported + rather than guessed at: holding it would mean choosing a later line for + it, and choosing wrong loses a flow with no error anywhere. + """ + chunk, fds = recvfds(self.sock, 65536, 4) + if not chunk: + for stray in fds: + # Closed rather than leaked: nothing can name it now. + os.close(stray) + raise ConnectionError("daemon closed the connection") + self.buf += chunk + + produced = 0 + while b"\n" in self.buf: + line, self.buf = self.buf.split(b"\n", 1) + self.lines.append([line, None]) + produced += 1 + + if not fds: + return + # This protocol never sends two at once. Extras are closed rather than + # left open with no owner. + for stray in fds[1:]: + os.close(stray) + if produced == 0: + os.close(fds[0]) + raise Protocol("a descriptor arrived on a read that completed no line") + self.lines[-1][1] = fds[0] + + def accept(self, listener: str) -> tuple[dict, int]: + """Take the next arriving flow off a stored listener descriptor. + + One recvmsg, one arrival: SOCK_SEQPACKET means the message carries + exactly its own descriptor, so the association rule the RPC socket needs + does not arise here. The payload has no trailing newline, because the + message boundary is the framing. + """ + sock = self.held(listener, LISTENER) + data, fds = recvfds(sock, 65536, 1) + if not fds: + raise Protocol(f"{listener!r} produced an arrival with no descriptor") + if not data: + os.close(fds[0]) + raise Protocol(f"{listener!r} produced a descriptor with no arrival") + return json.loads(data), fds[0] + + def held(self, name: str, want: str) -> socket.socket: + """Return a stored descriptor, refusing one of the wrong kind.""" + if name not in self.fds: + raise Protocol(f"no descriptor named {name!r}") + sock, role = self.fds[name] + if role != want: + raise Protocol(f"{name!r} is a {role} descriptor, not a {want} one") + return sock + + def keep(self, name: str, fd: int, role: str) -> None: + """Store a received descriptor under `name`, wrapped for its kind.""" + sock = socket.socket(socket.AF_UNIX, socket.SOCK_SEQPACKET, fileno=fd) + sock.settimeout(self.timeout) + self.fds[name] = (sock, role) + + def close(self) -> None: + """Close every descriptor, then the connection itself.""" + for sock, _role in self.fds.values(): + sock.close() + self.sock.close() + + +def substitute(value: Any, flows: dict[str, int]) -> Any: + """Replace every "@name" with the flow identifier stored under `name`.""" + if isinstance(value, str) and value.startswith("@"): + name = value[1:] + if name not in flows: + raise KeyError(f"no flow captured as {name!r}") + return flows[name] + if isinstance(value, dict): + return {key: substitute(item, flows) for key, item in value.items()} + if isinstance(value, list): + return [substitute(item, flows) for item in value] + return value + + +def dig(value: Any, dotted: str) -> Any: + """Read a dotted path out of a decoded reply, or None where it is absent.""" + for key in dotted.split("."): + if not isinstance(value, dict) or key not in value: + return None + value = value[key] + return value + + +def check(reply: dict, expect: dict) -> list[str]: + """Return one message per expectation the reply does not satisfy.""" + problems = [] + for dotted, wanted in expect.items(): + got = dig(reply, dotted) + if got != wanted: + problems.append(f"{dotted}: wanted {wanted!r}, got {got!r}") + return problems + + +def store(client: Client, step: dict, body: dict, fd: int | None) -> list[str]: + """Store what a step asked to keep, reporting a descriptor nobody named.""" + problems: list[str] = [] + + keep = step.get("keep_flow") + if keep is not None: + # A reply nests the identifier under `data`; an arrival message is the + # object itself. One reader for both, because a step should not have to + # know which produced it. + flow = dig(body, "data.flow_id") + if flow is None: + flow = body.get("flow_id") + if flow is None: + problems.append("keep_flow: nothing carried a flow_id") + else: + client.flows[keep] = flow + + wanted = [(step.get("keep_fd"), FLOW), (step.get("keep_listener"), LISTENER)] + named = [(name, role) for name, role in wanted if name is not None] + if len(named) > 1: + if fd is not None: + os.close(fd) + problems.append("a step named both keep_fd and keep_listener") + elif named and fd is None: + problems.append(f"{named[0][0]!r}: no descriptor arrived to keep") + elif named: + client.keep(named[0][0], fd, named[0][1]) + elif fd is not None: + # Leaving it unnamed would leak a flow or a held port for the rest of + # the run, with nothing to say so. + os.close(fd) + problems.append("a descriptor arrived that the step did not name") + + return problems + + +def run_rpc(client: Client, step: dict) -> list[str]: + """Send one command and report what did not hold.""" + command = step["command"] + try: + params = substitute(step.get("params"), client.flows) + expect = substitute(step.get("expect", {}), client.flows) + except KeyError as error: + return [str(error)] + + reply, fd = client.call(command, params) + problems = check(reply, expect) + problems += store(client, step, reply, fd) + + if problems: + problems.append(f"reply: {json.dumps(reply)}") + return problems + + +def run_accept(client: Client, step: dict) -> list[str]: + """Take one arrival off a listener and report what did not hold.""" + try: + arrival, fd = client.accept(step["accept"]) + except socket.timeout: + return [f"timed out waiting for an arrival on {step['accept']!r}"] + + problems = check(arrival, substitute(step.get("expect", {}), client.flows)) + problems += store(client, step, arrival, fd) + + if problems: + problems.append(f"arrival: {json.dumps(arrival)}") + return problems + + +def run_sleep(step: dict) -> list[str]: + """Hold every descriptor open for a while, failing nothing.""" + time.sleep(float(step["sleep"])) + return [] + + +def run_flow(client: Client, step: dict) -> list[str]: + """Operate on a stored descriptor and report what did not hold.""" + name = step["fd"] + if name not in client.fds: + return [f"no descriptor named {name!r}"] + flow, _role = client.fds[name] + + problems: list[str] = [] + + if "readable" in step: + ready, _, _ = select.select([flow], [], [], 0.25) + got = bool(ready) + if got != step["readable"]: + problems.append(f"readable: wanted {step['readable']}, got {got}") + + if "write" in step: + payload = bytes.fromhex(step["write"]) + for _ in range(step.get("repeat", 1)): + flow.send(payload) + + if "read" in step: + wanted = bytes.fromhex(step["expect_bytes"]) if "expect_bytes" in step else None + sizes = [] + for index in range(step["read"]): + try: + got = flow.recv(65536) + except socket.timeout: + problems.append(f"read {index}: timed out waiting for a datagram") + break + sizes.append(len(got)) + if wanted is not None and got != wanted: + problems.append( + f"read {index}: wanted {wanted.hex()}, got {got.hex()}" + ) + if "sizes" in step and sizes != step["sizes"]: + problems.append(f"sizes: wanted {step['sizes']}, got {sizes}") + + if step.get("close"): + flow.close() + del client.fds[name] + + return problems + + +def label_of(step: dict) -> str: + """The name a step is reported under, which callers wait on by substring.""" + if "fd" in step: + return f"fd {step['fd']}" + if "accept" in step: + return f"accept {step['accept']}" + if "sleep" in step: + return f"sleep {step['sleep']}" + return step.get("command", "?") + + +def main() -> int: + """Run the script against the socket and report every failing step.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--socket", required=True, help="native API socket path") + group = parser.add_mutually_exclusive_group(required=True) + group.add_argument("--script", help="steps as a JSON list") + group.add_argument("--script-file", help="file holding the steps as a JSON list") + parser.add_argument( + "--timeout", + type=float, + default=5.0, + help="socket timeout in seconds (default: 5)", + ) + args = parser.parse_args() + + text = args.script + if text is None: + with open(args.script_file, encoding="utf-8") as handle: + text = handle.read() + steps = json.loads(text) + + try: + client = Client(args.socket, args.timeout) + except OSError as error: + print(f"connect to {args.socket} failed: {error}", file=sys.stderr) + return 2 + + failures = 0 + try: + for index, step in enumerate(steps): + label = label_of(step) + try: + if "fd" in step: + problems = run_flow(client, step) + elif "accept" in step: + problems = run_accept(client, step) + elif "sleep" in step: + problems = run_sleep(step) + else: + problems = run_rpc(client, step) + except (OSError, ConnectionError, Protocol, json.JSONDecodeError) as error: + print(f"step {index} ({label}): {error}", file=sys.stderr) + return 2 + + if problems: + failures += 1 + print(f"step {index} ({label}) FAILED", file=sys.stderr) + for problem in problems: + print(f" {problem}", file=sys.stderr) + else: + print(f"step {index} ({label}) ok") + finally: + client.close() + + return 1 if failures else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/testing/native-api/control.py b/testing/native-api/control.py new file mode 100755 index 00000000..a67bc2ac --- /dev/null +++ b/testing/native-api/control.py @@ -0,0 +1,152 @@ +#!/usr/bin/env python3 +"""Control socket client for the native API checks. + +Speaks the control socket's line-delimited JSON protocol: one request line, one +response line, one command per connection. That is a different protocol from the +native API's — whose connection outlives its first command and carries +descriptors — which is why this is a separate tool rather than another step type +in `client.py`. + +It runs in a container for the same reason the native API client does: both +sockets are bound 0o770 root:fips, so the host user running the harness cannot +open either, while a client container mounting the same directory runs as root +and can. + +Usage: + control.py --socket PATH --command show_native_flows + control.py --socket PATH --command show_native_flows --expect data.flows.0.local_port=4501 + +An expectation is `dotted.path=value` or `dotted.path>value`. A path segment of +digits indexes a list, so `data.flows.0.state` reads the first flow's state. The +value is parsed as JSON where it parses and taken as a plain string where it +does not, so both `4501` and `established` say what they look like. `>` compares +numerically and is how a check asserts a counter moved without pinning a total +that the wire is free to reach by more than one datagram. + +Exit 0 when every expectation holds, 1 when one does not, 2 on a connection or +protocol failure. The reply is printed either way, because a check that missed +one field needs the whole object to say why. +""" + +from __future__ import annotations + +import argparse +import json +import socket +import sys +from typing import Any + +MISSING = object() +"""Distinguishes a path that is absent from one whose value is JSON null.""" + + +def query(path: str, command: str, timeout: float) -> dict: + """Send one command over its own connection and return the decoded reply.""" + conn = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + conn.settimeout(timeout) + try: + conn.connect(path) + conn.sendall((json.dumps({"command": command}) + "\n").encode()) + buffer = b"" + while b"\n" not in buffer: + chunk = conn.recv(65536) + if not chunk: + raise ConnectionError("the control socket closed before replying") + buffer += chunk + return json.loads(buffer.split(b"\n", 1)[0].decode()) + finally: + conn.close() + + +def dig(value: Any, dotted: str) -> Any: + """Read a dotted path out of a reply, indexing lists on a numeric segment.""" + for key in dotted.split("."): + if isinstance(value, list): + if not key.isdigit() or int(key) >= len(value): + return MISSING + value = value[int(key)] + elif isinstance(value, dict) and key in value: + value = value[key] + else: + return MISSING + return value + + +def parse(text: str) -> Any: + """Read an expectation's value as JSON, falling back to a plain string.""" + try: + return json.loads(text) + except json.JSONDecodeError: + return text + + +def split(expectation: str) -> tuple[str, str, Any]: + """Split `path=value` or `path>value` on whichever operator comes first.""" + cuts = [(expectation.index(op), op) for op in ("=", ">") if op in expectation] + if not cuts: + raise ValueError(f"{expectation!r} carries neither '=' nor '>'") + at, op = min(cuts) + return expectation[:at], op, parse(expectation[at + 1:]) + + +def check(reply: dict, expectation: str) -> str | None: + """Return a message when the expectation does not hold, else None.""" + dotted, op, wanted = split(expectation) + got = dig(reply, dotted) + if got is MISSING: + return f"{dotted}: wanted {op}{wanted!r}, but the path is absent" + if op == "=": + return None if got == wanted else f"{dotted}: wanted {wanted!r}, got {got!r}" + if isinstance(got, bool) or not isinstance(got, (int, float)) or got <= wanted: + return f"{dotted}: wanted more than {wanted!r}, got {got!r}" + return None + + +def main() -> int: + """Query the control socket and report every expectation that did not hold.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--socket", required=True, help="control socket path") + parser.add_argument("--command", required=True, help="control command to send") + parser.add_argument( + "--expect", + action="append", + default=[], + metavar="PATH=VALUE", + help="dotted-path expectation; repeatable", + ) + parser.add_argument( + "--timeout", + type=float, + default=5.0, + help="socket timeout in seconds (default: 5)", + ) + args = parser.parse_args() + + try: + reply = query(args.socket, args.command, args.timeout) + except (OSError, ConnectionError, json.JSONDecodeError) as error: + print(f"{args.command} on {args.socket} failed: {error}", file=sys.stderr) + return 2 + + print(json.dumps(reply)) + + if reply.get("status") != "ok": + print(f"{args.command} answered {reply.get('status')!r}", file=sys.stderr) + return 1 + + try: + problems = [ + message + for message in (check(reply, expectation) for expectation in args.expect) + if message is not None + ] + except ValueError as error: + print(f" {error}", file=sys.stderr) + return 2 + for message in problems: + print(f" {message}", file=sys.stderr) + return 1 if problems else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/testing/native-api/node-api-off.yaml b/testing/native-api/node-api-off.yaml new file mode 100644 index 00000000..47aae158 --- /dev/null +++ b/testing/native-api/node-api-off.yaml @@ -0,0 +1,28 @@ +# Node configuration with the native API explicitly disabled. +# +# The companion to node.yaml, used by the check that no socket appears when the +# feature is off. It names the same socket_path deliberately: if the disable +# were ignored, the socket would land exactly where the check looks. + +node: + identity: + nsec: "2102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f20" + + native_api: + enabled: false + socket_path: "/run/fips/api.sock" + # Named for the same reason as socket_path: everything node.yaml turns on + # is turned on here too, so `enabled` is the only difference between the + # two and the only thing the check can be observing. + debug_commands: true + +tun: + enabled: false + +dns: + enabled: false + +transports: + udp: + bind_addr: "0.0.0.0:2121" + mtu: 1472 diff --git a/testing/native-api/node-debug-off.yaml b/testing/native-api/node-debug-off.yaml new file mode 100644 index 00000000..172c5d1d --- /dev/null +++ b/testing/native-api/node-debug-off.yaml @@ -0,0 +1,27 @@ +# Node configuration with the native API on and the debug commands off. +# +# The companion to node.yaml, used by the check that inject, stats and arrive +# are refused where the gate is closed. It differs from node.yaml in one thing: +# `debug_commands` is not named at all, so what the check observes is the +# packaged default rather than an explicit false. A change that flipped that +# default would then red this check, which is the whole point of leaving the +# key out. + +node: + identity: + nsec: "2102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f20" + + native_api: + enabled: true + socket_path: "/run/fips/api.sock" + +tun: + enabled: false + +dns: + enabled: false + +transports: + udp: + bind_addr: "0.0.0.0:2121" + mtu: 1472 diff --git a/testing/native-api/node.yaml b/testing/native-api/node.yaml new file mode 100644 index 00000000..a8473286 --- /dev/null +++ b/testing/native-api/node.yaml @@ -0,0 +1,32 @@ +# Node configuration for the native datagram API checks. +# +# One node, no peers, no TUN and no DNS: increment A1 exercises the API socket +# itself, so the node needs to start and bind the socket and nothing else. +# Dropping TUN is what lets the container run without NET_ADMIN or +# /dev/net/tun, which keeps the check cheap and portable. + +node: + identity: + nsec: "2102030405060708090a0b0c0d0e0f101112131415161718191a1b1c1d1e1f20" + + native_api: + enabled: true + # Named explicitly rather than left to the default resolver, because the + # check bind-mounts this directory from the host in order to read the + # socket's mode and to reach it from the client container. + socket_path: "/run/fips/api.sock" + # inject, stats and arrive. Off in a packaged node, and on here because + # nine of the checks below drive the receive and dispatch paths through + # them. node-debug-off.yaml is the companion that leaves them off. + debug_commands: true + +tun: + enabled: false + +dns: + enabled: false + +transports: + udp: + bind_addr: "0.0.0.0:2121" + mtu: 1472 diff --git a/testing/native-api/test.sh b/testing/native-api/test.sh new file mode 100755 index 00000000..df9cae68 --- /dev/null +++ b/testing/native-api/test.sh @@ -0,0 +1,1417 @@ +#!/bin/bash +# Native datagram API checks (experimental feature). +# +# The client runs in its own container and reaches the daemon through a +# bind-mounted /run/fips. That is the real deployment shape: a separate process +# with its own filesystem opening the socket, rather than a test speaking to the +# daemon from inside its own container. It is also what makes the access policy +# observable — the host sees the socket file and can read its mode. +# +# The setup protocol is two commands and two descriptors. `connect` hands back +# a flow's SOCK_SEQPACKET half; `listen` hands back the listener's own, which is +# what makes a listener pollable and makes accepting a recvmsg on it rather than +# a round trip on the command socket. There are no events and no accept or +# reject command: an arriving flow is one message on the listener descriptor, +# carrying that flow's descriptor with it, and refusing a flow is close(2). +# +# Every check that receives a descriptor names it. The harness treats a +# descriptor nothing named as a failure rather than ignoring it, because an +# unnamed one is a flow or a held port dropped with nothing to say so. +# +# A peer is named by npub everywhere a client can see it, in a connect reply and +# in an arrival alike. The 16-byte node address is the wire's and appears only +# on the operator surface, which is why the control-socket check below asserts +# both and the client checks assert only the npub. +# +# The observability check reads the daemon's own view back out through the +# control socket. That needs nothing added to the image: the control socket is +# enabled by default and lands in the same bind-mounted /run/fips, and the image +# already carries python3 for the API client. +# +# inject, stats and arrive are debug commands, gated on +# node.native_api.debug_commands and off in a packaged node. node.yaml turns +# them on because nine checks here drive them; check_debug_commands_gated runs a +# second node that leaves the key alone and asserts all three are refused there. +# +# The echo check runs examples/native-echo.rs, which is built against the +# crate's own client module rather than against the line protocol. That makes it +# the only check here where both ends are programs a user could have written. +# examples/native-surface.rs is the second such program: it walks the client's +# whole public surface against a live daemon and asserts what each item reports, +# which is the coverage a check driven from the Python line-protocol client +# cannot give. Both image paths carry both binaries; see resolve_image and +# testing/docker/Dockerfile. +# +# Usage: ./test.sh +# +# Image: FIPS_TEST_IMAGE if set (this is what ci-local.sh passes, and there is +# deliberately no fips-test:latest to fall back on). Otherwise a minimal image +# is built from locally compiled binaries, which is the standalone path. + +set -uo pipefail + +SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" +REPO_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" +LABEL="com.corganlabs.fips-ci=1" +NODE_NAME="fips-native-api-node-$$" +GATED_NAME="fips-native-api-gated-$$" +SURFACE="fips-native-surface-$$" +SOCK_DIR="$(mktemp -d)" +IMAGE="" +BUILT_IMAGE="" + +# Two valid npubs the checks use as peers. Nothing is ever sent to them: the +# wire is not connected, and the arrival command only names them. They must +# decode, because connect and arrive resolve an npub to a node address. +PEER="npub1sjlh2c3x9w7kjsqg2ay080n2lff2uvt325vpan33ke34rn8l5jcqawh57m" +PEER2="npub1n9lpnv0592cc2ps6nm0ca3qls642vx7yjsv35rkxqzj2vgds52sqgpverl" + +# How many assertions one run of examples/native-surface.rs records. +# +# Deliberately brittle. The binary prints the recorder's own counter rather than +# a literal, so comparing against this number catches an assertion block that +# stopped running as well as one that failed: a #[cfg] gate that no longer +# matches, or an early return, still exits 0 and still prints the completion +# line, and only the count betrays it. Adding an assertion there must therefore +# force an edit here, which is the friction that keeps the two from drifting +# apart in silence. +SURFACE_ASSERTIONS=31 + +PASS=0 +FAIL=0 + +log() { echo "=== $*"; } +pass() { echo " PASS: $*"; PASS=$((PASS + 1)); } +fail() { echo " FAIL: $*"; FAIL=$((FAIL + 1)); } + +cleanup() { + teardown_two_nodes + docker rm -f "$NODE_NAME" >/dev/null 2>&1 + docker rm -f "${NODE_NAME}-off" >/dev/null 2>&1 + docker rm -f "$GATED_NAME" >/dev/null 2>&1 + docker rm -f "$SURFACE" >/dev/null 2>&1 + [[ -n "$BUILT_IMAGE" ]] && docker rmi -f "$BUILT_IMAGE" >/dev/null 2>&1 + rm -rf "$SOCK_DIR" + return 0 +} +trap cleanup EXIT + +# ───────────────────────────────────────────────────────────────────── +# Image +# ───────────────────────────────────────────────────────────────────── + +# Print a source file newer than `built`, and nothing when none is. The second +# argument, when given, is the example source `built` was compiled from. +# +# Every binary this run depends on is built from src/ and Cargo.toml, and is +# read alongside the harness client in this directory. The Python is +# bind-mounted live rather than built into the image, so a redesigned harness +# would otherwise run against a stale daemon without complaint, which is +# exactly the verdict this guard exists to prevent. +# +# An example is probed against its OWN .rs and no other, which is what keeps +# the guard usable rather than merely correct. Cargo does not relink +# target/release/fips when only an example changes, so probing the daemon +# against all of examples/ leaves it permanently older than a just-edited +# example: the run refuses, and the `cargo build --release --bins --examples` +# the refusal prescribes does not clear the condition. Editing native-surface.rs +# is the routine case now, so that deadlock would be hit on nearly every +# iteration. +stale_source() { + local built="$1" own="${2:-}" + find "$REPO_ROOT/src" "$REPO_ROOT/Cargo.toml" -newer "$built" -print -quit 2>/dev/null + find "$SCRIPT_DIR" -name '*.py' -newer "$built" -print -quit 2>/dev/null + [[ -n "$own" && "$own" -nt "$built" ]] && echo "$own" + return 0 +} + +resolve_image() { + if [[ -n "${FIPS_TEST_IMAGE:-}" ]]; then + IMAGE="$FIPS_TEST_IMAGE" + log "Using FIPS_TEST_IMAGE=$IMAGE" + return 0 + fi + + # All three binaries come from one profile directory. A release daemon + # paired with a debug echo server would be two builds of two trees, which is + # the mixture the staleness guard below exists to refuse. + local profile="" + for candidate in "$REPO_ROOT/target/release" "$REPO_ROOT/target/debug"; do + [[ -x "$candidate/fips" && -x "$candidate/examples/native-echo" \ + && -x "$candidate/examples/native-surface" ]] \ + && { profile="$candidate"; break; } + done + if [[ -z "$profile" ]]; then + echo "No FIPS_TEST_IMAGE, and no profile holds fips with the native-echo and native-surface examples." >&2 + echo "Build them first: cargo build --release --bins --examples" >&2 + return 1 + fi + local binary="$profile/fips" + local echo_binary="$profile/examples/native-echo" + local walk_binary="$profile/examples/native-surface" + + # A stale binary is the worst outcome available here: the checks would run, + # pass or fail against code that is not the working tree's, and report a + # verdict about the wrong thing. Refuse rather than warn. + local newest_source + newest_source="$(stale_source "$binary" + stale_source "$echo_binary" "$REPO_ROOT/examples/native-echo.rs" + stale_source "$walk_binary" "$REPO_ROOT/examples/native-surface.rs")" + # The probes can each name the same file; report it once. + newest_source="${newest_source%%$'\n'*}" + if [[ -n "$newest_source" ]]; then + echo "$(basename "$profile") binaries are older than $newest_source" >&2 + echo "Rebuild before running: cargo build --release --bins --examples" >&2 + return 1 + fi + + BUILT_IMAGE="fips-native-api-test:$$" + log "Building $BUILT_IMAGE from $(basename "$profile") binaries" + local context + context="$(mktemp -d)" + cp "$binary" "$context/fips" + cp "$echo_binary" "$context/native-echo" + cp "$walk_binary" "$context/native-surface" + cat > "$context/Dockerfile" <<'DOCKERFILE' +FROM debian:trixie-slim +# libdbus-1-3 and libsystemd0 are the daemon's dynamic dependencies (BLE and +# journal integration). Without them the binary does not start, and a check +# that only looks for a socket would read that as a feature failure. +RUN apt-get update && \ + apt-get install -y --no-install-recommends \ + python3 ca-certificates libdbus-1-3 libsystemd0 && \ + rm -rf /var/lib/apt/lists/* +COPY fips /usr/local/bin/fips +# The echo check runs native-echo as one end of a two-node exchange and the +# surface walk runs native-surface against a single node, so the standalone +# image needs both exactly as the CI image does. +COPY native-echo /usr/local/bin/native-echo +COPY native-surface /usr/local/bin/native-surface +RUN chmod +x /usr/local/bin/fips /usr/local/bin/native-echo /usr/local/bin/native-surface +DOCKERFILE + docker build -t "$BUILT_IMAGE" --label "$LABEL" "$context" --quiet >/dev/null + local status=$? + rm -rf "$context" + [[ $status -ne 0 ]] && return 1 + IMAGE="$BUILT_IMAGE" + return 0 +} + +# ───────────────────────────────────────────────────────────────────── +# Daemon +# ───────────────────────────────────────────────────────────────────── + +# Start a node with the given config, into the given container name. +start_node() { + local name="$1" config="$2" sockdir="$3" + docker rm -f "$name" >/dev/null 2>&1 + docker run -d --name "$name" --label "$LABEL" \ + -v "$sockdir:/run/fips" \ + -v "$SCRIPT_DIR/$config:/etc/fips/fips.yaml:ro" \ + --entrypoint /usr/local/bin/fips \ + "$IMAGE" --config /etc/fips/fips.yaml >/dev/null +} + +# Wait up to `timeout` seconds for the socket file to appear. +wait_for_socket() { + local path="$1" timeout="${2:-20}" waited=0 + while [[ $waited -lt $timeout ]]; do + [[ -S "$path" ]] && return 0 + sleep 1 + waited=$((waited + 1)) + done + return 1 +} + +# Wait up to `timeout` seconds for a node container to report itself running. +# +# "Running" means the daemon logged its startup line, not merely that the +# container is up: a container whose process is crash-looping is still "up" for +# a moment at a time. +wait_for_running() { + local name="$1" timeout="${2:-20}" waited=0 + while [[ $waited -lt $timeout ]]; do + if [[ "$(docker inspect -f '{{.State.Running}}' "$name" 2>/dev/null)" == "true" ]] \ + && docker logs "$name" 2>&1 | grep -q "Node started"; then + return 0 + fi + sleep 1 + waited=$((waited + 1)) + done + return 1 +} + +# Run a client script in its own container against the socket. +run_client() { + docker run --rm --label "$LABEL" \ + -v "$SOCK_DIR:/run/fips" \ + -v "$SCRIPT_DIR:/harness:ro" \ + --entrypoint python3 \ + "$IMAGE" -u /harness/client.py --socket /run/fips/api.sock --script "$1" +} + +# ───────────────────────────────────────────────────────────────────── +# Checks +# ───────────────────────────────────────────────────────────────────── + +check_socket_appears() { + log "The socket appears when the API is enabled" + start_node "$NODE_NAME" node.yaml "$SOCK_DIR" + if wait_for_socket "$SOCK_DIR/api.sock"; then + pass "socket bound at /run/fips/api.sock" + else + fail "socket never appeared" + docker logs "$NODE_NAME" 2>&1 | tail -20 + return 1 + fi +} + +check_socket_mode() { + log "The socket carries the group-access mode" + local mode + mode="$(stat -c '%a' "$SOCK_DIR/api.sock" 2>/dev/null)" + if [[ "$mode" == "770" ]]; then + pass "mode is 0770" + else + # A failure here means the access policy did not run, which would leave + # the socket readable by anyone who can reach the directory. + fail "mode is $mode, wanted 770" + fi +} + +check_commands_answered() { + log "Every command is answered on one connection, in command order" + # Several commands travel over a single connection, which is the property + # that separates this socket from the control socket's one-shot shape. Both + # setup commands carry a descriptor, so this is also where the descriptor + # association rule bites: two descriptor-bearing replies on one connection + # is the shape that would let a reader hand one reply the other's flow. + # + # The daemon's own re-encode of the peer is asserted rather than an echo of + # what was sent, because that is what makes a connect reply and an arrival + # name one peer the same way. + local script='[ + {"command":"listen","params":{"local_port":4242},"keep_listener":"L", + "expect":{"status":"ok","data.local_port":4242,"data.backlog":16}}, + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":4242,"local_port":5000}, + "keep_fd":"a", + "expect":{"status":"ok","data.local_port":5000,"data.remote_port":4242, + "data.peer":"'"$PEER"'"}}, + {"command":"teleport","expect":{"status":"error","data.errno":"EINVAL"}} + ]' + if run_client "$script"; then + pass "both setup commands and a refusal answered over one connection" + else + fail "a command was not answered as expected" + fi +} + +check_ephemeral_allocation() { + log "Port 0 and an absent port both ask for an ephemeral one" + # One rule covers both commands, and a listener that asked for 0 has nowhere + # but the reply to learn the port it got, which is getsockname after bind(2) + # with port 0. + # + # The three ports asserted are the first three the node ever hands out: the + # allocator is a cursor on the node that only moves forward, so this check + # has to run before anything else that asks for an ephemeral port. + local script='[ + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":4242}, + "keep_fd":"a","expect":{"status":"ok","data.local_port":49152}}, + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":4242,"local_port":0}, + "keep_fd":"b","expect":{"status":"ok","data.local_port":49153}}, + {"command":"listen","params":{"local_port":0},"keep_listener":"L", + "expect":{"status":"ok","data.local_port":49154}} + ]' + if run_client "$script"; then + pass "connect and listen both allocate from the ephemeral range" + else + fail "ephemeral allocation did not answer as expected" + fi +} + +check_reserved_ports_refused() { + log "Reserved port tiers are refused" + # Port 256 is the IPv6 shim. A client that could bind it, or name it on a + # peer, would be injecting into an IPv6 plane it does not own. + # + # The errno is asserted, not merely the refusal: it is the contract a + # binding turns into what bind(2) returns, and EADDRNOTAVAIL rather than + # EACCES because no client, however privileged, may hold port 256. + local script='[ + {"command":"listen","params":{"local_port":256}, + "expect":{"status":"error","data.errno":"EADDRNOTAVAIL"}}, + {"command":"listen","params":{"local_port":80}, + "expect":{"status":"error","data.errno":"EADDRNOTAVAIL"}}, + {"command":"listen","params":{"local_port":1023}, + "expect":{"status":"error","data.errno":"EADDRNOTAVAIL"}}, + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":256}, + "expect":{"status":"error","data.errno":"EADDRNOTAVAIL"}}, + {"command":"listen","params":{"local_port":1024},"keep_listener":"L", + "expect":{"status":"ok"}} + ]' + if run_client "$script"; then + pass "the reserved tiers are refused as EADDRNOTAVAIL and 1024 is not" + else + fail "port tier policy did not hold" + fi +} + +check_bad_input_survives() { + log "A bad command does not end the connection" + # The healthy path after each refusal is the point: a client that mistypes + # one command must not lose the flows its connection holds. + local script='[ + {"command":"teleport","expect":{"status":"error","data.errno":"EINVAL"}}, + {"command":"listen","expect":{"status":"error","data.errno":"EINVAL"}}, + {"command":"listen","params":{"local_port":4242},"keep_listener":"L", + "expect":{"status":"ok"}} + ]' + if run_client "$script"; then + pass "the connection survives an unknown command and missing params" + else + fail "the connection did not survive a bad command" + fi +} + +check_descriptor_carries_datagrams() { + log "A descriptor arrives with the reply and carries datagrams to the daemon" + # Three writes must reach the daemon as three datagrams. The byte count + # matters as much as the datagram count: a boundary loss would show up as + # one datagram of nine bytes rather than three of three. + local script='[ + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":4242}, + "keep_fd":"a","keep_flow":"a","expect":{"status":"ok"}}, + {"fd":"a","write":"00ff10","repeat":3}, + {"command":"stats","params":{"flow_id":"@a"}, + "expect":{"status":"ok","data.rx_datagrams":3,"data.rx_bytes":9,"data.closed":false}} + ]' + if run_client "$script"; then + pass "three datagrams crossed the descriptor and the daemon counted them" + else + fail "the daemon did not observe what the client wrote" + fi +} + +check_receive_direction() { + log "The daemon can write datagrams the client receives whole" + # 00ff10 includes a byte that is not valid UTF-8, so this also shows the + # path carries arbitrary bytes rather than text. + local script='[ + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":4242}, + "keep_fd":"a","keep_flow":"a","expect":{"status":"ok"}}, + {"fd":"a","readable":false}, + {"command":"inject","params":{"flow_id":"@a","data":"00ff10","repeat":3}, + "expect":{"status":"ok","data.datagrams":3,"data.bytes":9}}, + {"fd":"a","readable":true}, + {"fd":"a","read":3,"expect_bytes":"00ff10","sizes":[3,3,3]} + ]' + if run_client "$script"; then + pass "poll reported readability and three datagrams arrived whole" + else + fail "the receive direction did not behave" + fi +} + +check_close_reaches_the_daemon() { + log "Closing the descriptor releases the flow at the daemon" + # Without this the daemon would hold a flow for a client that is gone, which + # is the leak the whole close path exists to prevent. There is no close + # command and never was: closing the descriptor is the signal. + # + # The node forgets a released flow, so `stats` answers for it the way it + # answers for any name it does not hold. The step before the close is what + # makes that discriminating: the same flow answered a moment earlier, so the + # refusal afterwards can only be the release. The sleep is the task hop + # between the daemon reading end of file and giving the entry back. + local script='[ + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":4242}, + "keep_fd":"a","keep_flow":"a","expect":{"status":"ok"}}, + {"fd":"a","write":"aa"}, + {"command":"stats","params":{"flow_id":"@a"}, + "expect":{"status":"ok","data.closed":false,"data.rx_datagrams":1}}, + {"fd":"a","close":true}, + {"sleep":1}, + {"command":"stats","params":{"flow_id":"@a"},"expect":{"status":"error"}} + ]' + if run_client "$script"; then + pass "the daemon saw the close and gave the flow back" + else + fail "the daemon did not release the closed flow" + fi +} + +check_flows_are_independent() { + log "Two flows on one connection stay separate" + local script='[ + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":4242}, + "keep_fd":"a","keep_flow":"a","expect":{"status":"ok"}}, + {"command":"connect","params":{"peer":"'"$PEER2"'","remote_port":4243}, + "keep_fd":"b","keep_flow":"b","expect":{"status":"ok"}}, + {"fd":"a","write":"11","repeat":2}, + {"command":"stats","params":{"flow_id":"@a"},"expect":{"data.rx_datagrams":2}}, + {"command":"stats","params":{"flow_id":"@b"},"expect":{"data.rx_datagrams":0}}, + {"command":"inject","params":{"flow_id":"@b","data":"22"},"expect":{"status":"ok"}}, + {"fd":"b","read":1,"expect_bytes":"22"}, + {"fd":"a","readable":false} + ]' + if run_client "$script"; then + pass "each flow saw only its own traffic" + else + fail "traffic crossed between flows" + fi +} + +check_unknown_flow_refused() { + log "A flow no node holds, and bad hex, are refused" + # The debug commands are addressed node-wide rather than per connection, + # because a flow belongs to the node and not to the connection that opened + # it. So the refusal here is about a flow that does not exist, not about + # ownership: any process that can open this socket can name any flow on the + # node, which is why both commands are behind a gate that is off by default. + local script='[ + {"command":"stats","params":{"flow_id":99},"expect":{"status":"error"}}, + {"command":"inject","params":{"flow_id":99,"data":"00"},"expect":{"status":"error"}}, + {"command":"inject","params":{"flow_id":1,"data":"zz"},"expect":{"status":"error"}} + ]' + if run_client "$script"; then + pass "an absent flow and bad hex are refused" + else + fail "an absent flow or bad hex was accepted" + fi +} + +# A valid npub the checks use as a peer. Nothing is ever sent to it: the wire is +# not connected, and the arrival command only names it. + +check_port_ownership() { + log "A local port has one owner, and the descriptor is what holds it" + # The registry lives on the node, so one port has one owner whichever client + # asked. What releases it changed: the port is held by the listener's + # descriptor, not by the connection, so closing that descriptor is the + # unbind and closing the connection is not. + local first='[ + {"command":"listen","params":{"local_port":4300},"keep_listener":"L", + "expect":{"status":"ok"}}, + {"command":"listen","params":{"local_port":4300}, + "expect":{"status":"error","data.errno":"EADDRINUSE"}}, + {"command":"connect","params":{"peer":"'"$PEER"'","remote_port":9000,"local_port":4300}, + "expect":{"status":"error","data.errno":"EADDRINUSE"}}, + {"fd":"L","close":true}, + {"sleep":1}, + {"command":"listen","params":{"local_port":4300},"keep_listener":"M", + "expect":{"status":"ok"}} + ]' + if run_client "$first"; then + pass "a held port is refused twice, and closing the listener unbinds it" + else + fail "a port was handed out twice, or a closed listener kept it" + return + fi + + # That client's process is gone, so the kernel closed the descriptor it + # still held and the same unbind path ran without its cooperation. + local second='[ + {"command":"listen","params":{"local_port":4300},"keep_listener":"L", + "expect":{"status":"ok"}} + ]' + if run_client "$second"; then + pass "the port came back when its client exited" + else + fail "a client that exited left its port held" + fi +} + +check_listen_accept_deliver() { + log "An arrival reaches the listener descriptor with its flow and what it held" + # The arrival command drives the same dispatch the FSP receive path does, so + # this exercises the rule before the wire is involved. + # + # Three things are asserted that the accept command used to answer for. The + # arrival names the peer by npub, which is the address its session + # authenticated and not a hash of it. It carries this node's own npub, so an + # accepted flow can answer getsockname without consulting its listener. And + # the datagram the node held before any descriptor existed is already on the + # flow's descriptor when the client reads the arrival, which is the ordering + # the hand-off guarantees: losing it drops a peer'"'"'s opening message. + local script='[ + {"command":"listen","params":{"local_port":4301},"keep_listener":"L", + "expect":{"status":"ok","data.local_port":4301}}, + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5000,"dst_port":4301,"data":"00ff10"}, + "expect":{"status":"ok","data.outcome":"announced"}}, + {"accept":"L","keep_fd":"a","keep_flow":"a", + "expect":{"local_port":4301,"remote_port":5000,"peer":"'"$PEER"'","held":1, + "max_payload":1362}}, + {"fd":"a","read":1,"expect_bytes":"00ff10"}, + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5000,"dst_port":4301,"data":"aabb"}, + "expect":{"status":"ok","data.outcome":"delivered"}}, + {"fd":"a","read":1,"expect_bytes":"aabb"} + ]' + if run_client "$script"; then + pass "the held datagram and the next one both reached the client" + else + fail "the listen and accept path did not deliver" + fi +} + +check_listener_is_pollable() { + log "A listener descriptor is pollable, and reports a waiting arrival" + # The point of the listener being a descriptor rather than a registration: + # it joins a client'"'"'s existing poll, select or epoll loop with no new + # mechanism, and accept is a recvmsg on it. Nothing else here proves that — + # every other listener check discovers an arrival by blocking in accept, + # which a registration could have offered too. + # + # The not-readable step before the arrival is what makes the readable one + # mean something: without it, a descriptor that was readable from the moment + # it was created would satisfy the check. + local script='[ + {"command":"listen","params":{"local_port":4305},"keep_listener":"L", + "expect":{"status":"ok"}}, + {"fd":"L","readable":false}, + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5100,"dst_port":4305,"data":"aa"}, + "expect":{"status":"ok","data.outcome":"announced"}}, + {"fd":"L","readable":true}, + {"accept":"L","keep_fd":"a","expect":{"local_port":4305,"remote_port":5100}}, + {"fd":"L","readable":false}, + {"fd":"a","read":1,"expect_bytes":"aa"} + ]' + if run_client "$script"; then + pass "poll reported the listener unreadable, then readable, then unreadable" + else + fail "the listener descriptor did not report readability" + fi +} + +check_dispatch_order() { + log "Dispatch prefers an established flow, then a listener, then a drop" + # The flow key is the peer and both ports, so the second arrival on one key + # must reach the flow that key already names rather than announce a second + # time; a third arrival differing only in source port is a different key and + # must announce. The listener being unreadable in between is what says the + # established flow won: without it, a daemon that announced everything would + # satisfy every other step here. + local script='[ + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5000,"dst_port":4302,"data":"aa"}, + "expect":{"status":"ok","data.outcome":"dropped: no listener or flow on that port"}}, + {"command":"listen","params":{"local_port":4302},"keep_listener":"L", + "expect":{"status":"ok"}}, + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5000,"dst_port":4302,"data":"aa"}, + "expect":{"status":"ok","data.outcome":"announced"}}, + {"accept":"L","keep_fd":"a","expect":{"local_port":4302,"remote_port":5000}}, + {"fd":"a","read":1,"expect_bytes":"aa"}, + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5000,"dst_port":4302,"data":"bb"}, + "expect":{"status":"ok","data.outcome":"delivered"}}, + {"fd":"a","read":1,"expect_bytes":"bb"}, + {"fd":"L","readable":false}, + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5001,"dst_port":4302,"data":"cc"}, + "expect":{"status":"ok","data.outcome":"announced"}}, + {"accept":"L","keep_fd":"b","expect":{"local_port":4302,"remote_port":5001}}, + {"fd":"b","read":1,"expect_bytes":"cc"} + ]' + if run_client "$script"; then + pass "an unheld port drops, a listener announces, then its flow receives" + else + fail "the dispatch order did not hold" + fi +} + +check_backlog_is_not_the_clients_bound() { + log "A listener that never accepts is not bounded at its backlog" + # This is what the backlog stopped meaning. It bounds flows the rx_loop has + # announced and the daemon has not yet wired, and the daemon drains that + # queue itself: a pending entry gives up its backlog claim as soon as the + # listener task promotes it, which happens whether or not the client ever + # calls recvmsg. So a client that binds a listener and reads nothing does + # not stop at 16. + # + # What does bound it after that is the listener'"'"'s send buffer and the + # node-wide max_flows, neither of which is a number a shell check can + # produce deterministically; the drop paths for both are covered by the + # daemon'"'"'s own tests. What is asserted here is the change: twenty + # arrivals at an unread listener, on a backlog of sixteen, and every one + # announced. + # + # The flow key is (peer, remote port, local port), so varying the source + # port makes each arrival a separate flow without needing distinct npubs. + local steps='[{"command":"listen","params":{"local_port":4303},"keep_listener":"L","expect":{"status":"ok"}}' + local port + for port in $(seq 6000 6019); do + steps="$steps,{\"command\":\"arrive\",\"params\":{\"peer\":\"$PEER\",\"src_port\":$port,\"dst_port\":4303,\"data\":\"aa\"},\"expect\":{\"data.outcome\":\"announced\"}}" + done + steps="$steps]" + + if run_client "$steps" >/dev/null; then + pass "20 arrivals announced at a listener that read none of them" + else + fail "an unread listener refused an arrival before its send buffer filled" + run_client "$steps" 2>&1 | grep -A 3 FAILED | head -8 + fi +} + +check_refusing_a_flow_frees_it() { + log "Refusing an accepted flow is closing its descriptor, and that frees it" + # There is no reject command: Berkeley has exactly one way to refuse a + # connection and so does this. Keeping a second would let a client refuse a + # flow two indistinguishable ways. + # + # Two things have to follow the close, and the second is what makes this + # more than a repeat of the connected-flow close: the node forgets the flow, + # and the registry entry and its key go with it, so the very same key + # announces a new flow afterwards rather than delivering into the dead one. + local script='[ + {"command":"listen","params":{"local_port":4304},"keep_listener":"L", + "expect":{"status":"ok"}}, + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5000,"dst_port":4304,"data":"aa"}, + "expect":{"status":"ok","data.outcome":"announced"}}, + {"accept":"L","keep_fd":"a","keep_flow":"a","expect":{"local_port":4304}}, + {"command":"stats","params":{"flow_id":"@a"},"expect":{"status":"ok"}}, + {"fd":"a","close":true}, + {"sleep":1}, + {"command":"stats","params":{"flow_id":"@a"},"expect":{"status":"error"}}, + {"command":"arrive","params":{"peer":"'"$PEER"'","src_port":5000,"dst_port":4304,"data":"bb"}, + "expect":{"status":"ok","data.outcome":"announced"}}, + {"accept":"L","keep_fd":"b","expect":{"local_port":4304,"remote_port":5000}}, + {"fd":"b","read":1,"expect_bytes":"bb"} + ]' + if run_client "$script"; then + pass "a refused flow is gone and its key is free to arrive again" + else + fail "closing a refused flow did not release it" + fi +} + +# The refusal a gated debug command earns, one per command name. +gate_refusal() { + printf "'%s' is a debug command and is disabled; set node.native_api.debug_commands to enable it" "$1" +} + +check_debug_commands_gated() { + log "The debug commands are refused where the gate is closed" + local gated_dir + gated_dir="$(mktemp -d)" + start_node "$GATED_NAME" node-debug-off.yaml "$gated_dir" + if ! wait_for_socket "$gated_dir/api.sock"; then + fail "the gated node never bound its socket, so nothing here proves anything" + docker logs "$GATED_NAME" 2>&1 | tail -20 + docker rm -f "$GATED_NAME" >/dev/null 2>&1 + rm -rf "$gated_dir" + return 1 + fi + + # Every command below names a flow and a port this connection really holds, + # opened one step earlier. That is what makes the refusals discriminating: + # ungated, all three succeed, so an error here can only have come from the + # gate. The message is asserted whole for the same reason — a refusal that + # said "no such flow" would otherwise pass as a refusal. + local gated_script + gated_script="$(cat <&1 | tail -20 + fi + + docker rm -f "$GATED_NAME" >/dev/null 2>&1 + rm -rf "$gated_dir" + + # The other half of the guard: the same three commands, same shape, against + # the node that has the key on. A gate that refused a legitimate run would + # be no use, and nothing above would notice. + local open_script + open_script="$(cat <