mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-05 19:18:25 +00:00
ethernet-churn failed on master and next alike, as a tree that did not converge. The daemon was not the cause: the harness left ring links down and reported them restored, and under host load those dead links lined up until every link was down at once. Finding that turned up several more harness defects, fixed together here. The iface-binding suite built its host veth names from FIPS_CI_NAME_SUFFIX, and an interface name gets fifteen characters. On a runner that sets the suffix to a timestamp and a pid, ip(8) refused the name before the first pair existed. GitHub's job does not set the suffix, so it passed there. The names now use the four-hex-character token from sim.naming, as the chaos simulation and the NAT topology script already do, and the reaper in ci-cleanup.sh matches the new shape under both the scoped and the unscoped sweep. A churned node's restart recreates each veth pair it shared with its neighbours. A stopped container's network namespace can outlive the stop by about two minutes, and while it does, renaming the survivor's new end fails with "File exists". The harness ignored that, read the old interface's MAC, logged success, and left the new end down. The restore now deletes any interface holding the final or temporary name first, checks every add, move and rename, and waits for both ends to report operstate up. A restore that still fails raises as a harness fault. A container PID docker cannot report now raises instead of reading as "not running", and a pair is deferred only for a neighbour churn itself stopped. The survivor's end of a recreated link also gets its netem parameters back; before, that direction ran unshaped. The runner hands one down-node set to every manager, and node churn and traffic stored it as `down_nodes or set()`. The set is empty when they are built, so each kept a private copy: traffic started iperf3 on stopped containers, and netem and link flaps tried to shape them. Every manager and event schedule drew from one random stream in wall-clock order, so host load changed which node churn stopped next. Each consumer now has its own stream derived from the seed. The topology and ephemeral node choice stay on the seed's own stream, so generated topologies do not change, but every other runtime draw does. The final tree snapshot now waits for three consecutive agreeing reads, five seconds apart and bounded at ninety seconds, instead of being taken the moment the stopped nodes were restored. A red chaos scenario lost its results directory with the worktree the CI worker deletes. Each scenario's results are now scoped to the run, and a red prints its status, assertions, final tree and each node's log tail into the run log. ethernet-churn's baseline had been calibrated on the broken restore. On the fixed harness, sixteen runs across master-line and next-line code, twelve of them under contention and across three seeds, all ended with 4 nodes answering, 1 root and 3 parented, and the scenario now asserts exactly that. The scenario loader also checked the parented floor against one root only; it now checks it against max_roots, since a mesh with R roots can parent at most n - R nodes.
231 lines
8.0 KiB
Python
231 lines
8.0 KiB
Python
"""Node churn simulation via docker stop/start.
|
|
|
|
Simulates node loss and recovery by stopping and restarting containers.
|
|
When a node restarts, its FIPS process starts fresh — all peer
|
|
connections, sessions, and routing state are lost. Peers detect the
|
|
loss via handshake/link timeouts and reconverge the spanning tree.
|
|
|
|
After restart, netem rules must be re-applied since tc state is lost
|
|
when the container stops.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import random
|
|
import time
|
|
from collections import deque
|
|
from dataclasses import dataclass, field
|
|
|
|
from .docker_exec import docker_exec_quiet, is_container_running
|
|
from .scenario import NodeChurnConfig
|
|
from .topology import SimTopology
|
|
|
|
log = logging.getLogger(__name__)
|
|
|
|
|
|
@dataclass
|
|
class NodeState:
|
|
node_id: str
|
|
is_down: bool = False
|
|
down_since: float | None = None
|
|
restore_at: float | None = None
|
|
|
|
|
|
class NodeManager:
|
|
"""Manages node stop/start lifecycle."""
|
|
|
|
def __init__(
|
|
self,
|
|
topology: SimTopology,
|
|
config: NodeChurnConfig,
|
|
rng: random.Random,
|
|
netem_mgr=None,
|
|
down_nodes: set[str] | None = None,
|
|
veth_mgr=None,
|
|
on_node_restart=None,
|
|
):
|
|
self.topology = topology
|
|
self.config = config
|
|
self.rng = rng
|
|
self.netem_mgr = netem_mgr
|
|
self.veth_mgr = veth_mgr
|
|
# `is not None`, not `or`: the runner passes its shared set while it
|
|
# is still empty, and an empty set is falsy, so `or` replaced it with
|
|
# a private one and no other manager ever saw a node go down.
|
|
self.down_nodes = down_nodes if down_nodes is not None else set()
|
|
self.on_node_restart = on_node_restart
|
|
self.node_states: dict[str, NodeState] = {
|
|
nid: NodeState(node_id=nid) for nid in topology.nodes
|
|
}
|
|
|
|
@property
|
|
def down_count(self) -> int:
|
|
return sum(1 for ns in self.node_states.values() if ns.is_down)
|
|
|
|
def maybe_kill(self):
|
|
"""Attempt to stop a random node."""
|
|
if self.down_count >= self.config.max_down_nodes:
|
|
log.debug(
|
|
"At max_down_nodes (%d), skipping churn",
|
|
self.config.max_down_nodes,
|
|
)
|
|
return
|
|
|
|
# Candidate: any node that's currently up
|
|
up_nodes = [nid for nid, ns in self.node_states.items() if not ns.is_down]
|
|
if not up_nodes:
|
|
return
|
|
|
|
self.rng.shuffle(up_nodes)
|
|
|
|
for node_id in up_nodes:
|
|
if self.config.protect_connectivity and self._would_disconnect(node_id):
|
|
log.debug("Skipping %s (would disconnect graph)", node_id)
|
|
continue
|
|
|
|
down_duration = self.rng.uniform(
|
|
self.config.down_duration_secs.min,
|
|
self.config.down_duration_secs.max,
|
|
)
|
|
self._stop_node(node_id, down_duration)
|
|
return
|
|
|
|
log.debug("No safe node to kill (all would disconnect)")
|
|
|
|
def restore_expired(self):
|
|
"""Restart nodes whose down duration has expired."""
|
|
now = time.time()
|
|
for nid, state in self.node_states.items():
|
|
if state.is_down and state.restore_at and now >= state.restore_at:
|
|
self._start_node(nid)
|
|
|
|
def restore_all(self):
|
|
"""Restart all stopped nodes (for teardown — needed for log collection)."""
|
|
for nid, state in list(self.node_states.items()):
|
|
if state.is_down:
|
|
self._start_node(nid)
|
|
|
|
def _stop_node(self, node_id: str, duration: float):
|
|
"""Stop a container, and record it as down only if it stopped.
|
|
|
|
A `docker stop` that exits non-zero -- the container already gone,
|
|
the daemon refusing -- once left the node marked down anyway. Every
|
|
figure the scenario reasons with is derived from that mark:
|
|
`down_count` and so the `max_down_nodes` cap, `_would_disconnect`
|
|
and so the `protect_connectivity` guard, and the shared
|
|
`down_nodes` set that netem, links and traffic all skip. Returning
|
|
before the mutation keeps the model equal to the mesh and lets the
|
|
next churn tick retry the node.
|
|
"""
|
|
container = self.topology.container_name(node_id)
|
|
docker_exec_quiet(container, "kill 1", timeout=5) # SIGTERM to PID 1
|
|
# Use docker stop with a short grace period
|
|
import subprocess
|
|
|
|
result = subprocess.run(
|
|
["docker", "stop", "-t", "2", container],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=15,
|
|
)
|
|
if result.returncode != 0:
|
|
log.warning(
|
|
"Failed to stop %s: %s; leaving %s marked up",
|
|
container, result.stderr.strip(), node_id,
|
|
)
|
|
return
|
|
|
|
now = time.time()
|
|
state = self.node_states[node_id]
|
|
state.is_down = True
|
|
state.down_since = now
|
|
state.restore_at = now + duration
|
|
|
|
# Tell other managers to skip this node
|
|
self.down_nodes.add(node_id)
|
|
|
|
log.info("Node STOPPED: %s (restore in %.0fs)", node_id, duration)
|
|
|
|
def _start_node(self, node_id: str):
|
|
"""Start a stopped container and re-apply netem."""
|
|
container = self.topology.container_name(node_id)
|
|
import subprocess
|
|
|
|
result = subprocess.run(
|
|
["docker", "start", container],
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=15,
|
|
)
|
|
if result.returncode != 0:
|
|
log.warning("Failed to start %s: %s", container, result.stderr)
|
|
return
|
|
|
|
state = self.node_states[node_id]
|
|
down_for = time.time() - state.down_since if state.down_since else 0
|
|
state.is_down = False
|
|
state.down_since = None
|
|
state.restore_at = None
|
|
|
|
# Clear the down-node flag before re-applying netem
|
|
self.down_nodes.discard(node_id)
|
|
|
|
log.info("Node STARTED: %s (was down %.0fs)", node_id, down_for)
|
|
|
|
# Re-create veth pairs (container restart destroys netns)
|
|
if self.veth_mgr:
|
|
time.sleep(1)
|
|
# Churn's own record, not the shared set: netem's safety net also
|
|
# adds to that set, on a docker hiccup or a crash churn did not
|
|
# cause, and a pair deferred on that basis would never be rebuilt.
|
|
stopped = {nid for nid, ns in self.node_states.items() if ns.is_down}
|
|
self.veth_mgr.setup_node(node_id, stopped)
|
|
|
|
# Re-apply netem after a brief delay for the container to initialize
|
|
if self.netem_mgr:
|
|
if not self.veth_mgr:
|
|
time.sleep(1)
|
|
self.netem_mgr.setup_node(node_id)
|
|
|
|
# Notify callback (e.g., refresh npub for ephemeral identity nodes)
|
|
if self.on_node_restart:
|
|
self.on_node_restart(node_id)
|
|
|
|
def _would_disconnect(self, node_id: str) -> bool:
|
|
"""Check if removing this node (plus currently-down nodes) disconnects the graph.
|
|
|
|
Builds the subgraph of currently-up nodes (excluding the candidate)
|
|
and checks connectivity via BFS.
|
|
"""
|
|
# Active nodes: up and not the candidate
|
|
active_nodes = set()
|
|
for nid, state in self.node_states.items():
|
|
if not state.is_down and nid != node_id:
|
|
active_nodes.add(nid)
|
|
|
|
if len(active_nodes) <= 1:
|
|
return True # Can't remove from a 1-node graph
|
|
|
|
# Build adjacency for active nodes only
|
|
adj: dict[str, list[str]] = {nid: [] for nid in active_nodes}
|
|
for a, b in self.topology.edges:
|
|
if a in active_nodes and b in active_nodes:
|
|
adj[a].append(b)
|
|
adj[b].append(a)
|
|
|
|
# BFS
|
|
start = next(iter(active_nodes))
|
|
visited = set()
|
|
queue = deque([start])
|
|
while queue:
|
|
node = queue.popleft()
|
|
if node in visited:
|
|
continue
|
|
visited.add(node)
|
|
for neighbor in adj[node]:
|
|
if neighbor not in visited:
|
|
queue.append(neighbor)
|
|
|
|
return len(visited) < len(active_nodes)
|