Wait for restored chaos nodes to answer before the final snapshot

churn-mixed-10 failed its baseline with "9 node(s) answered (need 10)"
when the host was busy. The node missing from the final snapshot had just
been restarted by the teardown restore, and the snapshot queried it before
its control socket was open. In every churn-mixed-10 run the scenario's
seed restarts n01 at that point; its control socket opened 2.7-5.5 s after
the restart across 21 runs at three load levels, while the snapshot
followed the restore by 2-6 s. So the suite passed or failed on which came
first, and load only made the restart slower. Reproduced once in 17 runs
with 24 CPU hogs plus I/O on a 12-CPU host.

Teardown now waits, after restoring nodes, until every node's control
socket answers, bounded at 60 s (11 times the slowest start measured), and
then takes the snapshot. The floor of 10 answering nodes is unchanged. A
node that never answers is still reported absent: with one node stopped for
good just before the wait, the wait gave up after 61.7 s and the baseline
failed with 9 answered. Five more runs of the chaos set under the same
load were green, the wait taking 1.6-3.6 s.
This commit is contained in:
Johnathan Corgan
2026-09-19 17:21:45 +00:00
parent 0ed841b4a5
commit 2308e01ad4
+41 -1
View File
@@ -24,7 +24,12 @@ from .assertions import (
)
from .compose import generate_compose
from .config_gen import write_configs
from .control import snapshot_all_congestion, snapshot_all_mmp, snapshot_all_trees
from .control import (
query_status,
snapshot_all_congestion,
snapshot_all_mmp,
snapshot_all_trees,
)
from .docker_exec import docker_compose, existing_containers, force_remove
from .link_swap import LinkSwapManager
from .links import LinkManager
@@ -41,6 +46,10 @@ from .veth import VethManager
log = logging.getLogger(__name__)
# Upper bound on the wait for restored nodes' control sockets before the
# final snapshot. See _wait_control_sockets.
CONTROL_WAIT_SECS = 60
class SimRunner:
def __init__(self, scenario: Scenario):
@@ -698,6 +707,11 @@ class SimRunner:
log.info("Restoring stopped nodes...")
self.node_mgr.restore_all()
# A node restarted just above can take several seconds to open
# its control socket, and a snapshot taken before then records
# it as absent. Wait for every node to answer first.
self._wait_control_sockets(CONTROL_WAIT_SECS)
# Collect iperf3 throughput results before containers stop
if self.traffic_mgr:
iperf_results = self.traffic_mgr.collect_results()
@@ -838,6 +852,32 @@ class SimRunner:
return result
def _wait_control_sockets(self, timeout: float) -> None:
"""Wait until every node's control socket answers, up to timeout.
Bounded, and it never fails the run itself: a node that still does
not answer is left to the final snapshot, where the assertions see
it as absent.
"""
start = time.monotonic()
waiting = sorted(self.topology.nodes)
while True:
waiting = [
nid for nid in waiting
if query_status(self.topology.container_name(nid)) is None
]
elapsed = time.monotonic() - start
if not waiting or elapsed >= timeout:
break
time.sleep(2)
if waiting:
log.warning(
"Control socket wait: %s still not answering after %.1fs",
", ".join(waiting), elapsed,
)
else:
log.info("Control socket wait: all nodes answered after %.1fs", elapsed)
def _take_snapshot(self, label: str):
"""Query all nodes via control socket and save tree/MMP/congestion snapshots."""
if not self.topology: