mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-05 19:18:25 +00:00
Wait for restored chaos nodes to answer before the final snapshot
churn-mixed-10 failed its baseline with "9 node(s) answered (need 10)" when the host was busy. The node missing from the final snapshot had just been restarted by the teardown restore, and the snapshot queried it before its control socket was open. In every churn-mixed-10 run the scenario's seed restarts n01 at that point; its control socket opened 2.7-5.5 s after the restart across 21 runs at three load levels, while the snapshot followed the restore by 2-6 s. So the suite passed or failed on which came first, and load only made the restart slower. Reproduced once in 17 runs with 24 CPU hogs plus I/O on a 12-CPU host. Teardown now waits, after restoring nodes, until every node's control socket answers, bounded at 60 s (11 times the slowest start measured), and then takes the snapshot. The floor of 10 answering nodes is unchanged. A node that never answers is still reported absent: with one node stopped for good just before the wait, the wait gave up after 61.7 s and the baseline failed with 9 answered. Five more runs of the chaos set under the same load were green, the wait taking 1.6-3.6 s.
This commit is contained in:
@@ -24,7 +24,12 @@ from .assertions import (
|
|||||||
)
|
)
|
||||||
from .compose import generate_compose
|
from .compose import generate_compose
|
||||||
from .config_gen import write_configs
|
from .config_gen import write_configs
|
||||||
from .control import snapshot_all_congestion, snapshot_all_mmp, snapshot_all_trees
|
from .control import (
|
||||||
|
query_status,
|
||||||
|
snapshot_all_congestion,
|
||||||
|
snapshot_all_mmp,
|
||||||
|
snapshot_all_trees,
|
||||||
|
)
|
||||||
from .docker_exec import docker_compose, existing_containers, force_remove
|
from .docker_exec import docker_compose, existing_containers, force_remove
|
||||||
from .link_swap import LinkSwapManager
|
from .link_swap import LinkSwapManager
|
||||||
from .links import LinkManager
|
from .links import LinkManager
|
||||||
@@ -41,6 +46,10 @@ from .veth import VethManager
|
|||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Upper bound on the wait for restored nodes' control sockets before the
|
||||||
|
# final snapshot. See _wait_control_sockets.
|
||||||
|
CONTROL_WAIT_SECS = 60
|
||||||
|
|
||||||
|
|
||||||
class SimRunner:
|
class SimRunner:
|
||||||
def __init__(self, scenario: Scenario):
|
def __init__(self, scenario: Scenario):
|
||||||
@@ -698,6 +707,11 @@ class SimRunner:
|
|||||||
log.info("Restoring stopped nodes...")
|
log.info("Restoring stopped nodes...")
|
||||||
self.node_mgr.restore_all()
|
self.node_mgr.restore_all()
|
||||||
|
|
||||||
|
# A node restarted just above can take several seconds to open
|
||||||
|
# its control socket, and a snapshot taken before then records
|
||||||
|
# it as absent. Wait for every node to answer first.
|
||||||
|
self._wait_control_sockets(CONTROL_WAIT_SECS)
|
||||||
|
|
||||||
# Collect iperf3 throughput results before containers stop
|
# Collect iperf3 throughput results before containers stop
|
||||||
if self.traffic_mgr:
|
if self.traffic_mgr:
|
||||||
iperf_results = self.traffic_mgr.collect_results()
|
iperf_results = self.traffic_mgr.collect_results()
|
||||||
@@ -838,6 +852,32 @@ class SimRunner:
|
|||||||
|
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
def _wait_control_sockets(self, timeout: float) -> None:
|
||||||
|
"""Wait until every node's control socket answers, up to timeout.
|
||||||
|
|
||||||
|
Bounded, and it never fails the run itself: a node that still does
|
||||||
|
not answer is left to the final snapshot, where the assertions see
|
||||||
|
it as absent.
|
||||||
|
"""
|
||||||
|
start = time.monotonic()
|
||||||
|
waiting = sorted(self.topology.nodes)
|
||||||
|
while True:
|
||||||
|
waiting = [
|
||||||
|
nid for nid in waiting
|
||||||
|
if query_status(self.topology.container_name(nid)) is None
|
||||||
|
]
|
||||||
|
elapsed = time.monotonic() - start
|
||||||
|
if not waiting or elapsed >= timeout:
|
||||||
|
break
|
||||||
|
time.sleep(2)
|
||||||
|
if waiting:
|
||||||
|
log.warning(
|
||||||
|
"Control socket wait: %s still not answering after %.1fs",
|
||||||
|
", ".join(waiting), elapsed,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
log.info("Control socket wait: all nodes answered after %.1fs", elapsed)
|
||||||
|
|
||||||
def _take_snapshot(self, label: str):
|
def _take_snapshot(self, label: str):
|
||||||
"""Query all nodes via control socket and save tree/MMP/congestion snapshots."""
|
"""Query all nodes via control socket and save tree/MMP/congestion snapshots."""
|
||||||
if not self.topology:
|
if not self.topology:
|
||||||
|
|||||||
Reference in New Issue
Block a user