mirror of
https://github.com/jmcorgan/fips.git
synced 2026-08-10 08:37:02 +00:00
Test cost-based parent selection and kernel-drop detection as unit tests
The cost-selection chaos scenarios (cost-reeval, cost-avoidance, cost-stability, depth-vs-cost, mixed-technology, bottleneck-parent) tested TreeState::evaluate_parent's decision logic through a Docker mesh that could not exercise it reliably: the tree roots at whichever node holds the smallest NodeAddr, MMP link costs take several measurement windows to settle, and the parent hold-down plus hysteresis timing all confound the outcome. A deterministic link-cost flap still produced zero periodic parent switches in a full run. Replace those six scenarios with deterministic unit tests in src/tree/tests.rs that drive evaluate_parent directly: cheaper-link selection at equal depth, switch-on-cost-change, hysteresis suppressing a marginal change while allowing a significant one, and the depth-versus-cost effective-depth tradeoff. Each is constructed so that breaking the cost or hysteresis logic makes it fail. The congestion kernel-drop signal (SO_RXQ_OVFL) cannot be provoked deterministically in Docker: a fresh daemon reader keeps up with container-speed traffic, so the socket receive queue never overflows (an unshaped run with a 4 KB buffer and heavy traffic recorded zero drops on every node). Extract the drop-detection edge -- read the cumulative counter, fire an event only on the transition into a new drop burst -- into TransportDropState::observe_drops and unit-test it directly. congestion-stress keeps its ECN and MMP congestion-signal assertions, which do need the real shaped bottleneck queue. Remove the retired scenarios from both CI runners and update the chaos README.
This commit is contained in:
@@ -417,13 +417,10 @@ impl Node {
|
||||
for (&tid, transport) in &self.transports {
|
||||
let congestion = transport.congestion();
|
||||
let state = self.transport_drops.entry(tid).or_default();
|
||||
if let Some(current) = congestion.recv_drops {
|
||||
let new_drops = current > state.prev_drops;
|
||||
if new_drops && !state.dropping {
|
||||
new_drop_events.push(tid);
|
||||
}
|
||||
state.dropping = new_drops;
|
||||
state.prev_drops = current;
|
||||
if let Some(current) = congestion.recv_drops
|
||||
&& state.observe_drops(current)
|
||||
{
|
||||
new_drop_events.push(tid);
|
||||
}
|
||||
}
|
||||
for tid in new_drop_events {
|
||||
|
||||
@@ -266,6 +266,32 @@ struct TransportDropState {
|
||||
dropping: bool,
|
||||
}
|
||||
|
||||
impl TransportDropState {
|
||||
/// Fold a new cumulative `recv_drops` sample into the state and report
|
||||
/// whether it marks the *transition* into a dropping condition.
|
||||
///
|
||||
/// Returns true only on the edge where the cumulative `SO_RXQ_OVFL`
|
||||
/// counter rose since the previous sample **and** the transport was not
|
||||
/// already flagged as dropping. That edge is what `kernel_drop_events`
|
||||
/// counts: a first observation of a new drop burst, not every sample in
|
||||
/// which the counter happens to be non-zero. A sample with no rise
|
||||
/// clears the flag, so a later rise counts as a fresh event.
|
||||
///
|
||||
/// Pure and sans-IO by design: the tick handler reads the kernel
|
||||
/// counter from the socket and does the logging, but the detection
|
||||
/// decision lives here so it can be tested without a socket, a
|
||||
/// transport, or a running node — which is the only way it can be
|
||||
/// tested at all, since the kernel drop itself cannot be provoked
|
||||
/// deterministically.
|
||||
fn observe_drops(&mut self, current: u64) -> bool {
|
||||
let rose = current > self.prev_drops;
|
||||
let new_event = rose && !self.dropping;
|
||||
self.dropping = rose;
|
||||
self.prev_drops = current;
|
||||
new_event
|
||||
}
|
||||
}
|
||||
|
||||
/// State for a link waiting for transport-level connection establishment.
|
||||
///
|
||||
/// For connection-oriented transports (TCP, Tor), the transport connect runs
|
||||
|
||||
@@ -1995,3 +1995,40 @@ async fn handle_msg1_admits_existing_peer_at_cap() {
|
||||
"rate limiter must rebalance after the (bypass-admitted) handler returns"
|
||||
);
|
||||
}
|
||||
|
||||
// ===== Transport kernel-drop detection (sans-IO) =====
|
||||
//
|
||||
// The drop-detection edge-detector, tested directly. It replaces the
|
||||
// congestion-drops docker scenario, which could not provoke SO_RXQ_OVFL
|
||||
// deterministically (a fresh daemon reader keeps up with container-speed
|
||||
// traffic, so the kernel never overflows the socket queue). The kernel
|
||||
// dropping datagrams is not FIPS behaviour to test; the FIPS behaviour is
|
||||
// reading the SO_RXQ_OVFL counter and firing kernel_drop_events on the
|
||||
// transition into a new drop burst, which is exactly this decision.
|
||||
|
||||
#[test]
|
||||
fn test_transport_drop_state_fires_on_edge_and_rearms() {
|
||||
let mut s = TransportDropState::default();
|
||||
// Cumulative counter still 0: no rise, no event.
|
||||
assert!(!s.observe_drops(0));
|
||||
// First rise (0 -> 5): a new drop burst is observed, so it fires.
|
||||
assert!(s.observe_drops(5));
|
||||
// Counter keeps rising (5 -> 9) but we are already dropping: this is
|
||||
// the "first observed" contract, so it must NOT fire again.
|
||||
assert!(!s.observe_drops(9));
|
||||
// A sample with no further rise clears the dropping flag (no event).
|
||||
assert!(!s.observe_drops(9));
|
||||
// A later rise (9 -> 12) is a fresh burst and fires again.
|
||||
assert!(s.observe_drops(12));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_transport_drop_state_steady_counter_fires_once() {
|
||||
let mut s = TransportDropState::default();
|
||||
// A cumulative counter that jumps once and then holds steady must
|
||||
// register exactly one event, not one per sample — otherwise a single
|
||||
// historical drop burst would report congestion forever.
|
||||
assert!(s.observe_drops(7));
|
||||
assert!(!s.observe_drops(7));
|
||||
assert!(!s.observe_drops(7));
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user