Show the host load behind each failed suite in the local CI summary

Several local CI reds have come from suites whose timing floors are
sensitive to host contention, typically when two runs overlapped. The
summary gave no way to tell such a red from any other, so each one had to
be investigated from scratch.

ci-local.sh now samples CPU pressure (/proc/pressure/cpu, "some" stall
time) and the number of other CI runs' containers every 5 s for the whole
run, and writes a begin/end window around each suite: one end line per
sequential suite from record(), a begin/end pair per chaos scenario from
run_chaos (which runs once per scenario on the parallel and the --only
path), and a barrier after the chaos block. testing/lib/load-annotate.py
reads that log and prints, under each failed suite, the stall over its
window, the peak number of foreign runs, and LOAD-DEGRADED at or above
15%, not load-degraded below it, or load unknown when the window or the
samples are missing. A run-level line gives the peak stall and peak
foreign runs. The annotation changes no verdict and no exit code.

15% comes from measurement on the 12-CPU host: green chaos sets with no
foreign load ran at 4-11% stall, loaded sets at 39-70%, and the one
reproduced load-sensitive red at 63%. Setting FIPS_CI_LOAD_LOG keeps the
sample log after the run; otherwise teardown removes it.
This commit is contained in:
Johnathan Corgan
2026-09-19 17:21:45 +00:00
parent 2308e01ad4
commit fa5c7aaadf
2 changed files with 242 additions and 0 deletions
+61
View File
@@ -101,6 +101,14 @@
# A preempting CI worker maps 130/143 → "cancelled" (discard, do not record a
# failing commit), 0 → green, any other non-zero → red.
#
# Host load annotation: every failed suite in the summary is followed by the
# CPU pressure stall (/proc/pressure/cpu) over that suite's run and the peak
# number of other CI runs' containers, labelled LOAD-DEGRADED at or above
# CI_LOAD_THRESHOLD percent (testing/lib/load-annotate.py). It is context for
# reading a red and changes no verdict or exit code. The samples go to
# $FIPS_CI_LOAD_LOG if set (kept after the run), else to a per-run file that
# teardown removes.
#
# ── CI parity invariant ─────────────────────────────────────────────────────
# This local default suite set and the GitHub integration matrix
# (.github/workflows/ci.yml) MUST run the same integration suites, EXCEPT for
@@ -307,6 +315,9 @@ OVERALL=0
record() {
local name="$1" rc="$2"
RESULTS["$name"]=$rc
# A chaos suite's window is written by run_chaos, which runs once per
# scenario; record() runs twice for a parallel one (child and parent).
[[ "$name" == chaos-* ]] || load_mark end "$name"
if [[ $rc -ne 0 ]]; then
OVERALL=1
fail "$name"
@@ -385,6 +396,38 @@ readonly CI_RUN_NAME_SUFFIX
# while adding the cross-run prefix that scopes the reap.
ci_project() { printf '%s_%s' "$CI_PROJECT_PREFIX" "$1"; }
# ── Host load samples ─────────────────────────────────────────────────────
#
# Stall percent at or above which a failed suite is labelled LOAD-DEGRADED.
# Measured 2026-09-19 on core-vm (12 CPUs): green chaos sets with no foreign
# load ran at 4-11% CPU stall, and the one reproduced load-sensitive red
# (churn-mixed-10 answering 9 of 10) ran at 63%, under 24 CPU hogs plus I/O.
CI_LOAD_THRESHOLD=15
if [[ -n "${FIPS_CI_LOAD_LOG:-}" ]]; then
CI_LOAD_LOG="$FIPS_CI_LOAD_LOG"
CI_LOAD_LOG_OWNED=0
else
CI_LOAD_LOG="${TMPDIR:-/tmp}/fips-ci-load-${CI_RUN_ID}.log"
CI_LOAD_LOG_OWNED=1
fi
CI_LOAD_PID=""
# Append one window line (begin/end <suite>, or barrier) to the load log.
load_mark() { echo "$(date +%s.%N) $*" >> "$CI_LOAD_LOG" 2>/dev/null || true; }
# Sample CPU pressure and foreign CI runs every 5 s until killed.
load_sampler() {
local psi runs
load_mark barrier
while true; do
psi="$(awk '/^some/ { for (i = 1; i <= NF; i++) if ($i ~ /^total=/) { sub("total=", "", $i); print $i } }' /proc/pressure/cpu 2>/dev/null)"
runs="$(docker ps --filter "label=$CI_LABEL" --format '{{.Label "com.corganlabs.fips-ci.run"}}' 2>/dev/null \
| sort -u | grep -cvx -e "$CI_RUN_ID" -e '')"
echo "$(date +%s.%N) psi ${psi:-none} runs ${runs:-0} load1 $(cut -d' ' -f1 /proc/loadavg)" >> "$CI_LOAD_LOG"
sleep 5
done
}
# The name suffix one chaos scenario runs under. Its container names, its
# generated-config directory and the token in its host veth names all derive
# from this, so teardown recomputes it to know which interfaces are ours. Reads
@@ -404,6 +447,13 @@ ci_teardown() {
CI_CLEANED=1
local run_status="${1:-1}"
# 0. The load sampler first, so it stops writing before its log goes.
if [[ -n "$CI_LOAD_PID" ]]; then
kill "$CI_LOAD_PID" 2>/dev/null || true
wait "$CI_LOAD_PID" 2>/dev/null || true
fi
[[ "$CI_LOAD_LOG_OWNED" -eq 1 ]] && rm -f "$CI_LOAD_LOG"
# 1. Propagate to parallel chaos children and reap them (bounded).
if [[ ${#CI_CHAOS_PIDS[@]} -gt 0 ]]; then
kill -TERM "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true
@@ -693,6 +743,7 @@ run_chaos() {
suffix="$(ci_chaos_suffix "$name")"
local -x FIPS_CI_NAME_SUFFIX="$suffix"
load_mark begin "chaos-$name"
info "[chaos/$name] Running simulation"
if bash testing/chaos/scripts/chaos.sh "$@" 2>&1; then
rc=0
@@ -700,6 +751,7 @@ run_chaos() {
rc=1
fi
load_mark end "chaos-$name"
record "chaos-$name" $rc
# record() ends in pass()/fail(), which are echoes, so it returns 0 for any
@@ -1284,6 +1336,9 @@ run_integration() {
# All chaos children have been waited on; clear so a later signal does
# not try to kill already-reaped PIDs.
CI_CHAOS_PIDS=()
# The next sequential suite's window starts here, not at the last
# suite before the chaos block.
load_mark barrier
fi
# Sidecar
@@ -1377,10 +1432,13 @@ print_summary() {
else
failed=$((failed + 1))
echo -e " ${RED}✗${RESET} $name"
python3 "$SCRIPT_DIR/lib/load-annotate.py" "$CI_LOAD_LOG" \
--threshold "$CI_LOAD_THRESHOLD" "$name" 2>&1 | sed 's/^/ /'
fi
done
echo ""
python3 "$SCRIPT_DIR/lib/load-annotate.py" "$CI_LOAD_LOG" --run 2>&1 | sed 's/^/ /'
echo -e " ${BOLD}Total: $total Passed: $passed Failed: $failed${RESET}"
echo ""
@@ -1495,6 +1553,9 @@ main() {
stage "FIPS Local CI"
info "Project root: $PROJECT_ROOT"
load_sampler &
CI_LOAD_PID=$!
# Above the mode branches deliberately, so --only, --test-only and
# --build-only are gated too: a divergence invalidates any claim that a
# local run means what a GitHub run means, whichever subset was asked for.
+181
View File
@@ -0,0 +1,181 @@
#!/usr/bin/env python3
"""Annotate failed local-CI suites with the host load during their run.
ci-local.sh samples CPU pressure (/proc/pressure/cpu, "some" total stall
microseconds) and the count of other CI runs' containers every few
seconds into a load log, and writes window lines around each suite. This
reads that log and prints, for each failed suite named on the command
line, the stall percentage over the suite's window, the peak number of
foreign runs, and a label:
LOAD-DEGRADED stall at or above the threshold
not load-degraded stall below it
load unknown (...) the window or the samples needed to say are missing
The label is context for a human reading a red. It changes no verdict:
a red stays a red. Unknown is never reported as "not load-degraded".
Log lines (whitespace-separated, first field a Unix epoch):
<t> psi <some_total_us|none> runs <n> load1 <x>
<t> barrier
<t> begin <suite> (chaos suites only)
<t> end <suite>
A chaos suite's window runs from its begin to its end. Any other suite's
window runs from the latest preceding "end" of a non-chaos suite or
"barrier" line up to its own end.
Usage:
load-annotate.py <load-log> --threshold <pct> <failed-suite>...
load-annotate.py <load-log> --run
"""
from __future__ import annotations
import argparse
import sys
from dataclasses import dataclass
@dataclass
class Sample:
"""One sampler line."""
t: float
psi: int | None
runs: int | None
def parse(path: str):
"""Return (samples, events) from a load log; events keep file order."""
samples: list[Sample] = []
events: list[tuple[float, str, str | None]] = []
with open(path) as f:
for line in f:
parts = line.split()
if len(parts) < 2:
continue
try:
t = float(parts[0])
except ValueError:
continue
kind = parts[1]
if kind == "psi":
fields = dict(zip(parts[1::2], parts[2::2]))
psi = fields.get("psi")
runs = fields.get("runs")
samples.append(Sample(
t,
int(psi) if psi and psi.isdigit() else None,
int(runs) if runs and runs.isdigit() else None,
))
elif kind == "barrier":
events.append((t, "barrier", None))
elif kind in ("begin", "end") and len(parts) >= 3:
events.append((t, kind, parts[2]))
return samples, events
def window(name: str, events) -> tuple[float, float] | str:
"""Return (start, end) for a suite, or the reason there is none."""
if name.startswith("chaos-"):
begins = [e for e in events if e[1] == "begin" and e[2] == name]
ends = [e for e in events if e[1] == "end" and e[2] == name]
if not begins and not ends:
return "no window"
if len(begins) != 1 or len(ends) != 1 or ends[0][0] < begins[0][0]:
return "ambiguous window"
return begins[0][0], ends[0][0]
idx = [i for i, e in enumerate(events) if e[1] == "end" and e[2] == name]
if not idx:
return "no window"
if len(idx) > 1:
return "ambiguous window"
end_i = idx[0]
start = None
for e in reversed(events[:end_i]):
if e[1] == "barrier" or (e[1] == "end" and not e[2].startswith("chaos-")):
start = e[0]
break
if start is None:
return "no window"
return start, events[end_i][0]
def stall(samples: list[Sample], start: float, end: float) -> float | str:
"""Stall percent over [start, end], or the reason it cannot be given."""
psi = [s for s in samples if s.psi is not None]
if not psi:
return "no PSI samples"
before = [s for s in psi if s.t <= start]
after = [s for s in psi if s.t >= end]
s0 = before[-1] if before else next((s for s in psi if s.t >= start), None)
s1 = after[0] if after else next((s for s in reversed(psi) if s.t <= end), None)
if s0 is None or s1 is None or s1.t - s0.t <= 0:
return "too few PSI samples in window"
return (s1.psi - s0.psi) / ((s1.t - s0.t) * 1e6) * 100
def peak_runs(samples: list[Sample], start: float, end: float) -> int | None:
"""Largest foreign-run count sampled inside the window."""
runs = [s.runs for s in samples if start <= s.t <= end and s.runs is not None]
return max(runs) if runs else None
def annotate(name: str, samples, events, threshold: float) -> str:
"""One annotation line for a failed suite."""
w = window(name, events)
if isinstance(w, str):
return f"{name}: load unknown ({w})"
start, end = w
pct = stall(samples, start, end)
if isinstance(pct, str):
return f"{name}: load unknown ({pct})"
runs = peak_runs(samples, start, end)
label = "LOAD-DEGRADED" if pct >= threshold else "not load-degraded"
runs_txt = "?" if runs is None else str(runs)
return (f"{name}: {label} (cpu stall {pct:.1f}% over {end - start:.0f}s, "
f"threshold {threshold:g}%; peak foreign CI runs {runs_txt})")
def run_line(samples) -> str:
"""Run-level summary: peak interval stall and peak foreign runs."""
psi = [s for s in samples if s.psi is not None]
peaks = [
(b.psi - a.psi) / ((b.t - a.t) * 1e6) * 100
for a, b in zip(psi, psi[1:]) if b.t > a.t
]
runs = [s.runs for s in samples if s.runs is not None]
stall_txt = f"{max(peaks):.1f}%" if peaks else "unknown (no PSI samples)"
runs_txt = str(max(runs)) if runs else "unknown"
return f"host load: peak cpu stall {stall_txt}; peak foreign CI runs {runs_txt}"
def main(argv: list[str]) -> int:
"""Entry point."""
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
ap.add_argument("log")
ap.add_argument("--threshold", type=float)
ap.add_argument("--run", action="store_true")
ap.add_argument("suites", nargs="*")
a = ap.parse_intermixed_args(argv)
try:
samples, events = parse(a.log)
except OSError as exc:
for name in a.suites:
print(f"{name}: load unknown (load log unreadable: {exc.strerror})")
if a.run:
print("host load: unknown (load log unreadable)")
return 0
if a.run:
print(run_line(samples))
if a.suites and a.threshold is None:
ap.error("--threshold is required with suite names")
for name in a.suites:
print(annotate(name, samples, events, a.threshold))
return 0
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))