diff --git a/testing/ci-local.sh b/testing/ci-local.sh index ecab9d5e..cdaa413d 100755 --- a/testing/ci-local.sh +++ b/testing/ci-local.sh @@ -101,6 +101,14 @@ # A preempting CI worker maps 130/143 → "cancelled" (discard, do not record a # failing commit), 0 → green, any other non-zero → red. # +# Host load annotation: every failed suite in the summary is followed by the +# CPU pressure stall (/proc/pressure/cpu) over that suite's run and the peak +# number of other CI runs' containers, labelled LOAD-DEGRADED at or above +# CI_LOAD_THRESHOLD percent (testing/lib/load-annotate.py). It is context for +# reading a red and changes no verdict or exit code. The samples go to +# $FIPS_CI_LOAD_LOG if set (kept after the run), else to a per-run file that +# teardown removes. +# # ── CI parity invariant ───────────────────────────────────────────────────── # This local default suite set and the GitHub integration matrix # (.github/workflows/ci.yml) MUST run the same integration suites, EXCEPT for @@ -307,6 +315,9 @@ OVERALL=0 record() { local name="$1" rc="$2" RESULTS["$name"]=$rc + # A chaos suite's window is written by run_chaos, which runs once per + # scenario; record() runs twice for a parallel one (child and parent). + [[ "$name" == chaos-* ]] || load_mark end "$name" if [[ $rc -ne 0 ]]; then OVERALL=1 fail "$name" @@ -385,6 +396,38 @@ readonly CI_RUN_NAME_SUFFIX # while adding the cross-run prefix that scopes the reap. ci_project() { printf '%s_%s' "$CI_PROJECT_PREFIX" "$1"; } +# ── Host load samples ───────────────────────────────────────────────────── +# +# Stall percent at or above which a failed suite is labelled LOAD-DEGRADED. +# Measured 2026-09-19 on core-vm (12 CPUs): green chaos sets with no foreign +# load ran at 4-11% CPU stall, and the one reproduced load-sensitive red +# (churn-mixed-10 answering 9 of 10) ran at 63%, under 24 CPU hogs plus I/O. +CI_LOAD_THRESHOLD=15 +if [[ -n "${FIPS_CI_LOAD_LOG:-}" ]]; then + CI_LOAD_LOG="$FIPS_CI_LOAD_LOG" + CI_LOAD_LOG_OWNED=0 +else + CI_LOAD_LOG="${TMPDIR:-/tmp}/fips-ci-load-${CI_RUN_ID}.log" + CI_LOAD_LOG_OWNED=1 +fi +CI_LOAD_PID="" + +# Append one window line (begin/end , or barrier) to the load log. +load_mark() { echo "$(date +%s.%N) $*" >> "$CI_LOAD_LOG" 2>/dev/null || true; } + +# Sample CPU pressure and foreign CI runs every 5 s until killed. +load_sampler() { + local psi runs + load_mark barrier + while true; do + psi="$(awk '/^some/ { for (i = 1; i <= NF; i++) if ($i ~ /^total=/) { sub("total=", "", $i); print $i } }' /proc/pressure/cpu 2>/dev/null)" + runs="$(docker ps --filter "label=$CI_LABEL" --format '{{.Label "com.corganlabs.fips-ci.run"}}' 2>/dev/null \ + | sort -u | grep -cvx -e "$CI_RUN_ID" -e '')" + echo "$(date +%s.%N) psi ${psi:-none} runs ${runs:-0} load1 $(cut -d' ' -f1 /proc/loadavg)" >> "$CI_LOAD_LOG" + sleep 5 + done +} + # The name suffix one chaos scenario runs under. Its container names, its # generated-config directory and the token in its host veth names all derive # from this, so teardown recomputes it to know which interfaces are ours. Reads @@ -404,6 +447,13 @@ ci_teardown() { CI_CLEANED=1 local run_status="${1:-1}" + # 0. The load sampler first, so it stops writing before its log goes. + if [[ -n "$CI_LOAD_PID" ]]; then + kill "$CI_LOAD_PID" 2>/dev/null || true + wait "$CI_LOAD_PID" 2>/dev/null || true + fi + [[ "$CI_LOAD_LOG_OWNED" -eq 1 ]] && rm -f "$CI_LOAD_LOG" + # 1. Propagate to parallel chaos children and reap them (bounded). if [[ ${#CI_CHAOS_PIDS[@]} -gt 0 ]]; then kill -TERM "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true @@ -693,6 +743,7 @@ run_chaos() { suffix="$(ci_chaos_suffix "$name")" local -x FIPS_CI_NAME_SUFFIX="$suffix" + load_mark begin "chaos-$name" info "[chaos/$name] Running simulation" if bash testing/chaos/scripts/chaos.sh "$@" 2>&1; then rc=0 @@ -700,6 +751,7 @@ run_chaos() { rc=1 fi + load_mark end "chaos-$name" record "chaos-$name" $rc # record() ends in pass()/fail(), which are echoes, so it returns 0 for any @@ -1284,6 +1336,9 @@ run_integration() { # All chaos children have been waited on; clear so a later signal does # not try to kill already-reaped PIDs. CI_CHAOS_PIDS=() + # The next sequential suite's window starts here, not at the last + # suite before the chaos block. + load_mark barrier fi # Sidecar @@ -1377,10 +1432,13 @@ print_summary() { else failed=$((failed + 1)) echo -e " ${RED}✗${RESET} $name" + python3 "$SCRIPT_DIR/lib/load-annotate.py" "$CI_LOAD_LOG" \ + --threshold "$CI_LOAD_THRESHOLD" "$name" 2>&1 | sed 's/^/ /' fi done echo "" + python3 "$SCRIPT_DIR/lib/load-annotate.py" "$CI_LOAD_LOG" --run 2>&1 | sed 's/^/ /' echo -e " ${BOLD}Total: $total Passed: $passed Failed: $failed${RESET}" echo "" @@ -1495,6 +1553,9 @@ main() { stage "FIPS Local CI" info "Project root: $PROJECT_ROOT" + load_sampler & + CI_LOAD_PID=$! + # Above the mode branches deliberately, so --only, --test-only and # --build-only are gated too: a divergence invalidates any claim that a # local run means what a GitHub run means, whichever subset was asked for. diff --git a/testing/lib/load-annotate.py b/testing/lib/load-annotate.py new file mode 100644 index 00000000..d58e46a6 --- /dev/null +++ b/testing/lib/load-annotate.py @@ -0,0 +1,181 @@ +#!/usr/bin/env python3 +"""Annotate failed local-CI suites with the host load during their run. + +ci-local.sh samples CPU pressure (/proc/pressure/cpu, "some" total stall +microseconds) and the count of other CI runs' containers every few +seconds into a load log, and writes window lines around each suite. This +reads that log and prints, for each failed suite named on the command +line, the stall percentage over the suite's window, the peak number of +foreign runs, and a label: + + LOAD-DEGRADED stall at or above the threshold + not load-degraded stall below it + load unknown (...) the window or the samples needed to say are missing + +The label is context for a human reading a red. It changes no verdict: +a red stays a red. Unknown is never reported as "not load-degraded". + +Log lines (whitespace-separated, first field a Unix epoch): + + psi runs load1 + barrier + begin (chaos suites only) + end + +A chaos suite's window runs from its begin to its end. Any other suite's +window runs from the latest preceding "end" of a non-chaos suite or +"barrier" line up to its own end. + +Usage: + load-annotate.py --threshold ... + load-annotate.py --run +""" + +from __future__ import annotations + +import argparse +import sys +from dataclasses import dataclass + + +@dataclass +class Sample: + """One sampler line.""" + + t: float + psi: int | None + runs: int | None + + +def parse(path: str): + """Return (samples, events) from a load log; events keep file order.""" + samples: list[Sample] = [] + events: list[tuple[float, str, str | None]] = [] + with open(path) as f: + for line in f: + parts = line.split() + if len(parts) < 2: + continue + try: + t = float(parts[0]) + except ValueError: + continue + kind = parts[1] + if kind == "psi": + fields = dict(zip(parts[1::2], parts[2::2])) + psi = fields.get("psi") + runs = fields.get("runs") + samples.append(Sample( + t, + int(psi) if psi and psi.isdigit() else None, + int(runs) if runs and runs.isdigit() else None, + )) + elif kind == "barrier": + events.append((t, "barrier", None)) + elif kind in ("begin", "end") and len(parts) >= 3: + events.append((t, kind, parts[2])) + return samples, events + + +def window(name: str, events) -> tuple[float, float] | str: + """Return (start, end) for a suite, or the reason there is none.""" + if name.startswith("chaos-"): + begins = [e for e in events if e[1] == "begin" and e[2] == name] + ends = [e for e in events if e[1] == "end" and e[2] == name] + if not begins and not ends: + return "no window" + if len(begins) != 1 or len(ends) != 1 or ends[0][0] < begins[0][0]: + return "ambiguous window" + return begins[0][0], ends[0][0] + idx = [i for i, e in enumerate(events) if e[1] == "end" and e[2] == name] + if not idx: + return "no window" + if len(idx) > 1: + return "ambiguous window" + end_i = idx[0] + start = None + for e in reversed(events[:end_i]): + if e[1] == "barrier" or (e[1] == "end" and not e[2].startswith("chaos-")): + start = e[0] + break + if start is None: + return "no window" + return start, events[end_i][0] + + +def stall(samples: list[Sample], start: float, end: float) -> float | str: + """Stall percent over [start, end], or the reason it cannot be given.""" + psi = [s for s in samples if s.psi is not None] + if not psi: + return "no PSI samples" + before = [s for s in psi if s.t <= start] + after = [s for s in psi if s.t >= end] + s0 = before[-1] if before else next((s for s in psi if s.t >= start), None) + s1 = after[0] if after else next((s for s in reversed(psi) if s.t <= end), None) + if s0 is None or s1 is None or s1.t - s0.t <= 0: + return "too few PSI samples in window" + return (s1.psi - s0.psi) / ((s1.t - s0.t) * 1e6) * 100 + + +def peak_runs(samples: list[Sample], start: float, end: float) -> int | None: + """Largest foreign-run count sampled inside the window.""" + runs = [s.runs for s in samples if start <= s.t <= end and s.runs is not None] + return max(runs) if runs else None + + +def annotate(name: str, samples, events, threshold: float) -> str: + """One annotation line for a failed suite.""" + w = window(name, events) + if isinstance(w, str): + return f"{name}: load unknown ({w})" + start, end = w + pct = stall(samples, start, end) + if isinstance(pct, str): + return f"{name}: load unknown ({pct})" + runs = peak_runs(samples, start, end) + label = "LOAD-DEGRADED" if pct >= threshold else "not load-degraded" + runs_txt = "?" if runs is None else str(runs) + return (f"{name}: {label} (cpu stall {pct:.1f}% over {end - start:.0f}s, " + f"threshold {threshold:g}%; peak foreign CI runs {runs_txt})") + + +def run_line(samples) -> str: + """Run-level summary: peak interval stall and peak foreign runs.""" + psi = [s for s in samples if s.psi is not None] + peaks = [ + (b.psi - a.psi) / ((b.t - a.t) * 1e6) * 100 + for a, b in zip(psi, psi[1:]) if b.t > a.t + ] + runs = [s.runs for s in samples if s.runs is not None] + stall_txt = f"{max(peaks):.1f}%" if peaks else "unknown (no PSI samples)" + runs_txt = str(max(runs)) if runs else "unknown" + return f"host load: peak cpu stall {stall_txt}; peak foreign CI runs {runs_txt}" + + +def main(argv: list[str]) -> int: + """Entry point.""" + ap = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + ap.add_argument("log") + ap.add_argument("--threshold", type=float) + ap.add_argument("--run", action="store_true") + ap.add_argument("suites", nargs="*") + a = ap.parse_intermixed_args(argv) + try: + samples, events = parse(a.log) + except OSError as exc: + for name in a.suites: + print(f"{name}: load unknown (load log unreadable: {exc.strerror})") + if a.run: + print("host load: unknown (load log unreadable)") + return 0 + if a.run: + print(run_line(samples)) + if a.suites and a.threshold is None: + ap.error("--threshold is required with suite names") + for name in a.suites: + print(annotate(name, samples, events, a.threshold)) + return 0 + + +if __name__ == "__main__": + sys.exit(main(sys.argv[1:]))