mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-05 11:08:25 +00:00
Show the host load behind each failed suite in the local CI summary
Several local CI reds have come from suites whose timing floors are sensitive to host contention, typically when two runs overlapped. The summary gave no way to tell such a red from any other, so each one had to be investigated from scratch. ci-local.sh now samples CPU pressure (/proc/pressure/cpu, "some" stall time) and the number of other CI runs' containers every 5 s for the whole run, and writes a begin/end window around each suite: one end line per sequential suite from record(), a begin/end pair per chaos scenario from run_chaos (which runs once per scenario on the parallel and the --only path), and a barrier after the chaos block. testing/lib/load-annotate.py reads that log and prints, under each failed suite, the stall over its window, the peak number of foreign runs, and LOAD-DEGRADED at or above 15%, not load-degraded below it, or load unknown when the window or the samples are missing. A run-level line gives the peak stall and peak foreign runs. The annotation changes no verdict and no exit code. 15% comes from measurement on the 12-CPU host: green chaos sets with no foreign load ran at 4-11% stall, loaded sets at 39-70%, and the one reproduced load-sensitive red at 63%. Setting FIPS_CI_LOAD_LOG keeps the sample log after the run; otherwise teardown removes it.
This commit is contained in:
@@ -101,6 +101,14 @@
|
|||||||
# A preempting CI worker maps 130/143 → "cancelled" (discard, do not record a
|
# A preempting CI worker maps 130/143 → "cancelled" (discard, do not record a
|
||||||
# failing commit), 0 → green, any other non-zero → red.
|
# failing commit), 0 → green, any other non-zero → red.
|
||||||
#
|
#
|
||||||
|
# Host load annotation: every failed suite in the summary is followed by the
|
||||||
|
# CPU pressure stall (/proc/pressure/cpu) over that suite's run and the peak
|
||||||
|
# number of other CI runs' containers, labelled LOAD-DEGRADED at or above
|
||||||
|
# CI_LOAD_THRESHOLD percent (testing/lib/load-annotate.py). It is context for
|
||||||
|
# reading a red and changes no verdict or exit code. The samples go to
|
||||||
|
# $FIPS_CI_LOAD_LOG if set (kept after the run), else to a per-run file that
|
||||||
|
# teardown removes.
|
||||||
|
#
|
||||||
# ── CI parity invariant ─────────────────────────────────────────────────────
|
# ── CI parity invariant ─────────────────────────────────────────────────────
|
||||||
# This local default suite set and the GitHub integration matrix
|
# This local default suite set and the GitHub integration matrix
|
||||||
# (.github/workflows/ci.yml) MUST run the same integration suites, EXCEPT for
|
# (.github/workflows/ci.yml) MUST run the same integration suites, EXCEPT for
|
||||||
@@ -307,6 +315,9 @@ OVERALL=0
|
|||||||
record() {
|
record() {
|
||||||
local name="$1" rc="$2"
|
local name="$1" rc="$2"
|
||||||
RESULTS["$name"]=$rc
|
RESULTS["$name"]=$rc
|
||||||
|
# A chaos suite's window is written by run_chaos, which runs once per
|
||||||
|
# scenario; record() runs twice for a parallel one (child and parent).
|
||||||
|
[[ "$name" == chaos-* ]] || load_mark end "$name"
|
||||||
if [[ $rc -ne 0 ]]; then
|
if [[ $rc -ne 0 ]]; then
|
||||||
OVERALL=1
|
OVERALL=1
|
||||||
fail "$name"
|
fail "$name"
|
||||||
@@ -385,6 +396,38 @@ readonly CI_RUN_NAME_SUFFIX
|
|||||||
# while adding the cross-run prefix that scopes the reap.
|
# while adding the cross-run prefix that scopes the reap.
|
||||||
ci_project() { printf '%s_%s' "$CI_PROJECT_PREFIX" "$1"; }
|
ci_project() { printf '%s_%s' "$CI_PROJECT_PREFIX" "$1"; }
|
||||||
|
|
||||||
|
# ── Host load samples ─────────────────────────────────────────────────────
|
||||||
|
#
|
||||||
|
# Stall percent at or above which a failed suite is labelled LOAD-DEGRADED.
|
||||||
|
# Measured 2026-09-19 on core-vm (12 CPUs): green chaos sets with no foreign
|
||||||
|
# load ran at 4-11% CPU stall, and the one reproduced load-sensitive red
|
||||||
|
# (churn-mixed-10 answering 9 of 10) ran at 63%, under 24 CPU hogs plus I/O.
|
||||||
|
CI_LOAD_THRESHOLD=15
|
||||||
|
if [[ -n "${FIPS_CI_LOAD_LOG:-}" ]]; then
|
||||||
|
CI_LOAD_LOG="$FIPS_CI_LOAD_LOG"
|
||||||
|
CI_LOAD_LOG_OWNED=0
|
||||||
|
else
|
||||||
|
CI_LOAD_LOG="${TMPDIR:-/tmp}/fips-ci-load-${CI_RUN_ID}.log"
|
||||||
|
CI_LOAD_LOG_OWNED=1
|
||||||
|
fi
|
||||||
|
CI_LOAD_PID=""
|
||||||
|
|
||||||
|
# Append one window line (begin/end <suite>, or barrier) to the load log.
|
||||||
|
load_mark() { echo "$(date +%s.%N) $*" >> "$CI_LOAD_LOG" 2>/dev/null || true; }
|
||||||
|
|
||||||
|
# Sample CPU pressure and foreign CI runs every 5 s until killed.
|
||||||
|
load_sampler() {
|
||||||
|
local psi runs
|
||||||
|
load_mark barrier
|
||||||
|
while true; do
|
||||||
|
psi="$(awk '/^some/ { for (i = 1; i <= NF; i++) if ($i ~ /^total=/) { sub("total=", "", $i); print $i } }' /proc/pressure/cpu 2>/dev/null)"
|
||||||
|
runs="$(docker ps --filter "label=$CI_LABEL" --format '{{.Label "com.corganlabs.fips-ci.run"}}' 2>/dev/null \
|
||||||
|
| sort -u | grep -cvx -e "$CI_RUN_ID" -e '')"
|
||||||
|
echo "$(date +%s.%N) psi ${psi:-none} runs ${runs:-0} load1 $(cut -d' ' -f1 /proc/loadavg)" >> "$CI_LOAD_LOG"
|
||||||
|
sleep 5
|
||||||
|
done
|
||||||
|
}
|
||||||
|
|
||||||
# The name suffix one chaos scenario runs under. Its container names, its
|
# The name suffix one chaos scenario runs under. Its container names, its
|
||||||
# generated-config directory and the token in its host veth names all derive
|
# generated-config directory and the token in its host veth names all derive
|
||||||
# from this, so teardown recomputes it to know which interfaces are ours. Reads
|
# from this, so teardown recomputes it to know which interfaces are ours. Reads
|
||||||
@@ -404,6 +447,13 @@ ci_teardown() {
|
|||||||
CI_CLEANED=1
|
CI_CLEANED=1
|
||||||
local run_status="${1:-1}"
|
local run_status="${1:-1}"
|
||||||
|
|
||||||
|
# 0. The load sampler first, so it stops writing before its log goes.
|
||||||
|
if [[ -n "$CI_LOAD_PID" ]]; then
|
||||||
|
kill "$CI_LOAD_PID" 2>/dev/null || true
|
||||||
|
wait "$CI_LOAD_PID" 2>/dev/null || true
|
||||||
|
fi
|
||||||
|
[[ "$CI_LOAD_LOG_OWNED" -eq 1 ]] && rm -f "$CI_LOAD_LOG"
|
||||||
|
|
||||||
# 1. Propagate to parallel chaos children and reap them (bounded).
|
# 1. Propagate to parallel chaos children and reap them (bounded).
|
||||||
if [[ ${#CI_CHAOS_PIDS[@]} -gt 0 ]]; then
|
if [[ ${#CI_CHAOS_PIDS[@]} -gt 0 ]]; then
|
||||||
kill -TERM "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true
|
kill -TERM "${CI_CHAOS_PIDS[@]}" 2>/dev/null || true
|
||||||
@@ -693,6 +743,7 @@ run_chaos() {
|
|||||||
suffix="$(ci_chaos_suffix "$name")"
|
suffix="$(ci_chaos_suffix "$name")"
|
||||||
local -x FIPS_CI_NAME_SUFFIX="$suffix"
|
local -x FIPS_CI_NAME_SUFFIX="$suffix"
|
||||||
|
|
||||||
|
load_mark begin "chaos-$name"
|
||||||
info "[chaos/$name] Running simulation"
|
info "[chaos/$name] Running simulation"
|
||||||
if bash testing/chaos/scripts/chaos.sh "$@" 2>&1; then
|
if bash testing/chaos/scripts/chaos.sh "$@" 2>&1; then
|
||||||
rc=0
|
rc=0
|
||||||
@@ -700,6 +751,7 @@ run_chaos() {
|
|||||||
rc=1
|
rc=1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
load_mark end "chaos-$name"
|
||||||
record "chaos-$name" $rc
|
record "chaos-$name" $rc
|
||||||
|
|
||||||
# record() ends in pass()/fail(), which are echoes, so it returns 0 for any
|
# record() ends in pass()/fail(), which are echoes, so it returns 0 for any
|
||||||
@@ -1284,6 +1336,9 @@ run_integration() {
|
|||||||
# All chaos children have been waited on; clear so a later signal does
|
# All chaos children have been waited on; clear so a later signal does
|
||||||
# not try to kill already-reaped PIDs.
|
# not try to kill already-reaped PIDs.
|
||||||
CI_CHAOS_PIDS=()
|
CI_CHAOS_PIDS=()
|
||||||
|
# The next sequential suite's window starts here, not at the last
|
||||||
|
# suite before the chaos block.
|
||||||
|
load_mark barrier
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Sidecar
|
# Sidecar
|
||||||
@@ -1377,10 +1432,13 @@ print_summary() {
|
|||||||
else
|
else
|
||||||
failed=$((failed + 1))
|
failed=$((failed + 1))
|
||||||
echo -e " ${RED}✗${RESET} $name"
|
echo -e " ${RED}✗${RESET} $name"
|
||||||
|
python3 "$SCRIPT_DIR/lib/load-annotate.py" "$CI_LOAD_LOG" \
|
||||||
|
--threshold "$CI_LOAD_THRESHOLD" "$name" 2>&1 | sed 's/^/ /'
|
||||||
fi
|
fi
|
||||||
done
|
done
|
||||||
|
|
||||||
echo ""
|
echo ""
|
||||||
|
python3 "$SCRIPT_DIR/lib/load-annotate.py" "$CI_LOAD_LOG" --run 2>&1 | sed 's/^/ /'
|
||||||
echo -e " ${BOLD}Total: $total Passed: $passed Failed: $failed${RESET}"
|
echo -e " ${BOLD}Total: $total Passed: $passed Failed: $failed${RESET}"
|
||||||
echo ""
|
echo ""
|
||||||
|
|
||||||
@@ -1495,6 +1553,9 @@ main() {
|
|||||||
stage "FIPS Local CI"
|
stage "FIPS Local CI"
|
||||||
info "Project root: $PROJECT_ROOT"
|
info "Project root: $PROJECT_ROOT"
|
||||||
|
|
||||||
|
load_sampler &
|
||||||
|
CI_LOAD_PID=$!
|
||||||
|
|
||||||
# Above the mode branches deliberately, so --only, --test-only and
|
# Above the mode branches deliberately, so --only, --test-only and
|
||||||
# --build-only are gated too: a divergence invalidates any claim that a
|
# --build-only are gated too: a divergence invalidates any claim that a
|
||||||
# local run means what a GitHub run means, whichever subset was asked for.
|
# local run means what a GitHub run means, whichever subset was asked for.
|
||||||
|
|||||||
@@ -0,0 +1,181 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Annotate failed local-CI suites with the host load during their run.
|
||||||
|
|
||||||
|
ci-local.sh samples CPU pressure (/proc/pressure/cpu, "some" total stall
|
||||||
|
microseconds) and the count of other CI runs' containers every few
|
||||||
|
seconds into a load log, and writes window lines around each suite. This
|
||||||
|
reads that log and prints, for each failed suite named on the command
|
||||||
|
line, the stall percentage over the suite's window, the peak number of
|
||||||
|
foreign runs, and a label:
|
||||||
|
|
||||||
|
LOAD-DEGRADED stall at or above the threshold
|
||||||
|
not load-degraded stall below it
|
||||||
|
load unknown (...) the window or the samples needed to say are missing
|
||||||
|
|
||||||
|
The label is context for a human reading a red. It changes no verdict:
|
||||||
|
a red stays a red. Unknown is never reported as "not load-degraded".
|
||||||
|
|
||||||
|
Log lines (whitespace-separated, first field a Unix epoch):
|
||||||
|
|
||||||
|
<t> psi <some_total_us|none> runs <n> load1 <x>
|
||||||
|
<t> barrier
|
||||||
|
<t> begin <suite> (chaos suites only)
|
||||||
|
<t> end <suite>
|
||||||
|
|
||||||
|
A chaos suite's window runs from its begin to its end. Any other suite's
|
||||||
|
window runs from the latest preceding "end" of a non-chaos suite or
|
||||||
|
"barrier" line up to its own end.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
load-annotate.py <load-log> --threshold <pct> <failed-suite>...
|
||||||
|
load-annotate.py <load-log> --run
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import sys
|
||||||
|
from dataclasses import dataclass
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class Sample:
|
||||||
|
"""One sampler line."""
|
||||||
|
|
||||||
|
t: float
|
||||||
|
psi: int | None
|
||||||
|
runs: int | None
|
||||||
|
|
||||||
|
|
||||||
|
def parse(path: str):
|
||||||
|
"""Return (samples, events) from a load log; events keep file order."""
|
||||||
|
samples: list[Sample] = []
|
||||||
|
events: list[tuple[float, str, str | None]] = []
|
||||||
|
with open(path) as f:
|
||||||
|
for line in f:
|
||||||
|
parts = line.split()
|
||||||
|
if len(parts) < 2:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
t = float(parts[0])
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
kind = parts[1]
|
||||||
|
if kind == "psi":
|
||||||
|
fields = dict(zip(parts[1::2], parts[2::2]))
|
||||||
|
psi = fields.get("psi")
|
||||||
|
runs = fields.get("runs")
|
||||||
|
samples.append(Sample(
|
||||||
|
t,
|
||||||
|
int(psi) if psi and psi.isdigit() else None,
|
||||||
|
int(runs) if runs and runs.isdigit() else None,
|
||||||
|
))
|
||||||
|
elif kind == "barrier":
|
||||||
|
events.append((t, "barrier", None))
|
||||||
|
elif kind in ("begin", "end") and len(parts) >= 3:
|
||||||
|
events.append((t, kind, parts[2]))
|
||||||
|
return samples, events
|
||||||
|
|
||||||
|
|
||||||
|
def window(name: str, events) -> tuple[float, float] | str:
|
||||||
|
"""Return (start, end) for a suite, or the reason there is none."""
|
||||||
|
if name.startswith("chaos-"):
|
||||||
|
begins = [e for e in events if e[1] == "begin" and e[2] == name]
|
||||||
|
ends = [e for e in events if e[1] == "end" and e[2] == name]
|
||||||
|
if not begins and not ends:
|
||||||
|
return "no window"
|
||||||
|
if len(begins) != 1 or len(ends) != 1 or ends[0][0] < begins[0][0]:
|
||||||
|
return "ambiguous window"
|
||||||
|
return begins[0][0], ends[0][0]
|
||||||
|
idx = [i for i, e in enumerate(events) if e[1] == "end" and e[2] == name]
|
||||||
|
if not idx:
|
||||||
|
return "no window"
|
||||||
|
if len(idx) > 1:
|
||||||
|
return "ambiguous window"
|
||||||
|
end_i = idx[0]
|
||||||
|
start = None
|
||||||
|
for e in reversed(events[:end_i]):
|
||||||
|
if e[1] == "barrier" or (e[1] == "end" and not e[2].startswith("chaos-")):
|
||||||
|
start = e[0]
|
||||||
|
break
|
||||||
|
if start is None:
|
||||||
|
return "no window"
|
||||||
|
return start, events[end_i][0]
|
||||||
|
|
||||||
|
|
||||||
|
def stall(samples: list[Sample], start: float, end: float) -> float | str:
|
||||||
|
"""Stall percent over [start, end], or the reason it cannot be given."""
|
||||||
|
psi = [s for s in samples if s.psi is not None]
|
||||||
|
if not psi:
|
||||||
|
return "no PSI samples"
|
||||||
|
before = [s for s in psi if s.t <= start]
|
||||||
|
after = [s for s in psi if s.t >= end]
|
||||||
|
s0 = before[-1] if before else next((s for s in psi if s.t >= start), None)
|
||||||
|
s1 = after[0] if after else next((s for s in reversed(psi) if s.t <= end), None)
|
||||||
|
if s0 is None or s1 is None or s1.t - s0.t <= 0:
|
||||||
|
return "too few PSI samples in window"
|
||||||
|
return (s1.psi - s0.psi) / ((s1.t - s0.t) * 1e6) * 100
|
||||||
|
|
||||||
|
|
||||||
|
def peak_runs(samples: list[Sample], start: float, end: float) -> int | None:
|
||||||
|
"""Largest foreign-run count sampled inside the window."""
|
||||||
|
runs = [s.runs for s in samples if start <= s.t <= end and s.runs is not None]
|
||||||
|
return max(runs) if runs else None
|
||||||
|
|
||||||
|
|
||||||
|
def annotate(name: str, samples, events, threshold: float) -> str:
|
||||||
|
"""One annotation line for a failed suite."""
|
||||||
|
w = window(name, events)
|
||||||
|
if isinstance(w, str):
|
||||||
|
return f"{name}: load unknown ({w})"
|
||||||
|
start, end = w
|
||||||
|
pct = stall(samples, start, end)
|
||||||
|
if isinstance(pct, str):
|
||||||
|
return f"{name}: load unknown ({pct})"
|
||||||
|
runs = peak_runs(samples, start, end)
|
||||||
|
label = "LOAD-DEGRADED" if pct >= threshold else "not load-degraded"
|
||||||
|
runs_txt = "?" if runs is None else str(runs)
|
||||||
|
return (f"{name}: {label} (cpu stall {pct:.1f}% over {end - start:.0f}s, "
|
||||||
|
f"threshold {threshold:g}%; peak foreign CI runs {runs_txt})")
|
||||||
|
|
||||||
|
|
||||||
|
def run_line(samples) -> str:
|
||||||
|
"""Run-level summary: peak interval stall and peak foreign runs."""
|
||||||
|
psi = [s for s in samples if s.psi is not None]
|
||||||
|
peaks = [
|
||||||
|
(b.psi - a.psi) / ((b.t - a.t) * 1e6) * 100
|
||||||
|
for a, b in zip(psi, psi[1:]) if b.t > a.t
|
||||||
|
]
|
||||||
|
runs = [s.runs for s in samples if s.runs is not None]
|
||||||
|
stall_txt = f"{max(peaks):.1f}%" if peaks else "unknown (no PSI samples)"
|
||||||
|
runs_txt = str(max(runs)) if runs else "unknown"
|
||||||
|
return f"host load: peak cpu stall {stall_txt}; peak foreign CI runs {runs_txt}"
|
||||||
|
|
||||||
|
|
||||||
|
def main(argv: list[str]) -> int:
|
||||||
|
"""Entry point."""
|
||||||
|
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
||||||
|
ap.add_argument("log")
|
||||||
|
ap.add_argument("--threshold", type=float)
|
||||||
|
ap.add_argument("--run", action="store_true")
|
||||||
|
ap.add_argument("suites", nargs="*")
|
||||||
|
a = ap.parse_intermixed_args(argv)
|
||||||
|
try:
|
||||||
|
samples, events = parse(a.log)
|
||||||
|
except OSError as exc:
|
||||||
|
for name in a.suites:
|
||||||
|
print(f"{name}: load unknown (load log unreadable: {exc.strerror})")
|
||||||
|
if a.run:
|
||||||
|
print("host load: unknown (load log unreadable)")
|
||||||
|
return 0
|
||||||
|
if a.run:
|
||||||
|
print(run_line(samples))
|
||||||
|
if a.suites and a.threshold is None:
|
||||||
|
ap.error("--threshold is required with suite names")
|
||||||
|
for name in a.suites:
|
||||||
|
print(annotate(name, samples, events, a.threshold))
|
||||||
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main(sys.argv[1:]))
|
||||||
Reference in New Issue
Block a user