fleet alerting: fix netns naming in agent-health.sh + add fleet-alert-check.sh

agent-health.sh used bare node names for 'ip netns exec' but netns are

named warp-<node> since the NetVM layout; the 6189793 CDP-liveness check

always failed ('No such file or directory'), logging false CRITICALs and

kill -9'ing healthy browsers every 5 min. Use warp-$node.

fleet-alert-check.sh: new 5-min critical-condition detector (per-node CDP

liveness via warp-<node> netns). Consecutive-failure state machine:

page after 2 consecutive failures, re-page every 30 min while critical,

quiet-hours-aware (first alert always pages). Emits ALERT/RECOVERY

records to ~/.local/share/fleet-alert/outbox.jsonl for the container

fleet-alert-relay hook; best-effort box-ctl notify to healthy agents.

Session: sidechat/critical-alerting-pipeline
This commit is contained in:
operator-main
2026-10-04 20:09:36 +00:00
parent 8bb64c4826
commit 5ab157b3b0
2 changed files with 213 additions and 2 deletions
+2 -2
View File
@@ -21,7 +21,7 @@ check_cdp_port() {
# A browser can be running but not bound to CDP (zombie state). # A browser can be running but not bound to CDP (zombie state).
local node=$1 local node=$1
local cdp_port=$2 local cdp_port=$2
if sudo ip netns exec "$node" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then if sudo ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then
return 0 return 0
else else
return 1 return 1
@@ -61,7 +61,7 @@ restart_browser() {
done done
sleep 3 sleep 3
# Verify port is free before restart # Verify port is free before restart
if sudo ip netns exec "$agent" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then if sudo ip netns exec "warp-$agent" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then
echo "$(date -Iseconds) $agent: WARNING - port $cdp_port still bound after kill" >> "$LOG" echo "$(date -Iseconds) $agent: WARNING - port $cdp_port still bound after kill" >> "$LOG"
fi fi
# Restart via netvm-chrome.sh in its own systemd scope. # Restart via netvm-chrome.sh in its own systemd scope.
+211
View File
@@ -0,0 +1,211 @@
#!/bin/bash
# fleet-alert-check.sh — bl-side critical-condition detector for the fleet alerting pipeline.
#
# Closes the watchdog gap: agent-health.sh logs CRITICAL and restarts browsers,
# but NOTHING pages anyone. This script detects critical conditions, counts
# CONSECUTIVE failures, and emits alert records to an outbox that the
# container-side fleet-alert-relay hook picks up and pages to #lobby.
#
# Checks (bl-side only; VM-side checks live in the container relay):
# cdp:<node> headless Chromium CDP port not listening in the node's netns
# (catches zombie browsers: process alive, CDP not bound)
#
# Paging policy (env-overridable defaults — adjustable, not gates):
# FLEET_ALERT_THRESHOLD=2 consecutive failures before first page (~10 min at 5-min cadence)
# FLEET_ALERT_REALERT_MIN=30 re-page while still critical, at most every 30 min
# FLEET_ALERT_QUIET_HOURS="" e.g. "23:00-07:00" (bl local time); empty = page 24/7.
# First alert for a NEW incident always pages;
# quiet hours only suppress re-pages.
# FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify
# FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail
# (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip)
#
# State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters)
# Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay)
# Log: /tmp/fleet-alert-check.log
#
# Alert delivery legs:
# 1. outbox record -> container relay -> signed #lobby post (primary page)
# 2. best-effort `box-ctl.py notify` to currently-healthy agents (DM path needs a
# working browser; failures are logged, never fatal)
#
# Installed as user timer fleet-alert-check.timer (every 5 min), mirroring agent-health.timer.
set -uo pipefail
THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}"
REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}"
QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}"
DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}"
INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}"
STATE_DIR="${FLEET_ALERT_STATE_DIR:-$HOME/.local/share/fleet-alert}"
BIN="$(cd "$(dirname "$0")" && pwd)"
STATE="$STATE_DIR/state.json"
OUTBOX="$STATE_DIR/outbox.jsonl"
LOG="/tmp/fleet-alert-check.log"
mkdir -p "$STATE_DIR"
[ -f "$STATE" ] || echo '{}' > "$STATE"
touch "$OUTBOX"
NOW=$(date +%s)
log() { echo "$(date -Iseconds) $*" >> "$LOG"; }
# --- shared consecutive-failure state machine (also used by the container relay) ---
# usage: state_machine <cond> <failing 0|1> -> prints "<ACTION> <fails>"
# ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | NONE
state_machine() {
local cond="$1" failing="$2"
THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \
FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" <<'PYEOF'
import json, os, sys, time
state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3]
failing = failing_s == "1"
threshold = int(os.environ.get("THRESHOLD", "2"))
realert_min = int(os.environ.get("REALERT_MIN", "30"))
qh = os.environ.get("QUIET_HOURS", "")
dry = os.environ.get("FLEET_ALERT_DRY_RUN") == "1"
now = int(time.time())
def in_quiet(spec):
if not spec:
return False
try:
a, b = spec.split("-")
def m(s):
h, mi = s.split(":")
return int(h) * 60 + int(mi)
cur = time.localtime().tm_hour * 60 + time.localtime().tm_min
s, e = m(a), m(b)
return (s <= cur < e) if s <= e else (cur >= s or cur < e)
except Exception:
return False
try:
st = json.load(open(state_path))
except Exception:
st = {}
e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0}
action = "NONE"
if failing:
e["fails"] = int(e.get("fails", 0)) + 1
due = e["fails"] >= threshold and (
not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60
)
if due:
first = not e.get("alerted")
if not first and in_quiet(qh):
action = "SUPPRESSED"
else:
action = "ALERT_FIRST" if first else "ALERT_REALERT"
e["alerted"] = True
e["last_alert_ts"] = now
else:
if e.get("alerted"):
action = "RECOVERY"
e["fails"] = 0
e["alerted"] = False
st[cond] = e
if not dry:
json.dump(st, open(state_path, "w"))
print(action, e["fails"])
PYEOF
}
emit_record() { # kind cond detail consecutive
local kind="$1" cond="$2" detail="$3" consec="$4"
local id
id=$(tr -d '-' < /proc/sys/kernel/random/uuid | cut -c1-12)
local rec
rec=$(python3 -c 'import json,sys; print(json.dumps({"id":sys.argv[1],"ts":int(sys.argv[2]),"kind":sys.argv[3],"source":"bl","condition":sys.argv[4],"detail":sys.argv[5],"consecutive":int(sys.argv[6])}))' \
"$id" "$NOW" "$kind" "$cond" "$detail" "$consec")
if [ "$DRY_RUN" = "1" ]; then
log "DRY-RUN would emit: $rec"
else
echo "$rec" >> "$OUTBOX"
log "emitted $kind $cond x$consec"
fi
}
box_notify() {
# Best-effort DM to healthy agents via box-ctl. Never fatal.
# Fan-out runs in parallel with a per-notify timeout so one hung DM
# path can't stall the 5-minute check loop.
local cond="$1"
local detail="$2"
local msg="[fleet-alert] CRITICAL ${cond}: ${detail}"
msg="${msg:0:240}"
local agent pids=""
for agent in $HEALTHY_AGENTS; do
if [ "$DRY_RUN" = "1" ]; then
log "DRY-RUN would notify $agent"
continue
fi
(
if timeout 60 python3 "$BIN/box-ctl.py" notify "$agent" "$msg" >/dev/null 2>&1; then
log "notified $agent re $cond"
else
log "notify $agent failed re $cond (best-effort)"
fi
) &
pids="$pids $!"
done
local p
for p in $pids; do wait "$p" 2>/dev/null; done
}
injected() { # cond -> 0 if injected-fail
case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac
}
HEALTHY_AGENTS=""
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
[ -n "$node" ] && [ -n "$port" ] || continue
cond="cdp:$node"
procs=$(pgrep -f "chromium.*profiles/$node" 2>/dev/null | wc -l)
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
failing=0
HEALTHY_AGENTS="$HEALTHY_AGENTS $node"
else
failing=1
fi
injected "$cond" && failing=1
detail="CDP $port not listening in netns warp-$node (chromium procs=$procs)"
# NOTE: HEALTHY_AGENTS set inside the pipeline subshell is lost; recompute below.
read -r action fails < <(state_machine "$cond" "$failing")
case "$action" in
ALERT_FIRST|ALERT_REALERT)
emit_record "ALERT" "$cond" "$detail" "$fails"
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
;;
RECOVERY)
emit_record "RECOVERY" "$cond" "CDP $port listening again in netns warp-$node" "$fails"
;;
SUPPRESSED)
log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)"
;;
esac
done
# Recompute healthy agents in the main shell (pipeline subshell above can't export).
HEALTHY_AGENTS=""
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
[ -n "$node" ] && [ -n "$port" ] || continue
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
echo "$node"
fi
done > "$STATE_DIR/.healthy.tmp"
HEALTHY_AGENTS=$(tr '\n' ' ' < "$STATE_DIR/.healthy.tmp")
rm -f "$STATE_DIR/.healthy.tmp"
# Notify for this run's alerts (best effort). Skip entirely when nothing is healthy
# (notify needs a working browser via dm.py) or in dry-run.
if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then
while read -r cond; do
[ -n "$cond" ] && box_notify "$cond" "see #lobby for detail"
done < "$STATE_DIR/.alerts.tmp"
elif [ -f "$STATE_DIR/.alerts.tmp" ]; then
log "no healthy agents — box notify skipped (DM path needs a working browser)"
fi
rm -f "$STATE_DIR/.alerts.tmp"
tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG"
log "check complete"