5ab157b3b0
agent-health.sh used bare node names for 'ip netns exec' but netns are
named warp-<node> since the NetVM layout; the 6189793 CDP-liveness check
always failed ('No such file or directory'), logging false CRITICALs and
kill -9'ing healthy browsers every 5 min. Use warp-$node.
fleet-alert-check.sh: new 5-min critical-condition detector (per-node CDP
liveness via warp-<node> netns). Consecutive-failure state machine:
page after 2 consecutive failures, re-page every 30 min while critical,
quiet-hours-aware (first alert always pages). Emits ALERT/RECOVERY
records to ~/.local/share/fleet-alert/outbox.jsonl for the container
fleet-alert-relay hook; best-effort box-ctl notify to healthy agents.
Session: sidechat/critical-alerting-pipeline
212 lines
8.1 KiB
Bash
Executable File
212 lines
8.1 KiB
Bash
Executable File
#!/bin/bash
|
|
# fleet-alert-check.sh — bl-side critical-condition detector for the fleet alerting pipeline.
|
|
#
|
|
# Closes the watchdog gap: agent-health.sh logs CRITICAL and restarts browsers,
|
|
# but NOTHING pages anyone. This script detects critical conditions, counts
|
|
# CONSECUTIVE failures, and emits alert records to an outbox that the
|
|
# container-side fleet-alert-relay hook picks up and pages to #lobby.
|
|
#
|
|
# Checks (bl-side only; VM-side checks live in the container relay):
|
|
# cdp:<node> headless Chromium CDP port not listening in the node's netns
|
|
# (catches zombie browsers: process alive, CDP not bound)
|
|
#
|
|
# Paging policy (env-overridable defaults — adjustable, not gates):
|
|
# FLEET_ALERT_THRESHOLD=2 consecutive failures before first page (~10 min at 5-min cadence)
|
|
# FLEET_ALERT_REALERT_MIN=30 re-page while still critical, at most every 30 min
|
|
# FLEET_ALERT_QUIET_HOURS="" e.g. "23:00-07:00" (bl local time); empty = page 24/7.
|
|
# First alert for a NEW incident always pages;
|
|
# quiet hours only suppress re-pages.
|
|
# FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify
|
|
# FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail
|
|
# (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip)
|
|
#
|
|
# State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters)
|
|
# Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay)
|
|
# Log: /tmp/fleet-alert-check.log
|
|
#
|
|
# Alert delivery legs:
|
|
# 1. outbox record -> container relay -> signed #lobby post (primary page)
|
|
# 2. best-effort `box-ctl.py notify` to currently-healthy agents (DM path needs a
|
|
# working browser; failures are logged, never fatal)
|
|
#
|
|
# Installed as user timer fleet-alert-check.timer (every 5 min), mirroring agent-health.timer.
|
|
set -uo pipefail
|
|
|
|
THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}"
|
|
REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}"
|
|
QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}"
|
|
DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}"
|
|
INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}"
|
|
STATE_DIR="${FLEET_ALERT_STATE_DIR:-$HOME/.local/share/fleet-alert}"
|
|
BIN="$(cd "$(dirname "$0")" && pwd)"
|
|
STATE="$STATE_DIR/state.json"
|
|
OUTBOX="$STATE_DIR/outbox.jsonl"
|
|
LOG="/tmp/fleet-alert-check.log"
|
|
|
|
mkdir -p "$STATE_DIR"
|
|
[ -f "$STATE" ] || echo '{}' > "$STATE"
|
|
touch "$OUTBOX"
|
|
|
|
NOW=$(date +%s)
|
|
log() { echo "$(date -Iseconds) $*" >> "$LOG"; }
|
|
|
|
# --- shared consecutive-failure state machine (also used by the container relay) ---
|
|
# usage: state_machine <cond> <failing 0|1> -> prints "<ACTION> <fails>"
|
|
# ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | NONE
|
|
state_machine() {
|
|
local cond="$1" failing="$2"
|
|
THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \
|
|
FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" <<'PYEOF'
|
|
import json, os, sys, time
|
|
state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3]
|
|
failing = failing_s == "1"
|
|
threshold = int(os.environ.get("THRESHOLD", "2"))
|
|
realert_min = int(os.environ.get("REALERT_MIN", "30"))
|
|
qh = os.environ.get("QUIET_HOURS", "")
|
|
dry = os.environ.get("FLEET_ALERT_DRY_RUN") == "1"
|
|
now = int(time.time())
|
|
def in_quiet(spec):
|
|
if not spec:
|
|
return False
|
|
try:
|
|
a, b = spec.split("-")
|
|
def m(s):
|
|
h, mi = s.split(":")
|
|
return int(h) * 60 + int(mi)
|
|
cur = time.localtime().tm_hour * 60 + time.localtime().tm_min
|
|
s, e = m(a), m(b)
|
|
return (s <= cur < e) if s <= e else (cur >= s or cur < e)
|
|
except Exception:
|
|
return False
|
|
try:
|
|
st = json.load(open(state_path))
|
|
except Exception:
|
|
st = {}
|
|
e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0}
|
|
action = "NONE"
|
|
if failing:
|
|
e["fails"] = int(e.get("fails", 0)) + 1
|
|
due = e["fails"] >= threshold and (
|
|
not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60
|
|
)
|
|
if due:
|
|
first = not e.get("alerted")
|
|
if not first and in_quiet(qh):
|
|
action = "SUPPRESSED"
|
|
else:
|
|
action = "ALERT_FIRST" if first else "ALERT_REALERT"
|
|
e["alerted"] = True
|
|
e["last_alert_ts"] = now
|
|
else:
|
|
if e.get("alerted"):
|
|
action = "RECOVERY"
|
|
e["fails"] = 0
|
|
e["alerted"] = False
|
|
st[cond] = e
|
|
if not dry:
|
|
json.dump(st, open(state_path, "w"))
|
|
print(action, e["fails"])
|
|
PYEOF
|
|
}
|
|
|
|
emit_record() { # kind cond detail consecutive
|
|
local kind="$1" cond="$2" detail="$3" consec="$4"
|
|
local id
|
|
id=$(tr -d '-' < /proc/sys/kernel/random/uuid | cut -c1-12)
|
|
local rec
|
|
rec=$(python3 -c 'import json,sys; print(json.dumps({"id":sys.argv[1],"ts":int(sys.argv[2]),"kind":sys.argv[3],"source":"bl","condition":sys.argv[4],"detail":sys.argv[5],"consecutive":int(sys.argv[6])}))' \
|
|
"$id" "$NOW" "$kind" "$cond" "$detail" "$consec")
|
|
if [ "$DRY_RUN" = "1" ]; then
|
|
log "DRY-RUN would emit: $rec"
|
|
else
|
|
echo "$rec" >> "$OUTBOX"
|
|
log "emitted $kind $cond x$consec"
|
|
fi
|
|
}
|
|
|
|
box_notify() {
|
|
# Best-effort DM to healthy agents via box-ctl. Never fatal.
|
|
# Fan-out runs in parallel with a per-notify timeout so one hung DM
|
|
# path can't stall the 5-minute check loop.
|
|
local cond="$1"
|
|
local detail="$2"
|
|
local msg="[fleet-alert] CRITICAL ${cond}: ${detail}"
|
|
msg="${msg:0:240}"
|
|
local agent pids=""
|
|
for agent in $HEALTHY_AGENTS; do
|
|
if [ "$DRY_RUN" = "1" ]; then
|
|
log "DRY-RUN would notify $agent"
|
|
continue
|
|
fi
|
|
(
|
|
if timeout 60 python3 "$BIN/box-ctl.py" notify "$agent" "$msg" >/dev/null 2>&1; then
|
|
log "notified $agent re $cond"
|
|
else
|
|
log "notify $agent failed re $cond (best-effort)"
|
|
fi
|
|
) &
|
|
pids="$pids $!"
|
|
done
|
|
local p
|
|
for p in $pids; do wait "$p" 2>/dev/null; done
|
|
}
|
|
|
|
injected() { # cond -> 0 if injected-fail
|
|
case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac
|
|
}
|
|
|
|
HEALTHY_AGENTS=""
|
|
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
[ -n "$node" ] && [ -n "$port" ] || continue
|
|
cond="cdp:$node"
|
|
procs=$(pgrep -f "chromium.*profiles/$node" 2>/dev/null | wc -l)
|
|
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
|
|
failing=0
|
|
HEALTHY_AGENTS="$HEALTHY_AGENTS $node"
|
|
else
|
|
failing=1
|
|
fi
|
|
injected "$cond" && failing=1
|
|
detail="CDP $port not listening in netns warp-$node (chromium procs=$procs)"
|
|
# NOTE: HEALTHY_AGENTS set inside the pipeline subshell is lost; recompute below.
|
|
read -r action fails < <(state_machine "$cond" "$failing")
|
|
case "$action" in
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
emit_record "ALERT" "$cond" "$detail" "$fails"
|
|
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
|
;;
|
|
RECOVERY)
|
|
emit_record "RECOVERY" "$cond" "CDP $port listening again in netns warp-$node" "$fails"
|
|
;;
|
|
SUPPRESSED)
|
|
log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)"
|
|
;;
|
|
esac
|
|
done
|
|
|
|
# Recompute healthy agents in the main shell (pipeline subshell above can't export).
|
|
HEALTHY_AGENTS=""
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
[ -n "$node" ] && [ -n "$port" ] || continue
|
|
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
|
|
echo "$node"
|
|
fi
|
|
done > "$STATE_DIR/.healthy.tmp"
|
|
HEALTHY_AGENTS=$(tr '\n' ' ' < "$STATE_DIR/.healthy.tmp")
|
|
rm -f "$STATE_DIR/.healthy.tmp"
|
|
|
|
# Notify for this run's alerts (best effort). Skip entirely when nothing is healthy
|
|
# (notify needs a working browser via dm.py) or in dry-run.
|
|
if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then
|
|
while read -r cond; do
|
|
[ -n "$cond" ] && box_notify "$cond" "see #lobby for detail"
|
|
done < "$STATE_DIR/.alerts.tmp"
|
|
elif [ -f "$STATE_DIR/.alerts.tmp" ]; then
|
|
log "no healthy agents — box notify skipped (DM path needs a working browser)"
|
|
fi
|
|
rm -f "$STATE_DIR/.alerts.tmp"
|
|
|
|
tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG"
|
|
log "check complete"
|