#!/bin/bash # fleet-alert-check.sh — bl-side critical-condition detector for the fleet alerting pipeline. # # Closes the watchdog gap: agent-health.sh logs CRITICAL and restarts browsers, # but NOTHING pages anyone. This script detects critical conditions, counts # CONSECUTIVE failures, and emits alert records to an outbox that the # container-side fleet-alert-relay hook picks up and pages to #lobby. # # Checks (bl-side only; VM-side checks live in the container relay): # cdp: headless Chromium CDP port not listening in the node's netns # (catches zombie browsers: process alive, CDP not bound) # # Paging policy (env-overridable defaults — adjustable, not gates): # FLEET_ALERT_THRESHOLD=2 consecutive failures before first page (~10 min at 5-min cadence) # FLEET_ALERT_REALERT_MIN=30 re-page while still critical, at most every 30 min # FLEET_ALERT_QUIET_HOURS="" e.g. "23:00-07:00" (bl local time); empty = page 24/7. # First alert for a NEW incident always pages; # quiet hours only suppress re-pages. # FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify # FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail # (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip) # # State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters) # Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay) # Log: /tmp/fleet-alert-check.log # # Alert delivery legs: # 1. outbox record -> container relay -> signed #lobby post (primary page) # 2. best-effort `box-ctl.py notify` to currently-healthy agents (DM path needs a # working browser; failures are logged, never fatal) # # Installed as user timer fleet-alert-check.timer (every 5 min), mirroring agent-health.timer. set -uo pipefail THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}" REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}" QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}" DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}" INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}" STATE_DIR="${FLEET_ALERT_STATE_DIR:-$HOME/.local/share/fleet-alert}" BIN="$(cd "$(dirname "$0")" && pwd)" STATE="$STATE_DIR/state.json" OUTBOX="$STATE_DIR/outbox.jsonl" LOG="/tmp/fleet-alert-check.log" mkdir -p "$STATE_DIR" [ -f "$STATE" ] || echo '{}' > "$STATE" touch "$OUTBOX" NOW=$(date +%s) log() { echo "$(date -Iseconds) $*" >> "$LOG"; } # --- shared consecutive-failure state machine (also used by the container relay) --- # usage: state_machine -> prints " " # ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | NONE state_machine() { local cond="$1" failing="$2" THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \ FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" <<'PYEOF' import json, os, sys, time state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3] failing = failing_s == "1" threshold = int(os.environ.get("THRESHOLD", "2")) realert_min = int(os.environ.get("REALERT_MIN", "30")) qh = os.environ.get("QUIET_HOURS", "") dry = os.environ.get("FLEET_ALERT_DRY_RUN") == "1" now = int(time.time()) def in_quiet(spec): if not spec: return False try: a, b = spec.split("-") def m(s): h, mi = s.split(":") return int(h) * 60 + int(mi) cur = time.localtime().tm_hour * 60 + time.localtime().tm_min s, e = m(a), m(b) return (s <= cur < e) if s <= e else (cur >= s or cur < e) except Exception: return False try: st = json.load(open(state_path)) except Exception: st = {} e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0} action = "NONE" if failing: e["fails"] = int(e.get("fails", 0)) + 1 due = e["fails"] >= threshold and ( not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60 ) if due: first = not e.get("alerted") if not first and in_quiet(qh): action = "SUPPRESSED" else: action = "ALERT_FIRST" if first else "ALERT_REALERT" e["alerted"] = True e["last_alert_ts"] = now else: if e.get("alerted"): action = "RECOVERY" e["fails"] = 0 e["alerted"] = False st[cond] = e if not dry: json.dump(st, open(state_path, "w")) print(action, e["fails"]) PYEOF } emit_record() { # kind cond detail consecutive local kind="$1" cond="$2" detail="$3" consec="$4" local id id=$(tr -d '-' < /proc/sys/kernel/random/uuid | cut -c1-12) local rec rec=$(python3 -c 'import json,sys; print(json.dumps({"id":sys.argv[1],"ts":int(sys.argv[2]),"kind":sys.argv[3],"source":"bl","condition":sys.argv[4],"detail":sys.argv[5],"consecutive":int(sys.argv[6])}))' \ "$id" "$NOW" "$kind" "$cond" "$detail" "$consec") if [ "$DRY_RUN" = "1" ]; then log "DRY-RUN would emit: $rec" else echo "$rec" >> "$OUTBOX" log "emitted $kind $cond x$consec" fi } box_notify() { # Best-effort DM to healthy agents via box-ctl. Never fatal. # Fan-out runs in parallel with a per-notify timeout so one hung DM # path can't stall the 5-minute check loop. local cond="$1" local detail="$2" local msg="[fleet-alert] CRITICAL ${cond}: ${detail}" msg="${msg:0:240}" local agent pids="" for agent in $HEALTHY_AGENTS; do if [ "$DRY_RUN" = "1" ]; then log "DRY-RUN would notify $agent" continue fi ( if timeout 60 python3 "$BIN/box-ctl.py" notify "$agent" "$msg" >/dev/null 2>&1; then log "notified $agent re $cond" else log "notify $agent failed re $cond (best-effort)" fi ) & pids="$pids $!" done local p for p in $pids; do wait "$p" 2>/dev/null; done } injected() { # cond -> 0 if injected-fail case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac } HEALTHY_AGENTS="" "$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] && [ -n "$port" ] || continue cond="cdp:$node" procs=$(pgrep -f "chromium.*profiles/$node" 2>/dev/null | wc -l) if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then failing=0 HEALTHY_AGENTS="$HEALTHY_AGENTS $node" else failing=1 fi injected "$cond" && failing=1 detail="CDP $port not listening in netns warp-$node (chromium procs=$procs)" # NOTE: HEALTHY_AGENTS set inside the pipeline subshell is lost; recompute below. read -r action fails < <(state_machine "$cond" "$failing") case "$action" in ALERT_FIRST|ALERT_REALERT) emit_record "ALERT" "$cond" "$detail" "$fails" echo "$cond" >> "$STATE_DIR/.alerts.tmp" ;; RECOVERY) emit_record "RECOVERY" "$cond" "CDP $port listening again in netns warp-$node" "$fails" ;; SUPPRESSED) log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)" ;; esac done # --- Agent approval blockage detection (catches agents held up on approvals) --- "$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] || continue cond="approval:$node" pending_info=$(python3 -c " import sys sys.path.insert(0, '$BIN') import approvals info = approvals.inspect_node_approvals('$node') if info.get('has_pending'): print(f\"{info.get('ip') or 'unknown'}|{info.get('title') or ''}\") " 2>/dev/null || true) if [ -n "$pending_info" ]; then failing=1 target="${pending_info%%|*}" detail="Agent $node held up on browser approval for $target" else failing=0 detail="Agent $node approvals clear" fi injected "$cond" && failing=1 read -r action fails < <(state_machine "$cond" "$failing") case "$action" in ALERT_FIRST|ALERT_REALERT) emit_record "ALERT" "$cond" "$detail" "$fails" echo "$cond" >> "$STATE_DIR/.alerts.tmp" ;; RECOVERY) emit_record "RECOVERY" "$cond" "$detail" "$fails" ;; SUPPRESSED) log "$cond still critical x$fails — re-page suppressed" ;; esac done # --- Warp partition detection (2026-10-04) --- # A partitioned node has a live browser + CDP but no internet egress: the # chromebox watchdog sees a healthy browser while all automation fails. # Condition id: partition:. The detail names the node, the WireGuard # handshake age, the egress probe result, and the timestamp, and says # PARTITION explicitly so #lobby readers can tell a network partition from # a browser crash at a glance. Anti-spam comes from the shared consecutive- # failure state machine (2 consecutive failures before first page, re-page # at most every 30 min). WARP_PROBE_URL="${WARP_PROBE_URL:-https://1.1.1.1/cdn-cgi/trace}" warp_partition_probe() { # -> prints " " local node="$1" iface epoch now age_s iface=$(sudo -n ip netns exec "warp-$node" sh -c 'wg show interfaces 2>/dev/null | head -1') now=$(date +%s) epoch=$(sudo -n ip netns exec "warp-$node" wg show "$iface" latest-handshakes 2>/dev/null | awk '{print $2}') case "$epoch" in ''|*[!0-9]*) age_s="unknown" ;; *) age_s=$(( now - epoch )) ;; esac if sudo -n ip netns exec "warp-$node" curl -s -m 8 -o /dev/null "$WARP_PROBE_URL" 2>/dev/null; then echo "$age_s ok" else echo "$age_s FAIL" fi } "$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] && [ -n "$port" ] || continue cond="partition:$node" read -r hs_age probe_res < <(warp_partition_probe "$node") ts=$(date -u +%FT%TZ) if [ "$probe_res" = "ok" ]; then failing=0 detail="warp egress restored for $node at $ts (probe $WARP_PROBE_URL ok)" else failing=1 detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED)" fi if injected "$cond"; then failing=1 detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED) [INJECTED]" fi read -r action fails < <(state_machine "$cond" "$failing") case "$action" in ALERT_FIRST|ALERT_REALERT) emit_record "ALERT" "$cond" "$detail" "$fails" echo "$cond" >> "$STATE_DIR/.alerts.tmp" ;; RECOVERY) emit_record "RECOVERY" "$cond" "$detail" "$fails" ;; SUPPRESSED) log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)" ;; esac done # Recompute healthy agents in the main shell (pipeline subshell above can't export). HEALTHY_AGENTS="" "$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] && [ -n "$port" ] || continue if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then echo "$node" fi done > "$STATE_DIR/.healthy.tmp" HEALTHY_AGENTS=$(tr '\n' ' ' < "$STATE_DIR/.healthy.tmp") rm -f "$STATE_DIR/.healthy.tmp" # Notify for this run's alerts (best effort). Skip entirely when nothing is healthy # (notify needs a working browser via dm.py) or in dry-run. if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then while read -r cond; do [ -n "$cond" ] && box_notify "$cond" "see #lobby for detail" done < "$STATE_DIR/.alerts.tmp" elif [ -f "$STATE_DIR/.alerts.tmp" ]; then log "no healthy agents — box notify skipped (DM path needs a working browser)" fi rm -f "$STATE_DIR/.alerts.tmp" tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG" log "check complete" # Relay pending outbox records to #lobby with idempotency gates (posted watermark + content hash TTL) if [ "$DRY_RUN" -eq 0 ] && [ -x "$BIN/fleet-alert-relay.sh" ]; then "$BIN/fleet-alert-relay.sh" >> "$LOG" 2>&1 || true fi