#!/bin/bash # fleet-alert-check.sh — bl-side critical-condition detector for the fleet alerting pipeline. # # Closes the watchdog gap: agent-health.sh logs CRITICAL and restarts browsers, # but NOTHING pages anyone. This script detects critical conditions, counts # CONSECUTIVE failures, and emits alert records to an outbox that the # container-side fleet-alert-relay hook picks up and pages to #lobby. # # Checks (bl-side only; VM-side checks live in the container relay): # cdp: headless Chromium CDP port not listening in the node's netns # (catches zombie browsers: process alive, CDP not bound) # # Paging policy (env-overridable defaults — adjustable, not gates): # FLEET_ALERT_THRESHOLD=2 consecutive failures before first page (~10 min at 5-min cadence) # FLEET_ALERT_REALERT_MIN=30 re-page while still critical, at most every 30 min # FLEET_ALERT_QUIET_HOURS="" e.g. "23:00-07:00" (bl local time); empty = page 24/7. # First alert for a NEW incident always pages; # quiet hours only suppress re-pages. # FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify # FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail # (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip) # # State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters) # Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay) # Log: /tmp/fleet-alert-check.log # # Alert delivery legs: # 1. outbox record -> container relay -> signed #lobby post (primary page) # 2. best-effort `box-ctl.py notify` to currently-healthy agents (DM path needs a # working browser; failures are logged, never fatal) # # Installed as user timer fleet-alert-check.timer (every 5 min), mirroring agent-health.timer. set -uo pipefail THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}" REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}" # Approval/input-wait TTLs (seconds): conditions failing longer than this are # auto-expired (input waits dismissed, key requests denied) instead of paging # forever. Overridable per environment. INPUT_WAIT_TTL="${FLEET_ALERT_INPUT_WAIT_TTL:-1800}" BROWSER_APPROVAL_TTL="${FLEET_ALERT_BROWSER_APPROVAL_TTL:-1800}" QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}" DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}" INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}" STATE_DIR="${FLEET_ALERT_STATE_DIR:-$HOME/.local/share/fleet-alert}" BIN="$(cd "$(dirname "$0")" && pwd)" STATE="$STATE_DIR/state.json" OUTBOX="$STATE_DIR/outbox.jsonl" LOG="/tmp/fleet-alert-check.log" mkdir -p "$STATE_DIR" [ -f "$STATE" ] || echo '{}' > "$STATE" touch "$OUTBOX" NOW=$(date +%s) log() { echo "$(date -Iseconds) $*" >> "$LOG"; } # --- shared consecutive-failure state machine (also used by the container relay) --- # usage: state_machine [ttl_seconds] -> prints " " # When ttl_seconds > 0 and the condition has failed longer than the TTL, # prints "EXPIRED " so the caller can auto-resolve (dismiss/deny). # State entries track first_fail_ts (epoch of first consecutive failure). # ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | EXPIRED | NONE state_machine() { local cond="$1" failing="$2" ttl="${3:-0}" THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \ FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" "$ttl" <<'PYEOF' import json, os, sys, time state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3] ttl_seconds = int(sys.argv[4]) if len(sys.argv) > 4 else 0 failing = failing_s == "1" threshold = int(os.environ.get("THRESHOLD", "2")) realert_min = int(os.environ.get("REALERT_MIN", "30")) qh = os.environ.get("QUIET_HOURS", "") dry = os.environ.get("FLEET_ALERT_DRY_RUN") == "1" now = int(time.time()) def in_quiet(spec): if not spec: return False try: a, b = spec.split("-") def m(s): h, mi = s.split(":") return int(h) * 60 + int(mi) cur = time.localtime().tm_hour * 60 + time.localtime().tm_min s, e = m(a), m(b) return (s <= cur < e) if s <= e else (cur >= s or cur < e) except Exception: return False try: st = json.load(open(state_path)) except Exception: st = {} e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0} action = "NONE" if failing: if int(e.get("fails", 0)) == 0: e["first_fail_ts"] = now e["fails"] = int(e.get("fails", 0)) + 1 # TTL expiry: failing longer than ttl_seconds -> EXPIRED (caller auto-resolves) if ttl_seconds > 0 and now - int(e.get("first_fail_ts", now)) >= ttl_seconds: action = "EXPIRED" # Reset so a fresh incident starts clean after the caller resolves it e["fails"] = 0 e["alerted"] = False e.pop("first_fail_ts", None) else: due = e["fails"] >= threshold and ( not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60 ) if due: first = not e.get("alerted") if not first and in_quiet(qh): action = "SUPPRESSED" else: action = "ALERT_FIRST" if first else "ALERT_REALERT" e["alerted"] = True e["last_alert_ts"] = now else: if e.get("alerted"): action = "RECOVERY" e["fails"] = 0 e["alerted"] = False e.pop("first_fail_ts", None) st[cond] = e if not dry: json.dump(st, open(state_path, "w")) print(action, e["fails"]) PYEOF } emit_record() { # kind cond detail consecutive local kind="$1" cond="$2" detail="$3" consec="$4" local id id=$(tr -d '-' < /proc/sys/kernel/random/uuid | cut -c1-12) local rec rec=$(python3 -c 'import json,sys; print(json.dumps({"id":sys.argv[1],"ts":int(sys.argv[2]),"kind":sys.argv[3],"source":"bl","condition":sys.argv[4],"detail":sys.argv[5],"consecutive":int(sys.argv[6])}))' \ "$id" "$NOW" "$kind" "$cond" "$detail" "$consec") if [ "$DRY_RUN" = "1" ]; then log "DRY-RUN would emit: $rec" else echo "$rec" >> "$OUTBOX" log "emitted $kind $cond x$consec" fi } box_notify() { # Best-effort DM to healthy agents via box-ctl. Never fatal. # Fan-out runs in parallel with a per-notify timeout so one hung DM # path can't stall the 5-minute check loop. local cond="$1" local detail="$2" local msg="[fleet-alert] CRITICAL ${cond}: ${detail}" msg="${msg:0:240}" local agent pids="" for agent in $HEALTHY_AGENTS; do if [ "$DRY_RUN" = "1" ]; then log "DRY-RUN would notify $agent" continue fi ( if timeout 60 python3 "$BIN/box-ctl.py" notify "$agent" "$msg" >/dev/null 2>&1; then log "notified $agent re $cond" else log "notify $agent failed re $cond (best-effort)" fi ) & pids="$pids $!" done local p for p in $pids; do wait "$p" 2>/dev/null; done } notify_input_wait() { # Targeted DM for input waits (2026-10-05): DM ONLY the specific agent # whose session is waiting for human input -- not a broadcast to all # healthy agents. The #lobby post still fires via the relay leg for # human visibility; this DM ensures the responsible operator sees it # in their sidechat without digging through lobby noise. # Best-effort: never fatal to the 5-minute check loop. local node="$1" local detail="$2" local msg="[fleet-alert] INPUT WAIT: ${detail} -- reply: box approval reply ${node} \"\" or box notify ${node} \"\"" msg="${msg:0:900}" if [ "$DRY_RUN" = "1" ]; then log "DRY-RUN would DM $node re input_wait" return 0 fi if timeout 60 python3 "$BIN/box-ctl.py" notify "$node" "$msg" >/dev/null 2>&1; then log "input_wait targeted DM sent to $node" else log "input_wait DM to $node failed (best-effort, non-fatal)" fi } injected() { # cond -> 0 if injected-fail case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac } HEALTHY_AGENTS="" "$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] && [ -n "$port" ] || continue cond="cdp:$node" procs=$(pgrep -f "chromium.*profiles/$node" 2>/dev/null | wc -l) if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then failing=0 HEALTHY_AGENTS="$HEALTHY_AGENTS $node" else failing=1 fi injected "$cond" && failing=1 detail="CDP $port not listening in netns warp-$node (chromium procs=$procs)" # NOTE: HEALTHY_AGENTS set inside the pipeline subshell is lost; recompute below. read -r action fails < <(state_machine "$cond" "$failing") case "$action" in ALERT_FIRST|ALERT_REALERT) emit_record "ALERT" "$cond" "$detail" "$fails" echo "$cond" >> "$STATE_DIR/.alerts.tmp" ;; RECOVERY) emit_record "RECOVERY" "$cond" "CDP $port listening again in netns warp-$node" "$fails" ;; SUPPRESSED) log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)" ;; esac done # --- Agent approval blockage & input wait detection --- "$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] || continue node_data=$(python3 -c " import sys, json sys.path.insert(0, '$BIN') import approvals info = approvals.inspect_node_approvals('$node') out = { 'has_pending': info.get('has_pending', False), 'target': info.get('target') or info.get('ip') or 'unknown', 'title': info.get('title') or '', 'waits': info.get('input_waits') or [] } print(json.dumps(out)) " 2>/dev/null || echo '{"has_pending":false,"target":"unknown","title":"","waits":[]}') # 1. Egress permission dialog cond="approval:$node" has_pending=$(python3 -c "import json,sys; print(1 if json.loads(sys.argv[1]).get('has_pending') else 0)" "$node_data" 2>/dev/null || echo 0) if [ "$has_pending" = "1" ]; then failing=1 target=$(python3 -c "import json,sys; print(json.loads(sys.argv[1]).get('target','unknown'))" "$node_data" 2>/dev/null || echo unknown) detail="Agent $node held up on browser approval for $target" else failing=0 detail="Agent $node approvals clear" fi injected "$cond" && failing=1 read -r action fails < <(state_machine "$cond" "$failing" "$BROWSER_APPROVAL_TTL") case "$action" in ALERT_FIRST|ALERT_REALERT) emit_record "ALERT" "$cond" "$detail" "$fails" echo "$cond" >> "$STATE_DIR/.alerts.tmp" ;; RECOVERY) emit_record "RECOVERY" "$cond" "$detail" "$fails" ;; SUPPRESSED) log "$cond still critical x$fails — re-page suppressed" ;; EXPIRED) # Browser approval dialog exceeded BROWSER_APPROVAL_TTL without a # human decision: fail closed by denying it. log "$cond EXPIRED after ${BROWSER_APPROVAL_TTL}s without human decision — auto-denying (fail closed)" python3 - "$node" <<'PYEOF3' import sys, json sys.path.insert(0, "/home/super/Projects/NetVM/bin") import approvals node = sys.argv[1] print(json.dumps(approvals.deny_node_approval(node, caller="approval-ttl-expire"))) approvals.log_box_ctl("approval-expired", name=node, caller="approval-ttl-expire", extra={"note": "browser approval TTL elapsed; auto-denied (fail closed)"}) PYEOF3 emit_record "RECOVERY" "$cond" "Agent $node browser approval expired after ${BROWSER_APPROVAL_TTL}s; auto-denied" "$fails" ;; esac # 2. Sidebar task waiting on human input cond_in="input_wait:$node" wait_summary=$(python3 -c " import json,sys w = json.loads(sys.argv[1]).get('waits', []) if w: print('; '.join(f\"{item.get('task')}: {item.get('status')}\" for item in w)[:120]) " "$node_data" 2>/dev/null || true) if [ -n "$wait_summary" ]; then failing_in=1 detail_in="Agent $node task waiting for human input: $wait_summary" else failing_in=0 detail_in="Agent $node tasks running" fi injected "$cond_in" && failing_in=1 read -r action_in fails_in < <(state_machine "$cond_in" "$failing_in" "$INPUT_WAIT_TTL") case "$action_in" in ALERT_FIRST|ALERT_REALERT) emit_record "ALERT" "$cond_in" "$detail_in" "$fails_in" echo "$cond_in|$detail_in" >> "$STATE_DIR/.alerts.tmp" ;; RECOVERY) emit_record "RECOVERY" "$cond_in" "$detail_in" "$fails_in" ;; SUPPRESSED) log "$cond_in still critical x$fails_in — re-page suppressed" ;; EXPIRED) # Input wait exceeded INPUT_WAIT_TTL without human response: # auto-dismiss so the agent unblocks. Log the expiry and emit a # RECOVERY record (the wait is gone, not merely un-paged). log "$cond_in EXPIRED after ${INPUT_WAIT_TTL}s without human input — auto-dismissing" python3 - "$node" <<'PYEOF2' import sys sys.path.insert(0, "/home/super/Projects/NetVM/bin") import approvals, json node = sys.argv[1] info = approvals.inspect_node_approvals(node) for w in info.get("input_waits", []) or []: t = w.get("task") if t: approvals.mark_wait_responded(node, t, caller="approval-ttl-expire") approvals.log_box_ctl("approval-wait-expired", name=node, caller="approval-ttl-expire", extra={"note": "input wait TTL elapsed; auto-dismissed"}) print(json.dumps(approvals.dismiss_node_task(node, caller="approval-ttl-expire"))) PYEOF2 emit_record "RECOVERY" "$cond_in" "Agent $node input wait expired after ${INPUT_WAIT_TTL}s; auto-dismissed" "$fails_in" ;; esac done # --- Warp partition detection (2026-10-04) --- # A partitioned node has a live browser + CDP but no internet egress: the # chromebox watchdog sees a healthy browser while all automation fails. # Condition id: partition:. The detail names the node, the WireGuard # handshake age, the egress probe result, and the timestamp, and says # PARTITION explicitly so #lobby readers can tell a network partition from # a browser crash at a glance. Anti-spam comes from the shared consecutive- # failure state machine (2 consecutive failures before first page, re-page # at most every 30 min). WARP_PROBE_URL="${WARP_PROBE_URL:-https://1.1.1.1/cdn-cgi/trace}" warp_partition_probe() { # -> prints " " local node="$1" iface epoch now age_s iface=$(sudo -n ip netns exec "warp-$node" sh -c 'wg show interfaces 2>/dev/null | head -1') now=$(date +%s) epoch=$(sudo -n ip netns exec "warp-$node" wg show "$iface" latest-handshakes 2>/dev/null | awk '{print $2}') case "$epoch" in ''|*[!0-9]*) age_s="unknown" ;; *) age_s=$(( now - epoch )) ;; esac if sudo -n ip netns exec "warp-$node" curl -s -m 8 -o /dev/null "$WARP_PROBE_URL" 2>/dev/null; then echo "$age_s ok" else echo "$age_s FAIL" fi } "$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] && [ -n "$port" ] || continue cond="partition:$node" read -r hs_age probe_res < <(warp_partition_probe "$node") ts=$(date -u +%FT%TZ) if [ "$probe_res" = "ok" ]; then failing=0 detail="warp egress restored for $node at $ts (probe $WARP_PROBE_URL ok)" else failing=1 detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED)" fi if injected "$cond"; then failing=1 detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED) [INJECTED]" fi read -r action fails < <(state_machine "$cond" "$failing") case "$action" in ALERT_FIRST|ALERT_REALERT) emit_record "ALERT" "$cond" "$detail" "$fails" echo "$cond" >> "$STATE_DIR/.alerts.tmp" ;; RECOVERY) emit_record "RECOVERY" "$cond" "$detail" "$fails" ;; SUPPRESSED) log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)" ;; esac done # Recompute healthy agents in the main shell (pipeline subshell above can't export). HEALTHY_AGENTS="" "$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] && [ -n "$port" ] || continue if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then echo "$node" fi done > "$STATE_DIR/.healthy.tmp" HEALTHY_AGENTS=$(tr '\n' ' ' < "$STATE_DIR/.healthy.tmp") rm -f "$STATE_DIR/.healthy.tmp" # Notify for this run's alerts (best effort). Skip entirely when nothing is healthy # (notify needs a working browser via dm.py) or in dry-run. if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then while IFS= read -r line; do # alerts.tmp format: "cond" or "cond|detail" (input_wait carries detail) cond="${line%%|*}" detail="${line#*|}" [ "$detail" = "$line" ] && detail="" [ -n "$cond" ] || continue case "$cond" in input_wait:*) # Targeted: DM only the waiting agent, not a broadcast. node="${cond#input_wait:}" if [ -n "$detail" ]; then notify_input_wait "$node" "$detail" else box_notify "$cond" "see #lobby for detail" fi ;; *) box_notify "$cond" "see #lobby for detail" ;; esac done < "$STATE_DIR/.alerts.tmp" elif [ -f "$STATE_DIR/.alerts.tmp" ]; then log "no healthy agents — box notify skipped (DM path needs a working browser)" fi rm -f "$STATE_DIR/.alerts.tmp" tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG" log "check complete" # Relay pending outbox records to #lobby with idempotency gates (posted watermark + content hash TTL) if [ "$DRY_RUN" -eq 0 ] && [ -x "$BIN/fleet-alert-relay.sh" ]; then "$BIN/fleet-alert-relay.sh" >> "$LOG" 2>&1 || true fi