Files
box/bin/fleet-alert-check.sh
T

442 lines
18 KiB
Bash
Raw Normal View History

#!/bin/bash
# fleet-alert-check.sh — bl-side critical-condition detector for the fleet alerting pipeline.
#
# Closes the watchdog gap: agent-health.sh logs CRITICAL and restarts browsers,
# but NOTHING pages anyone. This script detects critical conditions, counts
# CONSECUTIVE failures, and emits alert records to an outbox that the
# container-side fleet-alert-relay hook picks up and pages to #lobby.
#
# Checks (bl-side only; VM-side checks live in the container relay):
# cdp:<node> headless Chromium CDP port not listening in the node's netns
# (catches zombie browsers: process alive, CDP not bound)
#
# Paging policy (env-overridable defaults — adjustable, not gates):
# FLEET_ALERT_THRESHOLD=2 consecutive failures before first page (~10 min at 5-min cadence)
# FLEET_ALERT_REALERT_MIN=30 re-page while still critical, at most every 30 min
# FLEET_ALERT_QUIET_HOURS="" e.g. "23:00-07:00" (bl local time); empty = page 24/7.
# First alert for a NEW incident always pages;
# quiet hours only suppress re-pages.
# FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify
# FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail
# (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip)
#
# State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters)
# Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay)
# Log: /tmp/fleet-alert-check.log
#
# Alert delivery legs:
# 1. outbox record -> container relay -> signed #lobby post (primary page)
# 2. best-effort `box-ctl.py notify` to currently-healthy agents (DM path needs a
# working browser; failures are logged, never fatal)
#
# Installed as user timer fleet-alert-check.timer (every 5 min), mirroring agent-health.timer.
set -uo pipefail
THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}"
REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}"
# Approval/input-wait TTLs (seconds): conditions failing longer than this are
# auto-expired (input waits dismissed, key requests denied) instead of paging
# forever. Overridable per environment.
INPUT_WAIT_TTL="${FLEET_ALERT_INPUT_WAIT_TTL:-1800}"
BROWSER_APPROVAL_TTL="${FLEET_ALERT_BROWSER_APPROVAL_TTL:-1800}"
QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}"
DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}"
INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}"
STATE_DIR="${FLEET_ALERT_STATE_DIR:-$HOME/.local/share/fleet-alert}"
BIN="$(cd "$(dirname "$0")" && pwd)"
STATE="$STATE_DIR/state.json"
OUTBOX="$STATE_DIR/outbox.jsonl"
LOG="/tmp/fleet-alert-check.log"
mkdir -p "$STATE_DIR"
[ -f "$STATE" ] || echo '{}' > "$STATE"
touch "$OUTBOX"
NOW=$(date +%s)
log() { echo "$(date -Iseconds) $*" >> "$LOG"; }
# --- shared consecutive-failure state machine (also used by the container relay) ---
# usage: state_machine <cond> <failing 0|1> [ttl_seconds] -> prints "<ACTION> <fails>"
# When ttl_seconds > 0 and the condition has failed longer than the TTL,
# prints "EXPIRED <fails>" so the caller can auto-resolve (dismiss/deny).
# State entries track first_fail_ts (epoch of first consecutive failure).
# ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | EXPIRED | NONE
state_machine() {
local cond="$1" failing="$2" ttl="${3:-0}"
THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \
FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" "$ttl" <<'PYEOF'
import json, os, sys, time
state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3]
ttl_seconds = int(sys.argv[4]) if len(sys.argv) > 4 else 0
failing = failing_s == "1"
threshold = int(os.environ.get("THRESHOLD", "2"))
realert_min = int(os.environ.get("REALERT_MIN", "30"))
qh = os.environ.get("QUIET_HOURS", "")
dry = os.environ.get("FLEET_ALERT_DRY_RUN") == "1"
now = int(time.time())
def in_quiet(spec):
if not spec:
return False
try:
a, b = spec.split("-")
def m(s):
h, mi = s.split(":")
return int(h) * 60 + int(mi)
cur = time.localtime().tm_hour * 60 + time.localtime().tm_min
s, e = m(a), m(b)
return (s <= cur < e) if s <= e else (cur >= s or cur < e)
except Exception:
return False
try:
st = json.load(open(state_path))
except Exception:
st = {}
e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0}
action = "NONE"
if failing:
if int(e.get("fails", 0)) == 0:
e["first_fail_ts"] = now
e["fails"] = int(e.get("fails", 0)) + 1
# TTL expiry: failing longer than ttl_seconds -> EXPIRED (caller auto-resolves)
if ttl_seconds > 0 and now - int(e.get("first_fail_ts", now)) >= ttl_seconds:
action = "EXPIRED"
# Reset so a fresh incident starts clean after the caller resolves it
e["fails"] = 0
e["alerted"] = False
e.pop("first_fail_ts", None)
else:
due = e["fails"] >= threshold and (
not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60
)
if due:
first = not e.get("alerted")
if not first and in_quiet(qh):
action = "SUPPRESSED"
else:
action = "ALERT_FIRST" if first else "ALERT_REALERT"
e["alerted"] = True
e["last_alert_ts"] = now
else:
if e.get("alerted"):
action = "RECOVERY"
e["fails"] = 0
e["alerted"] = False
e.pop("first_fail_ts", None)
st[cond] = e
if not dry:
json.dump(st, open(state_path, "w"))
print(action, e["fails"])
PYEOF
}
emit_record() { # kind cond detail consecutive
local kind="$1" cond="$2" detail="$3" consec="$4"
local id
id=$(tr -d '-' < /proc/sys/kernel/random/uuid | cut -c1-12)
local rec
rec=$(python3 -c 'import json,sys; print(json.dumps({"id":sys.argv[1],"ts":int(sys.argv[2]),"kind":sys.argv[3],"source":"bl","condition":sys.argv[4],"detail":sys.argv[5],"consecutive":int(sys.argv[6])}))' \
"$id" "$NOW" "$kind" "$cond" "$detail" "$consec")
if [ "$DRY_RUN" = "1" ]; then
log "DRY-RUN would emit: $rec"
else
echo "$rec" >> "$OUTBOX"
log "emitted $kind $cond x$consec"
fi
}
box_notify() {
# Best-effort DM to healthy agents via box-ctl. Never fatal.
# Fan-out runs in parallel with a per-notify timeout so one hung DM
# path can't stall the 5-minute check loop.
local cond="$1"
local detail="$2"
local msg="[fleet-alert] CRITICAL ${cond}: ${detail}"
msg="${msg:0:240}"
local agent pids=""
for agent in $HEALTHY_AGENTS; do
if [ "$DRY_RUN" = "1" ]; then
log "DRY-RUN would notify $agent"
continue
fi
(
if timeout 60 python3 "$BIN/box-ctl.py" notify "$agent" "$msg" >/dev/null 2>&1; then
log "notified $agent re $cond"
else
log "notify $agent failed re $cond (best-effort)"
fi
) &
pids="$pids $!"
done
local p
for p in $pids; do wait "$p" 2>/dev/null; done
}
notify_input_wait() {
# Targeted DM for input waits (2026-10-05): DM ONLY the specific agent
# whose session is waiting for human input -- not a broadcast to all
# healthy agents. The #lobby post still fires via the relay leg for
# human visibility; this DM ensures the responsible operator sees it
# in their sidechat without digging through lobby noise.
# Best-effort: never fatal to the 5-minute check loop.
local node="$1"
local detail="$2"
local msg="[fleet-alert] INPUT WAIT: ${detail} -- reply: box approval reply ${node} \"<msg>\" or box notify ${node} \"<msg>\""
msg="${msg:0:900}"
if [ "$DRY_RUN" = "1" ]; then
log "DRY-RUN would DM $node re input_wait"
return 0
fi
if timeout 60 python3 "$BIN/box-ctl.py" notify "$node" "$msg" >/dev/null 2>&1; then
log "input_wait targeted DM sent to $node"
else
log "input_wait DM to $node failed (best-effort, non-fatal)"
fi
}
injected() { # cond -> 0 if injected-fail
case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac
}
HEALTHY_AGENTS=""
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
[ -n "$node" ] && [ -n "$port" ] || continue
cond="cdp:$node"
procs=$(pgrep -f "chromium.*profiles/$node" 2>/dev/null | wc -l)
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
failing=0
HEALTHY_AGENTS="$HEALTHY_AGENTS $node"
else
failing=1
fi
injected "$cond" && failing=1
detail="CDP $port not listening in netns warp-$node (chromium procs=$procs)"
# NOTE: HEALTHY_AGENTS set inside the pipeline subshell is lost; recompute below.
read -r action fails < <(state_machine "$cond" "$failing")
case "$action" in
ALERT_FIRST|ALERT_REALERT)
emit_record "ALERT" "$cond" "$detail" "$fails"
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
;;
RECOVERY)
emit_record "RECOVERY" "$cond" "CDP $port listening again in netns warp-$node" "$fails"
;;
SUPPRESSED)
log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)"
;;
esac
done
# --- Agent approval blockage & input wait detection ---
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
[ -n "$node" ] || continue
node_data=$(python3 -c "
import sys, json
sys.path.insert(0, '$BIN')
import approvals
info = approvals.inspect_node_approvals('$node')
out = {
'has_pending': info.get('has_pending', False),
'target': info.get('target') or info.get('ip') or 'unknown',
'title': info.get('title') or '',
'waits': info.get('input_waits') or []
}
print(json.dumps(out))
" 2>/dev/null || echo '{"has_pending":false,"target":"unknown","title":"","waits":[]}')
# 1. Egress permission dialog
cond="approval:$node"
has_pending=$(python3 -c "import json,sys; print(1 if json.loads(sys.argv[1]).get('has_pending') else 0)" "$node_data" 2>/dev/null || echo 0)
if [ "$has_pending" = "1" ]; then
failing=1
target=$(python3 -c "import json,sys; print(json.loads(sys.argv[1]).get('target','unknown'))" "$node_data" 2>/dev/null || echo unknown)
detail="Agent $node held up on browser approval for $target"
else
failing=0
detail="Agent $node approvals clear"
fi
injected "$cond" && failing=1
read -r action fails < <(state_machine "$cond" "$failing" "$BROWSER_APPROVAL_TTL")
case "$action" in
ALERT_FIRST|ALERT_REALERT)
emit_record "ALERT" "$cond" "$detail" "$fails"
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
;;
RECOVERY)
emit_record "RECOVERY" "$cond" "$detail" "$fails"
;;
SUPPRESSED)
log "$cond still critical x$fails — re-page suppressed"
;;
EXPIRED)
# Browser approval dialog exceeded BROWSER_APPROVAL_TTL without a
# human decision: fail closed by denying it.
log "$cond EXPIRED after ${BROWSER_APPROVAL_TTL}s without human decision — auto-denying (fail closed)"
python3 - "$node" <<'PYEOF3'
import sys, json
sys.path.insert(0, "/home/super/Projects/NetVM/bin")
import approvals
node = sys.argv[1]
print(json.dumps(approvals.deny_node_approval(node, caller="approval-ttl-expire")))
approvals.log_box_ctl("approval-expired", name=node, caller="approval-ttl-expire",
extra={"note": "browser approval TTL elapsed; auto-denied (fail closed)"})
PYEOF3
emit_record "RECOVERY" "$cond" "Agent $node browser approval expired after ${BROWSER_APPROVAL_TTL}s; auto-denied" "$fails"
;;
esac
# 2. Sidebar task waiting on human input
cond_in="input_wait:$node"
wait_summary=$(python3 -c "
import json,sys
w = json.loads(sys.argv[1]).get('waits', [])
if w:
print('; '.join(f\"{item.get('task')}: {item.get('status')}\" for item in w)[:120])
" "$node_data" 2>/dev/null || true)
if [ -n "$wait_summary" ]; then
failing_in=1
detail_in="Agent $node task waiting for human input: $wait_summary"
else
failing_in=0
detail_in="Agent $node tasks running"
fi
injected "$cond_in" && failing_in=1
read -r action_in fails_in < <(state_machine "$cond_in" "$failing_in" "$INPUT_WAIT_TTL")
case "$action_in" in
ALERT_FIRST|ALERT_REALERT)
emit_record "ALERT" "$cond_in" "$detail_in" "$fails_in"
echo "$cond_in|$detail_in" >> "$STATE_DIR/.alerts.tmp"
;;
RECOVERY)
emit_record "RECOVERY" "$cond_in" "$detail_in" "$fails_in"
;;
SUPPRESSED)
log "$cond_in still critical x$fails_in — re-page suppressed"
;;
EXPIRED)
# Input wait exceeded INPUT_WAIT_TTL without human response:
# auto-dismiss so the agent unblocks. Log the expiry and emit a
# RECOVERY record (the wait is gone, not merely un-paged).
log "$cond_in EXPIRED after ${INPUT_WAIT_TTL}s without human input — auto-dismissing"
python3 - "$node" <<'PYEOF2'
import sys
sys.path.insert(0, "/home/super/Projects/NetVM/bin")
import approvals, json
node = sys.argv[1]
info = approvals.inspect_node_approvals(node)
for w in info.get("input_waits", []) or []:
t = w.get("task")
if t:
approvals.mark_wait_responded(node, t, caller="approval-ttl-expire")
approvals.log_box_ctl("approval-wait-expired", name=node, caller="approval-ttl-expire",
extra={"note": "input wait TTL elapsed; auto-dismissed"})
print(json.dumps(approvals.dismiss_node_task(node, caller="approval-ttl-expire")))
PYEOF2
emit_record "RECOVERY" "$cond_in" "Agent $node input wait expired after ${INPUT_WAIT_TTL}s; auto-dismissed" "$fails_in"
;;
esac
done
# --- Warp partition detection (2026-10-04) ---
# A partitioned node has a live browser + CDP but no internet egress: the
# chromebox watchdog sees a healthy browser while all automation fails.
# Condition id: partition:<node>. The detail names the node, the WireGuard
# handshake age, the egress probe result, and the timestamp, and says
# PARTITION explicitly so #lobby readers can tell a network partition from
# a browser crash at a glance. Anti-spam comes from the shared consecutive-
# failure state machine (2 consecutive failures before first page, re-page
# at most every 30 min).
WARP_PROBE_URL="${WARP_PROBE_URL:-https://1.1.1.1/cdn-cgi/trace}"
warp_partition_probe() { # <node> -> prints "<handshake_age_s|unknown> <ok|FAIL>"
local node="$1" iface epoch now age_s
iface=$(sudo -n ip netns exec "warp-$node" sh -c 'wg show interfaces 2>/dev/null | head -1')
now=$(date +%s)
epoch=$(sudo -n ip netns exec "warp-$node" wg show "$iface" latest-handshakes 2>/dev/null | awk '{print $2}')
case "$epoch" in ''|*[!0-9]*) age_s="unknown" ;; *) age_s=$(( now - epoch )) ;; esac
if sudo -n ip netns exec "warp-$node" curl -s -m 8 -o /dev/null "$WARP_PROBE_URL" 2>/dev/null; then
echo "$age_s ok"
else
echo "$age_s FAIL"
fi
}
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
[ -n "$node" ] && [ -n "$port" ] || continue
cond="partition:$node"
read -r hs_age probe_res < <(warp_partition_probe "$node")
ts=$(date -u +%FT%TZ)
if [ "$probe_res" = "ok" ]; then
failing=0
detail="warp egress restored for $node at $ts (probe $WARP_PROBE_URL ok)"
else
failing=1
detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED)"
fi
if injected "$cond"; then
failing=1
detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED) [INJECTED]"
fi
read -r action fails < <(state_machine "$cond" "$failing")
case "$action" in
ALERT_FIRST|ALERT_REALERT)
emit_record "ALERT" "$cond" "$detail" "$fails"
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
;;
RECOVERY)
emit_record "RECOVERY" "$cond" "$detail" "$fails"
;;
SUPPRESSED)
log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)"
;;
esac
done
# Recompute healthy agents in the main shell (pipeline subshell above can't export).
HEALTHY_AGENTS=""
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
[ -n "$node" ] && [ -n "$port" ] || continue
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
echo "$node"
fi
done > "$STATE_DIR/.healthy.tmp"
HEALTHY_AGENTS=$(tr '\n' ' ' < "$STATE_DIR/.healthy.tmp")
rm -f "$STATE_DIR/.healthy.tmp"
# Notify for this run's alerts (best effort). Skip entirely when nothing is healthy
# (notify needs a working browser via dm.py) or in dry-run.
if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then
while IFS= read -r line; do
# alerts.tmp format: "cond" or "cond|detail" (input_wait carries detail)
cond="${line%%|*}"
detail="${line#*|}"
[ "$detail" = "$line" ] && detail=""
[ -n "$cond" ] || continue
case "$cond" in
input_wait:*)
# Targeted: DM only the waiting agent, not a broadcast.
node="${cond#input_wait:}"
if [ -n "$detail" ]; then
notify_input_wait "$node" "$detail"
else
box_notify "$cond" "see #lobby for detail"
fi
;;
*)
box_notify "$cond" "see #lobby for detail"
;;
esac
done < "$STATE_DIR/.alerts.tmp"
elif [ -f "$STATE_DIR/.alerts.tmp" ]; then
log "no healthy agents — box notify skipped (DM path needs a working browser)"
fi
rm -f "$STATE_DIR/.alerts.tmp"
tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG"
log "check complete"
# Relay pending outbox records to #lobby with idempotency gates (posted watermark + content hash TTL)
if [ "$DRY_RUN" -eq 0 ] && [ -x "$BIN/fleet-alert-relay.sh" ]; then
"$BIN/fleet-alert-relay.sh" >> "$LOG" 2>&1 || true
fi