From 5ab157b3b09b8793a50009e080d2553e0cfeb73b Mon Sep 17 00:00:00 2001 From: operator-main Date: Sun, 4 Oct 2026 20:09:36 +0000 Subject: [PATCH] fleet alerting: fix netns naming in agent-health.sh + add fleet-alert-check.sh agent-health.sh used bare node names for 'ip netns exec' but netns are named warp- since the NetVM layout; the 6189793 CDP-liveness check always failed ('No such file or directory'), logging false CRITICALs and kill -9'ing healthy browsers every 5 min. Use warp-$node. fleet-alert-check.sh: new 5-min critical-condition detector (per-node CDP liveness via warp- netns). Consecutive-failure state machine: page after 2 consecutive failures, re-page every 30 min while critical, quiet-hours-aware (first alert always pages). Emits ALERT/RECOVERY records to ~/.local/share/fleet-alert/outbox.jsonl for the container fleet-alert-relay hook; best-effort box-ctl notify to healthy agents. Session: sidechat/critical-alerting-pipeline --- bin/agent-health.sh | 4 +- bin/fleet-alert-check.sh | 211 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 213 insertions(+), 2 deletions(-) create mode 100755 bin/fleet-alert-check.sh diff --git a/bin/agent-health.sh b/bin/agent-health.sh index 9f2a419..2ff56f7 100755 --- a/bin/agent-health.sh +++ b/bin/agent-health.sh @@ -21,7 +21,7 @@ check_cdp_port() { # A browser can be running but not bound to CDP (zombie state). local node=$1 local cdp_port=$2 - if sudo ip netns exec "$node" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then + if sudo ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then return 0 else return 1 @@ -61,7 +61,7 @@ restart_browser() { done sleep 3 # Verify port is free before restart - if sudo ip netns exec "$agent" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then + if sudo ip netns exec "warp-$agent" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then echo "$(date -Iseconds) $agent: WARNING - port $cdp_port still bound after kill" >> "$LOG" fi # Restart via netvm-chrome.sh in its own systemd scope. diff --git a/bin/fleet-alert-check.sh b/bin/fleet-alert-check.sh new file mode 100755 index 0000000..39af8b0 --- /dev/null +++ b/bin/fleet-alert-check.sh @@ -0,0 +1,211 @@ +#!/bin/bash +# fleet-alert-check.sh — bl-side critical-condition detector for the fleet alerting pipeline. +# +# Closes the watchdog gap: agent-health.sh logs CRITICAL and restarts browsers, +# but NOTHING pages anyone. This script detects critical conditions, counts +# CONSECUTIVE failures, and emits alert records to an outbox that the +# container-side fleet-alert-relay hook picks up and pages to #lobby. +# +# Checks (bl-side only; VM-side checks live in the container relay): +# cdp: headless Chromium CDP port not listening in the node's netns +# (catches zombie browsers: process alive, CDP not bound) +# +# Paging policy (env-overridable defaults — adjustable, not gates): +# FLEET_ALERT_THRESHOLD=2 consecutive failures before first page (~10 min at 5-min cadence) +# FLEET_ALERT_REALERT_MIN=30 re-page while still critical, at most every 30 min +# FLEET_ALERT_QUIET_HOURS="" e.g. "23:00-07:00" (bl local time); empty = page 24/7. +# First alert for a NEW incident always pages; +# quiet hours only suppress re-pages. +# FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify +# FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail +# (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip) +# +# State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters) +# Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay) +# Log: /tmp/fleet-alert-check.log +# +# Alert delivery legs: +# 1. outbox record -> container relay -> signed #lobby post (primary page) +# 2. best-effort `box-ctl.py notify` to currently-healthy agents (DM path needs a +# working browser; failures are logged, never fatal) +# +# Installed as user timer fleet-alert-check.timer (every 5 min), mirroring agent-health.timer. +set -uo pipefail + +THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}" +REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}" +QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}" +DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}" +INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}" +STATE_DIR="${FLEET_ALERT_STATE_DIR:-$HOME/.local/share/fleet-alert}" +BIN="$(cd "$(dirname "$0")" && pwd)" +STATE="$STATE_DIR/state.json" +OUTBOX="$STATE_DIR/outbox.jsonl" +LOG="/tmp/fleet-alert-check.log" + +mkdir -p "$STATE_DIR" +[ -f "$STATE" ] || echo '{}' > "$STATE" +touch "$OUTBOX" + +NOW=$(date +%s) +log() { echo "$(date -Iseconds) $*" >> "$LOG"; } + +# --- shared consecutive-failure state machine (also used by the container relay) --- +# usage: state_machine -> prints " " +# ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | NONE +state_machine() { + local cond="$1" failing="$2" + THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \ + FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" <<'PYEOF' +import json, os, sys, time +state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3] +failing = failing_s == "1" +threshold = int(os.environ.get("THRESHOLD", "2")) +realert_min = int(os.environ.get("REALERT_MIN", "30")) +qh = os.environ.get("QUIET_HOURS", "") +dry = os.environ.get("FLEET_ALERT_DRY_RUN") == "1" +now = int(time.time()) +def in_quiet(spec): + if not spec: + return False + try: + a, b = spec.split("-") + def m(s): + h, mi = s.split(":") + return int(h) * 60 + int(mi) + cur = time.localtime().tm_hour * 60 + time.localtime().tm_min + s, e = m(a), m(b) + return (s <= cur < e) if s <= e else (cur >= s or cur < e) + except Exception: + return False +try: + st = json.load(open(state_path)) +except Exception: + st = {} +e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0} +action = "NONE" +if failing: + e["fails"] = int(e.get("fails", 0)) + 1 + due = e["fails"] >= threshold and ( + not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60 + ) + if due: + first = not e.get("alerted") + if not first and in_quiet(qh): + action = "SUPPRESSED" + else: + action = "ALERT_FIRST" if first else "ALERT_REALERT" + e["alerted"] = True + e["last_alert_ts"] = now +else: + if e.get("alerted"): + action = "RECOVERY" + e["fails"] = 0 + e["alerted"] = False +st[cond] = e +if not dry: + json.dump(st, open(state_path, "w")) +print(action, e["fails"]) +PYEOF +} + +emit_record() { # kind cond detail consecutive + local kind="$1" cond="$2" detail="$3" consec="$4" + local id + id=$(tr -d '-' < /proc/sys/kernel/random/uuid | cut -c1-12) + local rec + rec=$(python3 -c 'import json,sys; print(json.dumps({"id":sys.argv[1],"ts":int(sys.argv[2]),"kind":sys.argv[3],"source":"bl","condition":sys.argv[4],"detail":sys.argv[5],"consecutive":int(sys.argv[6])}))' \ + "$id" "$NOW" "$kind" "$cond" "$detail" "$consec") + if [ "$DRY_RUN" = "1" ]; then + log "DRY-RUN would emit: $rec" + else + echo "$rec" >> "$OUTBOX" + log "emitted $kind $cond x$consec" + fi +} + +box_notify() { + # Best-effort DM to healthy agents via box-ctl. Never fatal. + # Fan-out runs in parallel with a per-notify timeout so one hung DM + # path can't stall the 5-minute check loop. + local cond="$1" + local detail="$2" + local msg="[fleet-alert] CRITICAL ${cond}: ${detail}" + msg="${msg:0:240}" + local agent pids="" + for agent in $HEALTHY_AGENTS; do + if [ "$DRY_RUN" = "1" ]; then + log "DRY-RUN would notify $agent" + continue + fi + ( + if timeout 60 python3 "$BIN/box-ctl.py" notify "$agent" "$msg" >/dev/null 2>&1; then + log "notified $agent re $cond" + else + log "notify $agent failed re $cond (best-effort)" + fi + ) & + pids="$pids $!" + done + local p + for p in $pids; do wait "$p" 2>/dev/null; done +} + +injected() { # cond -> 0 if injected-fail + case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac +} + +HEALTHY_AGENTS="" + +"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do + [ -n "$node" ] && [ -n "$port" ] || continue + cond="cdp:$node" + procs=$(pgrep -f "chromium.*profiles/$node" 2>/dev/null | wc -l) + if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then + failing=0 + HEALTHY_AGENTS="$HEALTHY_AGENTS $node" + else + failing=1 + fi + injected "$cond" && failing=1 + detail="CDP $port not listening in netns warp-$node (chromium procs=$procs)" + # NOTE: HEALTHY_AGENTS set inside the pipeline subshell is lost; recompute below. + read -r action fails < <(state_machine "$cond" "$failing") + case "$action" in + ALERT_FIRST|ALERT_REALERT) + emit_record "ALERT" "$cond" "$detail" "$fails" + echo "$cond" >> "$STATE_DIR/.alerts.tmp" + ;; + RECOVERY) + emit_record "RECOVERY" "$cond" "CDP $port listening again in netns warp-$node" "$fails" + ;; + SUPPRESSED) + log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)" + ;; + esac +done + +# Recompute healthy agents in the main shell (pipeline subshell above can't export). +HEALTHY_AGENTS="" +"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do + [ -n "$node" ] && [ -n "$port" ] || continue + if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then + echo "$node" + fi +done > "$STATE_DIR/.healthy.tmp" +HEALTHY_AGENTS=$(tr '\n' ' ' < "$STATE_DIR/.healthy.tmp") +rm -f "$STATE_DIR/.healthy.tmp" + +# Notify for this run's alerts (best effort). Skip entirely when nothing is healthy +# (notify needs a working browser via dm.py) or in dry-run. +if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then + while read -r cond; do + [ -n "$cond" ] && box_notify "$cond" "see #lobby for detail" + done < "$STATE_DIR/.alerts.tmp" +elif [ -f "$STATE_DIR/.alerts.tmp" ]; then + log "no healthy agents — box notify skipped (DM path needs a working browser)" +fi +rm -f "$STATE_DIR/.alerts.tmp" + +tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG" +log "check complete"