2026-10-04 20:09:36 +00:00
|
|
|
#!/bin/bash
|
|
|
|
|
# fleet-alert-check.sh — bl-side critical-condition detector for the fleet alerting pipeline.
|
|
|
|
|
#
|
|
|
|
|
# Closes the watchdog gap: agent-health.sh logs CRITICAL and restarts browsers,
|
|
|
|
|
# but NOTHING pages anyone. This script detects critical conditions, counts
|
|
|
|
|
# CONSECUTIVE failures, and emits alert records to an outbox that the
|
|
|
|
|
# container-side fleet-alert-relay hook picks up and pages to #lobby.
|
|
|
|
|
#
|
|
|
|
|
# Checks (bl-side only; VM-side checks live in the container relay):
|
|
|
|
|
# cdp:<node> headless Chromium CDP port not listening in the node's netns
|
|
|
|
|
# (catches zombie browsers: process alive, CDP not bound)
|
|
|
|
|
#
|
|
|
|
|
# Paging policy (env-overridable defaults — adjustable, not gates):
|
|
|
|
|
# FLEET_ALERT_THRESHOLD=2 consecutive failures before first page (~10 min at 5-min cadence)
|
|
|
|
|
# FLEET_ALERT_REALERT_MIN=30 re-page while still critical, at most every 30 min
|
|
|
|
|
# FLEET_ALERT_QUIET_HOURS="" e.g. "23:00-07:00" (bl local time); empty = page 24/7.
|
|
|
|
|
# First alert for a NEW incident always pages;
|
|
|
|
|
# quiet hours only suppress re-pages.
|
|
|
|
|
# FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify
|
|
|
|
|
# FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail
|
|
|
|
|
# (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip)
|
2026-10-07 00:25:51 +00:00
|
|
|
# FLEET_BL_RELAY=1 re-enable the bl-side #lobby relay (default 0/off:
|
|
|
|
|
# the container-side hook is the live pager; running
|
|
|
|
|
# both double-posts every alert — 2026-10-06)
|
2026-10-04 20:09:36 +00:00
|
|
|
#
|
|
|
|
|
# State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters)
|
|
|
|
|
# Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay)
|
|
|
|
|
# Log: /tmp/fleet-alert-check.log
|
|
|
|
|
#
|
|
|
|
|
# Alert delivery legs:
|
|
|
|
|
# 1. outbox record -> container relay -> signed #lobby post (primary page)
|
|
|
|
|
# 2. best-effort `box-ctl.py notify` to currently-healthy agents (DM path needs a
|
|
|
|
|
# working browser; failures are logged, never fatal)
|
|
|
|
|
#
|
|
|
|
|
# Installed as user timer fleet-alert-check.timer (every 5 min), mirroring agent-health.timer.
|
|
|
|
|
set -uo pipefail
|
|
|
|
|
|
|
|
|
|
THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}"
|
|
|
|
|
REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}"
|
2026-10-06 00:49:46 +00:00
|
|
|
# Approval/input-wait TTLs (seconds): conditions failing longer than this are
|
|
|
|
|
# auto-expired (input waits dismissed, key requests denied) instead of paging
|
|
|
|
|
# forever. Overridable per environment.
|
|
|
|
|
INPUT_WAIT_TTL="${FLEET_ALERT_INPUT_WAIT_TTL:-1800}"
|
|
|
|
|
BROWSER_APPROVAL_TTL="${FLEET_ALERT_BROWSER_APPROVAL_TTL:-1800}"
|
2026-10-04 20:09:36 +00:00
|
|
|
QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}"
|
|
|
|
|
DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}"
|
|
|
|
|
INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}"
|
|
|
|
|
STATE_DIR="${FLEET_ALERT_STATE_DIR:-$HOME/.local/share/fleet-alert}"
|
|
|
|
|
BIN="$(cd "$(dirname "$0")" && pwd)"
|
|
|
|
|
STATE="$STATE_DIR/state.json"
|
|
|
|
|
OUTBOX="$STATE_DIR/outbox.jsonl"
|
|
|
|
|
LOG="/tmp/fleet-alert-check.log"
|
|
|
|
|
|
|
|
|
|
mkdir -p "$STATE_DIR"
|
|
|
|
|
[ -f "$STATE" ] || echo '{}' > "$STATE"
|
|
|
|
|
touch "$OUTBOX"
|
|
|
|
|
|
|
|
|
|
NOW=$(date +%s)
|
|
|
|
|
log() { echo "$(date -Iseconds) $*" >> "$LOG"; }
|
|
|
|
|
|
|
|
|
|
# --- shared consecutive-failure state machine (also used by the container relay) ---
|
2026-10-06 00:49:46 +00:00
|
|
|
# usage: state_machine <cond> <failing 0|1> [ttl_seconds] -> prints "<ACTION> <fails>"
|
|
|
|
|
# When ttl_seconds > 0 and the condition has failed longer than the TTL,
|
|
|
|
|
# prints "EXPIRED <fails>" so the caller can auto-resolve (dismiss/deny).
|
|
|
|
|
# State entries track first_fail_ts (epoch of first consecutive failure).
|
|
|
|
|
# ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | EXPIRED | NONE
|
2026-10-04 20:09:36 +00:00
|
|
|
state_machine() {
|
2026-10-06 00:49:46 +00:00
|
|
|
local cond="$1" failing="$2" ttl="${3:-0}"
|
2026-10-04 20:09:36 +00:00
|
|
|
THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \
|
2026-10-06 00:49:46 +00:00
|
|
|
FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" "$ttl" <<'PYEOF'
|
2026-10-04 20:09:36 +00:00
|
|
|
import json, os, sys, time
|
|
|
|
|
state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3]
|
2026-10-06 00:49:46 +00:00
|
|
|
ttl_seconds = int(sys.argv[4]) if len(sys.argv) > 4 else 0
|
2026-10-04 20:09:36 +00:00
|
|
|
failing = failing_s == "1"
|
|
|
|
|
threshold = int(os.environ.get("THRESHOLD", "2"))
|
|
|
|
|
realert_min = int(os.environ.get("REALERT_MIN", "30"))
|
|
|
|
|
qh = os.environ.get("QUIET_HOURS", "")
|
|
|
|
|
dry = os.environ.get("FLEET_ALERT_DRY_RUN") == "1"
|
|
|
|
|
now = int(time.time())
|
|
|
|
|
def in_quiet(spec):
|
|
|
|
|
if not spec:
|
|
|
|
|
return False
|
|
|
|
|
try:
|
|
|
|
|
a, b = spec.split("-")
|
|
|
|
|
def m(s):
|
|
|
|
|
h, mi = s.split(":")
|
|
|
|
|
return int(h) * 60 + int(mi)
|
|
|
|
|
cur = time.localtime().tm_hour * 60 + time.localtime().tm_min
|
|
|
|
|
s, e = m(a), m(b)
|
|
|
|
|
return (s <= cur < e) if s <= e else (cur >= s or cur < e)
|
|
|
|
|
except Exception:
|
|
|
|
|
return False
|
|
|
|
|
try:
|
|
|
|
|
st = json.load(open(state_path))
|
|
|
|
|
except Exception:
|
|
|
|
|
st = {}
|
|
|
|
|
e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0}
|
|
|
|
|
action = "NONE"
|
|
|
|
|
if failing:
|
2026-10-06 00:49:46 +00:00
|
|
|
if int(e.get("fails", 0)) == 0:
|
|
|
|
|
e["first_fail_ts"] = now
|
2026-10-04 20:09:36 +00:00
|
|
|
e["fails"] = int(e.get("fails", 0)) + 1
|
2026-10-06 00:49:46 +00:00
|
|
|
# TTL expiry: failing longer than ttl_seconds -> EXPIRED (caller auto-resolves)
|
|
|
|
|
if ttl_seconds > 0 and now - int(e.get("first_fail_ts", now)) >= ttl_seconds:
|
|
|
|
|
action = "EXPIRED"
|
|
|
|
|
# Reset so a fresh incident starts clean after the caller resolves it
|
|
|
|
|
e["fails"] = 0
|
|
|
|
|
e["alerted"] = False
|
|
|
|
|
e.pop("first_fail_ts", None)
|
|
|
|
|
else:
|
|
|
|
|
due = e["fails"] >= threshold and (
|
|
|
|
|
not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60
|
|
|
|
|
)
|
|
|
|
|
if due:
|
|
|
|
|
first = not e.get("alerted")
|
|
|
|
|
if not first and in_quiet(qh):
|
|
|
|
|
action = "SUPPRESSED"
|
|
|
|
|
else:
|
|
|
|
|
action = "ALERT_FIRST" if first else "ALERT_REALERT"
|
|
|
|
|
e["alerted"] = True
|
|
|
|
|
e["last_alert_ts"] = now
|
2026-10-04 20:09:36 +00:00
|
|
|
else:
|
|
|
|
|
if e.get("alerted"):
|
|
|
|
|
action = "RECOVERY"
|
|
|
|
|
e["fails"] = 0
|
|
|
|
|
e["alerted"] = False
|
2026-10-06 00:49:46 +00:00
|
|
|
e.pop("first_fail_ts", None)
|
2026-10-04 20:09:36 +00:00
|
|
|
st[cond] = e
|
|
|
|
|
if not dry:
|
|
|
|
|
json.dump(st, open(state_path, "w"))
|
|
|
|
|
print(action, e["fails"])
|
|
|
|
|
PYEOF
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
emit_record() { # kind cond detail consecutive
|
|
|
|
|
local kind="$1" cond="$2" detail="$3" consec="$4"
|
|
|
|
|
local id
|
|
|
|
|
id=$(tr -d '-' < /proc/sys/kernel/random/uuid | cut -c1-12)
|
|
|
|
|
local rec
|
|
|
|
|
rec=$(python3 -c 'import json,sys; print(json.dumps({"id":sys.argv[1],"ts":int(sys.argv[2]),"kind":sys.argv[3],"source":"bl","condition":sys.argv[4],"detail":sys.argv[5],"consecutive":int(sys.argv[6])}))' \
|
|
|
|
|
"$id" "$NOW" "$kind" "$cond" "$detail" "$consec")
|
|
|
|
|
if [ "$DRY_RUN" = "1" ]; then
|
|
|
|
|
log "DRY-RUN would emit: $rec"
|
|
|
|
|
else
|
|
|
|
|
echo "$rec" >> "$OUTBOX"
|
|
|
|
|
log "emitted $kind $cond x$consec"
|
|
|
|
|
fi
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
box_notify() {
|
|
|
|
|
# Best-effort DM to healthy agents via box-ctl. Never fatal.
|
|
|
|
|
# Fan-out runs in parallel with a per-notify timeout so one hung DM
|
|
|
|
|
# path can't stall the 5-minute check loop.
|
|
|
|
|
local cond="$1"
|
|
|
|
|
local detail="$2"
|
|
|
|
|
local msg="[fleet-alert] CRITICAL ${cond}: ${detail}"
|
|
|
|
|
msg="${msg:0:240}"
|
|
|
|
|
local agent pids=""
|
|
|
|
|
for agent in $HEALTHY_AGENTS; do
|
|
|
|
|
if [ "$DRY_RUN" = "1" ]; then
|
|
|
|
|
log "DRY-RUN would notify $agent"
|
|
|
|
|
continue
|
|
|
|
|
fi
|
|
|
|
|
(
|
|
|
|
|
if timeout 60 python3 "$BIN/box-ctl.py" notify "$agent" "$msg" >/dev/null 2>&1; then
|
|
|
|
|
log "notified $agent re $cond"
|
|
|
|
|
else
|
|
|
|
|
log "notify $agent failed re $cond (best-effort)"
|
|
|
|
|
fi
|
|
|
|
|
) &
|
|
|
|
|
pids="$pids $!"
|
|
|
|
|
done
|
|
|
|
|
local p
|
|
|
|
|
for p in $pids; do wait "$p" 2>/dev/null; done
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-06 00:49:46 +00:00
|
|
|
notify_input_wait() {
|
|
|
|
|
# Targeted DM for input waits (2026-10-05): DM ONLY the specific agent
|
|
|
|
|
# whose session is waiting for human input -- not a broadcast to all
|
|
|
|
|
# healthy agents. The #lobby post still fires via the relay leg for
|
|
|
|
|
# human visibility; this DM ensures the responsible operator sees it
|
|
|
|
|
# in their sidechat without digging through lobby noise.
|
|
|
|
|
# Best-effort: never fatal to the 5-minute check loop.
|
|
|
|
|
local node="$1"
|
|
|
|
|
local detail="$2"
|
|
|
|
|
local msg="[fleet-alert] INPUT WAIT: ${detail} -- reply: box approval reply ${node} \"<msg>\" or box notify ${node} \"<msg>\""
|
|
|
|
|
msg="${msg:0:900}"
|
|
|
|
|
if [ "$DRY_RUN" = "1" ]; then
|
|
|
|
|
log "DRY-RUN would DM $node re input_wait"
|
|
|
|
|
return 0
|
|
|
|
|
fi
|
|
|
|
|
if timeout 60 python3 "$BIN/box-ctl.py" notify "$node" "$msg" >/dev/null 2>&1; then
|
|
|
|
|
log "input_wait targeted DM sent to $node"
|
|
|
|
|
else
|
|
|
|
|
log "input_wait DM to $node failed (best-effort, non-fatal)"
|
|
|
|
|
fi
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-04 20:09:36 +00:00
|
|
|
injected() { # cond -> 0 if injected-fail
|
|
|
|
|
case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
HEALTHY_AGENTS=""
|
|
|
|
|
|
|
|
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
|
|
|
[ -n "$node" ] && [ -n "$port" ] || continue
|
|
|
|
|
cond="cdp:$node"
|
|
|
|
|
procs=$(pgrep -f "chromium.*profiles/$node" 2>/dev/null | wc -l)
|
|
|
|
|
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
|
|
|
|
|
failing=0
|
|
|
|
|
HEALTHY_AGENTS="$HEALTHY_AGENTS $node"
|
|
|
|
|
else
|
|
|
|
|
failing=1
|
|
|
|
|
fi
|
|
|
|
|
injected "$cond" && failing=1
|
|
|
|
|
detail="CDP $port not listening in netns warp-$node (chromium procs=$procs)"
|
|
|
|
|
# NOTE: HEALTHY_AGENTS set inside the pipeline subshell is lost; recompute below.
|
|
|
|
|
read -r action fails < <(state_machine "$cond" "$failing")
|
|
|
|
|
case "$action" in
|
|
|
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
|
|
|
emit_record "ALERT" "$cond" "$detail" "$fails"
|
|
|
|
|
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
|
|
|
|
;;
|
|
|
|
|
RECOVERY)
|
|
|
|
|
emit_record "RECOVERY" "$cond" "CDP $port listening again in netns warp-$node" "$fails"
|
|
|
|
|
;;
|
|
|
|
|
SUPPRESSED)
|
|
|
|
|
log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)"
|
|
|
|
|
;;
|
|
|
|
|
esac
|
|
|
|
|
done
|
|
|
|
|
|
2026-10-05 20:18:47 +00:00
|
|
|
# --- Agent approval blockage & input wait detection ---
|
2026-10-05 17:42:01 +00:00
|
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
|
|
|
[ -n "$node" ] || continue
|
2026-10-05 20:18:47 +00:00
|
|
|
node_data=$(python3 -c "
|
|
|
|
|
import sys, json
|
2026-10-05 17:42:01 +00:00
|
|
|
sys.path.insert(0, '$BIN')
|
|
|
|
|
import approvals
|
|
|
|
|
info = approvals.inspect_node_approvals('$node')
|
2026-10-05 20:18:47 +00:00
|
|
|
out = {
|
|
|
|
|
'has_pending': info.get('has_pending', False),
|
|
|
|
|
'target': info.get('target') or info.get('ip') or 'unknown',
|
|
|
|
|
'title': info.get('title') or '',
|
|
|
|
|
'waits': info.get('input_waits') or []
|
|
|
|
|
}
|
|
|
|
|
print(json.dumps(out))
|
|
|
|
|
" 2>/dev/null || echo '{"has_pending":false,"target":"unknown","title":"","waits":[]}')
|
2026-10-05 17:42:01 +00:00
|
|
|
|
2026-10-05 20:18:47 +00:00
|
|
|
# 1. Egress permission dialog
|
|
|
|
|
cond="approval:$node"
|
|
|
|
|
has_pending=$(python3 -c "import json,sys; print(1 if json.loads(sys.argv[1]).get('has_pending') else 0)" "$node_data" 2>/dev/null || echo 0)
|
|
|
|
|
if [ "$has_pending" = "1" ]; then
|
2026-10-05 17:42:01 +00:00
|
|
|
failing=1
|
2026-10-05 20:18:47 +00:00
|
|
|
target=$(python3 -c "import json,sys; print(json.loads(sys.argv[1]).get('target','unknown'))" "$node_data" 2>/dev/null || echo unknown)
|
2026-10-05 17:42:01 +00:00
|
|
|
detail="Agent $node held up on browser approval for $target"
|
|
|
|
|
else
|
|
|
|
|
failing=0
|
|
|
|
|
detail="Agent $node approvals clear"
|
|
|
|
|
fi
|
|
|
|
|
injected "$cond" && failing=1
|
2026-10-06 00:49:46 +00:00
|
|
|
read -r action fails < <(state_machine "$cond" "$failing" "$BROWSER_APPROVAL_TTL")
|
2026-10-05 17:42:01 +00:00
|
|
|
case "$action" in
|
|
|
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
|
|
|
emit_record "ALERT" "$cond" "$detail" "$fails"
|
|
|
|
|
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
|
|
|
|
;;
|
|
|
|
|
RECOVERY)
|
|
|
|
|
emit_record "RECOVERY" "$cond" "$detail" "$fails"
|
|
|
|
|
;;
|
|
|
|
|
SUPPRESSED)
|
|
|
|
|
log "$cond still critical x$fails — re-page suppressed"
|
|
|
|
|
;;
|
2026-10-06 00:49:46 +00:00
|
|
|
EXPIRED)
|
|
|
|
|
# Browser approval dialog exceeded BROWSER_APPROVAL_TTL without a
|
|
|
|
|
# human decision: fail closed by denying it.
|
|
|
|
|
log "$cond EXPIRED after ${BROWSER_APPROVAL_TTL}s without human decision — auto-denying (fail closed)"
|
|
|
|
|
python3 - "$node" <<'PYEOF3'
|
|
|
|
|
import sys, json
|
|
|
|
|
sys.path.insert(0, "/home/super/Projects/NetVM/bin")
|
|
|
|
|
import approvals
|
|
|
|
|
node = sys.argv[1]
|
|
|
|
|
print(json.dumps(approvals.deny_node_approval(node, caller="approval-ttl-expire")))
|
|
|
|
|
approvals.log_box_ctl("approval-expired", name=node, caller="approval-ttl-expire",
|
|
|
|
|
extra={"note": "browser approval TTL elapsed; auto-denied (fail closed)"})
|
|
|
|
|
PYEOF3
|
|
|
|
|
emit_record "RECOVERY" "$cond" "Agent $node browser approval expired after ${BROWSER_APPROVAL_TTL}s; auto-denied" "$fails"
|
|
|
|
|
;;
|
2026-10-05 17:42:01 +00:00
|
|
|
esac
|
2026-10-05 20:18:47 +00:00
|
|
|
|
|
|
|
|
# 2. Sidebar task waiting on human input
|
|
|
|
|
cond_in="input_wait:$node"
|
|
|
|
|
wait_summary=$(python3 -c "
|
|
|
|
|
import json,sys
|
|
|
|
|
w = json.loads(sys.argv[1]).get('waits', [])
|
|
|
|
|
if w:
|
|
|
|
|
print('; '.join(f\"{item.get('task')}: {item.get('status')}\" for item in w)[:120])
|
|
|
|
|
" "$node_data" 2>/dev/null || true)
|
|
|
|
|
|
|
|
|
|
if [ -n "$wait_summary" ]; then
|
|
|
|
|
failing_in=1
|
|
|
|
|
detail_in="Agent $node task waiting for human input: $wait_summary"
|
|
|
|
|
else
|
|
|
|
|
failing_in=0
|
|
|
|
|
detail_in="Agent $node tasks running"
|
|
|
|
|
fi
|
|
|
|
|
injected "$cond_in" && failing_in=1
|
2026-10-06 00:49:46 +00:00
|
|
|
read -r action_in fails_in < <(state_machine "$cond_in" "$failing_in" "$INPUT_WAIT_TTL")
|
2026-10-05 20:18:47 +00:00
|
|
|
case "$action_in" in
|
|
|
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
|
|
|
emit_record "ALERT" "$cond_in" "$detail_in" "$fails_in"
|
2026-10-06 00:49:46 +00:00
|
|
|
echo "$cond_in|$detail_in" >> "$STATE_DIR/.alerts.tmp"
|
2026-10-05 20:18:47 +00:00
|
|
|
;;
|
|
|
|
|
RECOVERY)
|
|
|
|
|
emit_record "RECOVERY" "$cond_in" "$detail_in" "$fails_in"
|
|
|
|
|
;;
|
|
|
|
|
SUPPRESSED)
|
|
|
|
|
log "$cond_in still critical x$fails_in — re-page suppressed"
|
|
|
|
|
;;
|
2026-10-06 00:49:46 +00:00
|
|
|
EXPIRED)
|
|
|
|
|
# Input wait exceeded INPUT_WAIT_TTL without human response:
|
|
|
|
|
# auto-dismiss so the agent unblocks. Log the expiry and emit a
|
|
|
|
|
# RECOVERY record (the wait is gone, not merely un-paged).
|
|
|
|
|
log "$cond_in EXPIRED after ${INPUT_WAIT_TTL}s without human input — auto-dismissing"
|
|
|
|
|
python3 - "$node" <<'PYEOF2'
|
|
|
|
|
import sys
|
|
|
|
|
sys.path.insert(0, "/home/super/Projects/NetVM/bin")
|
|
|
|
|
import approvals, json
|
|
|
|
|
node = sys.argv[1]
|
|
|
|
|
info = approvals.inspect_node_approvals(node)
|
|
|
|
|
for w in info.get("input_waits", []) or []:
|
|
|
|
|
t = w.get("task")
|
|
|
|
|
if t:
|
|
|
|
|
approvals.mark_wait_responded(node, t, caller="approval-ttl-expire")
|
|
|
|
|
approvals.log_box_ctl("approval-wait-expired", name=node, caller="approval-ttl-expire",
|
|
|
|
|
extra={"note": "input wait TTL elapsed; auto-dismissed"})
|
|
|
|
|
print(json.dumps(approvals.dismiss_node_task(node, caller="approval-ttl-expire")))
|
|
|
|
|
PYEOF2
|
|
|
|
|
emit_record "RECOVERY" "$cond_in" "Agent $node input wait expired after ${INPUT_WAIT_TTL}s; auto-dismissed" "$fails_in"
|
|
|
|
|
;;
|
2026-10-05 20:18:47 +00:00
|
|
|
esac
|
2026-10-05 17:42:01 +00:00
|
|
|
done
|
|
|
|
|
|
2026-10-04 22:54:01 +00:00
|
|
|
# --- Warp partition detection (2026-10-04) ---
|
|
|
|
|
# A partitioned node has a live browser + CDP but no internet egress: the
|
|
|
|
|
# chromebox watchdog sees a healthy browser while all automation fails.
|
|
|
|
|
# Condition id: partition:<node>. The detail names the node, the WireGuard
|
|
|
|
|
# handshake age, the egress probe result, and the timestamp, and says
|
|
|
|
|
# PARTITION explicitly so #lobby readers can tell a network partition from
|
|
|
|
|
# a browser crash at a glance. Anti-spam comes from the shared consecutive-
|
|
|
|
|
# failure state machine (2 consecutive failures before first page, re-page
|
|
|
|
|
# at most every 30 min).
|
|
|
|
|
WARP_PROBE_URL="${WARP_PROBE_URL:-https://1.1.1.1/cdn-cgi/trace}"
|
|
|
|
|
warp_partition_probe() { # <node> -> prints "<handshake_age_s|unknown> <ok|FAIL>"
|
|
|
|
|
local node="$1" iface epoch now age_s
|
|
|
|
|
iface=$(sudo -n ip netns exec "warp-$node" sh -c 'wg show interfaces 2>/dev/null | head -1')
|
|
|
|
|
now=$(date +%s)
|
|
|
|
|
epoch=$(sudo -n ip netns exec "warp-$node" wg show "$iface" latest-handshakes 2>/dev/null | awk '{print $2}')
|
|
|
|
|
case "$epoch" in ''|*[!0-9]*) age_s="unknown" ;; *) age_s=$(( now - epoch )) ;; esac
|
|
|
|
|
if sudo -n ip netns exec "warp-$node" curl -s -m 8 -o /dev/null "$WARP_PROBE_URL" 2>/dev/null; then
|
|
|
|
|
echo "$age_s ok"
|
|
|
|
|
else
|
|
|
|
|
echo "$age_s FAIL"
|
|
|
|
|
fi
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
|
|
|
[ -n "$node" ] && [ -n "$port" ] || continue
|
|
|
|
|
cond="partition:$node"
|
|
|
|
|
read -r hs_age probe_res < <(warp_partition_probe "$node")
|
|
|
|
|
ts=$(date -u +%FT%TZ)
|
|
|
|
|
if [ "$probe_res" = "ok" ]; then
|
|
|
|
|
failing=0
|
|
|
|
|
detail="warp egress restored for $node at $ts (probe $WARP_PROBE_URL ok)"
|
|
|
|
|
else
|
|
|
|
|
failing=1
|
|
|
|
|
detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED)"
|
|
|
|
|
fi
|
|
|
|
|
if injected "$cond"; then
|
|
|
|
|
failing=1
|
|
|
|
|
detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED) [INJECTED]"
|
|
|
|
|
fi
|
|
|
|
|
read -r action fails < <(state_machine "$cond" "$failing")
|
|
|
|
|
case "$action" in
|
|
|
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
|
|
|
emit_record "ALERT" "$cond" "$detail" "$fails"
|
|
|
|
|
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
|
|
|
|
;;
|
|
|
|
|
RECOVERY)
|
|
|
|
|
emit_record "RECOVERY" "$cond" "$detail" "$fails"
|
|
|
|
|
;;
|
|
|
|
|
SUPPRESSED)
|
|
|
|
|
log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)"
|
|
|
|
|
;;
|
|
|
|
|
esac
|
|
|
|
|
done
|
|
|
|
|
|
2026-10-04 20:09:36 +00:00
|
|
|
# Recompute healthy agents in the main shell (pipeline subshell above can't export).
|
|
|
|
|
HEALTHY_AGENTS=""
|
|
|
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
|
|
|
[ -n "$node" ] && [ -n "$port" ] || continue
|
|
|
|
|
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
|
|
|
|
|
echo "$node"
|
|
|
|
|
fi
|
|
|
|
|
done > "$STATE_DIR/.healthy.tmp"
|
|
|
|
|
HEALTHY_AGENTS=$(tr '\n' ' ' < "$STATE_DIR/.healthy.tmp")
|
|
|
|
|
rm -f "$STATE_DIR/.healthy.tmp"
|
|
|
|
|
|
|
|
|
|
# Notify for this run's alerts (best effort). Skip entirely when nothing is healthy
|
|
|
|
|
# (notify needs a working browser via dm.py) or in dry-run.
|
|
|
|
|
if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then
|
2026-10-06 00:49:46 +00:00
|
|
|
while IFS= read -r line; do
|
|
|
|
|
# alerts.tmp format: "cond" or "cond|detail" (input_wait carries detail)
|
|
|
|
|
cond="${line%%|*}"
|
|
|
|
|
detail="${line#*|}"
|
|
|
|
|
[ "$detail" = "$line" ] && detail=""
|
|
|
|
|
[ -n "$cond" ] || continue
|
|
|
|
|
case "$cond" in
|
|
|
|
|
input_wait:*)
|
|
|
|
|
# Targeted: DM only the waiting agent, not a broadcast.
|
|
|
|
|
node="${cond#input_wait:}"
|
|
|
|
|
if [ -n "$detail" ]; then
|
|
|
|
|
notify_input_wait "$node" "$detail"
|
|
|
|
|
else
|
|
|
|
|
box_notify "$cond" "see #lobby for detail"
|
|
|
|
|
fi
|
|
|
|
|
;;
|
|
|
|
|
*)
|
|
|
|
|
box_notify "$cond" "see #lobby for detail"
|
|
|
|
|
;;
|
|
|
|
|
esac
|
2026-10-04 20:09:36 +00:00
|
|
|
done < "$STATE_DIR/.alerts.tmp"
|
|
|
|
|
elif [ -f "$STATE_DIR/.alerts.tmp" ]; then
|
|
|
|
|
log "no healthy agents — box notify skipped (DM path needs a working browser)"
|
|
|
|
|
fi
|
|
|
|
|
rm -f "$STATE_DIR/.alerts.tmp"
|
|
|
|
|
|
|
|
|
|
tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG"
|
|
|
|
|
log "check complete"
|
2026-10-05 17:41:04 +00:00
|
|
|
|
2026-10-07 00:25:51 +00:00
|
|
|
# Bl-side #lobby relay: DISABLED by default (FLEET_BL_RELAY=1 to re-enable).
|
|
|
|
|
# The container-side hook is the live pager; the bl relay never successfully
|
|
|
|
|
# posted (missing CHAT_KEYFILE) and enabling it now would double-post every
|
|
|
|
|
# alert in a second format. Re-enable only alongside retiring the container
|
|
|
|
|
# hook (and per the relay header, with opm sign-off).
|
|
|
|
|
if [ "${FLEET_BL_RELAY:-0}" = "1" ] && [ "$DRY_RUN" -eq 0 ] && [ -x "$BIN/fleet-alert-relay.sh" ]; then
|
2026-10-05 17:41:04 +00:00
|
|
|
"$BIN/fleet-alert-relay.sh" >> "$LOG" 2>&1 || true
|
|
|
|
|
fi
|