adfcd2e602
- box passkey [show|fetch] (+ muse passkey): documents VM-only passkey (/srv/box/passkey.txt, fallback /etc/netvm/passkey.txt on 34.139.37.135), probes VM over SSH with graceful fallback; --json supported. No secrets on bl. - approvals: request_key_approval / check_node_key_request; KEY_APPROVAL status surfaced in `box approvals check`; allow/deny resolve + audit to box-ctl.jsonl; never auto-approved. New `box approvals request-key <node> --reason`. - box lookup (summary|fleet|threads|unread|approvals|key|docs) and docs-lookup engine with lookup_internal/ database (docs_internal symlink). - muse-tmux: non-TTY attach falls back to scrollback capture; prune NameError fix. - box/muse passthrough for tmux/muse/docs; thread list/view alias + prefix resolve. - Docs: AGENTS.md, AGENT-TOOLING.md, BOX-WEB-SURFACE-GUIDE.md, README. - Tests: key-approval + passkey tests; sync stale sidechat UUIDs and manifest name. - .gitignore runtime trackers (subagent-sessions, conversation-nudge-tracker).
346 lines
13 KiB
Bash
Executable File
346 lines
13 KiB
Bash
Executable File
#!/bin/bash
|
|
# fleet-alert-check.sh — bl-side critical-condition detector for the fleet alerting pipeline.
|
|
#
|
|
# Closes the watchdog gap: agent-health.sh logs CRITICAL and restarts browsers,
|
|
# but NOTHING pages anyone. This script detects critical conditions, counts
|
|
# CONSECUTIVE failures, and emits alert records to an outbox that the
|
|
# container-side fleet-alert-relay hook picks up and pages to #lobby.
|
|
#
|
|
# Checks (bl-side only; VM-side checks live in the container relay):
|
|
# cdp:<node> headless Chromium CDP port not listening in the node's netns
|
|
# (catches zombie browsers: process alive, CDP not bound)
|
|
#
|
|
# Paging policy (env-overridable defaults — adjustable, not gates):
|
|
# FLEET_ALERT_THRESHOLD=2 consecutive failures before first page (~10 min at 5-min cadence)
|
|
# FLEET_ALERT_REALERT_MIN=30 re-page while still critical, at most every 30 min
|
|
# FLEET_ALERT_QUIET_HOURS="" e.g. "23:00-07:00" (bl local time); empty = page 24/7.
|
|
# First alert for a NEW incident always pages;
|
|
# quiet hours only suppress re-pages.
|
|
# FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify
|
|
# FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail
|
|
# (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip)
|
|
#
|
|
# State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters)
|
|
# Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay)
|
|
# Log: /tmp/fleet-alert-check.log
|
|
#
|
|
# Alert delivery legs:
|
|
# 1. outbox record -> container relay -> signed #lobby post (primary page)
|
|
# 2. best-effort `box-ctl.py notify` to currently-healthy agents (DM path needs a
|
|
# working browser; failures are logged, never fatal)
|
|
#
|
|
# Installed as user timer fleet-alert-check.timer (every 5 min), mirroring agent-health.timer.
|
|
set -uo pipefail
|
|
|
|
THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}"
|
|
REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}"
|
|
QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}"
|
|
DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}"
|
|
INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}"
|
|
STATE_DIR="${FLEET_ALERT_STATE_DIR:-$HOME/.local/share/fleet-alert}"
|
|
BIN="$(cd "$(dirname "$0")" && pwd)"
|
|
STATE="$STATE_DIR/state.json"
|
|
OUTBOX="$STATE_DIR/outbox.jsonl"
|
|
LOG="/tmp/fleet-alert-check.log"
|
|
|
|
mkdir -p "$STATE_DIR"
|
|
[ -f "$STATE" ] || echo '{}' > "$STATE"
|
|
touch "$OUTBOX"
|
|
|
|
NOW=$(date +%s)
|
|
log() { echo "$(date -Iseconds) $*" >> "$LOG"; }
|
|
|
|
# --- shared consecutive-failure state machine (also used by the container relay) ---
|
|
# usage: state_machine <cond> <failing 0|1> -> prints "<ACTION> <fails>"
|
|
# ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | NONE
|
|
state_machine() {
|
|
local cond="$1" failing="$2"
|
|
THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \
|
|
FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" <<'PYEOF'
|
|
import json, os, sys, time
|
|
state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3]
|
|
failing = failing_s == "1"
|
|
threshold = int(os.environ.get("THRESHOLD", "2"))
|
|
realert_min = int(os.environ.get("REALERT_MIN", "30"))
|
|
qh = os.environ.get("QUIET_HOURS", "")
|
|
dry = os.environ.get("FLEET_ALERT_DRY_RUN") == "1"
|
|
now = int(time.time())
|
|
def in_quiet(spec):
|
|
if not spec:
|
|
return False
|
|
try:
|
|
a, b = spec.split("-")
|
|
def m(s):
|
|
h, mi = s.split(":")
|
|
return int(h) * 60 + int(mi)
|
|
cur = time.localtime().tm_hour * 60 + time.localtime().tm_min
|
|
s, e = m(a), m(b)
|
|
return (s <= cur < e) if s <= e else (cur >= s or cur < e)
|
|
except Exception:
|
|
return False
|
|
try:
|
|
st = json.load(open(state_path))
|
|
except Exception:
|
|
st = {}
|
|
e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0}
|
|
action = "NONE"
|
|
if failing:
|
|
e["fails"] = int(e.get("fails", 0)) + 1
|
|
due = e["fails"] >= threshold and (
|
|
not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60
|
|
)
|
|
if due:
|
|
first = not e.get("alerted")
|
|
if not first and in_quiet(qh):
|
|
action = "SUPPRESSED"
|
|
else:
|
|
action = "ALERT_FIRST" if first else "ALERT_REALERT"
|
|
e["alerted"] = True
|
|
e["last_alert_ts"] = now
|
|
else:
|
|
if e.get("alerted"):
|
|
action = "RECOVERY"
|
|
e["fails"] = 0
|
|
e["alerted"] = False
|
|
st[cond] = e
|
|
if not dry:
|
|
json.dump(st, open(state_path, "w"))
|
|
print(action, e["fails"])
|
|
PYEOF
|
|
}
|
|
|
|
emit_record() { # kind cond detail consecutive
|
|
local kind="$1" cond="$2" detail="$3" consec="$4"
|
|
local id
|
|
id=$(tr -d '-' < /proc/sys/kernel/random/uuid | cut -c1-12)
|
|
local rec
|
|
rec=$(python3 -c 'import json,sys; print(json.dumps({"id":sys.argv[1],"ts":int(sys.argv[2]),"kind":sys.argv[3],"source":"bl","condition":sys.argv[4],"detail":sys.argv[5],"consecutive":int(sys.argv[6])}))' \
|
|
"$id" "$NOW" "$kind" "$cond" "$detail" "$consec")
|
|
if [ "$DRY_RUN" = "1" ]; then
|
|
log "DRY-RUN would emit: $rec"
|
|
else
|
|
echo "$rec" >> "$OUTBOX"
|
|
log "emitted $kind $cond x$consec"
|
|
fi
|
|
}
|
|
|
|
box_notify() {
|
|
# Best-effort DM to healthy agents via box-ctl. Never fatal.
|
|
# Fan-out runs in parallel with a per-notify timeout so one hung DM
|
|
# path can't stall the 5-minute check loop.
|
|
local cond="$1"
|
|
local detail="$2"
|
|
local msg="[fleet-alert] CRITICAL ${cond}: ${detail}"
|
|
msg="${msg:0:240}"
|
|
local agent pids=""
|
|
for agent in $HEALTHY_AGENTS; do
|
|
if [ "$DRY_RUN" = "1" ]; then
|
|
log "DRY-RUN would notify $agent"
|
|
continue
|
|
fi
|
|
(
|
|
if timeout 60 python3 "$BIN/box-ctl.py" notify "$agent" "$msg" >/dev/null 2>&1; then
|
|
log "notified $agent re $cond"
|
|
else
|
|
log "notify $agent failed re $cond (best-effort)"
|
|
fi
|
|
) &
|
|
pids="$pids $!"
|
|
done
|
|
local p
|
|
for p in $pids; do wait "$p" 2>/dev/null; done
|
|
}
|
|
|
|
injected() { # cond -> 0 if injected-fail
|
|
case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac
|
|
}
|
|
|
|
HEALTHY_AGENTS=""
|
|
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
[ -n "$node" ] && [ -n "$port" ] || continue
|
|
cond="cdp:$node"
|
|
procs=$(pgrep -f "chromium.*profiles/$node" 2>/dev/null | wc -l)
|
|
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
|
|
failing=0
|
|
HEALTHY_AGENTS="$HEALTHY_AGENTS $node"
|
|
else
|
|
failing=1
|
|
fi
|
|
injected "$cond" && failing=1
|
|
detail="CDP $port not listening in netns warp-$node (chromium procs=$procs)"
|
|
# NOTE: HEALTHY_AGENTS set inside the pipeline subshell is lost; recompute below.
|
|
read -r action fails < <(state_machine "$cond" "$failing")
|
|
case "$action" in
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
emit_record "ALERT" "$cond" "$detail" "$fails"
|
|
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
|
;;
|
|
RECOVERY)
|
|
emit_record "RECOVERY" "$cond" "CDP $port listening again in netns warp-$node" "$fails"
|
|
;;
|
|
SUPPRESSED)
|
|
log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)"
|
|
;;
|
|
esac
|
|
done
|
|
|
|
# --- Agent approval blockage & input wait detection ---
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
[ -n "$node" ] || continue
|
|
node_data=$(python3 -c "
|
|
import sys, json
|
|
sys.path.insert(0, '$BIN')
|
|
import approvals
|
|
info = approvals.inspect_node_approvals('$node')
|
|
out = {
|
|
'has_pending': info.get('has_pending', False),
|
|
'target': info.get('target') or info.get('ip') or 'unknown',
|
|
'title': info.get('title') or '',
|
|
'waits': info.get('input_waits') or []
|
|
}
|
|
print(json.dumps(out))
|
|
" 2>/dev/null || echo '{"has_pending":false,"target":"unknown","title":"","waits":[]}')
|
|
|
|
# 1. Egress permission dialog
|
|
cond="approval:$node"
|
|
has_pending=$(python3 -c "import json,sys; print(1 if json.loads(sys.argv[1]).get('has_pending') else 0)" "$node_data" 2>/dev/null || echo 0)
|
|
if [ "$has_pending" = "1" ]; then
|
|
failing=1
|
|
target=$(python3 -c "import json,sys; print(json.loads(sys.argv[1]).get('target','unknown'))" "$node_data" 2>/dev/null || echo unknown)
|
|
detail="Agent $node held up on browser approval for $target"
|
|
else
|
|
failing=0
|
|
detail="Agent $node approvals clear"
|
|
fi
|
|
injected "$cond" && failing=1
|
|
read -r action fails < <(state_machine "$cond" "$failing")
|
|
case "$action" in
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
emit_record "ALERT" "$cond" "$detail" "$fails"
|
|
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
|
;;
|
|
RECOVERY)
|
|
emit_record "RECOVERY" "$cond" "$detail" "$fails"
|
|
;;
|
|
SUPPRESSED)
|
|
log "$cond still critical x$fails — re-page suppressed"
|
|
;;
|
|
esac
|
|
|
|
# 2. Sidebar task waiting on human input
|
|
cond_in="input_wait:$node"
|
|
wait_summary=$(python3 -c "
|
|
import json,sys
|
|
w = json.loads(sys.argv[1]).get('waits', [])
|
|
if w:
|
|
print('; '.join(f\"{item.get('task')}: {item.get('status')}\" for item in w)[:120])
|
|
" "$node_data" 2>/dev/null || true)
|
|
|
|
if [ -n "$wait_summary" ]; then
|
|
failing_in=1
|
|
detail_in="Agent $node task waiting for human input: $wait_summary"
|
|
else
|
|
failing_in=0
|
|
detail_in="Agent $node tasks running"
|
|
fi
|
|
injected "$cond_in" && failing_in=1
|
|
read -r action_in fails_in < <(state_machine "$cond_in" "$failing_in")
|
|
case "$action_in" in
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
emit_record "ALERT" "$cond_in" "$detail_in" "$fails_in"
|
|
echo "$cond_in" >> "$STATE_DIR/.alerts.tmp"
|
|
;;
|
|
RECOVERY)
|
|
emit_record "RECOVERY" "$cond_in" "$detail_in" "$fails_in"
|
|
;;
|
|
SUPPRESSED)
|
|
log "$cond_in still critical x$fails_in — re-page suppressed"
|
|
;;
|
|
esac
|
|
done
|
|
|
|
# --- Warp partition detection (2026-10-04) ---
|
|
# A partitioned node has a live browser + CDP but no internet egress: the
|
|
# chromebox watchdog sees a healthy browser while all automation fails.
|
|
# Condition id: partition:<node>. The detail names the node, the WireGuard
|
|
# handshake age, the egress probe result, and the timestamp, and says
|
|
# PARTITION explicitly so #lobby readers can tell a network partition from
|
|
# a browser crash at a glance. Anti-spam comes from the shared consecutive-
|
|
# failure state machine (2 consecutive failures before first page, re-page
|
|
# at most every 30 min).
|
|
WARP_PROBE_URL="${WARP_PROBE_URL:-https://1.1.1.1/cdn-cgi/trace}"
|
|
warp_partition_probe() { # <node> -> prints "<handshake_age_s|unknown> <ok|FAIL>"
|
|
local node="$1" iface epoch now age_s
|
|
iface=$(sudo -n ip netns exec "warp-$node" sh -c 'wg show interfaces 2>/dev/null | head -1')
|
|
now=$(date +%s)
|
|
epoch=$(sudo -n ip netns exec "warp-$node" wg show "$iface" latest-handshakes 2>/dev/null | awk '{print $2}')
|
|
case "$epoch" in ''|*[!0-9]*) age_s="unknown" ;; *) age_s=$(( now - epoch )) ;; esac
|
|
if sudo -n ip netns exec "warp-$node" curl -s -m 8 -o /dev/null "$WARP_PROBE_URL" 2>/dev/null; then
|
|
echo "$age_s ok"
|
|
else
|
|
echo "$age_s FAIL"
|
|
fi
|
|
}
|
|
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
[ -n "$node" ] && [ -n "$port" ] || continue
|
|
cond="partition:$node"
|
|
read -r hs_age probe_res < <(warp_partition_probe "$node")
|
|
ts=$(date -u +%FT%TZ)
|
|
if [ "$probe_res" = "ok" ]; then
|
|
failing=0
|
|
detail="warp egress restored for $node at $ts (probe $WARP_PROBE_URL ok)"
|
|
else
|
|
failing=1
|
|
detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED)"
|
|
fi
|
|
if injected "$cond"; then
|
|
failing=1
|
|
detail="PARTITION $node: warp egress down at $ts (handshake ${hs_age}s ago, probe $WARP_PROBE_URL FAILED) [INJECTED]"
|
|
fi
|
|
read -r action fails < <(state_machine "$cond" "$failing")
|
|
case "$action" in
|
|
ALERT_FIRST|ALERT_REALERT)
|
|
emit_record "ALERT" "$cond" "$detail" "$fails"
|
|
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
|
;;
|
|
RECOVERY)
|
|
emit_record "RECOVERY" "$cond" "$detail" "$fails"
|
|
;;
|
|
SUPPRESSED)
|
|
log "$cond still critical x$fails — re-page suppressed by quiet hours ($QUIET_HOURS)"
|
|
;;
|
|
esac
|
|
done
|
|
|
|
# Recompute healthy agents in the main shell (pipeline subshell above can't export).
|
|
HEALTHY_AGENTS=""
|
|
"$BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
[ -n "$node" ] && [ -n "$port" ] || continue
|
|
if sudo -n ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$port "; then
|
|
echo "$node"
|
|
fi
|
|
done > "$STATE_DIR/.healthy.tmp"
|
|
HEALTHY_AGENTS=$(tr '\n' ' ' < "$STATE_DIR/.healthy.tmp")
|
|
rm -f "$STATE_DIR/.healthy.tmp"
|
|
|
|
# Notify for this run's alerts (best effort). Skip entirely when nothing is healthy
|
|
# (notify needs a working browser via dm.py) or in dry-run.
|
|
if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then
|
|
while read -r cond; do
|
|
[ -n "$cond" ] && box_notify "$cond" "see #lobby for detail"
|
|
done < "$STATE_DIR/.alerts.tmp"
|
|
elif [ -f "$STATE_DIR/.alerts.tmp" ]; then
|
|
log "no healthy agents — box notify skipped (DM path needs a working browser)"
|
|
fi
|
|
rm -f "$STATE_DIR/.alerts.tmp"
|
|
|
|
tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG"
|
|
log "check complete"
|
|
|
|
# Relay pending outbox records to #lobby with idempotency gates (posted watermark + content hash TTL)
|
|
if [ "$DRY_RUN" -eq 0 ] && [ -x "$BIN/fleet-alert-relay.sh" ]; then
|
|
"$BIN/fleet-alert-relay.sh" >> "$LOG" 2>&1 || true
|
|
fi
|