Files
box/bin/chromebox-watchdog.sh
T

226 lines
11 KiB
Bash
Executable File

#!/usr/bin/env bash
# chromebox-watchdog.sh [profile] — keep a chrome-box profile alive and healthy.
# Checks: 1) chromium process for the profile is running,
# 2) CDP responds and the chat page is present.
# If unhealthy: kill any stale chrome for the profile and relaunch headless
# via netvm-chrome.sh (profile dir persists session/cookies — the process is
# disposable, the state is not). Mirrors operator-646's container
# recover-after-rebuild.sh philosophy.
# Runs every 2 min via systemd timer chromebox-watchdog-<profile>.timer.
set -euo pipefail
export XDG_RUNTIME_DIR="${XDG_RUNTIME_DIR:-/run/user/$(id -u)}"
export DBUS_SESSION_BUS_ADDRESS="${DBUS_SESSION_BUS_ADDRESS:-unix:path=${XDG_RUNTIME_DIR}/bus}"
# Prevent overlapping runs: the timer fires every 2 min but a relaunch
# (kill + sleep 25 + chrome startup + page load) can exceed that, and two
# concurrent runs kill each others chrome (observed 2026-10-03: pip flapped
# with simultaneous "relaunch OK" and "relaunch FAILED").
LOCK="/tmp/chromebox-watchdog-${1:-pip}.lock"
# Tests source this file with CHROMEBOX_WATCHDOG_LIB_ONLY=1: they resolve
# ports and call helpers without running checks, so no lock is needed.
if [ "${CHROMEBOX_WATCHDOG_LIB_ONLY:-}" != "1" ]; then
exec 9>"$LOCK"
if ! flock -n 9; then
echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2
exit 0
fi
fi
PROFILE="${1:-pip}"
NETVM_BIN="/home/super/Projects/NetVM/bin"
LOG="/home/super/Projects/NetVM/chromebox-watchdog.log"
CHROME_LOG="/home/super/Projects/NetVM/chromebox-${PROFILE}.log"
# Rotate a log file past 10MB (keep one generation)
rotate_log() {
local f="$1"
[ -f "$f" ] || return 0
local sz
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
if [ "$sz" -gt 10485760 ]; then
mv -f "$f" "$f.1"
echo "[$(date -u +%FT%TZ)] [$PROFILE] log rotated" > "$f"
fi
}
rotate_log "$LOG"
# CHROME_LOG rotation happens after PROFILE is set (see below)
# Ports come from the fleet registry, not a hardcoded list: every active
# node (def/dev included) gets supervision automatically. The old 4-profile
# case left dev/def unsupervised — a dead Warp tunnel paged forever with
# no auto-recovery (2026-10-06 dev outage).
CDP_PORT="$("$NETVM_BIN/netvm-registry.py" "$PROFILE" 2>/dev/null)" || {
echo "unknown profile: $PROFILE" >&2
exit 1
}
[ -n "$CDP_PORT" ] || { echo "unknown profile: $PROFILE" >&2; exit 1; }
rotate_log "$CHROME_LOG"
log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; }
cdp_list() {
"$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null
}
# HEALTH_FAIL_REASON is set by healthy() on failure: which stage broke.
HEALTH_FAIL_REASON=""
healthy() {
HEALTH_FAIL_REASON=""
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 \
|| { HEALTH_FAIL_REASON="no chromium process for profile"; return 1; }
local list
list="$(cdp_list)" \
|| { HEALTH_FAIL_REASON="CDP unreachable on :$CDP_PORT"; return 1; }
[[ "$list" == *'"type": "page"'* ]] \
|| { HEALTH_FAIL_REASON="CDP up but no page target in list"; return 1; }
[[ "$list" == *'"url": "https://muse.ai'* ]] \
|| { HEALTH_FAIL_REASON="CDP up but not on muse.ai"; return 1; }
warp_egress_healthy || return 1
return 0
}
# Stage 5: Warp egress health (2026-10-04). A partitioned browser (tunnel down,
# CDP green) passes stages 1-4 while being unable to reach muse.ai. Check the
# WireGuard handshake age and probe egress from inside the node's netns.
# Sets HEALTH_FAIL_REASON with a distinct "warp egress down" prefix so
# root-cause analysis can distinguish partitions from Chromium crashes.
warp_egress_healthy() {
local wg_out hs_line age_s code
wg_out="$(sudo -n ip netns exec "warp-${PROFILE}" wg show 2>/dev/null)" \
|| { HEALTH_FAIL_REASON="warp egress down (wg show failed in warp-${PROFILE})"; return 1; }
hs_line="$(printf '%s\n' "$wg_out" | grep -i "latest handshake" | head -1)"
[ -n "$hs_line" ] \
|| { HEALTH_FAIL_REASON="warp egress down (no WireGuard handshake in warp-${PROFILE})"; return 1; }
# "latest handshake: 1 minute, 41 seconds ago" -> total seconds
age_s="$(printf '%s\n' "$hs_line" | python3 -c '
import sys, re
s = sys.stdin.read()
m = re.search(r"(\d+)\s*hour", s); h = int(m.group(1)) if m else 0
m = re.search(r"(\d+)\s*minute", s); mi = int(m.group(1)) if m else 0
m = re.search(r"(\d+)\s*second", s); se = int(m.group(1)) if m else 0
print(h*3600 + mi*60 + se)
' 2>/dev/null)"
{ [ -n "$age_s" ] && [ "$age_s" -ge 0 ]; } 2>/dev/null \
|| { HEALTH_FAIL_REASON="warp egress down (unparseable handshake: $hs_line)"; return 1; }
[ "$age_s" -le 180 ] \
|| { HEALTH_FAIL_REASON="warp egress down (handshake ${age_s}s old in warp-${PROFILE})"; return 1; }
code="$(sudo -n ip netns exec "warp-${PROFILE}" curl -m 5 -s -o /dev/null -w "%{http_code}" "https://muse.ai/" 2>/dev/null)" \
|| { HEALTH_FAIL_REASON="warp egress down (egress probe curl failed in warp-${PROFILE})"; return 1; }
# Any 2xx/3xx means we reached muse.ai infra ("/" 307-redirects to the
# auth flow). The probe tests egress connectivity, not page content.
case "$code" in
2*|3*) return 0 ;;
*) HEALTH_FAIL_REASON="warp egress down (egress probe HTTP $code in warp-${PROFILE})"; return 1 ;;
esac
}
# Allow sourcing for tests without running checks.
if [ "${CHROMEBOX_WATCHDOG_LIB_ONLY:-}" = "1" ]; then
return 0 2>/dev/null || exit 0
fi
if healthy; then
# Auto-reconcile idle workers for healthy profiles
# DISABLED 2026-10-06 by operator-646: kpi auto-spawn ignores job schedule fields;
# find_pending_work_for_node returns the alphabetically-first definition every tick,
# re-spawning and re-noticing every ~2min (pip b01 BOX-AUTO-WORKER loop, 29+ copies).
# Watchdog health path untouched. Re-enable once the spawner is schedule-aware.
# python3 "$NETVM_BIN/super-cli.py" kpi auto-spawn --node "$PROFILE" >>"$LOG" 2>&1 || true
exit 0
fi
# Relaunch-loop guard (2026-10-04): if the main browser for this profile
# launched <2 min ago it's probably still starting up (CDP not yet bound).
# Relaunching now would kill a healthy-but-slow cold start via netvm-chrome.sh
# and reset the startup clock every cycle (observed: opm piled up 5 chromiums
# because the guard only protected the kill step, not the relaunch). Skip the
# entire cycle instead.
# NOTE: match only the main browser process (--remote-debugging-port present,
# no --type= flag). Renderer/gpu children (--type=renderer etc.) start later
# than the main process and must not satisfy this check.
recent_pid=""
for _pid in $(pgrep -f "chromium.*--remote-debugging-port=${CDP_PORT}([[:space:]]|$)" 2>/dev/null); do
# Skip child processes (renderer, gpu, etc.) — only the main browser counts
if ps -o args= -p "$_pid" 2>/dev/null | grep -q -- "--type="; then
continue
fi
_start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0)
_now=$(date +%s)
if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then
recent_pid="$_pid"
break
fi
done
if [ -n "$recent_pid" ]; then
log "browser launched recently (pid $recent_pid), skipping relaunch (probably still starting)"
exit 0
fi
# Warp-partition recovery (2026-10-04): relaunching Chrome cannot fix a dead
# Warp tunnel — the new browser would fail the same egress check and the
# watchdog would loop. If the failure is warp-egress, restart the tunnel first
# (netvm-node-up.sh is idempotent); only fall through to the Chrome relaunch
# if the tunnel does not recover.
case "$HEALTH_FAIL_REASON" in
"warp egress down"*)
log "warp partition detected ($HEALTH_FAIL_REASON), restarting tunnel via netvm-node-up.sh"
sudo -n "$NETVM_BIN/netvm-node-up.sh" "$PROFILE" >>"$LOG" 2>&1 || true
sleep 5
if healthy; then
log "tunnel restart recovered warp egress, chrome relaunch not needed"
exit 0
fi
log "tunnel restart did not recover egress ($HEALTH_FAIL_REASON), proceeding with chrome relaunch"
;;
esac
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
# bracket trick so pkill never matches its own command line
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
pkill -f "chromium.*$pat" 2>/dev/null || true
sleep 3
# Clean up stale singleton symlinks that break subsequent browser startup
rm -f "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonLock" \
"/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonSocket" \
"/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonCookie" 2>/dev/null || true
for _sc in $(systemctl --user list-units --type=scope --plain --no-legend 2>/dev/null | awk '{print $1}' | grep -E "^netvm-chrome-${PROFILE}-"); do
systemctl --user stop "$_sc" 2>/dev/null || true
done
# Relaunch in its own systemd scope so it survives this oneshot run.
# nohup/setsid do NOT escape: this timer's service uses KillMode=control-group
# and systemd SIGKILLs everything in the cgroup at teardown (observed
# 2026-10-03: every relaunch "recovered" then died seconds later). A transient
# scope escapes the service cgroup; the scoped process inherits these fds so
# the log redirect below still captures chromium's output.
# NOTE: systemd-run --scope WAITS for the scope's processes (even --no-block,
# verified 2026-10-03), so it must be backgrounded — the scope is an
# independent unit and outlives the wrapper.
# NOTE: --cdp-port is pinned explicitly. netvm-chrome.sh defaults to a
# hash-derived port (9222+...) which will NOT match the registry port that
# muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the
# launcher but is unreachable to the API (observed 2026-10-03: pip relaunched
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
# 9>&-: do NOT let the backgrounded launcher inherit the watchdog lock fd.
# Inherited flock fds wedge the lock forever (observed 2026-10-04: opm/muse
# launchers held their profile lock for 100+ min, every later watchdog run
# skipped as "another run in progress" — the watchdog was silently dead).
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
>>"$CHROME_LOG" 2>&1 < /dev/null 9>&- &
# Retry with backoff (2026-10-04): cold starts (fresh egress IP, Cloudflare
# handshake) can take >25s for the page title to appear. A single check after
# 25s kills working-but-slow browsers. Try up to 4 times, 15s apart (~60s
# window), logging each attempt. Only declare FAILED if all attempts fail.
_relaunch_ok=0
for _attempt in 1 2 3 4; do
sleep 15
if healthy; then
log "relaunch OK (attempt $_attempt)"
_relaunch_ok=1
break
fi
log "relaunch attempt $_attempt not healthy yet ($HEALTH_FAIL_REASON)"
done
if [ "$_relaunch_ok" -ne 1 ]; then
log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention"
exit 1
fi