Files
box/bin/chromebox-watchdog.sh
T

139 lines
5.8 KiB
Bash
Raw Normal View History

#!/usr/bin/env bash
# chromebox-watchdog.sh [profile] — keep a chrome-box profile alive and healthy.
# Checks: 1) chromium process for the profile is running,
# 2) CDP responds and the chat page is present.
# If unhealthy: kill any stale chrome for the profile and relaunch headless
# via netvm-chrome.sh (profile dir persists session/cookies — the process is
# disposable, the state is not). Mirrors operator-646's container
# recover-after-rebuild.sh philosophy.
# Runs every 2 min via systemd timer chromebox-watchdog-<profile>.timer.
set -euo pipefail
export XDG_RUNTIME_DIR="${XDG_RUNTIME_DIR:-/run/user/$(id -u)}"
export DBUS_SESSION_BUS_ADDRESS="${DBUS_SESSION_BUS_ADDRESS:-unix:path=${XDG_RUNTIME_DIR}/bus}"
# Prevent overlapping runs: the timer fires every 2 min but a relaunch
# (kill + sleep 25 + chrome startup + page load) can exceed that, and two
# concurrent runs kill each others chrome (observed 2026-10-03: pip flapped
# with simultaneous "relaunch OK" and "relaunch FAILED").
LOCK="/tmp/chromebox-watchdog-${1:-pip}.lock"
exec 9>"$LOCK"
if ! flock -n 9; then
echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2
exit 0
fi
PROFILE="${1:-pip}"
NETVM_BIN="/home/super/Projects/NetVM/bin"
LOG="/home/super/Projects/NetVM/chromebox-watchdog.log"
CHROME_LOG="/home/super/Projects/NetVM/chromebox-${PROFILE}.log"
# Rotate a log file past 10MB (keep one generation)
rotate_log() {
local f="$1"
[ -f "$f" ] || return 0
local sz
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
if [ "$sz" -gt 10485760 ]; then
mv -f "$f" "$f.1"
echo "[$(date -u +%FT%TZ)] [$PROFILE] log rotated" > "$f"
fi
}
rotate_log "$LOG"
# CHROME_LOG rotation happens after PROFILE is set (see below)
case "$PROFILE" in
muse) CDP_PORT=9410 ;;
pip) CDP_PORT=9420 ;;
646) CDP_PORT=9430 ;;
opm) CDP_PORT=9440 ;;
*) echo "unknown profile: $PROFILE" >&2; exit 1 ;;
esac
rotate_log "$CHROME_LOG"
log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; }
cdp_list() {
"$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null
}
# HEALTH_FAIL_REASON is set by healthy() on failure: which stage broke.
HEALTH_FAIL_REASON=""
healthy() {
HEALTH_FAIL_REASON=""
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 \
|| { HEALTH_FAIL_REASON="no chromium process for profile"; return 1; }
local list
list="$(cdp_list)" \
|| { HEALTH_FAIL_REASON="CDP unreachable on :$CDP_PORT"; return 1; }
echo "$list" | grep -q '"type": "page"' \
|| { HEALTH_FAIL_REASON="CDP up but no page target in list"; return 1; }
echo "$list" | grep -E -q '"url": "https://muse\.ai' \
|| { HEALTH_FAIL_REASON="CDP up but not on muse.ai"; return 1; }
return 0
}
if healthy; then
exit 0
fi
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
# Kill-loop guard (2026-10-04): if a chromium for this profile launched <2 min
# ago it's probably still loading the page — killing it just restarts the
# loop (observed: 646 killed twice while the page was still loading).
recent_pid=""
for _pid in $(pgrep -f "chromium.*profiles/${PROFILE}/" 2>/dev/null); do
_start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0)
_now=$(date +%s)
if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then
recent_pid="$_pid"
break
fi
done
if [ -n "$recent_pid" ]; then
log "browser launched recently (pid $recent_pid), skipping kill (probably still starting)"
exit 0
fi
# bracket trick so pkill never matches its own command line
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
pkill -f "chromium.*$pat" 2>/dev/null || true
sleep 3
# Clean up stale singleton symlinks that break subsequent browser startup
rm -f "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonLock" \
"/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonSocket" \
"/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonCookie" 2>/dev/null || true
# Relaunch in its own systemd scope so it survives this oneshot run.
# nohup/setsid do NOT escape: this timer's service uses KillMode=control-group
# and systemd SIGKILLs everything in the cgroup at teardown (observed
# 2026-10-03: every relaunch "recovered" then died seconds later). A transient
# scope escapes the service cgroup; the scoped process inherits these fds so
# the log redirect below still captures chromium's output.
# NOTE: systemd-run --scope WAITS for the scope's processes (even --no-block,
# verified 2026-10-03), so it must be backgrounded — the scope is an
# independent unit and outlives the wrapper.
# NOTE: --cdp-port is pinned explicitly. netvm-chrome.sh defaults to a
# hash-derived port (9222+...) which will NOT match the registry port that
# muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the
# launcher but is unreachable to the API (observed 2026-10-03: pip relaunched
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
>>"$CHROME_LOG" 2>&1 < /dev/null &
# Retry with backoff (2026-10-04): cold starts (fresh egress IP, Cloudflare
# handshake) can take >25s for the page title to appear. A single check after
# 25s kills working-but-slow browsers. Try up to 4 times, 15s apart (~60s
# window), logging each attempt. Only declare FAILED if all attempts fail.
_relaunch_ok=0
for _attempt in 1 2 3 4; do
sleep 15
if healthy; then
log "relaunch OK (attempt $_attempt)"
_relaunch_ok=1
break
fi
log "relaunch attempt $_attempt not healthy yet ($HEALTH_FAIL_REASON)"
done
if [ "$_relaunch_ok" -ne 1 ]; then
log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention"
exit 1
fi