#!/usr/bin/env bash # chromebox-watchdog.sh [profile] — keep a chrome-box profile alive and healthy. # Checks: 1) chromium process for the profile is running, # 2) CDP responds and the chat page is present. # If unhealthy: kill any stale chrome for the profile and relaunch headless # via netvm-chrome.sh (profile dir persists session/cookies — the process is # disposable, the state is not). Mirrors operator-646's container # recover-after-rebuild.sh philosophy. # Runs every 2 min via systemd timer chromebox-watchdog-.timer. set -euo pipefail export XDG_RUNTIME_DIR="${XDG_RUNTIME_DIR:-/run/user/$(id -u)}" export DBUS_SESSION_BUS_ADDRESS="${DBUS_SESSION_BUS_ADDRESS:-unix:path=${XDG_RUNTIME_DIR}/bus}" # Prevent overlapping runs: the timer fires every 2 min but a relaunch # (kill + sleep 25 + chrome startup + page load) can exceed that, and two # concurrent runs kill each others chrome (observed 2026-10-03: pip flapped # with simultaneous "relaunch OK" and "relaunch FAILED"). LOCK="/tmp/chromebox-watchdog-${1:-pip}.lock" exec 9>"$LOCK" if ! flock -n 9; then echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2 exit 0 fi PROFILE="${1:-pip}" NETVM_BIN="/home/super/Projects/NetVM/bin" LOG="/home/super/Projects/NetVM/chromebox-watchdog.log" CHROME_LOG="/home/super/Projects/NetVM/chromebox-${PROFILE}.log" # Rotate a log file past 10MB (keep one generation) rotate_log() { local f="$1" [ -f "$f" ] || return 0 local sz sz=$(stat -c%s "$f" 2>/dev/null || echo 0) if [ "$sz" -gt 10485760 ]; then mv -f "$f" "$f.1" echo "[$(date -u +%FT%TZ)] [$PROFILE] log rotated" > "$f" fi } rotate_log "$LOG" # CHROME_LOG rotation happens after PROFILE is set (see below) case "$PROFILE" in muse) CDP_PORT=9410 ;; pip) CDP_PORT=9420 ;; 646) CDP_PORT=9430 ;; opm) CDP_PORT=9440 ;; *) echo "unknown profile: $PROFILE" >&2; exit 1 ;; esac rotate_log "$CHROME_LOG" log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; } cdp_list() { "$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null } # HEALTH_FAIL_REASON is set by healthy() on failure: which stage broke. HEALTH_FAIL_REASON="" healthy() { HEALTH_FAIL_REASON="" pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 \ || { HEALTH_FAIL_REASON="no chromium process for profile"; return 1; } local list list="$(cdp_list)" \ || { HEALTH_FAIL_REASON="CDP unreachable on :$CDP_PORT"; return 1; } echo "$list" | grep -q '"type": "page"' \ || { HEALTH_FAIL_REASON="CDP up but no page target in list"; return 1; } echo "$list" | grep -E -q '"url": "https://muse\.ai' \ || { HEALTH_FAIL_REASON="CDP up but not on muse.ai"; return 1; } return 0 } if healthy; then exit 0 fi log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox" # Kill-loop guard (2026-10-04): if a chromium for this profile launched <2 min # ago it's probably still loading the page — killing it just restarts the # loop (observed: 646 killed twice while the page was still loading). recent_pid="" for _pid in $(pgrep -f "chromium.*profiles/${PROFILE}/" 2>/dev/null); do _start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0) _now=$(date +%s) if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then recent_pid="$_pid" break fi done if [ -n "$recent_pid" ]; then log "browser launched recently (pid $recent_pid), skipping kill (probably still starting)" exit 0 fi # bracket trick so pkill never matches its own command line pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/" pkill -f "chromium.*$pat" 2>/dev/null || true sleep 3 # Clean up stale singleton symlinks that break subsequent browser startup rm -f "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonLock" \ "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonSocket" \ "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonCookie" 2>/dev/null || true # Relaunch in its own systemd scope so it survives this oneshot run. # nohup/setsid do NOT escape: this timer's service uses KillMode=control-group # and systemd SIGKILLs everything in the cgroup at teardown (observed # 2026-10-03: every relaunch "recovered" then died seconds later). A transient # scope escapes the service cgroup; the scoped process inherits these fds so # the log redirect below still captures chromium's output. # NOTE: systemd-run --scope WAITS for the scope's processes (even --no-block, # verified 2026-10-03), so it must be backgrounded — the scope is an # independent unit and outlives the wrapper. # NOTE: --cdp-port is pinned explicitly. netvm-chrome.sh defaults to a # hash-derived port (9222+...) which will NOT match the registry port that # muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the # launcher but is unreachable to the API (observed 2026-10-03: pip relaunched # on 9278 instead of 9420, watchdog looped on "relaunch FAILED"). systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \ "$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \ >>"$CHROME_LOG" 2>&1 < /dev/null & # Retry with backoff (2026-10-04): cold starts (fresh egress IP, Cloudflare # handshake) can take >25s for the page title to appear. A single check after # 25s kills working-but-slow browsers. Try up to 4 times, 15s apart (~60s # window), logging each attempt. Only declare FAILED if all attempts fail. _relaunch_ok=0 for _attempt in 1 2 3 4; do sleep 15 if healthy; then log "relaunch OK (attempt $_attempt)" _relaunch_ok=1 break fi log "relaunch attempt $_attempt not healthy yet ($HEALTH_FAIL_REASON)" done if [ "$_relaunch_ok" -ne 1 ]; then log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention" exit 1 fi