#!/usr/bin/env bash # chromebox-watchdog.sh [profile] — keep a chrome-box profile alive and healthy. # Checks: 1) chromium process for the profile is running, # 2) CDP responds and the chat page is present. # If unhealthy: kill any stale chrome for the profile and relaunch headless # via netvm-chrome.sh (profile dir persists session/cookies — the process is # disposable, the state is not). Mirrors operator-646's container # recover-after-rebuild.sh philosophy. # Runs every 2 min via systemd timer chromebox-watchdog-.timer. set -euo pipefail # Prevent overlapping runs: the timer fires every 2 min but a relaunch # (kill + sleep 25 + chrome startup + page load) can exceed that, and two # concurrent runs kill each others chrome (observed 2026-10-03: pip flapped # with simultaneous "relaunch OK" and "relaunch FAILED"). LOCK="/tmp/chromebox-watchdog-${1:-pip}.lock" exec 9>"$LOCK" if ! flock -n 9; then echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2 exit 0 fi PROFILE="${1:-pip}" NETVM_BIN="/home/super/Projects/NetVM/bin" LOG="/home/super/Projects/NetVM/chromebox-watchdog.log" case "$PROFILE" in muse) CDP_PORT=9410 ;; pip) CDP_PORT=9420 ;; 646) CDP_PORT=9430 ;; opm) CDP_PORT=9440 ;; *) echo "unknown profile: $PROFILE" >&2; exit 1 ;; esac log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; } cdp_list() { "$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null } healthy() { pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 || return 1 local list list="$(cdp_list)" || return 1 echo "$list" | grep -q '"title": "Chat' || return 1 return 0 } if healthy; then exit 0 fi log "unhealthy, relaunching chromebox" # bracket trick so pkill never matches its own command line pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/" pkill -f "chromium.*$pat" 2>/dev/null || true sleep 3 # Relaunch in its own systemd scope so it survives this oneshot run. # nohup/setsid do NOT escape: this timer's service uses KillMode=control-group # and systemd SIGKILLs everything in the cgroup at teardown (observed # 2026-10-03: every relaunch "recovered" then died seconds later). A transient # scope escapes the service cgroup; the scoped process inherits these fds so # the log redirect below still captures chromium's output. # NOTE: systemd-run --scope WAITS for the scope's processes (even --no-block, # verified 2026-10-03), so it must be backgrounded — the scope is an # independent unit and outlives the wrapper. # NOTE: --cdp-port is pinned explicitly. netvm-chrome.sh defaults to a # hash-derived port (9222+...) which will NOT match the registry port that # muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the # launcher but is unreachable to the API (observed 2026-10-03: pip relaunched # on 9278 instead of 9420, watchdog looped on "relaunch FAILED"). systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \ "$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \ >>"$LOG" 2>&1 < /dev/null & sleep 25 if healthy; then log "relaunch OK" else log "relaunch FAILED — needs operator attention" exit 1 fi