2026-10-04 01:10:45 +00:00
|
|
|
#!/usr/bin/env bash
|
|
|
|
|
# chromebox-watchdog.sh [profile] — keep a chrome-box profile alive and healthy.
|
|
|
|
|
# Checks: 1) chromium process for the profile is running,
|
|
|
|
|
# 2) CDP responds and the chat page is present.
|
|
|
|
|
# If unhealthy: kill any stale chrome for the profile and relaunch headless
|
|
|
|
|
# via netvm-chrome.sh (profile dir persists session/cookies — the process is
|
|
|
|
|
# disposable, the state is not). Mirrors operator-646's container
|
|
|
|
|
# recover-after-rebuild.sh philosophy.
|
|
|
|
|
# Runs every 2 min via systemd timer chromebox-watchdog-<profile>.timer.
|
|
|
|
|
set -euo pipefail
|
2026-10-04 16:34:25 +00:00
|
|
|
export XDG_RUNTIME_DIR="${XDG_RUNTIME_DIR:-/run/user/$(id -u)}"
|
|
|
|
|
export DBUS_SESSION_BUS_ADDRESS="${DBUS_SESSION_BUS_ADDRESS:-unix:path=${XDG_RUNTIME_DIR}/bus}"
|
2026-10-04 01:10:45 +00:00
|
|
|
# Prevent overlapping runs: the timer fires every 2 min but a relaunch
|
|
|
|
|
# (kill + sleep 25 + chrome startup + page load) can exceed that, and two
|
|
|
|
|
# concurrent runs kill each others chrome (observed 2026-10-03: pip flapped
|
|
|
|
|
# with simultaneous "relaunch OK" and "relaunch FAILED").
|
|
|
|
|
LOCK="/tmp/chromebox-watchdog-${1:-pip}.lock"
|
|
|
|
|
exec 9>"$LOCK"
|
|
|
|
|
if ! flock -n 9; then
|
|
|
|
|
echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2
|
|
|
|
|
exit 0
|
|
|
|
|
fi
|
|
|
|
|
|
|
|
|
|
PROFILE="${1:-pip}"
|
|
|
|
|
NETVM_BIN="/home/super/Projects/NetVM/bin"
|
|
|
|
|
LOG="/home/super/Projects/NetVM/chromebox-watchdog.log"
|
2026-10-04 12:40:56 +00:00
|
|
|
CHROME_LOG="/home/super/Projects/NetVM/chromebox-${PROFILE}.log"
|
|
|
|
|
|
|
|
|
|
# Rotate a log file past 10MB (keep one generation)
|
|
|
|
|
rotate_log() {
|
|
|
|
|
local f="$1"
|
|
|
|
|
[ -f "$f" ] || return 0
|
|
|
|
|
local sz
|
|
|
|
|
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
|
|
|
|
|
if [ "$sz" -gt 10485760 ]; then
|
|
|
|
|
mv -f "$f" "$f.1"
|
|
|
|
|
echo "[$(date -u +%FT%TZ)] [$PROFILE] log rotated" > "$f"
|
|
|
|
|
fi
|
|
|
|
|
}
|
|
|
|
|
rotate_log "$LOG"
|
|
|
|
|
# CHROME_LOG rotation happens after PROFILE is set (see below)
|
2026-10-04 01:10:45 +00:00
|
|
|
|
|
|
|
|
case "$PROFILE" in
|
|
|
|
|
muse) CDP_PORT=9410 ;;
|
|
|
|
|
pip) CDP_PORT=9420 ;;
|
|
|
|
|
646) CDP_PORT=9430 ;;
|
|
|
|
|
opm) CDP_PORT=9440 ;;
|
|
|
|
|
*) echo "unknown profile: $PROFILE" >&2; exit 1 ;;
|
|
|
|
|
esac
|
2026-10-04 12:40:56 +00:00
|
|
|
rotate_log "$CHROME_LOG"
|
2026-10-04 01:10:45 +00:00
|
|
|
|
|
|
|
|
log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; }
|
|
|
|
|
|
|
|
|
|
cdp_list() {
|
|
|
|
|
"$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null
|
|
|
|
|
}
|
|
|
|
|
|
2026-10-04 12:40:56 +00:00
|
|
|
# HEALTH_FAIL_REASON is set by healthy() on failure: which stage broke.
|
|
|
|
|
HEALTH_FAIL_REASON=""
|
2026-10-04 01:10:45 +00:00
|
|
|
healthy() {
|
2026-10-04 12:40:56 +00:00
|
|
|
HEALTH_FAIL_REASON=""
|
|
|
|
|
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 \
|
|
|
|
|
|| { HEALTH_FAIL_REASON="no chromium process for profile"; return 1; }
|
2026-10-04 01:10:45 +00:00
|
|
|
local list
|
2026-10-04 12:40:56 +00:00
|
|
|
list="$(cdp_list)" \
|
|
|
|
|
|| { HEALTH_FAIL_REASON="CDP unreachable on :$CDP_PORT"; return 1; }
|
2026-10-04 16:39:30 +00:00
|
|
|
echo "$list" | grep -q '"type": "page"' \
|
|
|
|
|
|| { HEALTH_FAIL_REASON="CDP up but no page target in list"; return 1; }
|
|
|
|
|
echo "$list" | grep -E -q '"url": "https://muse\.ai' \
|
|
|
|
|
|| { HEALTH_FAIL_REASON="CDP up but not on muse.ai"; return 1; }
|
2026-10-04 01:10:45 +00:00
|
|
|
return 0
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if healthy; then
|
|
|
|
|
exit 0
|
|
|
|
|
fi
|
|
|
|
|
|
2026-10-04 12:40:56 +00:00
|
|
|
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
|
2026-10-04 13:17:13 +00:00
|
|
|
# Kill-loop guard (2026-10-04): if a chromium for this profile launched <2 min
|
|
|
|
|
# ago it's probably still loading the page — killing it just restarts the
|
|
|
|
|
# loop (observed: 646 killed twice while the page was still loading).
|
|
|
|
|
recent_pid=""
|
|
|
|
|
for _pid in $(pgrep -f "chromium.*profiles/${PROFILE}/" 2>/dev/null); do
|
|
|
|
|
_start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0)
|
|
|
|
|
_now=$(date +%s)
|
|
|
|
|
if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then
|
|
|
|
|
recent_pid="$_pid"
|
|
|
|
|
break
|
|
|
|
|
fi
|
|
|
|
|
done
|
|
|
|
|
if [ -n "$recent_pid" ]; then
|
|
|
|
|
log "browser launched recently (pid $recent_pid), skipping kill (probably still starting)"
|
|
|
|
|
exit 0
|
|
|
|
|
fi
|
2026-10-04 01:10:45 +00:00
|
|
|
# bracket trick so pkill never matches its own command line
|
|
|
|
|
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
|
|
|
|
|
pkill -f "chromium.*$pat" 2>/dev/null || true
|
|
|
|
|
sleep 3
|
2026-10-04 16:34:25 +00:00
|
|
|
# Clean up stale singleton symlinks that break subsequent browser startup
|
|
|
|
|
rm -f "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonLock" \
|
|
|
|
|
"/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonSocket" \
|
|
|
|
|
"/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonCookie" 2>/dev/null || true
|
|
|
|
|
|
2026-10-04 01:10:45 +00:00
|
|
|
# Relaunch in its own systemd scope so it survives this oneshot run.
|
|
|
|
|
# nohup/setsid do NOT escape: this timer's service uses KillMode=control-group
|
|
|
|
|
# and systemd SIGKILLs everything in the cgroup at teardown (observed
|
|
|
|
|
# 2026-10-03: every relaunch "recovered" then died seconds later). A transient
|
|
|
|
|
# scope escapes the service cgroup; the scoped process inherits these fds so
|
|
|
|
|
# the log redirect below still captures chromium's output.
|
|
|
|
|
# NOTE: systemd-run --scope WAITS for the scope's processes (even --no-block,
|
|
|
|
|
# verified 2026-10-03), so it must be backgrounded — the scope is an
|
|
|
|
|
# independent unit and outlives the wrapper.
|
|
|
|
|
# NOTE: --cdp-port is pinned explicitly. netvm-chrome.sh defaults to a
|
|
|
|
|
# hash-derived port (9222+...) which will NOT match the registry port that
|
|
|
|
|
# muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the
|
|
|
|
|
# launcher but is unreachable to the API (observed 2026-10-03: pip relaunched
|
|
|
|
|
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
|
|
|
|
|
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
|
|
|
|
|
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
|
2026-10-04 12:40:56 +00:00
|
|
|
>>"$CHROME_LOG" 2>&1 < /dev/null &
|
2026-10-04 13:17:13 +00:00
|
|
|
# Retry with backoff (2026-10-04): cold starts (fresh egress IP, Cloudflare
|
|
|
|
|
# handshake) can take >25s for the page title to appear. A single check after
|
|
|
|
|
# 25s kills working-but-slow browsers. Try up to 4 times, 15s apart (~60s
|
|
|
|
|
# window), logging each attempt. Only declare FAILED if all attempts fail.
|
|
|
|
|
_relaunch_ok=0
|
|
|
|
|
for _attempt in 1 2 3 4; do
|
|
|
|
|
sleep 15
|
|
|
|
|
if healthy; then
|
|
|
|
|
log "relaunch OK (attempt $_attempt)"
|
|
|
|
|
_relaunch_ok=1
|
|
|
|
|
break
|
|
|
|
|
fi
|
|
|
|
|
log "relaunch attempt $_attempt not healthy yet ($HEALTH_FAIL_REASON)"
|
|
|
|
|
done
|
|
|
|
|
if [ "$_relaunch_ok" -ne 1 ]; then
|
2026-10-04 12:40:56 +00:00
|
|
|
log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention"
|
2026-10-04 01:10:45 +00:00
|
|
|
exit 1
|
|
|
|
|
fi
|