345eb09559
echo "$list" | grep -q under set -o pipefail exits 141 whenever grep matches before echo finishes writing, so healthy browsers were reported 'CDP up but no page target' and killed every 2 min fleet-wide (load 19+). Use [[ == *glob* ]] (no pipe, no race) for the page/muse.ai stages, and grep -c (reads to EOF, never early-exits) for the netns check.
209 lines
9.7 KiB
Bash
Executable File
209 lines
9.7 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# chromebox-watchdog.sh [profile] — keep a chrome-box profile alive and healthy.
|
|
# Checks: 1) chromium process for the profile is running,
|
|
# 2) CDP responds and the chat page is present.
|
|
# If unhealthy: kill any stale chrome for the profile and relaunch headless
|
|
# via netvm-chrome.sh (profile dir persists session/cookies — the process is
|
|
# disposable, the state is not). Mirrors operator-646's container
|
|
# recover-after-rebuild.sh philosophy.
|
|
# Runs every 2 min via systemd timer chromebox-watchdog-<profile>.timer.
|
|
set -euo pipefail
|
|
export XDG_RUNTIME_DIR="${XDG_RUNTIME_DIR:-/run/user/$(id -u)}"
|
|
export DBUS_SESSION_BUS_ADDRESS="${DBUS_SESSION_BUS_ADDRESS:-unix:path=${XDG_RUNTIME_DIR}/bus}"
|
|
# Prevent overlapping runs: the timer fires every 2 min but a relaunch
|
|
# (kill + sleep 25 + chrome startup + page load) can exceed that, and two
|
|
# concurrent runs kill each others chrome (observed 2026-10-03: pip flapped
|
|
# with simultaneous "relaunch OK" and "relaunch FAILED").
|
|
LOCK="/tmp/chromebox-watchdog-${1:-pip}.lock"
|
|
exec 9>"$LOCK"
|
|
if ! flock -n 9; then
|
|
echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2
|
|
exit 0
|
|
fi
|
|
|
|
PROFILE="${1:-pip}"
|
|
NETVM_BIN="/home/super/Projects/NetVM/bin"
|
|
LOG="/home/super/Projects/NetVM/chromebox-watchdog.log"
|
|
CHROME_LOG="/home/super/Projects/NetVM/chromebox-${PROFILE}.log"
|
|
|
|
# Rotate a log file past 10MB (keep one generation)
|
|
rotate_log() {
|
|
local f="$1"
|
|
[ -f "$f" ] || return 0
|
|
local sz
|
|
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
|
|
if [ "$sz" -gt 10485760 ]; then
|
|
mv -f "$f" "$f.1"
|
|
echo "[$(date -u +%FT%TZ)] [$PROFILE] log rotated" > "$f"
|
|
fi
|
|
}
|
|
rotate_log "$LOG"
|
|
# CHROME_LOG rotation happens after PROFILE is set (see below)
|
|
|
|
case "$PROFILE" in
|
|
muse) CDP_PORT=9410 ;;
|
|
pip) CDP_PORT=9420 ;;
|
|
646) CDP_PORT=9430 ;;
|
|
opm) CDP_PORT=9440 ;;
|
|
*) echo "unknown profile: $PROFILE" >&2; exit 1 ;;
|
|
esac
|
|
rotate_log "$CHROME_LOG"
|
|
|
|
log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; }
|
|
|
|
cdp_list() {
|
|
"$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null
|
|
}
|
|
|
|
# HEALTH_FAIL_REASON is set by healthy() on failure: which stage broke.
|
|
HEALTH_FAIL_REASON=""
|
|
healthy() {
|
|
HEALTH_FAIL_REASON=""
|
|
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 \
|
|
|| { HEALTH_FAIL_REASON="no chromium process for profile"; return 1; }
|
|
local list
|
|
list="$(cdp_list)" \
|
|
|| { HEALTH_FAIL_REASON="CDP unreachable on :$CDP_PORT"; return 1; }
|
|
[[ "$list" == *'"type": "page"'* ]] \
|
|
|| { HEALTH_FAIL_REASON="CDP up but no page target in list"; return 1; }
|
|
[[ "$list" == *'"url": "https://muse.ai'* ]] \
|
|
|| { HEALTH_FAIL_REASON="CDP up but not on muse.ai"; return 1; }
|
|
warp_egress_healthy || return 1
|
|
return 0
|
|
}
|
|
|
|
# Stage 5: Warp egress health (2026-10-04). A partitioned browser (tunnel down,
|
|
# CDP green) passes stages 1-4 while being unable to reach muse.ai. Check the
|
|
# WireGuard handshake age and probe egress from inside the node's netns.
|
|
# Sets HEALTH_FAIL_REASON with a distinct "warp egress down" prefix so
|
|
# root-cause analysis can distinguish partitions from Chromium crashes.
|
|
warp_egress_healthy() {
|
|
local wg_out hs_line age_s code
|
|
wg_out="$(sudo -n ip netns exec "warp-${PROFILE}" wg show 2>/dev/null)" \
|
|
|| { HEALTH_FAIL_REASON="warp egress down (wg show failed in warp-${PROFILE})"; return 1; }
|
|
hs_line="$(printf '%s\n' "$wg_out" | grep -i "latest handshake" | head -1)"
|
|
[ -n "$hs_line" ] \
|
|
|| { HEALTH_FAIL_REASON="warp egress down (no WireGuard handshake in warp-${PROFILE})"; return 1; }
|
|
# "latest handshake: 1 minute, 41 seconds ago" -> total seconds
|
|
age_s="$(printf '%s\n' "$hs_line" | python3 -c '
|
|
import sys, re
|
|
s = sys.stdin.read()
|
|
m = re.search(r"(\d+)\s*hour", s); h = int(m.group(1)) if m else 0
|
|
m = re.search(r"(\d+)\s*minute", s); mi = int(m.group(1)) if m else 0
|
|
m = re.search(r"(\d+)\s*second", s); se = int(m.group(1)) if m else 0
|
|
print(h*3600 + mi*60 + se)
|
|
' 2>/dev/null)"
|
|
{ [ -n "$age_s" ] && [ "$age_s" -ge 0 ]; } 2>/dev/null \
|
|
|| { HEALTH_FAIL_REASON="warp egress down (unparseable handshake: $hs_line)"; return 1; }
|
|
[ "$age_s" -le 180 ] \
|
|
|| { HEALTH_FAIL_REASON="warp egress down (handshake ${age_s}s old in warp-${PROFILE})"; return 1; }
|
|
code="$(sudo -n ip netns exec "warp-${PROFILE}" curl -m 5 -s -o /dev/null -w "%{http_code}" "https://muse.ai/" 2>/dev/null)" \
|
|
|| { HEALTH_FAIL_REASON="warp egress down (egress probe curl failed in warp-${PROFILE})"; return 1; }
|
|
# Any 2xx/3xx means we reached muse.ai infra ("/" 307-redirects to the
|
|
# auth flow). The probe tests egress connectivity, not page content.
|
|
case "$code" in
|
|
2*|3*) return 0 ;;
|
|
*) HEALTH_FAIL_REASON="warp egress down (egress probe HTTP $code in warp-${PROFILE})"; return 1 ;;
|
|
esac
|
|
}
|
|
|
|
if healthy; then
|
|
exit 0
|
|
fi
|
|
|
|
# Relaunch-loop guard (2026-10-04): if the main browser for this profile
|
|
# launched <2 min ago it's probably still starting up (CDP not yet bound).
|
|
# Relaunching now would kill a healthy-but-slow cold start via netvm-chrome.sh
|
|
# and reset the startup clock every cycle (observed: opm piled up 5 chromiums
|
|
# because the guard only protected the kill step, not the relaunch). Skip the
|
|
# entire cycle instead.
|
|
# NOTE: match only the main browser process (--remote-debugging-port present,
|
|
# no --type= flag). Renderer/gpu children (--type=renderer etc.) start later
|
|
# than the main process and must not satisfy this check.
|
|
recent_pid=""
|
|
for _pid in $(pgrep -f "chromium.*--remote-debugging-port=${CDP_PORT}([[:space:]]|$)" 2>/dev/null); do
|
|
# Skip child processes (renderer, gpu, etc.) — only the main browser counts
|
|
if ps -o args= -p "$_pid" 2>/dev/null | grep -q -- "--type="; then
|
|
continue
|
|
fi
|
|
_start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0)
|
|
_now=$(date +%s)
|
|
if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then
|
|
recent_pid="$_pid"
|
|
break
|
|
fi
|
|
done
|
|
if [ -n "$recent_pid" ]; then
|
|
log "browser launched recently (pid $recent_pid), skipping relaunch (probably still starting)"
|
|
exit 0
|
|
fi
|
|
# Warp-partition recovery (2026-10-04): relaunching Chrome cannot fix a dead
|
|
# Warp tunnel — the new browser would fail the same egress check and the
|
|
# watchdog would loop. If the failure is warp-egress, restart the tunnel first
|
|
# (netvm-node-up.sh is idempotent); only fall through to the Chrome relaunch
|
|
# if the tunnel does not recover.
|
|
case "$HEALTH_FAIL_REASON" in
|
|
"warp egress down"*)
|
|
log "warp partition detected ($HEALTH_FAIL_REASON), restarting tunnel via netvm-node-up.sh"
|
|
sudo -n "$NETVM_BIN/netvm-node-up.sh" "$PROFILE" >>"$LOG" 2>&1 || true
|
|
sleep 5
|
|
if healthy; then
|
|
log "tunnel restart recovered warp egress, chrome relaunch not needed"
|
|
exit 0
|
|
fi
|
|
log "tunnel restart did not recover egress ($HEALTH_FAIL_REASON), proceeding with chrome relaunch"
|
|
;;
|
|
esac
|
|
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
|
|
# bracket trick so pkill never matches its own command line
|
|
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
|
|
pkill -f "chromium.*$pat" 2>/dev/null || true
|
|
sleep 3
|
|
# Clean up stale singleton symlinks that break subsequent browser startup
|
|
rm -f "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonLock" \
|
|
"/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonSocket" \
|
|
"/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chromium/SingletonCookie" 2>/dev/null || true
|
|
for _sc in $(systemctl --user list-units --type=scope --plain --no-legend 2>/dev/null | awk '{print $1}' | grep -E "^netvm-chrome-${PROFILE}-"); do
|
|
systemctl --user stop "$_sc" 2>/dev/null || true
|
|
done
|
|
|
|
# Relaunch in its own systemd scope so it survives this oneshot run.
|
|
# nohup/setsid do NOT escape: this timer's service uses KillMode=control-group
|
|
# and systemd SIGKILLs everything in the cgroup at teardown (observed
|
|
# 2026-10-03: every relaunch "recovered" then died seconds later). A transient
|
|
# scope escapes the service cgroup; the scoped process inherits these fds so
|
|
# the log redirect below still captures chromium's output.
|
|
# NOTE: systemd-run --scope WAITS for the scope's processes (even --no-block,
|
|
# verified 2026-10-03), so it must be backgrounded — the scope is an
|
|
# independent unit and outlives the wrapper.
|
|
# NOTE: --cdp-port is pinned explicitly. netvm-chrome.sh defaults to a
|
|
# hash-derived port (9222+...) which will NOT match the registry port that
|
|
# muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the
|
|
# launcher but is unreachable to the API (observed 2026-10-03: pip relaunched
|
|
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
|
|
# 9>&-: do NOT let the backgrounded launcher inherit the watchdog lock fd.
|
|
# Inherited flock fds wedge the lock forever (observed 2026-10-04: opm/muse
|
|
# launchers held their profile lock for 100+ min, every later watchdog run
|
|
# skipped as "another run in progress" — the watchdog was silently dead).
|
|
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
|
|
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
|
|
>>"$CHROME_LOG" 2>&1 < /dev/null 9>&- &
|
|
# Retry with backoff (2026-10-04): cold starts (fresh egress IP, Cloudflare
|
|
# handshake) can take >25s for the page title to appear. A single check after
|
|
# 25s kills working-but-slow browsers. Try up to 4 times, 15s apart (~60s
|
|
# window), logging each attempt. Only declare FAILED if all attempts fail.
|
|
_relaunch_ok=0
|
|
for _attempt in 1 2 3 4; do
|
|
sleep 15
|
|
if healthy; then
|
|
log "relaunch OK (attempt $_attempt)"
|
|
_relaunch_ok=1
|
|
break
|
|
fi
|
|
log "relaunch attempt $_attempt not healthy yet ($HEALTH_FAIL_REASON)"
|
|
done
|
|
if [ "$_relaunch_ok" -ne 1 ]; then
|
|
log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention"
|
|
exit 1
|
|
fi
|