chromebox-watchdog: fix kill loop on slow cold starts

Two fixes: (1) retry health check 4x with 15s gaps after relaunch instead of single 25s check; (2) skip kill if browser launched <2min ago (probably still starting). Prevents watchdog from killing a working-but-slow browser.

Session: sidechat/chromebox-ops
This commit is contained in:
operator-main
2026-10-04 13:17:13 +00:00
parent 5f0a77d04b
commit a32670731f
+31 -4
View File
@@ -72,6 +72,22 @@ if healthy; then
fi fi
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox" log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
# Kill-loop guard (2026-10-04): if a chromium for this profile launched <2 min
# ago it's probably still loading the page — killing it just restarts the
# loop (observed: 646 killed twice while the page was still loading).
recent_pid=""
for _pid in $(pgrep -f "chromium.*profiles/${PROFILE}/" 2>/dev/null); do
_start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0)
_now=$(date +%s)
if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then
recent_pid="$_pid"
break
fi
done
if [ -n "$recent_pid" ]; then
log "browser launched recently (pid $recent_pid), skipping kill (probably still starting)"
exit 0
fi
# bracket trick so pkill never matches its own command line # bracket trick so pkill never matches its own command line
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/" pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
pkill -f "chromium.*$pat" 2>/dev/null || true pkill -f "chromium.*$pat" 2>/dev/null || true
@@ -93,10 +109,21 @@ sleep 3
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \ systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \ "$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
>>"$CHROME_LOG" 2>&1 < /dev/null & >>"$CHROME_LOG" 2>&1 < /dev/null &
sleep 25 # Retry with backoff (2026-10-04): cold starts (fresh egress IP, Cloudflare
if healthy; then # handshake) can take >25s for the page title to appear. A single check after
log "relaunch OK" # 25s kills working-but-slow browsers. Try up to 4 times, 15s apart (~60s
else # window), logging each attempt. Only declare FAILED if all attempts fail.
_relaunch_ok=0
for _attempt in 1 2 3 4; do
sleep 15
if healthy; then
log "relaunch OK (attempt $_attempt)"
_relaunch_ok=1
break
fi
log "relaunch attempt $_attempt not healthy yet ($HEALTH_FAIL_REASON)"
done
if [ "$_relaunch_ok" -ne 1 ]; then
log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention" log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention"
exit 1 exit 1
fi fi