Fix supervisor SIGKILL loop killing restarted browsers (opm/pip)
Root cause: agent-health.service runs Type=oneshot with the default KillMode=control-group. restart_browser() spawned the replacement chromium with nohup under the service, so systemd SIGKILLed it the moment the service exited. Every 5-min tick: FAIL -> restart -> RECOVERED -> SIGKILL at teardown. opm and pip were permanently dark and dm.py read masked it (empty output, exit 0). Fix: launch replacements via systemd-run --user --scope (backgrounded) so the browser lives in a transient scope outside the service cgroup and survives teardown. setsid does NOT escape either. Note: systemd-run --scope waits for the scope even with --no-block (verified 2026-10-03), hence the backgrounding. Same fix in chromebox-watchdog.sh (timers currently off). dm.py: dm_read() now uses run_full() and prints a WARNING to stderr with rc + last error line instead of failing silently on empty reads. Trailers: Session: sidechat/opm-blind-fix
This commit is contained in:
Executable
+80
@@ -0,0 +1,80 @@
|
||||
#!/usr/bin/env bash
|
||||
# chromebox-watchdog.sh [profile] — keep a chrome-box profile alive and healthy.
|
||||
# Checks: 1) chromium process for the profile is running,
|
||||
# 2) CDP responds and the chat page is present.
|
||||
# If unhealthy: kill any stale chrome for the profile and relaunch headless
|
||||
# via netvm-chrome.sh (profile dir persists session/cookies — the process is
|
||||
# disposable, the state is not). Mirrors operator-646's container
|
||||
# recover-after-rebuild.sh philosophy.
|
||||
# Runs every 2 min via systemd timer chromebox-watchdog-<profile>.timer.
|
||||
set -euo pipefail
|
||||
# Prevent overlapping runs: the timer fires every 2 min but a relaunch
|
||||
# (kill + sleep 25 + chrome startup + page load) can exceed that, and two
|
||||
# concurrent runs kill each others chrome (observed 2026-10-03: pip flapped
|
||||
# with simultaneous "relaunch OK" and "relaunch FAILED").
|
||||
LOCK="/tmp/chromebox-watchdog-${1:-pip}.lock"
|
||||
exec 9>"$LOCK"
|
||||
if ! flock -n 9; then
|
||||
echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2
|
||||
exit 0
|
||||
fi
|
||||
|
||||
PROFILE="${1:-pip}"
|
||||
NETVM_BIN="/home/super/Projects/NetVM/bin"
|
||||
LOG="/home/super/Projects/NetVM/chromebox-watchdog.log"
|
||||
|
||||
case "$PROFILE" in
|
||||
muse) CDP_PORT=9410 ;;
|
||||
pip) CDP_PORT=9420 ;;
|
||||
646) CDP_PORT=9430 ;;
|
||||
opm) CDP_PORT=9440 ;;
|
||||
*) echo "unknown profile: $PROFILE" >&2; exit 1 ;;
|
||||
esac
|
||||
|
||||
log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; }
|
||||
|
||||
cdp_list() {
|
||||
"$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null
|
||||
}
|
||||
|
||||
healthy() {
|
||||
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 || return 1
|
||||
local list
|
||||
list="$(cdp_list)" || return 1
|
||||
echo "$list" | grep -q '"title": "Chat' || return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
if healthy; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
log "unhealthy, relaunching chromebox"
|
||||
# bracket trick so pkill never matches its own command line
|
||||
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
|
||||
pkill -f "chromium.*$pat" 2>/dev/null || true
|
||||
sleep 3
|
||||
# Relaunch in its own systemd scope so it survives this oneshot run.
|
||||
# nohup/setsid do NOT escape: this timer's service uses KillMode=control-group
|
||||
# and systemd SIGKILLs everything in the cgroup at teardown (observed
|
||||
# 2026-10-03: every relaunch "recovered" then died seconds later). A transient
|
||||
# scope escapes the service cgroup; the scoped process inherits these fds so
|
||||
# the log redirect below still captures chromium's output.
|
||||
# NOTE: systemd-run --scope WAITS for the scope's processes (even --no-block,
|
||||
# verified 2026-10-03), so it must be backgrounded — the scope is an
|
||||
# independent unit and outlives the wrapper.
|
||||
# NOTE: --cdp-port is pinned explicitly. netvm-chrome.sh defaults to a
|
||||
# hash-derived port (9222+...) which will NOT match the registry port that
|
||||
# muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the
|
||||
# launcher but is unreachable to the API (observed 2026-10-03: pip relaunched
|
||||
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
|
||||
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
|
||||
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
|
||||
>>"$LOG" 2>&1 < /dev/null &
|
||||
sleep 25
|
||||
if healthy; then
|
||||
log "relaunch OK"
|
||||
else
|
||||
log "relaunch FAILED — needs operator attention"
|
||||
exit 1
|
||||
fi
|
||||
Reference in New Issue
Block a user