Watchdog upgrades: stage-specific logging + new CDP relay watchdog

chromebox-watchdog.sh: HEALTH_FAIL_REASON pinpoints which health stage failed (no process / CDP unreachable / no Chat page); chromium stdout redirected to per-profile chromebox-<profile>.log; 10MB log rotation (one generation); Chat title match relaxed to .*Chat.

bin/cdp-relay-watchdog.sh (new): keeps per-node CDP relays alive. Two-stage check: (1) host veth IP assigned (fail-loud, no auto-fix — veth recreation touches WireGuard/iptables), (2) relay connectivity via curl to veth IP:port (never trust pidfiles — observed stale 2026-10-04). Restarts dead/misrouted relays in-netns. Runs via systemd timer every 5min. Pattern mirrors chromebox-watchdog.sh.

Session: sidechat/chromebox-ops
This commit is contained in:
operator-main
2026-10-04 12:40:56 +00:00
parent 3dd9d9ffaf
commit 343920e0c3
2 changed files with 149 additions and 6 deletions
+28 -6
View File
@@ -22,6 +22,21 @@ fi
PROFILE="${1:-pip}"
NETVM_BIN="/home/super/Projects/NetVM/bin"
LOG="/home/super/Projects/NetVM/chromebox-watchdog.log"
CHROME_LOG="/home/super/Projects/NetVM/chromebox-${PROFILE}.log"
# Rotate a log file past 10MB (keep one generation)
rotate_log() {
local f="$1"
[ -f "$f" ] || return 0
local sz
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
if [ "$sz" -gt 10485760 ]; then
mv -f "$f" "$f.1"
echo "[$(date -u +%FT%TZ)] [$PROFILE] log rotated" > "$f"
fi
}
rotate_log "$LOG"
# CHROME_LOG rotation happens after PROFILE is set (see below)
case "$PROFILE" in
muse) CDP_PORT=9410 ;;
@@ -30,6 +45,7 @@ case "$PROFILE" in
opm) CDP_PORT=9440 ;;
*) echo "unknown profile: $PROFILE" >&2; exit 1 ;;
esac
rotate_log "$CHROME_LOG"
log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; }
@@ -37,11 +53,17 @@ cdp_list() {
"$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null
}
# HEALTH_FAIL_REASON is set by healthy() on failure: which stage broke.
HEALTH_FAIL_REASON=""
healthy() {
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 || return 1
HEALTH_FAIL_REASON=""
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 \
|| { HEALTH_FAIL_REASON="no chromium process for profile"; return 1; }
local list
list="$(cdp_list)" || return 1
echo "$list" | grep -q '"title": "Chat' || return 1
list="$(cdp_list)" \
|| { HEALTH_FAIL_REASON="CDP unreachable on :$CDP_PORT"; return 1; }
echo "$list" | grep -q '"title": ".*Chat' \
|| { HEALTH_FAIL_REASON="CDP up but no Chat page in target list"; return 1; }
return 0
}
@@ -49,7 +71,7 @@ if healthy; then
exit 0
fi
log "unhealthy, relaunching chromebox"
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
# bracket trick so pkill never matches its own command line
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
pkill -f "chromium.*$pat" 2>/dev/null || true
@@ -70,11 +92,11 @@ sleep 3
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
>>"$LOG" 2>&1 < /dev/null &
>>"$CHROME_LOG" 2>&1 < /dev/null &
sleep 25
if healthy; then
log "relaunch OK"
else
log "relaunch FAILED — needs operator attention"
log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention"
exit 1
fi