Watchdog upgrades: stage-specific logging + new CDP relay watchdog
chromebox-watchdog.sh: HEALTH_FAIL_REASON pinpoints which health stage failed (no process / CDP unreachable / no Chat page); chromium stdout redirected to per-profile chromebox-<profile>.log; 10MB log rotation (one generation); Chat title match relaxed to .*Chat. bin/cdp-relay-watchdog.sh (new): keeps per-node CDP relays alive. Two-stage check: (1) host veth IP assigned (fail-loud, no auto-fix — veth recreation touches WireGuard/iptables), (2) relay connectivity via curl to veth IP:port (never trust pidfiles — observed stale 2026-10-04). Restarts dead/misrouted relays in-netns. Runs via systemd timer every 5min. Pattern mirrors chromebox-watchdog.sh. Session: sidechat/chromebox-ops
This commit is contained in:
Executable
+121
@@ -0,0 +1,121 @@
|
||||
#!/usr/bin/env bash
|
||||
# cdp-relay-watchdog.sh — keep per-node CDP relays alive and correctly routed.
|
||||
# Two-stage health check:
|
||||
# 1. Host veth IP must be assigned (veth_healthy). Without it the relay is
|
||||
# unreachable from the host no matter how many times we restart it —
|
||||
# observed 2026-10-04 (muse/pip veths existed but had no IPs). FAIL_LOUD
|
||||
# in the log; do NOT auto-fix (veth recreation touches WireGuard/iptables).
|
||||
# 2. Relay connectivity (relay_healthy): curl to veth IP:port, not pidfile
|
||||
# (which goes stale and lies — observed 2026-10-04).
|
||||
# If a relay is down or misrouted: kill it and restart via the exact
|
||||
# netvm-node-up.sh relay invocation inside the node's netns.
|
||||
# Runs every 5 min via systemd timer cdp-relay-watchdog.timer.
|
||||
# Pattern mirrors chromebox-watchdog.sh (stage-specific logging, rotation).
|
||||
set -euo pipefail
|
||||
LOCK="/tmp/cdp-relay-watchdog.lock"
|
||||
exec 9>"$LOCK"
|
||||
if ! flock -n 9; then
|
||||
echo "[$(date -u +%FT%TZ)] another relay watchdog run in progress, skipping" >&2
|
||||
exit 0
|
||||
fi
|
||||
|
||||
NETVM_BIN="/home/super/Projects/NetVM/bin"
|
||||
LOG="/home/super/Projects/NetVM/cdp-relay-watchdog.log"
|
||||
|
||||
rotate_log() {
|
||||
local f="$1"
|
||||
[ -f "$f" ] || return 0
|
||||
local sz
|
||||
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
|
||||
if [ "$sz" -gt 10485760 ]; then
|
||||
mv -f "$f" "$f.1"
|
||||
echo "[$(date -u +%FT%TZ)] log rotated" > "$f"
|
||||
fi
|
||||
}
|
||||
rotate_log "$LOG"
|
||||
log() { echo "[$(date -u +%FT%TZ)] $*" | tee -a "$LOG"; }
|
||||
|
||||
# node -> "veth_ip:port" via netvm-names.sh (hash-derived, don't hardcode)
|
||||
relay_target() {
|
||||
local node="$1"
|
||||
# shellcheck disable=SC1091
|
||||
. "$NETVM_BIN/netvm-names.sh"
|
||||
netvm_names "$node" || return 1
|
||||
# CDP_PORT_OVERRIDE pins registry ports; fall back to hash-derived
|
||||
local port="${CDP_PORT_OVERRIDE:-$CDP_PORT}"
|
||||
case "$node" in
|
||||
muse) port=9410 ;; pip) port=9420 ;; 646) port=9430 ;; opm) port=9440 ;;
|
||||
esac
|
||||
echo "$PEER_IP:$port"
|
||||
}
|
||||
node_port() { echo "${1##*:}"; }
|
||||
|
||||
# node -> "VETH GW" via netvm-names.sh
|
||||
node_veth() {
|
||||
local node="$1"
|
||||
# shellcheck disable=SC1091
|
||||
. "$NETVM_BIN/netvm-names.sh"
|
||||
netvm_names "$node" || return 1
|
||||
echo "$VETH $GW"
|
||||
}
|
||||
|
||||
# Is the host-side veth IP assigned? The relay listens on the netns-side peer
|
||||
# IP; the host reaches it via the veth interface's GW address. If the GW IP is
|
||||
# missing, the relay is unreachable from the host — restarting the relay is
|
||||
# pointless and masks the real problem.
|
||||
veth_healthy() {
|
||||
local node="$1" veth gw
|
||||
read -r veth gw <<< "$(node_veth "$node")" || return 1
|
||||
ip addr show dev "$veth" 2>/dev/null | grep -q "inet ${gw}/" || return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
relay_healthy() {
|
||||
local node="$1" target
|
||||
target="$(relay_target "$node")" || return 1
|
||||
curl -s -m 8 "http://$target/json/version" 2>/dev/null | grep -q '"Browser"' || return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
restart_relay() {
|
||||
local node="$1" target port veth_ip netns
|
||||
target="$(relay_target "$node")"
|
||||
veth_ip="${target%%:*}"
|
||||
port="${target##*:}"
|
||||
netns="warp-$node"
|
||||
# Kill any existing relay for this node's port (correct or not)
|
||||
pkill -f "netvm-cdp-relay.py .* $port 127.0.0.1 $port" 2>/dev/null || true
|
||||
sleep 2
|
||||
# Launch inside the netns, listening on the veth IP (host-reachable)
|
||||
sudo -n ip netns exec "$netns" setsid nohup python3 \
|
||||
"$NETVM_BIN/netvm-cdp-relay.py" "$veth_ip" "$port" 127.0.0.1 "$port" \
|
||||
>>"$LOG" 2>&1 < /dev/null &
|
||||
sleep 5
|
||||
if relay_healthy "$node"; then
|
||||
log "[$node] relay restarted OK on $target"
|
||||
return 0
|
||||
else
|
||||
log "[$node] relay restart FAILED on $target — needs operator attention"
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
FAILED=0
|
||||
for node in muse pip 646 opm; do
|
||||
# Stage 1: host veth IP. Fail loud, skip relay restart (pointless).
|
||||
if ! veth_healthy "$node"; then
|
||||
read -r veth gw <<< "$(node_veth "$node")"
|
||||
log "[$node] FAIL_LOUD: host veth $veth missing IP $gw — relay unreachable, needs netvm-node-up.sh $node (manual)"
|
||||
FAILED=1
|
||||
continue
|
||||
fi
|
||||
# Stage 2: relay connectivity.
|
||||
if relay_healthy "$node"; then
|
||||
continue
|
||||
fi
|
||||
target="$(relay_target "$node")"
|
||||
log "[$node] relay unhealthy on $target, restarting"
|
||||
restart_relay "$node" || FAILED=1
|
||||
done
|
||||
|
||||
exit $FAILED
|
||||
@@ -22,6 +22,21 @@ fi
|
||||
PROFILE="${1:-pip}"
|
||||
NETVM_BIN="/home/super/Projects/NetVM/bin"
|
||||
LOG="/home/super/Projects/NetVM/chromebox-watchdog.log"
|
||||
CHROME_LOG="/home/super/Projects/NetVM/chromebox-${PROFILE}.log"
|
||||
|
||||
# Rotate a log file past 10MB (keep one generation)
|
||||
rotate_log() {
|
||||
local f="$1"
|
||||
[ -f "$f" ] || return 0
|
||||
local sz
|
||||
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
|
||||
if [ "$sz" -gt 10485760 ]; then
|
||||
mv -f "$f" "$f.1"
|
||||
echo "[$(date -u +%FT%TZ)] [$PROFILE] log rotated" > "$f"
|
||||
fi
|
||||
}
|
||||
rotate_log "$LOG"
|
||||
# CHROME_LOG rotation happens after PROFILE is set (see below)
|
||||
|
||||
case "$PROFILE" in
|
||||
muse) CDP_PORT=9410 ;;
|
||||
@@ -30,6 +45,7 @@ case "$PROFILE" in
|
||||
opm) CDP_PORT=9440 ;;
|
||||
*) echo "unknown profile: $PROFILE" >&2; exit 1 ;;
|
||||
esac
|
||||
rotate_log "$CHROME_LOG"
|
||||
|
||||
log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; }
|
||||
|
||||
@@ -37,11 +53,17 @@ cdp_list() {
|
||||
"$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null
|
||||
}
|
||||
|
||||
# HEALTH_FAIL_REASON is set by healthy() on failure: which stage broke.
|
||||
HEALTH_FAIL_REASON=""
|
||||
healthy() {
|
||||
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 || return 1
|
||||
HEALTH_FAIL_REASON=""
|
||||
pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 \
|
||||
|| { HEALTH_FAIL_REASON="no chromium process for profile"; return 1; }
|
||||
local list
|
||||
list="$(cdp_list)" || return 1
|
||||
echo "$list" | grep -q '"title": "Chat' || return 1
|
||||
list="$(cdp_list)" \
|
||||
|| { HEALTH_FAIL_REASON="CDP unreachable on :$CDP_PORT"; return 1; }
|
||||
echo "$list" | grep -q '"title": ".*Chat' \
|
||||
|| { HEALTH_FAIL_REASON="CDP up but no Chat page in target list"; return 1; }
|
||||
return 0
|
||||
}
|
||||
|
||||
@@ -49,7 +71,7 @@ if healthy; then
|
||||
exit 0
|
||||
fi
|
||||
|
||||
log "unhealthy, relaunching chromebox"
|
||||
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
|
||||
# bracket trick so pkill never matches its own command line
|
||||
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
|
||||
pkill -f "chromium.*$pat" 2>/dev/null || true
|
||||
@@ -70,11 +92,11 @@ sleep 3
|
||||
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
|
||||
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
|
||||
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
|
||||
>>"$LOG" 2>&1 < /dev/null &
|
||||
>>"$CHROME_LOG" 2>&1 < /dev/null &
|
||||
sleep 25
|
||||
if healthy; then
|
||||
log "relaunch OK"
|
||||
else
|
||||
log "relaunch FAILED — needs operator attention"
|
||||
log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
Reference in New Issue
Block a user