From 343920e0c3f48cd9e04e51f1b83ef6c877bb20e6 Mon Sep 17 00:00:00 2001 From: operator-main Date: Sun, 4 Oct 2026 12:40:56 +0000 Subject: [PATCH] Watchdog upgrades: stage-specific logging + new CDP relay watchdog MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit chromebox-watchdog.sh: HEALTH_FAIL_REASON pinpoints which health stage failed (no process / CDP unreachable / no Chat page); chromium stdout redirected to per-profile chromebox-.log; 10MB log rotation (one generation); Chat title match relaxed to .*Chat. bin/cdp-relay-watchdog.sh (new): keeps per-node CDP relays alive. Two-stage check: (1) host veth IP assigned (fail-loud, no auto-fix — veth recreation touches WireGuard/iptables), (2) relay connectivity via curl to veth IP:port (never trust pidfiles — observed stale 2026-10-04). Restarts dead/misrouted relays in-netns. Runs via systemd timer every 5min. Pattern mirrors chromebox-watchdog.sh. Session: sidechat/chromebox-ops --- bin/cdp-relay-watchdog.sh | 121 ++++++++++++++++++++++++++++++++++++++ bin/chromebox-watchdog.sh | 34 +++++++++-- 2 files changed, 149 insertions(+), 6 deletions(-) create mode 100755 bin/cdp-relay-watchdog.sh diff --git a/bin/cdp-relay-watchdog.sh b/bin/cdp-relay-watchdog.sh new file mode 100755 index 0000000..948312a --- /dev/null +++ b/bin/cdp-relay-watchdog.sh @@ -0,0 +1,121 @@ +#!/usr/bin/env bash +# cdp-relay-watchdog.sh — keep per-node CDP relays alive and correctly routed. +# Two-stage health check: +# 1. Host veth IP must be assigned (veth_healthy). Without it the relay is +# unreachable from the host no matter how many times we restart it — +# observed 2026-10-04 (muse/pip veths existed but had no IPs). FAIL_LOUD +# in the log; do NOT auto-fix (veth recreation touches WireGuard/iptables). +# 2. Relay connectivity (relay_healthy): curl to veth IP:port, not pidfile +# (which goes stale and lies — observed 2026-10-04). +# If a relay is down or misrouted: kill it and restart via the exact +# netvm-node-up.sh relay invocation inside the node's netns. +# Runs every 5 min via systemd timer cdp-relay-watchdog.timer. +# Pattern mirrors chromebox-watchdog.sh (stage-specific logging, rotation). +set -euo pipefail +LOCK="/tmp/cdp-relay-watchdog.lock" +exec 9>"$LOCK" +if ! flock -n 9; then + echo "[$(date -u +%FT%TZ)] another relay watchdog run in progress, skipping" >&2 + exit 0 +fi + +NETVM_BIN="/home/super/Projects/NetVM/bin" +LOG="/home/super/Projects/NetVM/cdp-relay-watchdog.log" + +rotate_log() { + local f="$1" + [ -f "$f" ] || return 0 + local sz + sz=$(stat -c%s "$f" 2>/dev/null || echo 0) + if [ "$sz" -gt 10485760 ]; then + mv -f "$f" "$f.1" + echo "[$(date -u +%FT%TZ)] log rotated" > "$f" + fi +} +rotate_log "$LOG" +log() { echo "[$(date -u +%FT%TZ)] $*" | tee -a "$LOG"; } + +# node -> "veth_ip:port" via netvm-names.sh (hash-derived, don't hardcode) +relay_target() { + local node="$1" + # shellcheck disable=SC1091 + . "$NETVM_BIN/netvm-names.sh" + netvm_names "$node" || return 1 + # CDP_PORT_OVERRIDE pins registry ports; fall back to hash-derived + local port="${CDP_PORT_OVERRIDE:-$CDP_PORT}" + case "$node" in + muse) port=9410 ;; pip) port=9420 ;; 646) port=9430 ;; opm) port=9440 ;; + esac + echo "$PEER_IP:$port" +} +node_port() { echo "${1##*:}"; } + +# node -> "VETH GW" via netvm-names.sh +node_veth() { + local node="$1" + # shellcheck disable=SC1091 + . "$NETVM_BIN/netvm-names.sh" + netvm_names "$node" || return 1 + echo "$VETH $GW" +} + +# Is the host-side veth IP assigned? The relay listens on the netns-side peer +# IP; the host reaches it via the veth interface's GW address. If the GW IP is +# missing, the relay is unreachable from the host — restarting the relay is +# pointless and masks the real problem. +veth_healthy() { + local node="$1" veth gw + read -r veth gw <<< "$(node_veth "$node")" || return 1 + ip addr show dev "$veth" 2>/dev/null | grep -q "inet ${gw}/" || return 1 + return 0 +} + +relay_healthy() { + local node="$1" target + target="$(relay_target "$node")" || return 1 + curl -s -m 8 "http://$target/json/version" 2>/dev/null | grep -q '"Browser"' || return 1 + return 0 +} + +restart_relay() { + local node="$1" target port veth_ip netns + target="$(relay_target "$node")" + veth_ip="${target%%:*}" + port="${target##*:}" + netns="warp-$node" + # Kill any existing relay for this node's port (correct or not) + pkill -f "netvm-cdp-relay.py .* $port 127.0.0.1 $port" 2>/dev/null || true + sleep 2 + # Launch inside the netns, listening on the veth IP (host-reachable) + sudo -n ip netns exec "$netns" setsid nohup python3 \ + "$NETVM_BIN/netvm-cdp-relay.py" "$veth_ip" "$port" 127.0.0.1 "$port" \ + >>"$LOG" 2>&1 < /dev/null & + sleep 5 + if relay_healthy "$node"; then + log "[$node] relay restarted OK on $target" + return 0 + else + log "[$node] relay restart FAILED on $target — needs operator attention" + return 1 + fi +} + +FAILED=0 +for node in muse pip 646 opm; do + # Stage 1: host veth IP. Fail loud, skip relay restart (pointless). + if ! veth_healthy "$node"; then + read -r veth gw <<< "$(node_veth "$node")" + log "[$node] FAIL_LOUD: host veth $veth missing IP $gw — relay unreachable, needs netvm-node-up.sh $node (manual)" + FAILED=1 + continue + fi + # Stage 2: relay connectivity. + if relay_healthy "$node"; then + continue + fi + target="$(relay_target "$node")" + log "[$node] relay unhealthy on $target, restarting" + restart_relay "$node" || FAILED=1 +done + +exit $FAILED diff --git a/bin/chromebox-watchdog.sh b/bin/chromebox-watchdog.sh index 480cf40..c2f55b1 100755 --- a/bin/chromebox-watchdog.sh +++ b/bin/chromebox-watchdog.sh @@ -22,6 +22,21 @@ fi PROFILE="${1:-pip}" NETVM_BIN="/home/super/Projects/NetVM/bin" LOG="/home/super/Projects/NetVM/chromebox-watchdog.log" +CHROME_LOG="/home/super/Projects/NetVM/chromebox-${PROFILE}.log" + +# Rotate a log file past 10MB (keep one generation) +rotate_log() { + local f="$1" + [ -f "$f" ] || return 0 + local sz + sz=$(stat -c%s "$f" 2>/dev/null || echo 0) + if [ "$sz" -gt 10485760 ]; then + mv -f "$f" "$f.1" + echo "[$(date -u +%FT%TZ)] [$PROFILE] log rotated" > "$f" + fi +} +rotate_log "$LOG" +# CHROME_LOG rotation happens after PROFILE is set (see below) case "$PROFILE" in muse) CDP_PORT=9410 ;; @@ -30,6 +45,7 @@ case "$PROFILE" in opm) CDP_PORT=9440 ;; *) echo "unknown profile: $PROFILE" >&2; exit 1 ;; esac +rotate_log "$CHROME_LOG" log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; } @@ -37,11 +53,17 @@ cdp_list() { "$NETVM_BIN/netvm-exec.sh" "$PROFILE" -- curl -s -m 8 "http://127.0.0.1:$CDP_PORT/json/list" 2>/dev/null } +# HEALTH_FAIL_REASON is set by healthy() on failure: which stage broke. +HEALTH_FAIL_REASON="" healthy() { - pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 || return 1 + HEALTH_FAIL_REASON="" + pgrep -f "chromium.*profiles/${PROFILE}/" >/dev/null 2>&1 \ + || { HEALTH_FAIL_REASON="no chromium process for profile"; return 1; } local list - list="$(cdp_list)" || return 1 - echo "$list" | grep -q '"title": "Chat' || return 1 + list="$(cdp_list)" \ + || { HEALTH_FAIL_REASON="CDP unreachable on :$CDP_PORT"; return 1; } + echo "$list" | grep -q '"title": ".*Chat' \ + || { HEALTH_FAIL_REASON="CDP up but no Chat page in target list"; return 1; } return 0 } @@ -49,7 +71,7 @@ if healthy; then exit 0 fi -log "unhealthy, relaunching chromebox" +log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox" # bracket trick so pkill never matches its own command line pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/" pkill -f "chromium.*$pat" 2>/dev/null || true @@ -70,11 +92,11 @@ sleep 3 # on 9278 instead of 9420, watchdog looped on "relaunch FAILED"). systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \ "$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \ - >>"$LOG" 2>&1 < /dev/null & + >>"$CHROME_LOG" 2>&1 < /dev/null & sleep 25 if healthy; then log "relaunch OK" else - log "relaunch FAILED — needs operator attention" + log "relaunch FAILED ($HEALTH_FAIL_REASON) \u2014 needs operator attention" exit 1 fi