#!/usr/bin/env bash # cdp-relay-watchdog.sh — keep per-node CDP relays alive and correctly routed. # Two-stage health check: # 1. Host veth IP must be assigned (veth_healthy). Without it the relay is # unreachable from the host no matter how many times we restart it — # observed 2026-10-04 (muse/pip veths existed but had no IPs). FAIL_LOUD # in the log; do NOT auto-fix (veth recreation touches WireGuard/iptables). # 2. Relay connectivity (relay_healthy): curl to veth IP:port, not pidfile # (which goes stale and lies — observed 2026-10-04). # If a relay is down or misrouted: kill it and restart via the exact # netvm-node-up.sh relay invocation inside the node's netns. # Runs every 5 min via systemd timer cdp-relay-watchdog.timer. # Pattern mirrors chromebox-watchdog.sh (stage-specific logging, rotation). set -euo pipefail LOCK="/tmp/cdp-relay-watchdog.lock" # Tests source this file with CDP_RELAY_WATCHDOG_LIB_ONLY=1: they call # helpers without running checks, so no lock is needed. if [ "${CDP_RELAY_WATCHDOG_LIB_ONLY:-}" != "1" ]; then exec 9>"$LOCK" if ! flock -n 9; then echo "[$(date -u +%FT%TZ)] another relay watchdog run in progress, skipping" >&2 exit 0 fi fi NETVM_BIN="/home/super/Projects/NetVM/bin" LOG="/home/super/Projects/NetVM/cdp-relay-watchdog.log" rotate_log() { local f="$1" [ -f "$f" ] || return 0 local sz sz=$(stat -c%s "$f" 2>/dev/null || echo 0) if [ "$sz" -gt 10485760 ]; then mv -f "$f" "$f.1" echo "[$(date -u +%FT%TZ)] log rotated" > "$f" fi } rotate_log "$LOG" log() { echo "[$(date -u +%FT%TZ)] $*" | tee -a "$LOG"; } # node -> "veth_ip:port" via netvm-names.sh (hash-derived, don't hardcode) relay_target() { local node="$1" reg_port="" # shellcheck disable=SC1091 . "$NETVM_BIN/netvm-names.sh" netvm_names "$node" || return 1 # The registry is the source of truth for ports (new nodes propagate # automatically); netvm-names pinning is the fallback. if reg_port=$("$NETVM_BIN/netvm-registry.py" "$node" 2>/dev/null); then [ -n "$reg_port" ] && CDP_PORT="$reg_port" fi echo "$PEER_IP:$CDP_PORT" } node_port() { echo "${1##*:}"; } # node -> "VETH GW" via netvm-names.sh node_veth() { local node="$1" # shellcheck disable=SC1091 . "$NETVM_BIN/netvm-names.sh" netvm_names "$node" || return 1 echo "$VETH $GW" } # Is the host-side veth IP assigned? The relay listens on the netns-side peer # IP; the host reaches it via the veth interface's GW address. If the GW IP is # missing, the relay is unreachable from the host — restarting the relay is # pointless and masks the real problem. veth_healthy() { local node="$1" veth gw read -r veth gw <<< "$(node_veth "$node")" || return 1 ip addr show dev "$veth" 2>/dev/null | grep -q "inet ${gw}/" || return 1 return 0 } relay_healthy() { local node="$1" target target="$(relay_target "$node")" || return 1 curl -s -m 8 "http://$target/json/version" 2>/dev/null | grep -q '"Browser"' || return 1 return 0 } restart_relay() { local node="$1" target port veth_ip netns target="$(relay_target "$node")" veth_ip="${target%%:*}" port="${target##*:}" netns="warp-$node" # Kill any existing relay for this node's port (correct or not) # Relays are root-owned (started via sudo ip netns exec); the timer runs as # super, so the kill needs sudo too. Without it pkill fails EPERM silently # and the "restart" false-positives via SO_REUSEADDR double-bind. sudo -n pkill -f "netvm-cdp-relay.py .* $port 127.0.0.1 $port" 2>/dev/null || true sleep 2 # Launch inside the netns, listening on the veth IP (host-reachable) sudo -n ip netns exec "$netns" setsid nohup python3 \ "$NETVM_BIN/netvm-cdp-relay.py" "$veth_ip" "$port" 127.0.0.1 "$port" \ >>"$LOG" 2>&1 < /dev/null & sleep 5 if relay_healthy "$node"; then log "[$node] relay restarted OK on $target" return 0 else log "[$node] relay restart FAILED on $target — needs operator attention" return 1 fi } # Registry-driven node list: every active node gets relay supervision # (the old hardcoded 4-node list left def/dev unsupervised — 2026-10-06). watched_nodes() { "$NETVM_BIN/netvm-registry.py" 2>/dev/null | cut -d: -f1 } # Allow sourcing for tests without running checks. if [ "${CDP_RELAY_WATCHDOG_LIB_ONLY:-}" = "1" ]; then return 0 2>/dev/null || exit 0 fi FAILED=0 NODES="$(watched_nodes)" if [ -z "$NODES" ]; then log "FAIL_LOUD: node registry empty/unreadable, skipping run" exit 1 fi # shellcheck disable=SC2086 (intended word splitting: one node per word) for node in $NODES; do # Stage 1: host veth IP. Fail loud, skip relay restart (pointless). if ! veth_healthy "$node"; then read -r veth gw <<< "$(node_veth "$node")" log "[$node] FAIL_LOUD: host veth $veth missing IP $gw — relay unreachable, needs netvm-node-up.sh $node (manual)" FAILED=1 continue fi # Stage 2: relay connectivity. if relay_healthy "$node"; then continue fi target="$(relay_target "$node")" log "[$node] relay unhealthy on $target, restarting" restart_relay "$node" || FAILED=1 done exit $FAILED