Files
box/bin/cdp-relay-watchdog.sh
T

146 lines
5.0 KiB
Bash
Executable File

#!/usr/bin/env bash
# cdp-relay-watchdog.sh — keep per-node CDP relays alive and correctly routed.
# Two-stage health check:
# 1. Host veth IP must be assigned (veth_healthy). Without it the relay is
# unreachable from the host no matter how many times we restart it —
# observed 2026-10-04 (muse/pip veths existed but had no IPs). FAIL_LOUD
# in the log; do NOT auto-fix (veth recreation touches WireGuard/iptables).
# 2. Relay connectivity (relay_healthy): curl to veth IP:port, not pidfile
# (which goes stale and lies — observed 2026-10-04).
# If a relay is down or misrouted: kill it and restart via the exact
# netvm-node-up.sh relay invocation inside the node's netns.
# Runs every 5 min via systemd timer cdp-relay-watchdog.timer.
# Pattern mirrors chromebox-watchdog.sh (stage-specific logging, rotation).
set -euo pipefail
LOCK="/tmp/cdp-relay-watchdog.lock"
# Tests source this file with CDP_RELAY_WATCHDOG_LIB_ONLY=1: they call
# helpers without running checks, so no lock is needed.
if [ "${CDP_RELAY_WATCHDOG_LIB_ONLY:-}" != "1" ]; then
exec 9>"$LOCK"
if ! flock -n 9; then
echo "[$(date -u +%FT%TZ)] another relay watchdog run in progress, skipping" >&2
exit 0
fi
fi
NETVM_BIN="/home/super/Projects/NetVM/bin"
LOG="/home/super/Projects/NetVM/cdp-relay-watchdog.log"
rotate_log() {
local f="$1"
[ -f "$f" ] || return 0
local sz
sz=$(stat -c%s "$f" 2>/dev/null || echo 0)
if [ "$sz" -gt 10485760 ]; then
mv -f "$f" "$f.1"
echo "[$(date -u +%FT%TZ)] log rotated" > "$f"
fi
}
rotate_log "$LOG"
log() { echo "[$(date -u +%FT%TZ)] $*" | tee -a "$LOG"; }
# node -> "veth_ip:port" via netvm-names.sh (hash-derived, don't hardcode)
relay_target() {
local node="$1" reg_port=""
# shellcheck disable=SC1091
. "$NETVM_BIN/netvm-names.sh"
netvm_names "$node" || return 1
# The registry is the source of truth for ports (new nodes propagate
# automatically); netvm-names pinning is the fallback.
if reg_port=$("$NETVM_BIN/netvm-registry.py" "$node" 2>/dev/null); then
[ -n "$reg_port" ] && CDP_PORT="$reg_port"
fi
echo "$PEER_IP:$CDP_PORT"
}
node_port() { echo "${1##*:}"; }
# node -> "VETH GW" via netvm-names.sh
node_veth() {
local node="$1"
# shellcheck disable=SC1091
. "$NETVM_BIN/netvm-names.sh"
netvm_names "$node" || return 1
echo "$VETH $GW"
}
# Is the host-side veth IP assigned? The relay listens on the netns-side peer
# IP; the host reaches it via the veth interface's GW address. If the GW IP is
# missing, the relay is unreachable from the host — restarting the relay is
# pointless and masks the real problem.
veth_healthy() {
local node="$1" veth gw
read -r veth gw <<< "$(node_veth "$node")" || return 1
ip addr show dev "$veth" 2>/dev/null | grep -q "inet ${gw}/" || return 1
return 0
}
relay_healthy() {
local node="$1" target
target="$(relay_target "$node")" || return 1
curl -s -m 8 "http://$target/json/version" 2>/dev/null | grep -q '"Browser"' || return 1
return 0
}
restart_relay() {
local node="$1" target port veth_ip netns
target="$(relay_target "$node")"
veth_ip="${target%%:*}"
port="${target##*:}"
netns="warp-$node"
# Kill any existing relay for this node's port (correct or not)
# Relays are root-owned (started via sudo ip netns exec); the timer runs as
# super, so the kill needs sudo too. Without it pkill fails EPERM silently
# and the "restart" false-positives via SO_REUSEADDR double-bind.
sudo -n pkill -f "netvm-cdp-relay.py .* $port 127.0.0.1 $port" 2>/dev/null || true
sleep 2
# Launch inside the netns, listening on the veth IP (host-reachable)
sudo -n ip netns exec "$netns" setsid nohup python3 \
"$NETVM_BIN/netvm-cdp-relay.py" "$veth_ip" "$port" 127.0.0.1 "$port" \
>>"$LOG" 2>&1 < /dev/null &
sleep 5
if relay_healthy "$node"; then
log "[$node] relay restarted OK on $target"
return 0
else
log "[$node] relay restart FAILED on $target — needs operator attention"
return 1
fi
}
# Registry-driven node list: every active node gets relay supervision
# (the old hardcoded 4-node list left def/dev unsupervised — 2026-10-06).
watched_nodes() {
"$NETVM_BIN/netvm-registry.py" 2>/dev/null | cut -d: -f1
}
# Allow sourcing for tests without running checks.
if [ "${CDP_RELAY_WATCHDOG_LIB_ONLY:-}" = "1" ]; then
return 0 2>/dev/null || exit 0
fi
FAILED=0
NODES="$(watched_nodes)"
if [ -z "$NODES" ]; then
log "FAIL_LOUD: node registry empty/unreadable, skipping run"
exit 1
fi
# shellcheck disable=SC2086 (intended word splitting: one node per word)
for node in $NODES; do
# Stage 1: host veth IP. Fail loud, skip relay restart (pointless).
if ! veth_healthy "$node"; then
read -r veth gw <<< "$(node_veth "$node")"
log "[$node] FAIL_LOUD: host veth $veth missing IP $gw — relay unreachable, needs netvm-node-up.sh $node (manual)"
FAILED=1
continue
fi
# Stage 2: relay connectivity.
if relay_healthy "$node"; then
continue
fi
target="$(relay_target "$node")"
log "[$node] relay unhealthy on $target, restarting"
restart_relay "$node" || FAILED=1
done
exit $FAILED