From ae1bf17de561b2766e2d500ad08bd5cdc2b09fbe Mon Sep 17 00:00:00 2001 From: operator-main Date: Sun, 4 Oct 2026 18:22:10 +0000 Subject: [PATCH] Fix relay watchdog false-positive: sudo the pkill restart_relay() ran pkill without sudo, but relay processes are root-owned and the timer runs as User=super. The kill failed EPERM (silently swallowed by || true), the old relay kept running, and SO_REUSEADDR let the replacement double-bind the same port. The post-restart health check then passed and logged "restarted OK" when nothing was actually restarted. Adding sudo -n to the pkill, matching the sudo -n ip netns exec already used to start the relay. Session: sidechat/chromebox-ops --- bin/cdp-relay-watchdog.sh | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/bin/cdp-relay-watchdog.sh b/bin/cdp-relay-watchdog.sh index 948312a..e8b2dd3 100755 --- a/bin/cdp-relay-watchdog.sh +++ b/bin/cdp-relay-watchdog.sh @@ -84,7 +84,10 @@ restart_relay() { port="${target##*:}" netns="warp-$node" # Kill any existing relay for this node's port (correct or not) - pkill -f "netvm-cdp-relay.py .* $port 127.0.0.1 $port" 2>/dev/null || true + # Relays are root-owned (started via sudo ip netns exec); the timer runs as + # super, so the kill needs sudo too. Without it pkill fails EPERM silently + # and the "restart" false-positives via SO_REUSEADDR double-bind. + sudo -n pkill -f "netvm-cdp-relay.py .* $port 127.0.0.1 $port" 2>/dev/null || true sleep 2 # Launch inside the netns, listening on the veth IP (host-reachable) sudo -n ip netns exec "$netns" setsid nohup python3 \