agent-health.sh: Verify CDP port liveness, not just API success\n\nThe watchdog only checked if the API responded, missing zombie\nbrowsers (process alive but CDP not listening). Now explicitly\nchecks via ss -tln in the netns before trying the API.\nAlso: kill by exact PIDs instead of unreliable pkill patterns,\nand warn if port still bound after kill.

This commit is contained in:
operator-main
2026-10-04 02:23:20 +00:00
parent 5dddedb667
commit 6189793748
+28 -3
View File
@@ -16,12 +16,30 @@
NETVM_BIN="$(cd "$(dirname "$0")" && pwd)"
LOG="/tmp/agent-health.log"
check_cdp_port() {
# Verify CDP port is actually listening in the netns.
# A browser can be running but not bound to CDP (zombie state).
local node=$1
local cdp_port=$2
if sudo ip netns exec "$node" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then
return 0
else
return 1
fi
}
check_agent() {
local agent=$1
local node=$2
local cdp_port=$3
# Try API messages command (lightweight check)
# First: verify CDP port is listening (catches zombie browsers)
if ! check_cdp_port "$node" "$cdp_port"; then
echo "$(date -Iseconds) $agent: FAIL (cdp port $cdp_port not listening)" >> "$LOG"
return 1
fi
# Then: try API messages command (lightweight check)
if timeout 30 "$NETVM_BIN/netvm-exec.sh" "$node" -- python3 "$NETVM_BIN/muse-chat-api.py" --account "$agent" messages 1 > /dev/null 2>&1; then
echo "$(date -Iseconds) $agent: OK" >> "$LOG"
return 0
@@ -36,9 +54,16 @@ restart_browser() {
local cdp_port=$2
echo "$(date -Iseconds) $agent: restarting browser..." >> "$LOG"
# Kill existing
pkill -f "$agent.*$cdp_port" 2>/dev/null
# Kill existing by exact PIDs (pkill patterns are unreliable)
# Find chromium processes for this profile
for pid in $(pgrep -f "chromium.*profiles/$agent" 2>/dev/null); do
kill -9 "$pid" 2>/dev/null
done
sleep 3
# Verify port is free before restart
if sudo ip netns exec "$agent" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then
echo "$(date -Iseconds) $agent: WARNING - port $cdp_port still bound after kill" >> "$LOG"
fi
# Restart via netvm-chrome.sh in its own systemd scope.
# This oneshot service runs with KillMode=control-group, so anything
# spawned directly under it (nohup AND setsid both stay in the cgroup)