#!/bin/bash # GOLDEN PATH: container -> VM (34.139.37.135) -> bl (100.123.153.75) -> netns -> browser -> agent # This script is the operator's heartbeat. If it stops, agents go dark. # When debugging: trace each hop. Don't assume -- verify. # Operator health monitor for Muse agents on bl. # Checks each agent via API every 5 minutes. If unresponsive: # 1. Restart the browser # 2. Re-check # 3. Log failure if still down # # Run via systemd timer or cron: */5 * * * * /home/super/Projects/NetVM/bin/agent-health.sh # # Agents are defined in ACCOUNTS.md. This script reads the active ones. NETVM_BIN="$(cd "$(dirname "$0")" && pwd)" LOG="/tmp/agent-health.log" # 2026-10-04: per-node consecutive-API-failure counters. A single # muse-chat-api.py failure must not kill a healthy browser (observed # 2026-10-04 20:48:44 UTC: 646's browser killed on an API timeout while its # CDP port was still listening). Require 2 CONSECUTIVE API failures before # the kill path; the counter resets on any successful check. STATE_DIR="/tmp/agent-health-state" mkdir -p "$STATE_DIR" 2>/dev/null check_cdp_port() { # Verify CDP port is actually listening in the netns. # A browser can be running but not bound to CDP (zombie state). local node=$1 local cdp_port=$2 if sudo ip netns exec "warp-$node" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then return 0 else return 1 fi } # Return codes: 0 = healthy, 1 = API check failed (port OK), 2 = CDP port down. check_agent() { local agent=$1 local node=$2 local cdp_port=$3 # First: verify CDP port is listening (catches zombie browsers) if ! check_cdp_port "$node" "$cdp_port"; then echo "$(date -Iseconds) $agent: FAIL (cdp port $cdp_port not listening)" >> "$LOG" return 2 fi # Then: try API messages command (lightweight check) if timeout 30 "$NETVM_BIN/netvm-exec.sh" "$node" -- python3 "$NETVM_BIN/muse-chat-api.py" --account "$agent" messages 1 > /dev/null 2>&1; then echo "$(date -Iseconds) $agent: OK" >> "$LOG" return 0 else echo "$(date -Iseconds) $agent: FAIL (api timeout)" >> "$LOG" return 1 fi } # 2026-10-04: recent-relaunch guard. chromebox-watchdog.sh (2-min timer) and # this script (5-min timer) could otherwise kill each other's fresh browsers: # a browser just relaunched by the watchdog is still starting when this # script's API check times out on it. Skip the kill path when the main # browser process for this profile launched <2 min ago. Mirrors # chromebox-watchdog.sh's relaunch-loop guard idiom (main process only: # --remote-debugging-port present, no --type= flag; renderer/gpu children # start later than the main process and must not satisfy this check). recent_relaunch() { local cdp_port=$1 local _pid _start _now for _pid in $(pgrep -f "chromium.*--remote-debugging-port=${cdp_port}([[:space:]]|$)" 2>/dev/null); do # Skip child processes (renderer, gpu, etc.) — only the main browser counts if ps -o args= -p "$_pid" 2>/dev/null | grep -q -- "--type="; then continue fi _start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0) _now=$(date +%s) if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then return 0 fi done return 1 } restart_browser() { local agent=$1 local cdp_port=$2 echo "$(date -Iseconds) $agent: restarting browser..." >> "$LOG" # Kill existing by exact PIDs (pkill patterns are unreliable) # Find chromium processes for this profile for pid in $(pgrep -f "chromium.*profiles/$agent" 2>/dev/null); do kill -9 "$pid" 2>/dev/null done sleep 3 # Verify port is free before restart if sudo ip netns exec "warp-$agent" ss -tln 2>/dev/null | grep -q ":$cdp_port "; then echo "$(date -Iseconds) $agent: WARNING - port $cdp_port still bound after kill" >> "$LOG" fi # Restart via netvm-chrome.sh in its own systemd scope. # This oneshot service runs with KillMode=control-group, so anything # spawned directly under it (nohup AND setsid both stay in the cgroup) # is SIGKILLed when the service exits — observed 2026-10-03: every # restart "recovered" then died at service teardown, looping forever. # A transient scope escapes the service cgroup and survives. # NOTE: systemd-run --scope WAITS for the scope's processes (even with # --no-block, verified 2026-10-03), so background it — the scope itself # is an independent unit and outlives the wrapper. cd "$NETVM_BIN" systemd-run --user --scope --unit="netvm-chrome-$agent-$(date +%s)" \ ./netvm-chrome.sh --headless --cdp-port "$cdp_port" "$agent" https://muse.ai \ > "/tmp/bl-$agent.log" 2>&1 & sleep 15 echo "$(date -Iseconds) $agent: browser restarted" >> "$LOG" } # Main echo "=== Health check $(date -Iseconds) ===" >> "$LOG" check_one() { local agent=$1 local cdp_port=$2 # node == agent == profile (unified naming) check_agent "$agent" "$agent" "$cdp_port" local rc=$? if [ $rc -eq 0 ]; then # Healthy: reset the consecutive-API-failure counter. rm -f "$STATE_DIR/failcount-$agent" 2>/dev/null return 0 fi if [ $rc -eq 1 ]; then # API failed but CDP port is listening: this is the false-kill vector # (2026-10-04: single 30s API timeout killed 646's healthy browser). # Require 2 CONSECUTIVE API failures before killing. local count=0 local cf="$STATE_DIR/failcount-$agent" [ -f "$cf" ] && count=$(cat "$cf" 2>/dev/null || echo 0) count=$(( count + 1 )) if [ "$count" -lt 2 ]; then echo "$count" > "$cf" echo "$(date -Iseconds) $agent: API failure $count of 2, deferring kill" >> "$LOG" return 0 fi rm -f "$cf" 2>/dev/null else # CDP port not listening (rc=2): zombie browser, kill immediately as before. rm -f "$STATE_DIR/failcount-$agent" 2>/dev/null fi # De-conflict with chromebox-watchdog.sh: if the main browser for this # profile launched <2 min ago, the watchdog just relaunched it — skip the # kill path rather than racing it on a fresh cold start. if recent_relaunch "$cdp_port"; then echo "$(date -Iseconds) $agent: skipping kill (browser launched <2m ago, likely watchdog relaunch)" >> "$LOG" return 0 fi restart_browser "$agent" "$cdp_port" # 2026-10-04: post-restart re-check grace extended to ~60s total # (restart_browser sleeps 15s internally + 45s here), matching # chromebox-watchdog.sh's proven 60s retry window. Cold starts on a # loaded box (load ~7) need more than 20s before CDP/API respond. sleep 45 if ! check_agent "$agent" "$agent" "$cdp_port"; then echo "$(date -Iseconds) $agent: CRITICAL - still down after restart" >> "$LOG" # TODO: Alert operator (e.g., via board post or email) else echo "$(date -Iseconds) $agent: RECOVERED after restart" >> "$LOG" fi } # Every active node in the NODES.md registry gets checked — new nodes # propagate automatically, no per-node blocks to add. "$NETVM_BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] && [ -n "$port" ] && check_one "$node" "$port" done # Trim log (keep last 1000 lines) tail -1000 "$LOG" > "$LOG.tmp" && mv "$LOG.tmp" "$LOG"