2026-10-03 18:03:17 +00:00
#!/bin/bash
2026-10-03 18:04:57 +00:00
# GOLDEN PATH: container -> VM (34.139.37.135) -> bl (100.123.153.75) -> netns -> browser -> agent
# This script is the operator's heartbeat. If it stops, agents go dark.
# When debugging: trace each hop. Don't assume -- verify.
2026-10-03 18:03:17 +00:00
# Operator health monitor for Muse agents on bl.
# Checks each agent via API every 5 minutes. If unresponsive:
# 1. Restart the browser
# 2. Re-check
# 3. Log failure if still down
#
# Run via systemd timer or cron: */5 * * * * /home/super/Projects/NetVM/bin/agent-health.sh
#
# Agents are defined in ACCOUNTS.md. This script reads the active ones.
NETVM_BIN = " $( cd " $( dirname " $0 " ) " && pwd ) "
LOG = "/tmp/agent-health.log"
2026-10-04 22:46:44 +00:00
# 2026-10-04: per-node consecutive-API-failure counters. A single
# muse-chat-api.py failure must not kill a healthy browser (observed
# 2026-10-04 20:48:44 UTC: 646's browser killed on an API timeout while its
# CDP port was still listening). Require 2 CONSECUTIVE API failures before
# the kill path; the counter resets on any successful check.
STATE_DIR = "/tmp/agent-health-state"
mkdir -p " $STATE_DIR " 2>/dev/null
2026-10-04 02:23:20 +00:00
check_cdp_port( ) {
# Verify CDP port is actually listening in the netns.
# A browser can be running but not bound to CDP (zombie state).
local node = $1
local cdp_port = $2
2026-10-04 20:09:36 +00:00
if sudo ip netns exec " warp- $node " ss -tln 2>/dev/null | grep -q " : $cdp_port " ; then
2026-10-04 02:23:20 +00:00
return 0
else
return 1
fi
}
2026-10-04 22:46:44 +00:00
# Return codes: 0 = healthy, 1 = API check failed (port OK), 2 = CDP port down.
2026-10-03 18:03:17 +00:00
check_agent( ) {
local agent = $1
local node = $2
local cdp_port = $3
2026-10-04 02:23:20 +00:00
# First: verify CDP port is listening (catches zombie browsers)
if ! check_cdp_port " $node " " $cdp_port " ; then
echo " $( date -Iseconds) $agent : FAIL (cdp port $cdp_port not listening) " >> " $LOG "
2026-10-04 22:46:44 +00:00
return 2
2026-10-04 02:23:20 +00:00
fi
# Then: try API messages command (lightweight check)
2026-10-03 18:03:17 +00:00
if timeout 30 " $NETVM_BIN /netvm-exec.sh " " $node " -- python3 " $NETVM_BIN /muse-chat-api.py " --account " $agent " messages 1 > /dev/null 2>& 1; then
echo " $( date -Iseconds) $agent : OK " >> " $LOG "
return 0
else
echo " $( date -Iseconds) $agent : FAIL (api timeout) " >> " $LOG "
return 1
fi
}
2026-10-04 22:46:44 +00:00
# 2026-10-04: recent-relaunch guard. chromebox-watchdog.sh (2-min timer) and
# this script (5-min timer) could otherwise kill each other's fresh browsers:
# a browser just relaunched by the watchdog is still starting when this
# script's API check times out on it. Skip the kill path when the main
# browser process for this profile launched <2 min ago. Mirrors
# chromebox-watchdog.sh's relaunch-loop guard idiom (main process only:
# --remote-debugging-port present, no --type= flag; renderer/gpu children
# start later than the main process and must not satisfy this check).
recent_relaunch( ) {
local cdp_port = $1
local _pid _start _now
for _pid in $( pgrep -f " chromium.*--remote-debugging-port= ${ cdp_port } ([[:space:]]| $) " 2>/dev/null) ; do
# Skip child processes (renderer, gpu, etc.) — only the main browser counts
if ps -o args = -p " $_pid " 2>/dev/null | grep -q -- "--type=" ; then
continue
fi
_start = $( date -d " $( ps -o lstart = -p " $_pid " 2>/dev/null) " +%s 2>/dev/null || echo 0)
_now = $( date +%s)
if [ $(( _now - _start )) -lt 120 ] && [ " $_start " -gt 0 ] ; then
return 0
fi
done
return 1
}
2026-10-03 18:03:17 +00:00
restart_browser( ) {
local agent = $1
local cdp_port = $2
echo " $( date -Iseconds) $agent : restarting browser... " >> " $LOG "
2026-10-04 02:23:20 +00:00
# Kill existing by exact PIDs (pkill patterns are unreliable)
# Find chromium processes for this profile
for pid in $( pgrep -f " chromium.*profiles/ $agent " 2>/dev/null) ; do
kill -9 " $pid " 2>/dev/null
done
2026-10-03 18:03:17 +00:00
sleep 3
2026-10-04 02:23:20 +00:00
# Verify port is free before restart
2026-10-04 20:09:36 +00:00
if sudo ip netns exec " warp- $agent " ss -tln 2>/dev/null | grep -q " : $cdp_port " ; then
2026-10-04 02:23:20 +00:00
echo " $( date -Iseconds) $agent : WARNING - port $cdp_port still bound after kill " >> " $LOG "
fi
2026-10-04 01:10:45 +00:00
# Restart via netvm-chrome.sh in its own systemd scope.
# This oneshot service runs with KillMode=control-group, so anything
# spawned directly under it (nohup AND setsid both stay in the cgroup)
# is SIGKILLed when the service exits — observed 2026-10-03: every
# restart "recovered" then died at service teardown, looping forever.
# A transient scope escapes the service cgroup and survives.
# NOTE: systemd-run --scope WAITS for the scope's processes (even with
# --no-block, verified 2026-10-03), so background it — the scope itself
# is an independent unit and outlives the wrapper.
2026-10-03 18:03:17 +00:00
cd " $NETVM_BIN "
2026-10-04 01:10:45 +00:00
systemd-run --user --scope --unit= " netvm-chrome- $agent - $( date +%s) " \
./netvm-chrome.sh --headless --cdp-port " $cdp_port " " $agent " https://muse.ai \
> " /tmp/bl- $agent .log " 2>& 1 &
2026-10-03 18:03:17 +00:00
sleep 15
echo " $( date -Iseconds) $agent : browser restarted " >> " $LOG "
}
2026-10-06 18:16:14 +00:00
# 2026-10-06: restart circuit breaker. A restart that leaves the agent
# still failing is FUTILE (observed 2026-10-06: def's API check failed
# 155x while its CDP port was up; 57 kill+restart cycles murdered a
# healthy browser for an account-layer failure restarts cannot fix).
# After FUTILE_THRESHOLD consecutive futile restarts, open the circuit:
# stop killing/restarting and alert, until CIRCUIT_COOLDOWN seconds pass
# (one half-open probe restart) or any check succeeds. Manual reset:
# rm /tmp/agent-health-state/circuit-<agent> /tmp/agent-health-state/futile-<agent>
FUTILE_THRESHOLD = 3
CIRCUIT_COOLDOWN = 1800
# circuit_allows <agent>: return 0 if a restart may proceed now.
circuit_allows( ) {
local agent = $1 now opened retry_in
local cf = " $STATE_DIR /circuit- $agent "
[ -f " $cf " ] || return 0
opened = $( cat " $cf " 2>/dev/null || echo 0)
now = $( date +%s)
if [ $(( now - opened )) -ge $CIRCUIT_COOLDOWN ] ; then
echo " $( date -Iseconds) $agent : circuit half-open after ${ CIRCUIT_COOLDOWN } s cooldown, one probe restart " >> " $LOG "
return 0
fi
retry_in = $(( ( opened + CIRCUIT_COOLDOWN - now + 59 ) / 60 ))
echo " $( date -Iseconds) $agent : CIRCUIT OPEN - skipping kill/restart (restarts futile, probe retry in ~ ${ retry_in } m; manual reset: rm $cf ) " >> " $LOG "
return 1
}
# circuit_note_restart <agent> <ok|fail>: record a restart outcome.
circuit_note_restart( ) {
local agent = $1 outcome = $2 count = 0
local ff = " $STATE_DIR /futile- $agent " cf = " $STATE_DIR /circuit- $agent "
if [ " $outcome " = "ok" ] ; then
rm -f " $ff " " $cf " 2>/dev/null
return 0
fi
[ -f " $ff " ] && count = $( cat " $ff " 2>/dev/null || echo 0)
count = $(( count + 1 ))
echo " $count " > " $ff "
if [ " $count " -ge " $FUTILE_THRESHOLD " ] ; then
date +%s > " $cf "
local msg = " $agent : ALERT - $count consecutive futile restarts, circuit OPEN for ${ CIRCUIT_COOLDOWN } s (no more kills until probe; manual reset: rm $cf $ff ) "
echo " $( date -Iseconds) $msg " >> " $LOG "
echo " agent-health ALERT: $msg "
fi
}
# Allow sourcing for tests without running checks.
if [ " ${ AGENT_HEALTH_LIB_ONLY :- } " = "1" ] ; then
return 0 2>/dev/null || exit 0
fi
2026-10-03 18:03:17 +00:00
# Main
echo " === Health check $( date -Iseconds) === " >> " $LOG "
2026-10-03 21:28:36 +00:00
check_one( ) {
local agent = $1
local cdp_port = $2
# node == agent == profile (unified naming)
2026-10-04 22:46:44 +00:00
check_agent " $agent " " $agent " " $cdp_port "
local rc = $?
if [ $rc -eq 0 ] ; then
2026-10-06 18:16:14 +00:00
# Healthy: reset the consecutive-API-failure counter and close
# any open restart circuit.
rm -f " $STATE_DIR /failcount- $agent " " $STATE_DIR /futile- $agent " " $STATE_DIR /circuit- $agent " 2>/dev/null
2026-10-04 22:46:44 +00:00
return 0
fi
if [ $rc -eq 1 ] ; then
# API failed but CDP port is listening: this is the false-kill vector
# (2026-10-04: single 30s API timeout killed 646's healthy browser).
# Require 2 CONSECUTIVE API failures before killing.
local count = 0
local cf = " $STATE_DIR /failcount- $agent "
[ -f " $cf " ] && count = $( cat " $cf " 2>/dev/null || echo 0)
count = $(( count + 1 ))
if [ " $count " -lt 2 ] ; then
echo " $count " > " $cf "
echo " $( date -Iseconds) $agent : API failure $count of 2, deferring kill " >> " $LOG "
return 0
2026-10-03 21:28:36 +00:00
fi
2026-10-04 22:46:44 +00:00
rm -f " $cf " 2>/dev/null
else
# CDP port not listening (rc=2): zombie browser, kill immediately as before.
rm -f " $STATE_DIR /failcount- $agent " 2>/dev/null
fi
# De-conflict with chromebox-watchdog.sh: if the main browser for this
# profile launched <2 min ago, the watchdog just relaunched it — skip the
# kill path rather than racing it on a fresh cold start.
if recent_relaunch " $cdp_port " ; then
echo " $( date -Iseconds) $agent : skipping kill (browser launched <2m ago, likely watchdog relaunch) " >> " $LOG "
return 0
fi
2026-10-06 18:16:14 +00:00
# Circuit breaker: repeated futile restarts stop here until cooldown.
if ! circuit_allows " $agent " ; then
return 0
fi
2026-10-04 22:46:44 +00:00
restart_browser " $agent " " $cdp_port "
# 2026-10-04: post-restart re-check grace extended to ~60s total
# (restart_browser sleeps 15s internally + 45s here), matching
# chromebox-watchdog.sh's proven 60s retry window. Cold starts on a
# loaded box (load ~7) need more than 20s before CDP/API respond.
sleep 45
if ! check_agent " $agent " " $agent " " $cdp_port " ; then
echo " $( date -Iseconds) $agent : CRITICAL - still down after restart " >> " $LOG "
2026-10-06 18:16:14 +00:00
circuit_note_restart " $agent " fail
2026-10-04 22:46:44 +00:00
else
echo " $( date -Iseconds) $agent : RECOVERED after restart " >> " $LOG "
2026-10-06 18:16:14 +00:00
circuit_note_restart " $agent " ok
2026-10-03 18:03:17 +00:00
fi
2026-10-03 21:28:36 +00:00
}
2026-10-03 18:03:17 +00:00
2026-10-03 21:28:36 +00:00
# Every active node in the NODES.md registry gets checked — new nodes
# propagate automatically, no per-node blocks to add.
" $NETVM_BIN /netvm-registry.py " 2>/dev/null | while IFS = : read -r node port; do
[ -n " $node " ] && [ -n " $port " ] && check_one " $node " " $port "
done
2026-10-03 18:03:17 +00:00
# Trim log (keep last 1000 lines)
tail -1000 " $LOG " > " $LOG .tmp " && mv " $LOG .tmp " " $LOG "