2026-10-03 18:03:17 +00:00
|
|
|
#!/bin/bash
|
2026-10-03 18:04:57 +00:00
|
|
|
# GOLDEN PATH: container -> VM (34.139.37.135) -> bl (100.123.153.75) -> netns -> browser -> agent
|
|
|
|
|
# This script is the operator's heartbeat. If it stops, agents go dark.
|
|
|
|
|
# When debugging: trace each hop. Don't assume -- verify.
|
|
|
|
|
|
2026-10-03 18:03:17 +00:00
|
|
|
# Operator health monitor for Muse agents on bl.
|
|
|
|
|
# Checks each agent via API every 5 minutes. If unresponsive:
|
|
|
|
|
# 1. Restart the browser
|
|
|
|
|
# 2. Re-check
|
|
|
|
|
# 3. Log failure if still down
|
|
|
|
|
#
|
|
|
|
|
# Run via systemd timer or cron: */5 * * * * /home/super/Projects/NetVM/bin/agent-health.sh
|
|
|
|
|
#
|
|
|
|
|
# Agents are defined in ACCOUNTS.md. This script reads the active ones.
|
|
|
|
|
|
|
|
|
|
NETVM_BIN="$(cd "$(dirname "$0")" && pwd)"
|
|
|
|
|
LOG="/tmp/agent-health.log"
|
|
|
|
|
|
|
|
|
|
check_agent() {
|
|
|
|
|
local agent=$1
|
|
|
|
|
local node=$2
|
|
|
|
|
local cdp_port=$3
|
|
|
|
|
|
|
|
|
|
# Try API messages command (lightweight check)
|
|
|
|
|
if timeout 30 "$NETVM_BIN/netvm-exec.sh" "$node" -- python3 "$NETVM_BIN/muse-chat-api.py" --account "$agent" messages 1 > /dev/null 2>&1; then
|
|
|
|
|
echo "$(date -Iseconds) $agent: OK" >> "$LOG"
|
|
|
|
|
return 0
|
|
|
|
|
else
|
|
|
|
|
echo "$(date -Iseconds) $agent: FAIL (api timeout)" >> "$LOG"
|
|
|
|
|
return 1
|
|
|
|
|
fi
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
restart_browser() {
|
|
|
|
|
local agent=$1
|
|
|
|
|
local cdp_port=$2
|
|
|
|
|
|
|
|
|
|
echo "$(date -Iseconds) $agent: restarting browser..." >> "$LOG"
|
|
|
|
|
# Kill existing
|
|
|
|
|
pkill -f "$agent.*$cdp_port" 2>/dev/null
|
|
|
|
|
sleep 3
|
|
|
|
|
# Restart via netvm-chrome.sh
|
|
|
|
|
cd "$NETVM_BIN"
|
|
|
|
|
nohup ./netvm-chrome.sh --headless --cdp-port "$cdp_port" "$agent" https://muse.ai > "/tmp/bl-$agent.log" 2>&1 &
|
|
|
|
|
sleep 15
|
|
|
|
|
echo "$(date -Iseconds) $agent: browser restarted" >> "$LOG"
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
# Main
|
|
|
|
|
echo "=== Health check $(date -Iseconds) ===" >> "$LOG"
|
|
|
|
|
|
2026-10-03 21:28:36 +00:00
|
|
|
check_one() {
|
|
|
|
|
local agent=$1
|
|
|
|
|
local cdp_port=$2
|
|
|
|
|
# node == agent == profile (unified naming)
|
|
|
|
|
if ! check_agent "$agent" "$agent" "$cdp_port"; then
|
|
|
|
|
restart_browser "$agent" "$cdp_port"
|
|
|
|
|
sleep 5
|
|
|
|
|
if ! check_agent "$agent" "$agent" "$cdp_port"; then
|
|
|
|
|
echo "$(date -Iseconds) $agent: CRITICAL - still down after restart" >> "$LOG"
|
|
|
|
|
# TODO: Alert operator (e.g., via board post or email)
|
|
|
|
|
else
|
|
|
|
|
echo "$(date -Iseconds) $agent: RECOVERED after restart" >> "$LOG"
|
|
|
|
|
fi
|
2026-10-03 18:03:17 +00:00
|
|
|
fi
|
2026-10-03 21:28:36 +00:00
|
|
|
}
|
2026-10-03 18:03:17 +00:00
|
|
|
|
2026-10-03 21:28:36 +00:00
|
|
|
# Every active node in the NODES.md registry gets checked — new nodes
|
|
|
|
|
# propagate automatically, no per-node blocks to add.
|
|
|
|
|
"$NETVM_BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do
|
|
|
|
|
[ -n "$node" ] && [ -n "$port" ] && check_one "$node" "$port"
|
|
|
|
|
done
|
2026-10-03 18:03:17 +00:00
|
|
|
|
|
|
|
|
# Trim log (keep last 1000 lines)
|
|
|
|
|
tail -1000 "$LOG" > "$LOG.tmp" && mv "$LOG.tmp" "$LOG"
|