#!/bin/bash # GOLDEN PATH: container -> VM (34.139.37.135) -> bl (100.123.153.75) -> netns -> browser -> agent # This script is the operator's heartbeat. If it stops, agents go dark. # When debugging: trace each hop. Don't assume -- verify. # Operator health monitor for Muse agents on bl. # Checks each agent via API every 5 minutes. If unresponsive: # 1. Restart the browser # 2. Re-check # 3. Log failure if still down # # Run via systemd timer or cron: */5 * * * * /home/super/Projects/NetVM/bin/agent-health.sh # # Agents are defined in ACCOUNTS.md. This script reads the active ones. NETVM_BIN="$(cd "$(dirname "$0")" && pwd)" LOG="/tmp/agent-health.log" check_agent() { local agent=$1 local node=$2 local cdp_port=$3 # Try API messages command (lightweight check) if timeout 30 "$NETVM_BIN/netvm-exec.sh" "$node" -- python3 "$NETVM_BIN/muse-chat-api.py" --account "$agent" messages 1 > /dev/null 2>&1; then echo "$(date -Iseconds) $agent: OK" >> "$LOG" return 0 else echo "$(date -Iseconds) $agent: FAIL (api timeout)" >> "$LOG" return 1 fi } restart_browser() { local agent=$1 local cdp_port=$2 echo "$(date -Iseconds) $agent: restarting browser..." >> "$LOG" # Kill existing pkill -f "$agent.*$cdp_port" 2>/dev/null sleep 3 # Restart via netvm-chrome.sh cd "$NETVM_BIN" nohup ./netvm-chrome.sh --headless --cdp-port "$cdp_port" "$agent" https://muse.ai > "/tmp/bl-$agent.log" 2>&1 & sleep 15 echo "$(date -Iseconds) $agent: browser restarted" >> "$LOG" } # Main echo "=== Health check $(date -Iseconds) ===" >> "$LOG" check_one() { local agent=$1 local cdp_port=$2 # node == agent == profile (unified naming) if ! check_agent "$agent" "$agent" "$cdp_port"; then restart_browser "$agent" "$cdp_port" sleep 5 if ! check_agent "$agent" "$agent" "$cdp_port"; then echo "$(date -Iseconds) $agent: CRITICAL - still down after restart" >> "$LOG" # TODO: Alert operator (e.g., via board post or email) else echo "$(date -Iseconds) $agent: RECOVERED after restart" >> "$LOG" fi fi } # Every active node in the NODES.md registry gets checked — new nodes # propagate automatically, no per-node blocks to add. "$NETVM_BIN/netvm-registry.py" 2>/dev/null | while IFS=: read -r node port; do [ -n "$node" ] && [ -n "$port" ] && check_one "$node" "$port" done # Trim log (keep last 1000 lines) tail -1000 "$LOG" > "$LOG.tmp" && mv "$LOG.tmp" "$LOG"