Harden agent-health.sh: soften API-timeout kill path, add recent-relaunch guard
- Require 2 CONSECUTIVE muse-chat-api.py API failures before kill -9 (per-node counter in /tmp/agent-health-state, reset on success). A single 30s API timeout killed 646s healthy browser at 20:48:44 UTC while its CDP port was still listening. - Extend post-restart re-check grace to ~60s (15s internal + 45s), matching chromebox-watchdog.shs proven 60s retry window. - Add recent_relaunch() guard (mirrors chromebox-watchdog.sh idiom): skip the kill path when the main browser process launched <2 min ago, so the two watchdogs can not kill each others fresh browsers. - check_agent now returns 0/1/2 (healthy/api-fail/port-down); CDP-port failure still kills immediately. warp-$node checks untouched. Session: sidechat/chromebox-fixes
This commit is contained in:
+76
-4
@@ -16,6 +16,14 @@
|
|||||||
NETVM_BIN="$(cd "$(dirname "$0")" && pwd)"
|
NETVM_BIN="$(cd "$(dirname "$0")" && pwd)"
|
||||||
LOG="/tmp/agent-health.log"
|
LOG="/tmp/agent-health.log"
|
||||||
|
|
||||||
|
# 2026-10-04: per-node consecutive-API-failure counters. A single
|
||||||
|
# muse-chat-api.py failure must not kill a healthy browser (observed
|
||||||
|
# 2026-10-04 20:48:44 UTC: 646's browser killed on an API timeout while its
|
||||||
|
# CDP port was still listening). Require 2 CONSECUTIVE API failures before
|
||||||
|
# the kill path; the counter resets on any successful check.
|
||||||
|
STATE_DIR="/tmp/agent-health-state"
|
||||||
|
mkdir -p "$STATE_DIR" 2>/dev/null
|
||||||
|
|
||||||
check_cdp_port() {
|
check_cdp_port() {
|
||||||
# Verify CDP port is actually listening in the netns.
|
# Verify CDP port is actually listening in the netns.
|
||||||
# A browser can be running but not bound to CDP (zombie state).
|
# A browser can be running but not bound to CDP (zombie state).
|
||||||
@@ -28,6 +36,7 @@ check_cdp_port() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Return codes: 0 = healthy, 1 = API check failed (port OK), 2 = CDP port down.
|
||||||
check_agent() {
|
check_agent() {
|
||||||
local agent=$1
|
local agent=$1
|
||||||
local node=$2
|
local node=$2
|
||||||
@@ -36,7 +45,7 @@ check_agent() {
|
|||||||
# First: verify CDP port is listening (catches zombie browsers)
|
# First: verify CDP port is listening (catches zombie browsers)
|
||||||
if ! check_cdp_port "$node" "$cdp_port"; then
|
if ! check_cdp_port "$node" "$cdp_port"; then
|
||||||
echo "$(date -Iseconds) $agent: FAIL (cdp port $cdp_port not listening)" >> "$LOG"
|
echo "$(date -Iseconds) $agent: FAIL (cdp port $cdp_port not listening)" >> "$LOG"
|
||||||
return 1
|
return 2
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Then: try API messages command (lightweight check)
|
# Then: try API messages command (lightweight check)
|
||||||
@@ -49,6 +58,31 @@ check_agent() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# 2026-10-04: recent-relaunch guard. chromebox-watchdog.sh (2-min timer) and
|
||||||
|
# this script (5-min timer) could otherwise kill each other's fresh browsers:
|
||||||
|
# a browser just relaunched by the watchdog is still starting when this
|
||||||
|
# script's API check times out on it. Skip the kill path when the main
|
||||||
|
# browser process for this profile launched <2 min ago. Mirrors
|
||||||
|
# chromebox-watchdog.sh's relaunch-loop guard idiom (main process only:
|
||||||
|
# --remote-debugging-port present, no --type= flag; renderer/gpu children
|
||||||
|
# start later than the main process and must not satisfy this check).
|
||||||
|
recent_relaunch() {
|
||||||
|
local cdp_port=$1
|
||||||
|
local _pid _start _now
|
||||||
|
for _pid in $(pgrep -f "chromium.*--remote-debugging-port=${cdp_port}([[:space:]]|$)" 2>/dev/null); do
|
||||||
|
# Skip child processes (renderer, gpu, etc.) — only the main browser counts
|
||||||
|
if ps -o args= -p "$_pid" 2>/dev/null | grep -q -- "--type="; then
|
||||||
|
continue
|
||||||
|
fi
|
||||||
|
_start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0)
|
||||||
|
_now=$(date +%s)
|
||||||
|
if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
restart_browser() {
|
restart_browser() {
|
||||||
local agent=$1
|
local agent=$1
|
||||||
local cdp_port=$2
|
local cdp_port=$2
|
||||||
@@ -88,16 +122,54 @@ check_one() {
|
|||||||
local agent=$1
|
local agent=$1
|
||||||
local cdp_port=$2
|
local cdp_port=$2
|
||||||
# node == agent == profile (unified naming)
|
# node == agent == profile (unified naming)
|
||||||
if ! check_agent "$agent" "$agent" "$cdp_port"; then
|
check_agent "$agent" "$agent" "$cdp_port"
|
||||||
|
local rc=$?
|
||||||
|
|
||||||
|
if [ $rc -eq 0 ]; then
|
||||||
|
# Healthy: reset the consecutive-API-failure counter.
|
||||||
|
rm -f "$STATE_DIR/failcount-$agent" 2>/dev/null
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [ $rc -eq 1 ]; then
|
||||||
|
# API failed but CDP port is listening: this is the false-kill vector
|
||||||
|
# (2026-10-04: single 30s API timeout killed 646's healthy browser).
|
||||||
|
# Require 2 CONSECUTIVE API failures before killing.
|
||||||
|
local count=0
|
||||||
|
local cf="$STATE_DIR/failcount-$agent"
|
||||||
|
[ -f "$cf" ] && count=$(cat "$cf" 2>/dev/null || echo 0)
|
||||||
|
count=$(( count + 1 ))
|
||||||
|
if [ "$count" -lt 2 ]; then
|
||||||
|
echo "$count" > "$cf"
|
||||||
|
echo "$(date -Iseconds) $agent: API failure $count of 2, deferring kill" >> "$LOG"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
rm -f "$cf" 2>/dev/null
|
||||||
|
else
|
||||||
|
# CDP port not listening (rc=2): zombie browser, kill immediately as before.
|
||||||
|
rm -f "$STATE_DIR/failcount-$agent" 2>/dev/null
|
||||||
|
fi
|
||||||
|
|
||||||
|
# De-conflict with chromebox-watchdog.sh: if the main browser for this
|
||||||
|
# profile launched <2 min ago, the watchdog just relaunched it — skip the
|
||||||
|
# kill path rather than racing it on a fresh cold start.
|
||||||
|
if recent_relaunch "$cdp_port"; then
|
||||||
|
echo "$(date -Iseconds) $agent: skipping kill (browser launched <2m ago, likely watchdog relaunch)" >> "$LOG"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
|
|
||||||
restart_browser "$agent" "$cdp_port"
|
restart_browser "$agent" "$cdp_port"
|
||||||
sleep 5
|
# 2026-10-04: post-restart re-check grace extended to ~60s total
|
||||||
|
# (restart_browser sleeps 15s internally + 45s here), matching
|
||||||
|
# chromebox-watchdog.sh's proven 60s retry window. Cold starts on a
|
||||||
|
# loaded box (load ~7) need more than 20s before CDP/API respond.
|
||||||
|
sleep 45
|
||||||
if ! check_agent "$agent" "$agent" "$cdp_port"; then
|
if ! check_agent "$agent" "$agent" "$cdp_port"; then
|
||||||
echo "$(date -Iseconds) $agent: CRITICAL - still down after restart" >> "$LOG"
|
echo "$(date -Iseconds) $agent: CRITICAL - still down after restart" >> "$LOG"
|
||||||
# TODO: Alert operator (e.g., via board post or email)
|
# TODO: Alert operator (e.g., via board post or email)
|
||||||
else
|
else
|
||||||
echo "$(date -Iseconds) $agent: RECOVERED after restart" >> "$LOG"
|
echo "$(date -Iseconds) $agent: RECOVERED after restart" >> "$LOG"
|
||||||
fi
|
fi
|
||||||
fi
|
|
||||||
}
|
}
|
||||||
|
|
||||||
# Every active node in the NODES.md registry gets checked — new nodes
|
# Every active node in the NODES.md registry gets checked — new nodes
|
||||||
|
|||||||
Reference in New Issue
Block a user