From 76f854da6bbbb81806d3e578f8d8ed02a943f52c Mon Sep 17 00:00:00 2001 From: operator-main Date: Sun, 4 Oct 2026 22:46:44 +0000 Subject: [PATCH] Harden agent-health.sh: soften API-timeout kill path, add recent-relaunch guard - Require 2 CONSECUTIVE muse-chat-api.py API failures before kill -9 (per-node counter in /tmp/agent-health-state, reset on success). A single 30s API timeout killed 646s healthy browser at 20:48:44 UTC while its CDP port was still listening. - Extend post-restart re-check grace to ~60s (15s internal + 45s), matching chromebox-watchdog.shs proven 60s retry window. - Add recent_relaunch() guard (mirrors chromebox-watchdog.sh idiom): skip the kill path when the main browser process launched <2 min ago, so the two watchdogs can not kill each others fresh browsers. - check_agent now returns 0/1/2 (healthy/api-fail/port-down); CDP-port failure still kills immediately. warp-$node checks untouched. Session: sidechat/chromebox-fixes --- bin/agent-health.sh | 90 ++++++++++++++++++++++++++++++++++++++++----- 1 file changed, 81 insertions(+), 9 deletions(-) diff --git a/bin/agent-health.sh b/bin/agent-health.sh index 2ff56f7..5bb7d0c 100755 --- a/bin/agent-health.sh +++ b/bin/agent-health.sh @@ -16,6 +16,14 @@ NETVM_BIN="$(cd "$(dirname "$0")" && pwd)" LOG="/tmp/agent-health.log" +# 2026-10-04: per-node consecutive-API-failure counters. A single +# muse-chat-api.py failure must not kill a healthy browser (observed +# 2026-10-04 20:48:44 UTC: 646's browser killed on an API timeout while its +# CDP port was still listening). Require 2 CONSECUTIVE API failures before +# the kill path; the counter resets on any successful check. +STATE_DIR="/tmp/agent-health-state" +mkdir -p "$STATE_DIR" 2>/dev/null + check_cdp_port() { # Verify CDP port is actually listening in the netns. # A browser can be running but not bound to CDP (zombie state). @@ -28,6 +36,7 @@ check_cdp_port() { fi } +# Return codes: 0 = healthy, 1 = API check failed (port OK), 2 = CDP port down. check_agent() { local agent=$1 local node=$2 @@ -36,7 +45,7 @@ check_agent() { # First: verify CDP port is listening (catches zombie browsers) if ! check_cdp_port "$node" "$cdp_port"; then echo "$(date -Iseconds) $agent: FAIL (cdp port $cdp_port not listening)" >> "$LOG" - return 1 + return 2 fi # Then: try API messages command (lightweight check) @@ -49,6 +58,31 @@ check_agent() { fi } +# 2026-10-04: recent-relaunch guard. chromebox-watchdog.sh (2-min timer) and +# this script (5-min timer) could otherwise kill each other's fresh browsers: +# a browser just relaunched by the watchdog is still starting when this +# script's API check times out on it. Skip the kill path when the main +# browser process for this profile launched <2 min ago. Mirrors +# chromebox-watchdog.sh's relaunch-loop guard idiom (main process only: +# --remote-debugging-port present, no --type= flag; renderer/gpu children +# start later than the main process and must not satisfy this check). +recent_relaunch() { + local cdp_port=$1 + local _pid _start _now + for _pid in $(pgrep -f "chromium.*--remote-debugging-port=${cdp_port}([[:space:]]|$)" 2>/dev/null); do + # Skip child processes (renderer, gpu, etc.) — only the main browser counts + if ps -o args= -p "$_pid" 2>/dev/null | grep -q -- "--type="; then + continue + fi + _start=$(date -d "$(ps -o lstart= -p "$_pid" 2>/dev/null)" +%s 2>/dev/null || echo 0) + _now=$(date +%s) + if [ $(( _now - _start )) -lt 120 ] && [ "$_start" -gt 0 ]; then + return 0 + fi + done + return 1 +} + restart_browser() { local agent=$1 local cdp_port=$2 @@ -88,15 +122,53 @@ check_one() { local agent=$1 local cdp_port=$2 # node == agent == profile (unified naming) - if ! check_agent "$agent" "$agent" "$cdp_port"; then - restart_browser "$agent" "$cdp_port" - sleep 5 - if ! check_agent "$agent" "$agent" "$cdp_port"; then - echo "$(date -Iseconds) $agent: CRITICAL - still down after restart" >> "$LOG" - # TODO: Alert operator (e.g., via board post or email) - else - echo "$(date -Iseconds) $agent: RECOVERED after restart" >> "$LOG" + check_agent "$agent" "$agent" "$cdp_port" + local rc=$? + + if [ $rc -eq 0 ]; then + # Healthy: reset the consecutive-API-failure counter. + rm -f "$STATE_DIR/failcount-$agent" 2>/dev/null + return 0 + fi + + if [ $rc -eq 1 ]; then + # API failed but CDP port is listening: this is the false-kill vector + # (2026-10-04: single 30s API timeout killed 646's healthy browser). + # Require 2 CONSECUTIVE API failures before killing. + local count=0 + local cf="$STATE_DIR/failcount-$agent" + [ -f "$cf" ] && count=$(cat "$cf" 2>/dev/null || echo 0) + count=$(( count + 1 )) + if [ "$count" -lt 2 ]; then + echo "$count" > "$cf" + echo "$(date -Iseconds) $agent: API failure $count of 2, deferring kill" >> "$LOG" + return 0 fi + rm -f "$cf" 2>/dev/null + else + # CDP port not listening (rc=2): zombie browser, kill immediately as before. + rm -f "$STATE_DIR/failcount-$agent" 2>/dev/null + fi + + # De-conflict with chromebox-watchdog.sh: if the main browser for this + # profile launched <2 min ago, the watchdog just relaunched it — skip the + # kill path rather than racing it on a fresh cold start. + if recent_relaunch "$cdp_port"; then + echo "$(date -Iseconds) $agent: skipping kill (browser launched <2m ago, likely watchdog relaunch)" >> "$LOG" + return 0 + fi + + restart_browser "$agent" "$cdp_port" + # 2026-10-04: post-restart re-check grace extended to ~60s total + # (restart_browser sleeps 15s internally + 45s here), matching + # chromebox-watchdog.sh's proven 60s retry window. Cold starts on a + # loaded box (load ~7) need more than 20s before CDP/API respond. + sleep 45 + if ! check_agent "$agent" "$agent" "$cdp_port"; then + echo "$(date -Iseconds) $agent: CRITICAL - still down after restart" >> "$LOG" + # TODO: Alert operator (e.g., via board post or email) + else + echo "$(date -Iseconds) $agent: RECOVERED after restart" >> "$LOG" fi }