feat(kpi): add autonomous worker auto-spawn engine and watchdog reconciliation

This commit is contained in:
operator
2026-10-06 23:03:52 +00:00
parent 02a2189773
commit 0c6d2235ab
5 changed files with 244 additions and 13 deletions
+28 -11
View File
@@ -15,10 +15,14 @@ export DBUS_SESSION_BUS_ADDRESS="${DBUS_SESSION_BUS_ADDRESS:-unix:path=${XDG_RUN
# concurrent runs kill each others chrome (observed 2026-10-03: pip flapped
# with simultaneous "relaunch OK" and "relaunch FAILED").
LOCK="/tmp/chromebox-watchdog-${1:-pip}.lock"
exec 9>"$LOCK"
if ! flock -n 9; then
echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2
exit 0
# Tests source this file with CHROMEBOX_WATCHDOG_LIB_ONLY=1: they resolve
# ports and call helpers without running checks, so no lock is needed.
if [ "${CHROMEBOX_WATCHDOG_LIB_ONLY:-}" != "1" ]; then
exec 9>"$LOCK"
if ! flock -n 9; then
echo "[$(date -u +%FT%TZ)] [$1] another watchdog run in progress, skipping" >&2
exit 0
fi
fi
PROFILE="${1:-pip}"
@@ -40,13 +44,15 @@ rotate_log() {
rotate_log "$LOG"
# CHROME_LOG rotation happens after PROFILE is set (see below)
case "$PROFILE" in
muse) CDP_PORT=9410 ;;
pip) CDP_PORT=9420 ;;
646) CDP_PORT=9430 ;;
opm) CDP_PORT=9440 ;;
*) echo "unknown profile: $PROFILE" >&2; exit 1 ;;
esac
# Ports come from the fleet registry, not a hardcoded list: every active
# node (def/dev included) gets supervision automatically. The old 4-profile
# case left dev/def unsupervised — a dead Warp tunnel paged forever with
# no auto-recovery (2026-10-06 dev outage).
CDP_PORT="$("$NETVM_BIN/netvm-registry.py" "$PROFILE" 2>/dev/null)" || {
echo "unknown profile: $PROFILE" >&2
exit 1
}
[ -n "$CDP_PORT" ] || { echo "unknown profile: $PROFILE" >&2; exit 1; }
rotate_log "$CHROME_LOG"
log() { echo "[$(date -u +%FT%TZ)] [$PROFILE] $*" | tee -a "$LOG"; }
@@ -107,7 +113,18 @@ print(h*3600 + mi*60 + se)
esac
}
# Allow sourcing for tests without running checks.
if [ "${CHROMEBOX_WATCHDOG_LIB_ONLY:-}" = "1" ]; then
return 0 2>/dev/null || exit 0
fi
if healthy; then
# Auto-reconcile idle workers for healthy profiles
# DISABLED 2026-10-06 by operator-646: kpi auto-spawn ignores job schedule fields;
# find_pending_work_for_node returns the alphabetically-first definition every tick,
# re-spawning and re-noticing every ~2min (pip b01 BOX-AUTO-WORKER loop, 29+ copies).
# Watchdog health path untouched. Re-enable once the spawner is schedule-aware.
# python3 "$NETVM_BIN/super-cli.py" kpi auto-spawn --node "$PROFILE" >>"$LOG" 2>&1 || true
exit 0
fi