approvals: stale-wait cleanup, key-decision notify, TTLs, key/browser isolation

- bin/approvals.py: responded-wait filtering + auto-mark, key-decision sidechat-first notify (notified flag, --message, --allow-main-chat), TTL defaults (input 30m / browser 30m / key 2h), cross-type guard (browser actions cannot resolve key requests), sweep_expired_key_requests; restores check_node_key_request fallback in inspect_node_approvals

- bin/box-ctl.py + bin/super-cli.py (approvals hunks only): --message/--allow-main-chat passthrough on allow/deny, clear/clear-all actions, def sidechat routing; restores sys.exit(1) on dismiss failure

- bin/fleet-alert-check.sh: TTL-aware state_machine (EXPIRED action), auto-deny expired browser approvals (fail closed), auto-dismiss expired input waits, targeted per-agent DM for input waits

- bin/job-dispatch.py + bin/gravity.py: KEY_APPROVAL excluded from auto-approval

Reviewed by 5 independent reviewers (all APPROVE/APPROVE WITH NOTES); integration gate GO (17/17 tests). TUI hunks in super-cli.py intentionally excluded.
This commit is contained in:
operator
2026-10-06 00:49:46 +00:00
parent adfcd2e602
commit 740648e973
6 changed files with 515 additions and 48 deletions
+116 -20
View File
@@ -34,6 +34,11 @@ set -uo pipefail
THRESHOLD="${FLEET_ALERT_THRESHOLD:-2}"
REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}"
# Approval/input-wait TTLs (seconds): conditions failing longer than this are
# auto-expired (input waits dismissed, key requests denied) instead of paging
# forever. Overridable per environment.
INPUT_WAIT_TTL="${FLEET_ALERT_INPUT_WAIT_TTL:-1800}"
BROWSER_APPROVAL_TTL="${FLEET_ALERT_BROWSER_APPROVAL_TTL:-1800}"
QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}"
DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}"
INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}"
@@ -51,14 +56,18 @@ NOW=$(date +%s)
log() { echo "$(date -Iseconds) $*" >> "$LOG"; }
# --- shared consecutive-failure state machine (also used by the container relay) ---
# usage: state_machine <cond> <failing 0|1> -> prints "<ACTION> <fails>"
# ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | NONE
# usage: state_machine <cond> <failing 0|1> [ttl_seconds] -> prints "<ACTION> <fails>"
# When ttl_seconds > 0 and the condition has failed longer than the TTL,
# prints "EXPIRED <fails>" so the caller can auto-resolve (dismiss/deny).
# State entries track first_fail_ts (epoch of first consecutive failure).
# ACTION: ALERT_FIRST | ALERT_REALERT | RECOVERY | SUPPRESSED | EXPIRED | NONE
state_machine() {
local cond="$1" failing="$2"
local cond="$1" failing="$2" ttl="${3:-0}"
THRESHOLD="$THRESHOLD" REALERT_MIN="$REALERT_MIN" QUIET_HOURS="$QUIET_HOURS" \
FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" <<'PYEOF'
FLEET_ALERT_DRY_RUN="$DRY_RUN" python3 - "$STATE" "$cond" "$failing" "$ttl" <<'PYEOF'
import json, os, sys, time
state_path, cond, failing_s = sys.argv[1], sys.argv[2], sys.argv[3]
ttl_seconds = int(sys.argv[4]) if len(sys.argv) > 4 else 0
failing = failing_s == "1"
threshold = int(os.environ.get("THRESHOLD", "2"))
realert_min = int(os.environ.get("REALERT_MIN", "30"))
@@ -85,23 +94,34 @@ except Exception:
e = st.get(cond) or {"fails": 0, "alerted": False, "last_alert_ts": 0}
action = "NONE"
if failing:
if int(e.get("fails", 0)) == 0:
e["first_fail_ts"] = now
e["fails"] = int(e.get("fails", 0)) + 1
due = e["fails"] >= threshold and (
not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60
)
if due:
first = not e.get("alerted")
if not first and in_quiet(qh):
action = "SUPPRESSED"
else:
action = "ALERT_FIRST" if first else "ALERT_REALERT"
e["alerted"] = True
e["last_alert_ts"] = now
# TTL expiry: failing longer than ttl_seconds -> EXPIRED (caller auto-resolves)
if ttl_seconds > 0 and now - int(e.get("first_fail_ts", now)) >= ttl_seconds:
action = "EXPIRED"
# Reset so a fresh incident starts clean after the caller resolves it
e["fails"] = 0
e["alerted"] = False
e.pop("first_fail_ts", None)
else:
due = e["fails"] >= threshold and (
not e.get("alerted") or now - int(e.get("last_alert_ts", 0)) >= realert_min * 60
)
if due:
first = not e.get("alerted")
if not first and in_quiet(qh):
action = "SUPPRESSED"
else:
action = "ALERT_FIRST" if first else "ALERT_REALERT"
e["alerted"] = True
e["last_alert_ts"] = now
else:
if e.get("alerted"):
action = "RECOVERY"
e["fails"] = 0
e["alerted"] = False
e.pop("first_fail_ts", None)
st[cond] = e
if not dry:
json.dump(st, open(state_path, "w"))
@@ -151,6 +171,28 @@ box_notify() {
for p in $pids; do wait "$p" 2>/dev/null; done
}
notify_input_wait() {
# Targeted DM for input waits (2026-10-05): DM ONLY the specific agent
# whose session is waiting for human input -- not a broadcast to all
# healthy agents. The #lobby post still fires via the relay leg for
# human visibility; this DM ensures the responsible operator sees it
# in their sidechat without digging through lobby noise.
# Best-effort: never fatal to the 5-minute check loop.
local node="$1"
local detail="$2"
local msg="[fleet-alert] INPUT WAIT: ${detail} -- reply: box approval reply ${node} \"<msg>\" or box notify ${node} \"<msg>\""
msg="${msg:0:900}"
if [ "$DRY_RUN" = "1" ]; then
log "DRY-RUN would DM $node re input_wait"
return 0
fi
if timeout 60 python3 "$BIN/box-ctl.py" notify "$node" "$msg" >/dev/null 2>&1; then
log "input_wait targeted DM sent to $node"
else
log "input_wait DM to $node failed (best-effort, non-fatal)"
fi
}
injected() { # cond -> 0 if injected-fail
case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac
}
@@ -214,7 +256,7 @@ print(json.dumps(out))
detail="Agent $node approvals clear"
fi
injected "$cond" && failing=1
read -r action fails < <(state_machine "$cond" "$failing")
read -r action fails < <(state_machine "$cond" "$failing" "$BROWSER_APPROVAL_TTL")
case "$action" in
ALERT_FIRST|ALERT_REALERT)
emit_record "ALERT" "$cond" "$detail" "$fails"
@@ -226,6 +268,21 @@ print(json.dumps(out))
SUPPRESSED)
log "$cond still critical x$fails — re-page suppressed"
;;
EXPIRED)
# Browser approval dialog exceeded BROWSER_APPROVAL_TTL without a
# human decision: fail closed by denying it.
log "$cond EXPIRED after ${BROWSER_APPROVAL_TTL}s without human decision — auto-denying (fail closed)"
python3 - "$node" <<'PYEOF3'
import sys, json
sys.path.insert(0, "/home/super/Projects/NetVM/bin")
import approvals
node = sys.argv[1]
print(json.dumps(approvals.deny_node_approval(node, caller="approval-ttl-expire")))
approvals.log_box_ctl("approval-expired", name=node, caller="approval-ttl-expire",
extra={"note": "browser approval TTL elapsed; auto-denied (fail closed)"})
PYEOF3
emit_record "RECOVERY" "$cond" "Agent $node browser approval expired after ${BROWSER_APPROVAL_TTL}s; auto-denied" "$fails"
;;
esac
# 2. Sidebar task waiting on human input
@@ -245,11 +302,11 @@ if w:
detail_in="Agent $node tasks running"
fi
injected "$cond_in" && failing_in=1
read -r action_in fails_in < <(state_machine "$cond_in" "$failing_in")
read -r action_in fails_in < <(state_machine "$cond_in" "$failing_in" "$INPUT_WAIT_TTL")
case "$action_in" in
ALERT_FIRST|ALERT_REALERT)
emit_record "ALERT" "$cond_in" "$detail_in" "$fails_in"
echo "$cond_in" >> "$STATE_DIR/.alerts.tmp"
echo "$cond_in|$detail_in" >> "$STATE_DIR/.alerts.tmp"
;;
RECOVERY)
emit_record "RECOVERY" "$cond_in" "$detail_in" "$fails_in"
@@ -257,6 +314,27 @@ if w:
SUPPRESSED)
log "$cond_in still critical x$fails_in — re-page suppressed"
;;
EXPIRED)
# Input wait exceeded INPUT_WAIT_TTL without human response:
# auto-dismiss so the agent unblocks. Log the expiry and emit a
# RECOVERY record (the wait is gone, not merely un-paged).
log "$cond_in EXPIRED after ${INPUT_WAIT_TTL}s without human input — auto-dismissing"
python3 - "$node" <<'PYEOF2'
import sys
sys.path.insert(0, "/home/super/Projects/NetVM/bin")
import approvals, json
node = sys.argv[1]
info = approvals.inspect_node_approvals(node)
for w in info.get("input_waits", []) or []:
t = w.get("task")
if t:
approvals.mark_wait_responded(node, t, caller="approval-ttl-expire")
approvals.log_box_ctl("approval-wait-expired", name=node, caller="approval-ttl-expire",
extra={"note": "input wait TTL elapsed; auto-dismissed"})
print(json.dumps(approvals.dismiss_node_task(node, caller="approval-ttl-expire")))
PYEOF2
emit_record "RECOVERY" "$cond_in" "Agent $node input wait expired after ${INPUT_WAIT_TTL}s; auto-dismissed" "$fails_in"
;;
esac
done
@@ -328,8 +406,26 @@ rm -f "$STATE_DIR/.healthy.tmp"
# Notify for this run's alerts (best effort). Skip entirely when nothing is healthy
# (notify needs a working browser via dm.py) or in dry-run.
if [ -n "$HEALTHY_AGENTS" ] && [ -f "$STATE_DIR/.alerts.tmp" ]; then
while read -r cond; do
[ -n "$cond" ] && box_notify "$cond" "see #lobby for detail"
while IFS= read -r line; do
# alerts.tmp format: "cond" or "cond|detail" (input_wait carries detail)
cond="${line%%|*}"
detail="${line#*|}"
[ "$detail" = "$line" ] && detail=""
[ -n "$cond" ] || continue
case "$cond" in
input_wait:*)
# Targeted: DM only the waiting agent, not a broadcast.
node="${cond#input_wait:}"
if [ -n "$detail" ]; then
notify_input_wait "$node" "$detail"
else
box_notify "$cond" "see #lobby for detail"
fi
;;
*)
box_notify "$cond" "see #lobby for detail"
;;
esac
done < "$STATE_DIR/.alerts.tmp"
elif [ -f "$STATE_DIR/.alerts.tmp" ]; then
log "no healthy agents — box notify skipped (DM path needs a working browser)"