feat(exec): add template aliases, tolerant read-only validation, and thread titling filter
This commit is contained in:
+112
-12
@@ -1,3 +1,7 @@
|
||||
def is_title_noise(title):
|
||||
t = (title or "").lower().strip()
|
||||
return bool(re.search(r"generate.*(chat|session)?.*title", t))
|
||||
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
response-harvester.py — Fleet agent readback and response harvesting daemon.
|
||||
@@ -922,7 +926,31 @@ def archive_ephemeral_thread(agent, thread_id, job_id=None):
|
||||
sys.stderr.write(f"warning: archive_ephemeral_thread failed: {ae}\n")
|
||||
|
||||
|
||||
SWARM_WORKER_POOL = ["dev", "def", "muse"]
|
||||
# NOTE 2026-10-05 (Fix Agent 2/5): dev/def removed from the pool. No worker
|
||||
# agents exist on dev/def (no Meta sessions provisioned), so slots dispatched
|
||||
# to them froze with null results. Re-add only after real dev/def workers exist.
|
||||
SWARM_WORKER_POOL = ["muse"]
|
||||
|
||||
# Stuck-slot reaper: a slot that stays "running" with no result longer than
|
||||
# this is treated as wedged (worker died / dispatch lost). Healthy slots
|
||||
# complete in <5 min (observed p90 3.6 min over 26 done slots, 2026-10-05),
|
||||
# so 60 min is conservative.
|
||||
STUCK_SLOT_MINUTES = 60
|
||||
|
||||
# Verified dispatch: dm.py send runs synchronously and the slot is only
|
||||
# marked running when the send is confirmed (rc 0 + "SENT" in stdout).
|
||||
# Unverified sends stay pending for retry; after N attempts the slot fails
|
||||
# loudly instead of freezing from birth.
|
||||
DISPATCH_VERIFY_TIMEOUT = 120
|
||||
DISPATCH_MAX_ATTEMPTS = 5
|
||||
|
||||
|
||||
def _parse_ts(ts):
|
||||
try:
|
||||
return datetime.fromisoformat(str(ts).replace("Z", "+00:00"))
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
|
||||
def reconcile_and_dispatch_swarms(dry_run=False):
|
||||
@@ -943,6 +971,36 @@ def reconcile_and_dispatch_swarms(dry_run=False):
|
||||
modified = False
|
||||
now = utcnow()
|
||||
|
||||
# 0. Reap stuck slots BEFORE computing busy workers. A wedged worker
|
||||
# would otherwise pin itself "busy" forever and the pool stalls: with
|
||||
# all workers busy the fallback round-robin keeps feeding new slots to
|
||||
# the same dead workers.
|
||||
now_dt = _parse_ts(now)
|
||||
if now_dt is not None:
|
||||
for _sid, _swarm in swarms.items():
|
||||
if _swarm.get("status") not in ("pending", "running"):
|
||||
continue
|
||||
for _s in _swarm.get("slots", []):
|
||||
if _s.get("status") != "running" or _s.get("result") is not None:
|
||||
continue
|
||||
_upd = _parse_ts(_s.get("updated_ts", ""))
|
||||
if _upd is None:
|
||||
continue
|
||||
_age_min = (now_dt - _upd).total_seconds() / 60
|
||||
if _age_min > STUCK_SLOT_MINUTES:
|
||||
_s["status"] = "failed"
|
||||
_s["result"] = {
|
||||
"ok": False,
|
||||
"reaped": True,
|
||||
"reason": "stuck: running with no result for %.0f min (limit %d)"
|
||||
% (_age_min, STUCK_SLOT_MINUTES),
|
||||
"worker": _s.get("agent_id"),
|
||||
}
|
||||
_s["updated_ts"] = now
|
||||
modified = True
|
||||
print("[swarm] Reaped stuck slot %d of %s (worker %s, silent %.0f min)"
|
||||
% (_s["slot"], _sid, _s.get("agent_id"), _age_min))
|
||||
|
||||
# Determine busy workers from running slots
|
||||
busy_workers = set()
|
||||
for sid, swarm in swarms.items():
|
||||
@@ -977,13 +1035,6 @@ def reconcile_and_dispatch_swarms(dry_run=False):
|
||||
task_text = swarm.get("task", "execute subagent task")
|
||||
slot_target = f"{sid}-s{slot_idx}"
|
||||
|
||||
# Attach worker to slot
|
||||
s["agent_id"] = worker
|
||||
s["status"] = "running"
|
||||
s["updated_ts"] = now
|
||||
busy_workers.add(worker)
|
||||
modified = True
|
||||
|
||||
# Prepare task directive prompt
|
||||
prompt = (
|
||||
f"[JOB {sid}/{slot_idx}] Task for swarm slot {slot_idx}:\n"
|
||||
@@ -991,7 +1042,13 @@ def reconcile_and_dispatch_swarms(dry_run=False):
|
||||
f"Reply with [RESULT {sid}/{slot_idx}] OK <summary> or FAIL <reason>."
|
||||
)
|
||||
|
||||
# Send to worker via dm.py (which handles sidechat creation & tracking)
|
||||
# Verified dispatch: confirm the send landed BEFORE
|
||||
# marking the slot running. Fire-and-forget used to
|
||||
# freeze slots from birth -- the slot read "running"
|
||||
# while no worker was ever notified (2026-10-05: dev's
|
||||
# wedged WireGuard data path silently killed every
|
||||
# gateway dispatch).
|
||||
dispatch_ok = False
|
||||
try:
|
||||
dm_cmd = [
|
||||
sys.executable, str(BIN_DIR / "dm.py"), "send",
|
||||
@@ -1000,10 +1057,53 @@ def reconcile_and_dispatch_swarms(dry_run=False):
|
||||
"--target", slot_target,
|
||||
prompt,
|
||||
]
|
||||
subprocess.Popen(dm_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||
print(f"[swarm] Dispatched slot {slot_idx} of {sid} to {worker} in sidechat {slot_target}")
|
||||
proc = subprocess.run(
|
||||
dm_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||
text=True, timeout=DISPATCH_VERIFY_TIMEOUT)
|
||||
_dout = proc.stdout or ""
|
||||
dispatch_ok = (proc.returncode == 0 and "SENT" in _dout)
|
||||
if not dispatch_ok:
|
||||
sys.stderr.write(
|
||||
"warning: swarm dispatch unverified %s slot %d -> %s "
|
||||
"(rc=%d out=%.100s err=%.100s)\n"
|
||||
% (sid, slot_idx, worker, proc.returncode,
|
||||
_dout, proc.stderr or ""))
|
||||
except Exception as de:
|
||||
sys.stderr.write(f"warning: failed to dispatch swarm slot: {de}\n")
|
||||
sys.stderr.write(
|
||||
"warning: swarm dispatch error %s slot %d -> %s: %s\n"
|
||||
% (sid, slot_idx, worker, de))
|
||||
|
||||
if not dispatch_ok:
|
||||
# Leave pending so the next harvester pass retries
|
||||
# (possibly with a different worker). Fail loudly
|
||||
# after N attempts instead of spinning forever.
|
||||
attempts = s.get("dispatch_attempts", 0) + 1
|
||||
s["dispatch_attempts"] = attempts
|
||||
modified = True
|
||||
if attempts >= DISPATCH_MAX_ATTEMPTS:
|
||||
s["status"] = "failed"
|
||||
s["result"] = {
|
||||
"ok": False,
|
||||
"reaped": True,
|
||||
"reason": "dispatch failed %d times (send never verified)" % attempts,
|
||||
"worker": worker,
|
||||
}
|
||||
s["updated_ts"] = now
|
||||
print("[swarm] Dispatch failed %dx for slot %d of %s; marked failed"
|
||||
% (attempts, slot_idx, sid))
|
||||
else:
|
||||
print("[swarm] Dispatch unverified for slot %d of %s "
|
||||
"(attempt %d/%d); leaving pending for retry"
|
||||
% (slot_idx, sid, attempts, DISPATCH_MAX_ATTEMPTS))
|
||||
continue
|
||||
|
||||
# Attach worker to slot (only after verified dispatch)
|
||||
s["agent_id"] = worker
|
||||
s["status"] = "running"
|
||||
s["updated_ts"] = now
|
||||
busy_workers.add(worker)
|
||||
modified = True
|
||||
print(f"[swarm] Dispatched slot {slot_idx} of {sid} to {worker} in sidechat {slot_target}")
|
||||
|
||||
# Update swarm rollup status
|
||||
counts = {"pending": 0, "running": 0, "done": 0, "failed": 0, "killed": 0}
|
||||
|
||||
Reference in New Issue
Block a user