feat(exec): add template aliases, tolerant read-only validation, and thread titling filter

This commit is contained in:
operator
2026-10-05 15:14:24 +00:00
parent 8e6dd1e892
commit 5ff32b5202
3 changed files with 188 additions and 17 deletions
+112 -12
View File
@@ -1,3 +1,7 @@
def is_title_noise(title):
t = (title or "").lower().strip()
return bool(re.search(r"generate.*(chat|session)?.*title", t))
#!/usr/bin/env python3
"""
response-harvester.py — Fleet agent readback and response harvesting daemon.
@@ -922,7 +926,31 @@ def archive_ephemeral_thread(agent, thread_id, job_id=None):
sys.stderr.write(f"warning: archive_ephemeral_thread failed: {ae}\n")
SWARM_WORKER_POOL = ["dev", "def", "muse"]
# NOTE 2026-10-05 (Fix Agent 2/5): dev/def removed from the pool. No worker
# agents exist on dev/def (no Meta sessions provisioned), so slots dispatched
# to them froze with null results. Re-add only after real dev/def workers exist.
SWARM_WORKER_POOL = ["muse"]
# Stuck-slot reaper: a slot that stays "running" with no result longer than
# this is treated as wedged (worker died / dispatch lost). Healthy slots
# complete in <5 min (observed p90 3.6 min over 26 done slots, 2026-10-05),
# so 60 min is conservative.
STUCK_SLOT_MINUTES = 60
# Verified dispatch: dm.py send runs synchronously and the slot is only
# marked running when the send is confirmed (rc 0 + "SENT" in stdout).
# Unverified sends stay pending for retry; after N attempts the slot fails
# loudly instead of freezing from birth.
DISPATCH_VERIFY_TIMEOUT = 120
DISPATCH_MAX_ATTEMPTS = 5
def _parse_ts(ts):
try:
return datetime.fromisoformat(str(ts).replace("Z", "+00:00"))
except Exception:
return None
def reconcile_and_dispatch_swarms(dry_run=False):
@@ -943,6 +971,36 @@ def reconcile_and_dispatch_swarms(dry_run=False):
modified = False
now = utcnow()
# 0. Reap stuck slots BEFORE computing busy workers. A wedged worker
# would otherwise pin itself "busy" forever and the pool stalls: with
# all workers busy the fallback round-robin keeps feeding new slots to
# the same dead workers.
now_dt = _parse_ts(now)
if now_dt is not None:
for _sid, _swarm in swarms.items():
if _swarm.get("status") not in ("pending", "running"):
continue
for _s in _swarm.get("slots", []):
if _s.get("status") != "running" or _s.get("result") is not None:
continue
_upd = _parse_ts(_s.get("updated_ts", ""))
if _upd is None:
continue
_age_min = (now_dt - _upd).total_seconds() / 60
if _age_min > STUCK_SLOT_MINUTES:
_s["status"] = "failed"
_s["result"] = {
"ok": False,
"reaped": True,
"reason": "stuck: running with no result for %.0f min (limit %d)"
% (_age_min, STUCK_SLOT_MINUTES),
"worker": _s.get("agent_id"),
}
_s["updated_ts"] = now
modified = True
print("[swarm] Reaped stuck slot %d of %s (worker %s, silent %.0f min)"
% (_s["slot"], _sid, _s.get("agent_id"), _age_min))
# Determine busy workers from running slots
busy_workers = set()
for sid, swarm in swarms.items():
@@ -977,13 +1035,6 @@ def reconcile_and_dispatch_swarms(dry_run=False):
task_text = swarm.get("task", "execute subagent task")
slot_target = f"{sid}-s{slot_idx}"
# Attach worker to slot
s["agent_id"] = worker
s["status"] = "running"
s["updated_ts"] = now
busy_workers.add(worker)
modified = True
# Prepare task directive prompt
prompt = (
f"[JOB {sid}/{slot_idx}] Task for swarm slot {slot_idx}:\n"
@@ -991,7 +1042,13 @@ def reconcile_and_dispatch_swarms(dry_run=False):
f"Reply with [RESULT {sid}/{slot_idx}] OK <summary> or FAIL <reason>."
)
# Send to worker via dm.py (which handles sidechat creation & tracking)
# Verified dispatch: confirm the send landed BEFORE
# marking the slot running. Fire-and-forget used to
# freeze slots from birth -- the slot read "running"
# while no worker was ever notified (2026-10-05: dev's
# wedged WireGuard data path silently killed every
# gateway dispatch).
dispatch_ok = False
try:
dm_cmd = [
sys.executable, str(BIN_DIR / "dm.py"), "send",
@@ -1000,10 +1057,53 @@ def reconcile_and_dispatch_swarms(dry_run=False):
"--target", slot_target,
prompt,
]
subprocess.Popen(dm_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
print(f"[swarm] Dispatched slot {slot_idx} of {sid} to {worker} in sidechat {slot_target}")
proc = subprocess.run(
dm_cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
text=True, timeout=DISPATCH_VERIFY_TIMEOUT)
_dout = proc.stdout or ""
dispatch_ok = (proc.returncode == 0 and "SENT" in _dout)
if not dispatch_ok:
sys.stderr.write(
"warning: swarm dispatch unverified %s slot %d -> %s "
"(rc=%d out=%.100s err=%.100s)\n"
% (sid, slot_idx, worker, proc.returncode,
_dout, proc.stderr or ""))
except Exception as de:
sys.stderr.write(f"warning: failed to dispatch swarm slot: {de}\n")
sys.stderr.write(
"warning: swarm dispatch error %s slot %d -> %s: %s\n"
% (sid, slot_idx, worker, de))
if not dispatch_ok:
# Leave pending so the next harvester pass retries
# (possibly with a different worker). Fail loudly
# after N attempts instead of spinning forever.
attempts = s.get("dispatch_attempts", 0) + 1
s["dispatch_attempts"] = attempts
modified = True
if attempts >= DISPATCH_MAX_ATTEMPTS:
s["status"] = "failed"
s["result"] = {
"ok": False,
"reaped": True,
"reason": "dispatch failed %d times (send never verified)" % attempts,
"worker": worker,
}
s["updated_ts"] = now
print("[swarm] Dispatch failed %dx for slot %d of %s; marked failed"
% (attempts, slot_idx, sid))
else:
print("[swarm] Dispatch unverified for slot %d of %s "
"(attempt %d/%d); leaving pending for retry"
% (slot_idx, sid, attempts, DISPATCH_MAX_ATTEMPTS))
continue
# Attach worker to slot (only after verified dispatch)
s["agent_id"] = worker
s["status"] = "running"
s["updated_ts"] = now
busy_workers.add(worker)
modified = True
print(f"[swarm] Dispatched slot {slot_idx} of {sid} to {worker} in sidechat {slot_target}")
# Update swarm rollup status
counts = {"pending": 0, "running": 0, "done": 0, "failed": 0, "killed": 0}