feat(autonomy): autonomous swarm worker orchestration, nudge triggers, and recursive loop

This commit is contained in:
operator
2026-10-05 04:32:06 +00:00
parent b0454289ae
commit b88209d7a1
4 changed files with 1636 additions and 4 deletions
+226 -1
View File
@@ -48,6 +48,8 @@ JOB_LOG = NETVM_ROOT / "job-log.jsonl"
FOLLOWUPS_FILE = NETVM_ROOT / "followups.json"
JOB_SIDECHATS_FILE = NETVM_ROOT / "job-sidechats.json"
WAKE_SIDECHATS_FILE = Path("/home/super/sidechat-wake/wake-sidechats.json")
SWARM_FILE = NETVM_ROOT / "swarms.json"
NUDGE_TRACKER_FILE = NETVM_ROOT / "conversation-nudge-tracker.json"
JOBS_DIR = NETVM_ROOT / "jobs"
DISPATCH_PY = BIN_DIR / "job-dispatch.py"
@@ -583,6 +585,8 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
tool_calls = parse_tool_calls(text)
if tool_calls and not dry_run:
for op, args in tool_calls:
if isinstance(args, dict) and "agent" not in args:
args["agent"] = agent
print(f"[{agent}] Executing tool '{op}' from message {mid[:8]} in thread {thread_name or thread_id[:8]}")
t_ok, t_res = execute_agent_tool(agent, op, args)
tool_record = {
@@ -628,7 +632,19 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
}
if not dry_run:
append_jsonl(JOB_LOG, job_record)
trigger_chain_next(job_id, result_text, success=not is_fail)
# Check if this is a swarm slot result: sw-YYYYMMDD-HHMMSS-xxxx/<slot>
if "/" in job_id and job_id.startswith("sw-"):
try:
s_sid, s_slot = job_id.split("/", 1)
s_proc = subprocess.Popen(
[sys.executable, str(BIN_DIR / "box-ctl.py"), "swarm-report", s_sid, s_slot],
stdin=subprocess.PIPE, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL
)
s_proc.communicate(input=json.dumps({"ok": not is_fail, "result": result_text}).encode("utf-8"))
except Exception as se:
sys.stderr.write(f"warning: failed to record swarm report: {se}\n")
else:
trigger_chain_next(job_id, result_text, success=not is_fail)
if not is_fail:
archive_ephemeral_thread(agent, thread_id, job_id=job_id)
clear_matching_followups(followups, agent, thread_id, mid, text,
@@ -640,6 +656,7 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
dry_run, job_id=job_id, verb=verb)
else:
clear_matching_followups(followups, agent, thread_id, mid, text, dry_run)
maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=dry_run)
return new_messages, new_wm, job_results
@@ -782,6 +799,212 @@ def archive_ephemeral_thread(agent, thread_id, job_id=None):
sys.stderr.write(f"warning: archive_ephemeral_thread failed: {ae}\n")
SWARM_WORKER_POOL = ["dev", "def", "muse"]
def reconcile_and_dispatch_swarms(dry_run=False):
"""
Autonomous Swarm Orchestrator:
1. Scans swarms.json for pending slots.
2. Dynamically allocates available auxiliary worker nodes (dev, def, muse).
3. Provisions ephemeral sidechat per slot and dispatches the task with [RESULT <swarm_id>/<slot>].
4. Upon completion of all slots, sends completion summary DM to originating coordinator.
"""
if not SWARM_FILE.exists() or dry_run:
return
try:
swarms = json.loads(SWARM_FILE.read_text(encoding="utf-8"))
except Exception:
return
modified = False
now = utcnow()
# Determine busy workers from running slots
busy_workers = set()
for sid, swarm in swarms.items():
if swarm.get("status") in ("pending", "running"):
for slot in swarm.get("slots", []):
if slot.get("status") == "running" and slot.get("agent_id"):
busy_workers.add(slot["agent_id"])
# Process each swarm
for sid, swarm in swarms.items():
st = swarm.get("status")
creator = swarm.get("created_by") or "646"
if creator not in ("646", "pip", "opm", "muse", "dev", "def"):
creator = "646"
# 1. Allocate & dispatch pending slots
if st in ("pending", "running"):
slots = swarm.get("slots", [])
for s in slots:
if s.get("status") == "pending":
# Find first available worker
worker = None
for w in SWARM_WORKER_POOL:
if w not in busy_workers:
worker = w
break
if not worker:
# Fallback round-robin across worker pool if all are busy
worker = SWARM_WORKER_POOL[s["slot"] % len(SWARM_WORKER_POOL)]
slot_idx = s["slot"]
task_text = swarm.get("task", "execute subagent task")
slot_target = f"{sid}-s{slot_idx}"
# Attach worker to slot
s["agent_id"] = worker
s["status"] = "running"
s["updated_ts"] = now
busy_workers.add(worker)
modified = True
# Prepare task directive prompt
prompt = (
f"[JOB {sid}/{slot_idx}] Task for swarm slot {slot_idx}:\n"
f"{task_text}\n\n"
f"Reply with [RESULT {sid}/{slot_idx}] OK <summary> or FAIL <reason>."
)
# Send to worker via dm.py (which handles sidechat creation & tracking)
try:
dm_cmd = [
sys.executable, str(BIN_DIR / "dm.py"), "send",
"--agent", creator,
"--to", worker,
"--target", slot_target,
prompt,
]
subprocess.Popen(dm_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
print(f"[swarm] Dispatched slot {slot_idx} of {sid} to {worker} in sidechat {slot_target}")
except Exception as de:
sys.stderr.write(f"warning: failed to dispatch swarm slot: {de}\n")
# Update swarm rollup status
counts = {"pending": 0, "running": 0, "done": 0, "failed": 0, "killed": 0}
for slot in slots:
counts[slot.get("status", "pending")] = counts.get(slot.get("status", "pending"), 0) + 1
if counts["pending"] + counts["running"] == 0:
swarm["status"] = "completed" if counts["failed"] == 0 else "partial"
swarm["updated_ts"] = now
modified = True
elif counts["running"] or counts["done"]:
swarm["status"] = "running"
swarm["updated_ts"] = now
modified = True
# 2. Check if newly completed/partial and notify coordinator
if swarm.get("status") in ("completed", "partial") and not swarm.get("notified_coordinator"):
swarm["notified_coordinator"] = True
modified = True
done_cnt = sum(1 for sl in swarm.get("slots", []) if sl.get("status") == "done")
total_cnt = len(swarm.get("slots", []))
summary_msg = (
f"[Swarm Report] Swarm {sid} ({swarm.get('label') or 'task'}) finished: "
f"{done_cnt}/{total_cnt} slots successful. "
f"Console: https://box.muse-dev.online/#dashboard"
)
try:
coord_dm = [
sys.executable, str(BIN_DIR / "dm.py"), "send",
"--agent", "box",
"--to", creator,
"--target", "main",
"--allow-main-chat",
summary_msg,
]
subprocess.Popen(coord_dm, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
print(f"[swarm] Delivered completion report for {sid} to coordinator {creator}")
except Exception as ce:
sys.stderr.write(f"warning: failed to notify swarm coordinator: {ce}\n")
if modified:
tmp_swarms = str(SWARM_FILE) + f".tmp.{os.getpid()}"
with open(tmp_swarms, "w", encoding="utf-8") as f:
json.dump(swarms, f, indent=2)
os.replace(tmp_swarms, SWARM_FILE)
def check_and_archive_terminal_swarms():
"""Check swarms.json and auto-archive ephemeral sidechats for completed or terminal swarms."""
if not SWARM_FILE.exists() or not JOB_SIDECHATS_FILE.exists():
return
try:
swarms_data = json.loads(SWARM_FILE.read_text(encoding="utf-8"))
sc_data = json.loads(JOB_SIDECHATS_FILE.read_text(encoding="utf-8"))
except Exception:
return
for sid, swarm in swarms_data.items():
st = swarm.get("status")
if st in ("completed", "partial", "killed"):
# Check slots or matching sidechats
for key, val in sc_data.items():
if isinstance(val, dict) and not val.get("archived") and val.get("type") != "persistent":
if sid in key or (swarm.get("label") and swarm.get("label") in key):
tu = val.get("thread_uuid")
ag = val.get("agent", "opm")
if tu:
archive_ephemeral_thread(ag, tu, job_id=sid)
def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=False):
"""
If an agent replies conversationally in a sidechat backed by a job or follow-up
without providing [RESULT <id>] or tool directives, deliver a terse 1-turn nudge footer.
"""
if dry_run or not thread_id or thread_id == "main":
return
if not re.fullmatch(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", thread_id.lower()):
return
# Look up active follow-ups or job association for this thread
followups = load_json_file(FOLLOWUPS_FILE)
matching_job_id = None
for f_id, f_rec in followups.items():
if f_rec.get("status") in ("pending", "acknowledged") and f_rec.get("thread_uuid") == thread_id:
matching_job_id = f_rec.get("job_id") or f_id
break
if not matching_job_id and JOB_SIDECHATS_FILE.exists():
try:
sc_state = json.loads(JOB_SIDECHATS_FILE.read_text(encoding="utf-8"))
for k, v in sc_state.items():
if isinstance(v, dict) and v.get("thread_uuid") == thread_id:
if k.startswith(("box-", "job-", "pipe-")):
matching_job_id = k
break
except Exception:
pass
if not matching_job_id:
return
tracker = load_json_file(NUDGE_TRACKER_FILE)
rec = tracker.get(thread_id, {})
if rec.get("nudged_for_mid") == mid or rec.get("nudge_count", 0) >= 1:
return
# Inject terse nudge footer
nudge_msg = f"[Nudge: To advance the workflow, reply with [RESULT {matching_job_id}] <outcome>]"
try:
import muse_hybrid
print(f"[{agent}] Injecting 1-turn nudge into {thread_name or thread_id[:8]} for job {matching_job_id}")
muse_hybrid.send_message(agent, nudge_msg, thread_id=thread_id, wait=0)
tracker[thread_id] = {
"ts": utcnow(),
"job_id": matching_job_id,
"nudged_for_mid": mid,
"nudge_count": rec.get("nudge_count", 0) + 1,
}
save_json_file(NUDGE_TRACKER_FILE, tracker)
except Exception as ne:
sys.stderr.write(f"warning: failed to deliver conversational nudge: {ne}\n")
# Chain deduplication
CHAINED_JOBS_FILE = Path(__file__).parent / "chained-jobs.json"
@@ -1081,6 +1304,8 @@ def harvest_cycle(target_agent=None, dry_run=False, output_json=False):
if not dry_run:
save_json_file(WATERMARKS_FILE, watermarks)
reconcile_and_dispatch_swarms()
check_and_archive_terminal_swarms()
# Output formatting
if output_json: