feat(autonomy): autonomous swarm worker orchestration, nudge triggers, and recursive loop
This commit is contained in:
+226
-1
@@ -48,6 +48,8 @@ JOB_LOG = NETVM_ROOT / "job-log.jsonl"
|
||||
FOLLOWUPS_FILE = NETVM_ROOT / "followups.json"
|
||||
JOB_SIDECHATS_FILE = NETVM_ROOT / "job-sidechats.json"
|
||||
WAKE_SIDECHATS_FILE = Path("/home/super/sidechat-wake/wake-sidechats.json")
|
||||
SWARM_FILE = NETVM_ROOT / "swarms.json"
|
||||
NUDGE_TRACKER_FILE = NETVM_ROOT / "conversation-nudge-tracker.json"
|
||||
JOBS_DIR = NETVM_ROOT / "jobs"
|
||||
DISPATCH_PY = BIN_DIR / "job-dispatch.py"
|
||||
|
||||
@@ -583,6 +585,8 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
||||
tool_calls = parse_tool_calls(text)
|
||||
if tool_calls and not dry_run:
|
||||
for op, args in tool_calls:
|
||||
if isinstance(args, dict) and "agent" not in args:
|
||||
args["agent"] = agent
|
||||
print(f"[{agent}] Executing tool '{op}' from message {mid[:8]} in thread {thread_name or thread_id[:8]}")
|
||||
t_ok, t_res = execute_agent_tool(agent, op, args)
|
||||
tool_record = {
|
||||
@@ -628,7 +632,19 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
||||
}
|
||||
if not dry_run:
|
||||
append_jsonl(JOB_LOG, job_record)
|
||||
trigger_chain_next(job_id, result_text, success=not is_fail)
|
||||
# Check if this is a swarm slot result: sw-YYYYMMDD-HHMMSS-xxxx/<slot>
|
||||
if "/" in job_id and job_id.startswith("sw-"):
|
||||
try:
|
||||
s_sid, s_slot = job_id.split("/", 1)
|
||||
s_proc = subprocess.Popen(
|
||||
[sys.executable, str(BIN_DIR / "box-ctl.py"), "swarm-report", s_sid, s_slot],
|
||||
stdin=subprocess.PIPE, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL
|
||||
)
|
||||
s_proc.communicate(input=json.dumps({"ok": not is_fail, "result": result_text}).encode("utf-8"))
|
||||
except Exception as se:
|
||||
sys.stderr.write(f"warning: failed to record swarm report: {se}\n")
|
||||
else:
|
||||
trigger_chain_next(job_id, result_text, success=not is_fail)
|
||||
if not is_fail:
|
||||
archive_ephemeral_thread(agent, thread_id, job_id=job_id)
|
||||
clear_matching_followups(followups, agent, thread_id, mid, text,
|
||||
@@ -640,6 +656,7 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
||||
dry_run, job_id=job_id, verb=verb)
|
||||
else:
|
||||
clear_matching_followups(followups, agent, thread_id, mid, text, dry_run)
|
||||
maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=dry_run)
|
||||
|
||||
return new_messages, new_wm, job_results
|
||||
|
||||
@@ -782,6 +799,212 @@ def archive_ephemeral_thread(agent, thread_id, job_id=None):
|
||||
sys.stderr.write(f"warning: archive_ephemeral_thread failed: {ae}\n")
|
||||
|
||||
|
||||
SWARM_WORKER_POOL = ["dev", "def", "muse"]
|
||||
|
||||
|
||||
def reconcile_and_dispatch_swarms(dry_run=False):
|
||||
"""
|
||||
Autonomous Swarm Orchestrator:
|
||||
1. Scans swarms.json for pending slots.
|
||||
2. Dynamically allocates available auxiliary worker nodes (dev, def, muse).
|
||||
3. Provisions ephemeral sidechat per slot and dispatches the task with [RESULT <swarm_id>/<slot>].
|
||||
4. Upon completion of all slots, sends completion summary DM to originating coordinator.
|
||||
"""
|
||||
if not SWARM_FILE.exists() or dry_run:
|
||||
return
|
||||
try:
|
||||
swarms = json.loads(SWARM_FILE.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
return
|
||||
|
||||
modified = False
|
||||
now = utcnow()
|
||||
|
||||
# Determine busy workers from running slots
|
||||
busy_workers = set()
|
||||
for sid, swarm in swarms.items():
|
||||
if swarm.get("status") in ("pending", "running"):
|
||||
for slot in swarm.get("slots", []):
|
||||
if slot.get("status") == "running" and slot.get("agent_id"):
|
||||
busy_workers.add(slot["agent_id"])
|
||||
|
||||
# Process each swarm
|
||||
for sid, swarm in swarms.items():
|
||||
st = swarm.get("status")
|
||||
creator = swarm.get("created_by") or "646"
|
||||
if creator not in ("646", "pip", "opm", "muse", "dev", "def"):
|
||||
creator = "646"
|
||||
|
||||
# 1. Allocate & dispatch pending slots
|
||||
if st in ("pending", "running"):
|
||||
slots = swarm.get("slots", [])
|
||||
for s in slots:
|
||||
if s.get("status") == "pending":
|
||||
# Find first available worker
|
||||
worker = None
|
||||
for w in SWARM_WORKER_POOL:
|
||||
if w not in busy_workers:
|
||||
worker = w
|
||||
break
|
||||
if not worker:
|
||||
# Fallback round-robin across worker pool if all are busy
|
||||
worker = SWARM_WORKER_POOL[s["slot"] % len(SWARM_WORKER_POOL)]
|
||||
|
||||
slot_idx = s["slot"]
|
||||
task_text = swarm.get("task", "execute subagent task")
|
||||
slot_target = f"{sid}-s{slot_idx}"
|
||||
|
||||
# Attach worker to slot
|
||||
s["agent_id"] = worker
|
||||
s["status"] = "running"
|
||||
s["updated_ts"] = now
|
||||
busy_workers.add(worker)
|
||||
modified = True
|
||||
|
||||
# Prepare task directive prompt
|
||||
prompt = (
|
||||
f"[JOB {sid}/{slot_idx}] Task for swarm slot {slot_idx}:\n"
|
||||
f"{task_text}\n\n"
|
||||
f"Reply with [RESULT {sid}/{slot_idx}] OK <summary> or FAIL <reason>."
|
||||
)
|
||||
|
||||
# Send to worker via dm.py (which handles sidechat creation & tracking)
|
||||
try:
|
||||
dm_cmd = [
|
||||
sys.executable, str(BIN_DIR / "dm.py"), "send",
|
||||
"--agent", creator,
|
||||
"--to", worker,
|
||||
"--target", slot_target,
|
||||
prompt,
|
||||
]
|
||||
subprocess.Popen(dm_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||
print(f"[swarm] Dispatched slot {slot_idx} of {sid} to {worker} in sidechat {slot_target}")
|
||||
except Exception as de:
|
||||
sys.stderr.write(f"warning: failed to dispatch swarm slot: {de}\n")
|
||||
|
||||
# Update swarm rollup status
|
||||
counts = {"pending": 0, "running": 0, "done": 0, "failed": 0, "killed": 0}
|
||||
for slot in slots:
|
||||
counts[slot.get("status", "pending")] = counts.get(slot.get("status", "pending"), 0) + 1
|
||||
if counts["pending"] + counts["running"] == 0:
|
||||
swarm["status"] = "completed" if counts["failed"] == 0 else "partial"
|
||||
swarm["updated_ts"] = now
|
||||
modified = True
|
||||
elif counts["running"] or counts["done"]:
|
||||
swarm["status"] = "running"
|
||||
swarm["updated_ts"] = now
|
||||
modified = True
|
||||
|
||||
# 2. Check if newly completed/partial and notify coordinator
|
||||
if swarm.get("status") in ("completed", "partial") and not swarm.get("notified_coordinator"):
|
||||
swarm["notified_coordinator"] = True
|
||||
modified = True
|
||||
done_cnt = sum(1 for sl in swarm.get("slots", []) if sl.get("status") == "done")
|
||||
total_cnt = len(swarm.get("slots", []))
|
||||
summary_msg = (
|
||||
f"[Swarm Report] Swarm {sid} ({swarm.get('label') or 'task'}) finished: "
|
||||
f"{done_cnt}/{total_cnt} slots successful. "
|
||||
f"Console: https://box.muse-dev.online/#dashboard"
|
||||
)
|
||||
try:
|
||||
coord_dm = [
|
||||
sys.executable, str(BIN_DIR / "dm.py"), "send",
|
||||
"--agent", "box",
|
||||
"--to", creator,
|
||||
"--target", "main",
|
||||
"--allow-main-chat",
|
||||
summary_msg,
|
||||
]
|
||||
subprocess.Popen(coord_dm, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||
print(f"[swarm] Delivered completion report for {sid} to coordinator {creator}")
|
||||
except Exception as ce:
|
||||
sys.stderr.write(f"warning: failed to notify swarm coordinator: {ce}\n")
|
||||
|
||||
if modified:
|
||||
tmp_swarms = str(SWARM_FILE) + f".tmp.{os.getpid()}"
|
||||
with open(tmp_swarms, "w", encoding="utf-8") as f:
|
||||
json.dump(swarms, f, indent=2)
|
||||
os.replace(tmp_swarms, SWARM_FILE)
|
||||
|
||||
|
||||
def check_and_archive_terminal_swarms():
|
||||
"""Check swarms.json and auto-archive ephemeral sidechats for completed or terminal swarms."""
|
||||
if not SWARM_FILE.exists() or not JOB_SIDECHATS_FILE.exists():
|
||||
return
|
||||
try:
|
||||
swarms_data = json.loads(SWARM_FILE.read_text(encoding="utf-8"))
|
||||
sc_data = json.loads(JOB_SIDECHATS_FILE.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
return
|
||||
|
||||
for sid, swarm in swarms_data.items():
|
||||
st = swarm.get("status")
|
||||
if st in ("completed", "partial", "killed"):
|
||||
# Check slots or matching sidechats
|
||||
for key, val in sc_data.items():
|
||||
if isinstance(val, dict) and not val.get("archived") and val.get("type") != "persistent":
|
||||
if sid in key or (swarm.get("label") and swarm.get("label") in key):
|
||||
tu = val.get("thread_uuid")
|
||||
ag = val.get("agent", "opm")
|
||||
if tu:
|
||||
archive_ephemeral_thread(ag, tu, job_id=sid)
|
||||
|
||||
|
||||
|
||||
def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=False):
|
||||
"""
|
||||
If an agent replies conversationally in a sidechat backed by a job or follow-up
|
||||
without providing [RESULT <id>] or tool directives, deliver a terse 1-turn nudge footer.
|
||||
"""
|
||||
if dry_run or not thread_id or thread_id == "main":
|
||||
return
|
||||
if not re.fullmatch(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", thread_id.lower()):
|
||||
return
|
||||
|
||||
# Look up active follow-ups or job association for this thread
|
||||
followups = load_json_file(FOLLOWUPS_FILE)
|
||||
matching_job_id = None
|
||||
for f_id, f_rec in followups.items():
|
||||
if f_rec.get("status") in ("pending", "acknowledged") and f_rec.get("thread_uuid") == thread_id:
|
||||
matching_job_id = f_rec.get("job_id") or f_id
|
||||
break
|
||||
|
||||
if not matching_job_id and JOB_SIDECHATS_FILE.exists():
|
||||
try:
|
||||
sc_state = json.loads(JOB_SIDECHATS_FILE.read_text(encoding="utf-8"))
|
||||
for k, v in sc_state.items():
|
||||
if isinstance(v, dict) and v.get("thread_uuid") == thread_id:
|
||||
if k.startswith(("box-", "job-", "pipe-")):
|
||||
matching_job_id = k
|
||||
break
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if not matching_job_id:
|
||||
return
|
||||
|
||||
tracker = load_json_file(NUDGE_TRACKER_FILE)
|
||||
rec = tracker.get(thread_id, {})
|
||||
if rec.get("nudged_for_mid") == mid or rec.get("nudge_count", 0) >= 1:
|
||||
return
|
||||
|
||||
# Inject terse nudge footer
|
||||
nudge_msg = f"[Nudge: To advance the workflow, reply with [RESULT {matching_job_id}] <outcome>]"
|
||||
try:
|
||||
import muse_hybrid
|
||||
print(f"[{agent}] Injecting 1-turn nudge into {thread_name or thread_id[:8]} for job {matching_job_id}")
|
||||
muse_hybrid.send_message(agent, nudge_msg, thread_id=thread_id, wait=0)
|
||||
tracker[thread_id] = {
|
||||
"ts": utcnow(),
|
||||
"job_id": matching_job_id,
|
||||
"nudged_for_mid": mid,
|
||||
"nudge_count": rec.get("nudge_count", 0) + 1,
|
||||
}
|
||||
save_json_file(NUDGE_TRACKER_FILE, tracker)
|
||||
except Exception as ne:
|
||||
sys.stderr.write(f"warning: failed to deliver conversational nudge: {ne}\n")
|
||||
|
||||
|
||||
|
||||
# Chain deduplication
|
||||
CHAINED_JOBS_FILE = Path(__file__).parent / "chained-jobs.json"
|
||||
@@ -1081,6 +1304,8 @@ def harvest_cycle(target_agent=None, dry_run=False, output_json=False):
|
||||
|
||||
if not dry_run:
|
||||
save_json_file(WATERMARKS_FILE, watermarks)
|
||||
reconcile_and_dispatch_swarms()
|
||||
check_and_archive_terminal_swarms()
|
||||
|
||||
# Output formatting
|
||||
if output_json:
|
||||
|
||||
Reference in New Issue
Block a user