feat(autonomy): autonomous swarm worker orchestration, nudge triggers, and recursive loop
This commit is contained in:
+1168
-1
File diff suppressed because it is too large
Load Diff
@@ -562,6 +562,34 @@ def _cron_runs_build(a):
|
|||||||
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'job-list']
|
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'job-list']
|
||||||
|
|
||||||
|
|
||||||
|
def _cron_timer_start_validate(raw):
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
raise OpError('args must be an object')
|
||||||
|
allowed = {'name'}
|
||||||
|
for k in raw:
|
||||||
|
if k not in allowed:
|
||||||
|
raise OpError(f'unknown arg: {k}')
|
||||||
|
return {'name': _job_name(raw.get('name'))}
|
||||||
|
|
||||||
|
|
||||||
|
def _cron_timer_start_build(a):
|
||||||
|
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'timer-start', a['name']]
|
||||||
|
|
||||||
|
|
||||||
|
def _cron_timer_create_validate(raw):
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
raise OpError('args must be an object')
|
||||||
|
allowed = {'name'}
|
||||||
|
for k in raw:
|
||||||
|
if k not in allowed:
|
||||||
|
raise OpError(f'unknown arg: {k}')
|
||||||
|
return {'name': _job_name(raw.get('name'))}
|
||||||
|
|
||||||
|
|
||||||
|
def _cron_timer_create_build(a):
|
||||||
|
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'timer-create', a['name']]
|
||||||
|
|
||||||
|
|
||||||
def _vars_list_validate(raw):
|
def _vars_list_validate(raw):
|
||||||
if raw not in ({}, None):
|
if raw not in ({}, None):
|
||||||
raise OpError('vars.list takes no required args')
|
raise OpError('vars.list takes no required args')
|
||||||
@@ -705,6 +733,55 @@ def _service_restart_build(a):
|
|||||||
return [sys.executable, os.path.join(BIN_DIR, 'box-sys-op.py'), 'service.restart']
|
return [sys.executable, os.path.join(BIN_DIR, 'box-sys-op.py'), 'service.restart']
|
||||||
|
|
||||||
|
|
||||||
|
def _swarm_spawn_validate(raw):
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
raise OpError('args must be an object')
|
||||||
|
allowed = {'count', 'task', 'label'}
|
||||||
|
for k in raw:
|
||||||
|
if k not in allowed:
|
||||||
|
raise OpError(f'unknown arg: {k}')
|
||||||
|
count = _opt_int(raw.get('count', 1), 1, 50, 'count') or 1
|
||||||
|
task = _clean_message(raw.get('task'))
|
||||||
|
label = raw.get('label')
|
||||||
|
if label and not TARGET_RE.fullmatch(str(label)):
|
||||||
|
raise OpError('label must match safe identifier')
|
||||||
|
return {'count': count, 'task': task, 'label': str(label) if label else None}
|
||||||
|
|
||||||
|
|
||||||
|
def _swarm_spawn_build(a):
|
||||||
|
cmd = [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'swarm-spawn', str(a['count']), a['task']]
|
||||||
|
if a.get('label'):
|
||||||
|
cmd.extend(['--label', a['label']])
|
||||||
|
return cmd
|
||||||
|
|
||||||
|
|
||||||
|
def _swarm_status_validate(raw):
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
raise OpError('args must be an object')
|
||||||
|
allowed = {'swarm_id', 'id'}
|
||||||
|
for k in raw:
|
||||||
|
if k not in allowed:
|
||||||
|
raise OpError(f'unknown arg: {k}')
|
||||||
|
sid = raw.get('swarm_id') or raw.get('id')
|
||||||
|
if not isinstance(sid, str) or not TARGET_RE.fullmatch(sid):
|
||||||
|
raise OpError('swarm_id must match safe identifier')
|
||||||
|
return {'swarm_id': sid}
|
||||||
|
|
||||||
|
|
||||||
|
def _swarm_status_build(a):
|
||||||
|
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'swarm-status', a['swarm_id']]
|
||||||
|
|
||||||
|
|
||||||
|
def _swarm_list_validate(raw):
|
||||||
|
if raw not in ({}, None):
|
||||||
|
raise OpError('swarm.list takes no required args')
|
||||||
|
return {}
|
||||||
|
|
||||||
|
|
||||||
|
def _swarm_list_build(a):
|
||||||
|
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'swarm-list']
|
||||||
|
|
||||||
|
|
||||||
# op -> {validate, build, timeout, side_effecting, description}
|
# op -> {validate, build, timeout, side_effecting, description}
|
||||||
OPS = {
|
OPS = {
|
||||||
'dm.send': {
|
'dm.send': {
|
||||||
@@ -782,6 +859,16 @@ OPS = {
|
|||||||
'timeout': 300, 'side_effecting': True,
|
'timeout': 300, 'side_effecting': True,
|
||||||
'desc': 'Trigger on-demand execution of a scheduled job',
|
'desc': 'Trigger on-demand execution of a scheduled job',
|
||||||
},
|
},
|
||||||
|
'cron.timer_create': {
|
||||||
|
'validate': _cron_timer_create_validate, 'build': _cron_timer_create_build,
|
||||||
|
'timeout': 30, 'side_effecting': True,
|
||||||
|
'desc': 'Create systemd user timer unit for a job',
|
||||||
|
},
|
||||||
|
'cron.timer_start': {
|
||||||
|
'validate': _cron_timer_start_validate, 'build': _cron_timer_start_build,
|
||||||
|
'timeout': 30, 'side_effecting': True,
|
||||||
|
'desc': 'Enable and start systemd user timer unit for a job',
|
||||||
|
},
|
||||||
'vars.list': {
|
'vars.list': {
|
||||||
'validate': _vars_list_validate, 'build': _vars_list_build,
|
'validate': _vars_list_validate, 'build': _vars_list_build,
|
||||||
'timeout': 30, 'side_effecting': False,
|
'timeout': 30, 'side_effecting': False,
|
||||||
@@ -822,6 +909,21 @@ OPS = {
|
|||||||
'timeout': 30, 'side_effecting': True,
|
'timeout': 30, 'side_effecting': True,
|
||||||
'desc': 'Restart allowlisted fleet systemd service',
|
'desc': 'Restart allowlisted fleet systemd service',
|
||||||
},
|
},
|
||||||
|
'swarm.spawn': {
|
||||||
|
'validate': _swarm_spawn_validate, 'build': _swarm_spawn_build,
|
||||||
|
'timeout': 30, 'side_effecting': True,
|
||||||
|
'desc': 'Spawn autonomous subagent swarm slots managed by Box',
|
||||||
|
},
|
||||||
|
'swarm.status': {
|
||||||
|
'validate': _swarm_status_validate, 'build': _swarm_status_build,
|
||||||
|
'timeout': 30, 'side_effecting': False,
|
||||||
|
'desc': 'Inspect subagent swarm status and progress',
|
||||||
|
},
|
||||||
|
'swarm.list': {
|
||||||
|
'validate': _swarm_list_validate, 'build': _swarm_list_build,
|
||||||
|
'timeout': 30, 'side_effecting': False,
|
||||||
|
'desc': 'List all subagent swarms and their counts',
|
||||||
|
},
|
||||||
'exec.ping': {
|
'exec.ping': {
|
||||||
'validate': _health_validate,
|
'validate': _health_validate,
|
||||||
'build': lambda a: ['/bin/echo', 'PONG'],
|
'build': lambda a: ['/bin/echo', 'PONG'],
|
||||||
|
|||||||
+226
-1
@@ -48,6 +48,8 @@ JOB_LOG = NETVM_ROOT / "job-log.jsonl"
|
|||||||
FOLLOWUPS_FILE = NETVM_ROOT / "followups.json"
|
FOLLOWUPS_FILE = NETVM_ROOT / "followups.json"
|
||||||
JOB_SIDECHATS_FILE = NETVM_ROOT / "job-sidechats.json"
|
JOB_SIDECHATS_FILE = NETVM_ROOT / "job-sidechats.json"
|
||||||
WAKE_SIDECHATS_FILE = Path("/home/super/sidechat-wake/wake-sidechats.json")
|
WAKE_SIDECHATS_FILE = Path("/home/super/sidechat-wake/wake-sidechats.json")
|
||||||
|
SWARM_FILE = NETVM_ROOT / "swarms.json"
|
||||||
|
NUDGE_TRACKER_FILE = NETVM_ROOT / "conversation-nudge-tracker.json"
|
||||||
JOBS_DIR = NETVM_ROOT / "jobs"
|
JOBS_DIR = NETVM_ROOT / "jobs"
|
||||||
DISPATCH_PY = BIN_DIR / "job-dispatch.py"
|
DISPATCH_PY = BIN_DIR / "job-dispatch.py"
|
||||||
|
|
||||||
@@ -583,6 +585,8 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
|||||||
tool_calls = parse_tool_calls(text)
|
tool_calls = parse_tool_calls(text)
|
||||||
if tool_calls and not dry_run:
|
if tool_calls and not dry_run:
|
||||||
for op, args in tool_calls:
|
for op, args in tool_calls:
|
||||||
|
if isinstance(args, dict) and "agent" not in args:
|
||||||
|
args["agent"] = agent
|
||||||
print(f"[{agent}] Executing tool '{op}' from message {mid[:8]} in thread {thread_name or thread_id[:8]}")
|
print(f"[{agent}] Executing tool '{op}' from message {mid[:8]} in thread {thread_name or thread_id[:8]}")
|
||||||
t_ok, t_res = execute_agent_tool(agent, op, args)
|
t_ok, t_res = execute_agent_tool(agent, op, args)
|
||||||
tool_record = {
|
tool_record = {
|
||||||
@@ -628,7 +632,19 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
|||||||
}
|
}
|
||||||
if not dry_run:
|
if not dry_run:
|
||||||
append_jsonl(JOB_LOG, job_record)
|
append_jsonl(JOB_LOG, job_record)
|
||||||
trigger_chain_next(job_id, result_text, success=not is_fail)
|
# Check if this is a swarm slot result: sw-YYYYMMDD-HHMMSS-xxxx/<slot>
|
||||||
|
if "/" in job_id and job_id.startswith("sw-"):
|
||||||
|
try:
|
||||||
|
s_sid, s_slot = job_id.split("/", 1)
|
||||||
|
s_proc = subprocess.Popen(
|
||||||
|
[sys.executable, str(BIN_DIR / "box-ctl.py"), "swarm-report", s_sid, s_slot],
|
||||||
|
stdin=subprocess.PIPE, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL
|
||||||
|
)
|
||||||
|
s_proc.communicate(input=json.dumps({"ok": not is_fail, "result": result_text}).encode("utf-8"))
|
||||||
|
except Exception as se:
|
||||||
|
sys.stderr.write(f"warning: failed to record swarm report: {se}\n")
|
||||||
|
else:
|
||||||
|
trigger_chain_next(job_id, result_text, success=not is_fail)
|
||||||
if not is_fail:
|
if not is_fail:
|
||||||
archive_ephemeral_thread(agent, thread_id, job_id=job_id)
|
archive_ephemeral_thread(agent, thread_id, job_id=job_id)
|
||||||
clear_matching_followups(followups, agent, thread_id, mid, text,
|
clear_matching_followups(followups, agent, thread_id, mid, text,
|
||||||
@@ -640,6 +656,7 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
|||||||
dry_run, job_id=job_id, verb=verb)
|
dry_run, job_id=job_id, verb=verb)
|
||||||
else:
|
else:
|
||||||
clear_matching_followups(followups, agent, thread_id, mid, text, dry_run)
|
clear_matching_followups(followups, agent, thread_id, mid, text, dry_run)
|
||||||
|
maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=dry_run)
|
||||||
|
|
||||||
return new_messages, new_wm, job_results
|
return new_messages, new_wm, job_results
|
||||||
|
|
||||||
@@ -782,6 +799,212 @@ def archive_ephemeral_thread(agent, thread_id, job_id=None):
|
|||||||
sys.stderr.write(f"warning: archive_ephemeral_thread failed: {ae}\n")
|
sys.stderr.write(f"warning: archive_ephemeral_thread failed: {ae}\n")
|
||||||
|
|
||||||
|
|
||||||
|
SWARM_WORKER_POOL = ["dev", "def", "muse"]
|
||||||
|
|
||||||
|
|
||||||
|
def reconcile_and_dispatch_swarms(dry_run=False):
|
||||||
|
"""
|
||||||
|
Autonomous Swarm Orchestrator:
|
||||||
|
1. Scans swarms.json for pending slots.
|
||||||
|
2. Dynamically allocates available auxiliary worker nodes (dev, def, muse).
|
||||||
|
3. Provisions ephemeral sidechat per slot and dispatches the task with [RESULT <swarm_id>/<slot>].
|
||||||
|
4. Upon completion of all slots, sends completion summary DM to originating coordinator.
|
||||||
|
"""
|
||||||
|
if not SWARM_FILE.exists() or dry_run:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
swarms = json.loads(SWARM_FILE.read_text(encoding="utf-8"))
|
||||||
|
except Exception:
|
||||||
|
return
|
||||||
|
|
||||||
|
modified = False
|
||||||
|
now = utcnow()
|
||||||
|
|
||||||
|
# Determine busy workers from running slots
|
||||||
|
busy_workers = set()
|
||||||
|
for sid, swarm in swarms.items():
|
||||||
|
if swarm.get("status") in ("pending", "running"):
|
||||||
|
for slot in swarm.get("slots", []):
|
||||||
|
if slot.get("status") == "running" and slot.get("agent_id"):
|
||||||
|
busy_workers.add(slot["agent_id"])
|
||||||
|
|
||||||
|
# Process each swarm
|
||||||
|
for sid, swarm in swarms.items():
|
||||||
|
st = swarm.get("status")
|
||||||
|
creator = swarm.get("created_by") or "646"
|
||||||
|
if creator not in ("646", "pip", "opm", "muse", "dev", "def"):
|
||||||
|
creator = "646"
|
||||||
|
|
||||||
|
# 1. Allocate & dispatch pending slots
|
||||||
|
if st in ("pending", "running"):
|
||||||
|
slots = swarm.get("slots", [])
|
||||||
|
for s in slots:
|
||||||
|
if s.get("status") == "pending":
|
||||||
|
# Find first available worker
|
||||||
|
worker = None
|
||||||
|
for w in SWARM_WORKER_POOL:
|
||||||
|
if w not in busy_workers:
|
||||||
|
worker = w
|
||||||
|
break
|
||||||
|
if not worker:
|
||||||
|
# Fallback round-robin across worker pool if all are busy
|
||||||
|
worker = SWARM_WORKER_POOL[s["slot"] % len(SWARM_WORKER_POOL)]
|
||||||
|
|
||||||
|
slot_idx = s["slot"]
|
||||||
|
task_text = swarm.get("task", "execute subagent task")
|
||||||
|
slot_target = f"{sid}-s{slot_idx}"
|
||||||
|
|
||||||
|
# Attach worker to slot
|
||||||
|
s["agent_id"] = worker
|
||||||
|
s["status"] = "running"
|
||||||
|
s["updated_ts"] = now
|
||||||
|
busy_workers.add(worker)
|
||||||
|
modified = True
|
||||||
|
|
||||||
|
# Prepare task directive prompt
|
||||||
|
prompt = (
|
||||||
|
f"[JOB {sid}/{slot_idx}] Task for swarm slot {slot_idx}:\n"
|
||||||
|
f"{task_text}\n\n"
|
||||||
|
f"Reply with [RESULT {sid}/{slot_idx}] OK <summary> or FAIL <reason>."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Send to worker via dm.py (which handles sidechat creation & tracking)
|
||||||
|
try:
|
||||||
|
dm_cmd = [
|
||||||
|
sys.executable, str(BIN_DIR / "dm.py"), "send",
|
||||||
|
"--agent", creator,
|
||||||
|
"--to", worker,
|
||||||
|
"--target", slot_target,
|
||||||
|
prompt,
|
||||||
|
]
|
||||||
|
subprocess.Popen(dm_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||||
|
print(f"[swarm] Dispatched slot {slot_idx} of {sid} to {worker} in sidechat {slot_target}")
|
||||||
|
except Exception as de:
|
||||||
|
sys.stderr.write(f"warning: failed to dispatch swarm slot: {de}\n")
|
||||||
|
|
||||||
|
# Update swarm rollup status
|
||||||
|
counts = {"pending": 0, "running": 0, "done": 0, "failed": 0, "killed": 0}
|
||||||
|
for slot in slots:
|
||||||
|
counts[slot.get("status", "pending")] = counts.get(slot.get("status", "pending"), 0) + 1
|
||||||
|
if counts["pending"] + counts["running"] == 0:
|
||||||
|
swarm["status"] = "completed" if counts["failed"] == 0 else "partial"
|
||||||
|
swarm["updated_ts"] = now
|
||||||
|
modified = True
|
||||||
|
elif counts["running"] or counts["done"]:
|
||||||
|
swarm["status"] = "running"
|
||||||
|
swarm["updated_ts"] = now
|
||||||
|
modified = True
|
||||||
|
|
||||||
|
# 2. Check if newly completed/partial and notify coordinator
|
||||||
|
if swarm.get("status") in ("completed", "partial") and not swarm.get("notified_coordinator"):
|
||||||
|
swarm["notified_coordinator"] = True
|
||||||
|
modified = True
|
||||||
|
done_cnt = sum(1 for sl in swarm.get("slots", []) if sl.get("status") == "done")
|
||||||
|
total_cnt = len(swarm.get("slots", []))
|
||||||
|
summary_msg = (
|
||||||
|
f"[Swarm Report] Swarm {sid} ({swarm.get('label') or 'task'}) finished: "
|
||||||
|
f"{done_cnt}/{total_cnt} slots successful. "
|
||||||
|
f"Console: https://box.muse-dev.online/#dashboard"
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
coord_dm = [
|
||||||
|
sys.executable, str(BIN_DIR / "dm.py"), "send",
|
||||||
|
"--agent", "box",
|
||||||
|
"--to", creator,
|
||||||
|
"--target", "main",
|
||||||
|
"--allow-main-chat",
|
||||||
|
summary_msg,
|
||||||
|
]
|
||||||
|
subprocess.Popen(coord_dm, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||||
|
print(f"[swarm] Delivered completion report for {sid} to coordinator {creator}")
|
||||||
|
except Exception as ce:
|
||||||
|
sys.stderr.write(f"warning: failed to notify swarm coordinator: {ce}\n")
|
||||||
|
|
||||||
|
if modified:
|
||||||
|
tmp_swarms = str(SWARM_FILE) + f".tmp.{os.getpid()}"
|
||||||
|
with open(tmp_swarms, "w", encoding="utf-8") as f:
|
||||||
|
json.dump(swarms, f, indent=2)
|
||||||
|
os.replace(tmp_swarms, SWARM_FILE)
|
||||||
|
|
||||||
|
|
||||||
|
def check_and_archive_terminal_swarms():
|
||||||
|
"""Check swarms.json and auto-archive ephemeral sidechats for completed or terminal swarms."""
|
||||||
|
if not SWARM_FILE.exists() or not JOB_SIDECHATS_FILE.exists():
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
swarms_data = json.loads(SWARM_FILE.read_text(encoding="utf-8"))
|
||||||
|
sc_data = json.loads(JOB_SIDECHATS_FILE.read_text(encoding="utf-8"))
|
||||||
|
except Exception:
|
||||||
|
return
|
||||||
|
|
||||||
|
for sid, swarm in swarms_data.items():
|
||||||
|
st = swarm.get("status")
|
||||||
|
if st in ("completed", "partial", "killed"):
|
||||||
|
# Check slots or matching sidechats
|
||||||
|
for key, val in sc_data.items():
|
||||||
|
if isinstance(val, dict) and not val.get("archived") and val.get("type") != "persistent":
|
||||||
|
if sid in key or (swarm.get("label") and swarm.get("label") in key):
|
||||||
|
tu = val.get("thread_uuid")
|
||||||
|
ag = val.get("agent", "opm")
|
||||||
|
if tu:
|
||||||
|
archive_ephemeral_thread(ag, tu, job_id=sid)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=False):
|
||||||
|
"""
|
||||||
|
If an agent replies conversationally in a sidechat backed by a job or follow-up
|
||||||
|
without providing [RESULT <id>] or tool directives, deliver a terse 1-turn nudge footer.
|
||||||
|
"""
|
||||||
|
if dry_run or not thread_id or thread_id == "main":
|
||||||
|
return
|
||||||
|
if not re.fullmatch(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", thread_id.lower()):
|
||||||
|
return
|
||||||
|
|
||||||
|
# Look up active follow-ups or job association for this thread
|
||||||
|
followups = load_json_file(FOLLOWUPS_FILE)
|
||||||
|
matching_job_id = None
|
||||||
|
for f_id, f_rec in followups.items():
|
||||||
|
if f_rec.get("status") in ("pending", "acknowledged") and f_rec.get("thread_uuid") == thread_id:
|
||||||
|
matching_job_id = f_rec.get("job_id") or f_id
|
||||||
|
break
|
||||||
|
|
||||||
|
if not matching_job_id and JOB_SIDECHATS_FILE.exists():
|
||||||
|
try:
|
||||||
|
sc_state = json.loads(JOB_SIDECHATS_FILE.read_text(encoding="utf-8"))
|
||||||
|
for k, v in sc_state.items():
|
||||||
|
if isinstance(v, dict) and v.get("thread_uuid") == thread_id:
|
||||||
|
if k.startswith(("box-", "job-", "pipe-")):
|
||||||
|
matching_job_id = k
|
||||||
|
break
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
if not matching_job_id:
|
||||||
|
return
|
||||||
|
|
||||||
|
tracker = load_json_file(NUDGE_TRACKER_FILE)
|
||||||
|
rec = tracker.get(thread_id, {})
|
||||||
|
if rec.get("nudged_for_mid") == mid or rec.get("nudge_count", 0) >= 1:
|
||||||
|
return
|
||||||
|
|
||||||
|
# Inject terse nudge footer
|
||||||
|
nudge_msg = f"[Nudge: To advance the workflow, reply with [RESULT {matching_job_id}] <outcome>]"
|
||||||
|
try:
|
||||||
|
import muse_hybrid
|
||||||
|
print(f"[{agent}] Injecting 1-turn nudge into {thread_name or thread_id[:8]} for job {matching_job_id}")
|
||||||
|
muse_hybrid.send_message(agent, nudge_msg, thread_id=thread_id, wait=0)
|
||||||
|
tracker[thread_id] = {
|
||||||
|
"ts": utcnow(),
|
||||||
|
"job_id": matching_job_id,
|
||||||
|
"nudged_for_mid": mid,
|
||||||
|
"nudge_count": rec.get("nudge_count", 0) + 1,
|
||||||
|
}
|
||||||
|
save_json_file(NUDGE_TRACKER_FILE, tracker)
|
||||||
|
except Exception as ne:
|
||||||
|
sys.stderr.write(f"warning: failed to deliver conversational nudge: {ne}\n")
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
# Chain deduplication
|
# Chain deduplication
|
||||||
CHAINED_JOBS_FILE = Path(__file__).parent / "chained-jobs.json"
|
CHAINED_JOBS_FILE = Path(__file__).parent / "chained-jobs.json"
|
||||||
@@ -1081,6 +1304,8 @@ def harvest_cycle(target_agent=None, dry_run=False, output_json=False):
|
|||||||
|
|
||||||
if not dry_run:
|
if not dry_run:
|
||||||
save_json_file(WATERMARKS_FILE, watermarks)
|
save_json_file(WATERMARKS_FILE, watermarks)
|
||||||
|
reconcile_and_dispatch_swarms()
|
||||||
|
check_and_archive_terminal_swarms()
|
||||||
|
|
||||||
# Output formatting
|
# Output formatting
|
||||||
if output_json:
|
if output_json:
|
||||||
|
|||||||
+140
-2
@@ -276,9 +276,14 @@ def make_digest_id(agent):
|
|||||||
|
|
||||||
# In-band response-contract footer for ACTIONABLE digests. The verbs are matched
|
# In-band response-contract footer for ACTIONABLE digests. The verbs are matched
|
||||||
# by response-harvester.py to resolve followups: ACK/CLAIM acknowledge (nudge
|
# by response-harvester.py to resolve followups: ACK/CLAIM acknowledge (nudge
|
||||||
# suppression), RESULT/DECLINE/NO-ACTION close. ~94 chars, well under budget.
|
# suppression), RESULT/DECLINE/NO-ACTION close. The closing line enforces the
|
||||||
|
# recursive box->agent->box discipline: post the RESULT back in this thread
|
||||||
|
# (box records it and dispatches the next chained step); never DM the next
|
||||||
|
# agent directly. ~166 chars, within the 600-char digest budget.
|
||||||
CONTRACT_FOOTER = ("Reply: [ACK id] seen | [CLAIM id] mine | "
|
CONTRACT_FOOTER = ("Reply: [ACK id] seen | [CLAIM id] mine | "
|
||||||
"[RESULT id] done | [DECLINE id] | [NO-ACTION id]")
|
"[RESULT id] done | [DECLINE id] | [NO-ACTION id]. "
|
||||||
|
"Report back here. Box dispatches the next step; "
|
||||||
|
"do not DM the next agent directly.")
|
||||||
|
|
||||||
|
|
||||||
def send_prompt(sender, agent, sidechat, digest):
|
def send_prompt(sender, agent, sidechat, digest):
|
||||||
@@ -552,6 +557,63 @@ def check_subagents(agent, cfg, threads_meta):
|
|||||||
return alerts
|
return alerts
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Brain workspace — the main loop operates its thinking in the "main-loop
|
||||||
|
# brain" sidechat (opm's account). See bin/brain.py.
|
||||||
|
# User directive 2026-10-04: "we need main loop to operate its brains in
|
||||||
|
# side chat". Safety: brain posts carry [BRAIN], never [JOB]; never
|
||||||
|
# --expect-reply; own messages skipped on read; !loop only from authorized
|
||||||
|
# senders. All enforced in brain.py.
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
_BRAIN_LAST_IDS = {} # (agent, sidechat_name) -> last message_id seen
|
||||||
|
|
||||||
|
|
||||||
|
def read_brain_messages(agent, sidechat_name, since_ts):
|
||||||
|
"""read_fn adapter for brain.BrainWorkspace.
|
||||||
|
|
||||||
|
Resolves sidechat_name -> UUID via dm (never hardcoded; UUIDs rotate),
|
||||||
|
reads via the same muse_hybrid primitive the loop uses, returns
|
||||||
|
[{"sender", "text", "ts"}]. Tracks last message_id per (agent, name);
|
||||||
|
first run anchors at newest with no backfill (same policy as
|
||||||
|
new_messages for main chat).
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
import dm
|
||||||
|
uuid = dm.resolve_sidechat_target(sidechat_name, agent)
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
if not uuid:
|
||||||
|
return []
|
||||||
|
try:
|
||||||
|
import muse_hybrid
|
||||||
|
msgs, err = muse_hybrid.get_history(agent, thread_id=uuid, limit=10)
|
||||||
|
except Exception:
|
||||||
|
return []
|
||||||
|
if err or not msgs:
|
||||||
|
return []
|
||||||
|
key = (agent, sidechat_name)
|
||||||
|
last_id = _BRAIN_LAST_IDS.get(key)
|
||||||
|
ids = [m.get("message_id") or "msg-%s" % m.get("seq") for m in msgs]
|
||||||
|
if last_id is None:
|
||||||
|
# First run: anchor at newest, no backfill (old !loop commands
|
||||||
|
# must not fire on deploy).
|
||||||
|
_BRAIN_LAST_IDS[key] = ids[-1] if ids else None
|
||||||
|
return []
|
||||||
|
if last_id in ids:
|
||||||
|
new_msgs = msgs[ids.index(last_id) + 1:]
|
||||||
|
else:
|
||||||
|
# Watermark fell out of the read window: treat all as new.
|
||||||
|
# Safe: brain intake skips own messages and only authorized
|
||||||
|
# !loop senders can act.
|
||||||
|
new_msgs = msgs
|
||||||
|
_BRAIN_LAST_IDS[key] = ids[-1] if ids else last_id
|
||||||
|
now = time.time()
|
||||||
|
return [{"sender": m.get("role", "unknown"),
|
||||||
|
"text": m.get("text", ""),
|
||||||
|
"ts": now} for m in new_msgs]
|
||||||
|
|
||||||
|
|
||||||
def do_check(only_agent=None):
|
def do_check(only_agent=None):
|
||||||
st = load_state()
|
st = load_state()
|
||||||
cfg = get_config(st)
|
cfg = get_config(st)
|
||||||
@@ -564,6 +626,23 @@ def do_check(only_agent=None):
|
|||||||
results = {}
|
results = {}
|
||||||
prompted = 0
|
prompted = 0
|
||||||
errors = 0
|
errors = 0
|
||||||
|
quiet_held = 0
|
||||||
|
|
||||||
|
# Brain workspace: the loop thinks in the sidechat, takes !loop
|
||||||
|
# instructions there. Non-fatal: a brain failure must never break
|
||||||
|
# the tick.
|
||||||
|
brain = None
|
||||||
|
try:
|
||||||
|
from brain import BrainWorkspace
|
||||||
|
brain = BrainWorkspace(STATE_FILE, read_fn=read_brain_messages)
|
||||||
|
intake = brain.intake()
|
||||||
|
if intake.get("commands"):
|
||||||
|
log("brain: %d commands, %d acks, %d ignored-senders" % (
|
||||||
|
intake["commands"], intake.get("acks", 0),
|
||||||
|
intake.get("ignored_senders", 0)))
|
||||||
|
except Exception as e:
|
||||||
|
log("brain init/intake failed (non-fatal): %r" % e)
|
||||||
|
brain = None
|
||||||
|
|
||||||
import muse_hybrid
|
import muse_hybrid
|
||||||
|
|
||||||
@@ -581,6 +660,10 @@ def do_check(only_agent=None):
|
|||||||
if not enabled.get(agent, True):
|
if not enabled.get(agent, True):
|
||||||
results[agent] = {"ok": True, "new": 0, "disabled": True}
|
results[agent] = {"ok": True, "new": 0, "disabled": True}
|
||||||
continue
|
continue
|
||||||
|
if brain is not None and brain.ignored(agent):
|
||||||
|
log("%s: skipped (brain !loop ignore active)" % agent)
|
||||||
|
results[agent] = {"ok": True, "new": 0, "ignored": True}
|
||||||
|
continue
|
||||||
wm = agents_state.get(agent) or {}
|
wm = agents_state.get(agent) or {}
|
||||||
|
|
||||||
# 1. Main chat check
|
# 1. Main chat check
|
||||||
@@ -640,6 +723,18 @@ def do_check(only_agent=None):
|
|||||||
errors += 1
|
errors += 1
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
# Brain quiet mode: hold actionable escalations. Reads continue,
|
||||||
|
# digests are composed, but nothing escalates to the agent — the
|
||||||
|
# thinking note in the brain carries the activity instead.
|
||||||
|
if brain is not None and brain.quiet():
|
||||||
|
is_act, _urg = classify_digest(digest)
|
||||||
|
if is_act:
|
||||||
|
log("%s: quiet mode - digest held (not escalated)" % agent)
|
||||||
|
results[agent] = {"ok": True, "new": total_new,
|
||||||
|
"quiet_held": True}
|
||||||
|
quiet_held += 1
|
||||||
|
continue
|
||||||
|
|
||||||
sent, detail, digest_id, actionable = send_prompt(cfg["sender"], agent, sidechat, digest)
|
sent, detail, digest_id, actionable = send_prompt(cfg["sender"], agent, sidechat, digest)
|
||||||
if sent:
|
if sent:
|
||||||
prompted += 1
|
prompted += 1
|
||||||
@@ -653,6 +748,49 @@ def do_check(only_agent=None):
|
|||||||
results[agent] = {"ok": False, "error": detail, "new": total_new}
|
results[agent] = {"ok": False, "error": detail, "new": total_new}
|
||||||
errors += 1
|
errors += 1
|
||||||
|
|
||||||
|
# Brain: post the tick's thinking to the sidechat workspace, then
|
||||||
|
# persist brain state. Non-fatal on failure.
|
||||||
|
if brain is not None:
|
||||||
|
try:
|
||||||
|
seen = {}
|
||||||
|
escalated = []
|
||||||
|
for a in agents:
|
||||||
|
r = results.get(a, {})
|
||||||
|
seen[a] = (r.get("new", 0), 0, 0)
|
||||||
|
if r.get("prompted"):
|
||||||
|
wm_a = agents_state.get(a) or {}
|
||||||
|
did = wm_a.get("last_digest_id")
|
||||||
|
if did and wm_a.get("last_digest_actionable"):
|
||||||
|
escalated.append(did)
|
||||||
|
closure_rate = None
|
||||||
|
health = None
|
||||||
|
try:
|
||||||
|
health = digest_health()
|
||||||
|
closure_rate = (health or {}).get("closure_rate")
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
wm_epochs = {}
|
||||||
|
for a in agents:
|
||||||
|
ts_s = (agents_state.get(a) or {}).get("last_ts")
|
||||||
|
try:
|
||||||
|
if ts_s:
|
||||||
|
dt = datetime.fromisoformat(
|
||||||
|
ts_s.replace("Z", "+00:00"))
|
||||||
|
wm_epochs[a] = dt.timestamp()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
brain.post_thinking(
|
||||||
|
{"seen": seen, "escalated": escalated,
|
||||||
|
"skipped_info": quiet_held, "errors": errors,
|
||||||
|
"closure_rate": closure_rate},
|
||||||
|
cfg={a: bool(enabled.get(a, True)) for a in agents},
|
||||||
|
watermarks=wm_epochs,
|
||||||
|
health=health,
|
||||||
|
)
|
||||||
|
brain.save()
|
||||||
|
except Exception as e:
|
||||||
|
log("brain post_thinking failed (non-fatal): %r" % e)
|
||||||
|
|
||||||
# Save under the state lock with a fresh reload: an enable/disable may
|
# Save under the state lock with a fresh reload: an enable/disable may
|
||||||
# have landed during the slow chat reads; preserve its config changes
|
# have landed during the slow chat reads; preserve its config changes
|
||||||
# and only update the keys this run owns (watermarks, last_run/result).
|
# and only update the keys this run owns (watermarks, last_run/result).
|
||||||
|
|||||||
Reference in New Issue
Block a user