feat(autonomy): autonomous swarm worker orchestration, nudge triggers, and recursive loop

This commit is contained in:
operator
2026-10-05 04:32:06 +00:00
parent b0454289ae
commit b88209d7a1
4 changed files with 1636 additions and 4 deletions
+1168 -1
View File
File diff suppressed because it is too large Load Diff
+102
View File
@@ -562,6 +562,34 @@ def _cron_runs_build(a):
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'job-list'] return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'job-list']
def _cron_timer_start_validate(raw):
if not isinstance(raw, dict):
raise OpError('args must be an object')
allowed = {'name'}
for k in raw:
if k not in allowed:
raise OpError(f'unknown arg: {k}')
return {'name': _job_name(raw.get('name'))}
def _cron_timer_start_build(a):
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'timer-start', a['name']]
def _cron_timer_create_validate(raw):
if not isinstance(raw, dict):
raise OpError('args must be an object')
allowed = {'name'}
for k in raw:
if k not in allowed:
raise OpError(f'unknown arg: {k}')
return {'name': _job_name(raw.get('name'))}
def _cron_timer_create_build(a):
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'timer-create', a['name']]
def _vars_list_validate(raw): def _vars_list_validate(raw):
if raw not in ({}, None): if raw not in ({}, None):
raise OpError('vars.list takes no required args') raise OpError('vars.list takes no required args')
@@ -705,6 +733,55 @@ def _service_restart_build(a):
return [sys.executable, os.path.join(BIN_DIR, 'box-sys-op.py'), 'service.restart'] return [sys.executable, os.path.join(BIN_DIR, 'box-sys-op.py'), 'service.restart']
def _swarm_spawn_validate(raw):
if not isinstance(raw, dict):
raise OpError('args must be an object')
allowed = {'count', 'task', 'label'}
for k in raw:
if k not in allowed:
raise OpError(f'unknown arg: {k}')
count = _opt_int(raw.get('count', 1), 1, 50, 'count') or 1
task = _clean_message(raw.get('task'))
label = raw.get('label')
if label and not TARGET_RE.fullmatch(str(label)):
raise OpError('label must match safe identifier')
return {'count': count, 'task': task, 'label': str(label) if label else None}
def _swarm_spawn_build(a):
cmd = [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'swarm-spawn', str(a['count']), a['task']]
if a.get('label'):
cmd.extend(['--label', a['label']])
return cmd
def _swarm_status_validate(raw):
if not isinstance(raw, dict):
raise OpError('args must be an object')
allowed = {'swarm_id', 'id'}
for k in raw:
if k not in allowed:
raise OpError(f'unknown arg: {k}')
sid = raw.get('swarm_id') or raw.get('id')
if not isinstance(sid, str) or not TARGET_RE.fullmatch(sid):
raise OpError('swarm_id must match safe identifier')
return {'swarm_id': sid}
def _swarm_status_build(a):
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'swarm-status', a['swarm_id']]
def _swarm_list_validate(raw):
if raw not in ({}, None):
raise OpError('swarm.list takes no required args')
return {}
def _swarm_list_build(a):
return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'swarm-list']
# op -> {validate, build, timeout, side_effecting, description} # op -> {validate, build, timeout, side_effecting, description}
OPS = { OPS = {
'dm.send': { 'dm.send': {
@@ -782,6 +859,16 @@ OPS = {
'timeout': 300, 'side_effecting': True, 'timeout': 300, 'side_effecting': True,
'desc': 'Trigger on-demand execution of a scheduled job', 'desc': 'Trigger on-demand execution of a scheduled job',
}, },
'cron.timer_create': {
'validate': _cron_timer_create_validate, 'build': _cron_timer_create_build,
'timeout': 30, 'side_effecting': True,
'desc': 'Create systemd user timer unit for a job',
},
'cron.timer_start': {
'validate': _cron_timer_start_validate, 'build': _cron_timer_start_build,
'timeout': 30, 'side_effecting': True,
'desc': 'Enable and start systemd user timer unit for a job',
},
'vars.list': { 'vars.list': {
'validate': _vars_list_validate, 'build': _vars_list_build, 'validate': _vars_list_validate, 'build': _vars_list_build,
'timeout': 30, 'side_effecting': False, 'timeout': 30, 'side_effecting': False,
@@ -822,6 +909,21 @@ OPS = {
'timeout': 30, 'side_effecting': True, 'timeout': 30, 'side_effecting': True,
'desc': 'Restart allowlisted fleet systemd service', 'desc': 'Restart allowlisted fleet systemd service',
}, },
'swarm.spawn': {
'validate': _swarm_spawn_validate, 'build': _swarm_spawn_build,
'timeout': 30, 'side_effecting': True,
'desc': 'Spawn autonomous subagent swarm slots managed by Box',
},
'swarm.status': {
'validate': _swarm_status_validate, 'build': _swarm_status_build,
'timeout': 30, 'side_effecting': False,
'desc': 'Inspect subagent swarm status and progress',
},
'swarm.list': {
'validate': _swarm_list_validate, 'build': _swarm_list_build,
'timeout': 30, 'side_effecting': False,
'desc': 'List all subagent swarms and their counts',
},
'exec.ping': { 'exec.ping': {
'validate': _health_validate, 'validate': _health_validate,
'build': lambda a: ['/bin/echo', 'PONG'], 'build': lambda a: ['/bin/echo', 'PONG'],
+225
View File
@@ -48,6 +48,8 @@ JOB_LOG = NETVM_ROOT / "job-log.jsonl"
FOLLOWUPS_FILE = NETVM_ROOT / "followups.json" FOLLOWUPS_FILE = NETVM_ROOT / "followups.json"
JOB_SIDECHATS_FILE = NETVM_ROOT / "job-sidechats.json" JOB_SIDECHATS_FILE = NETVM_ROOT / "job-sidechats.json"
WAKE_SIDECHATS_FILE = Path("/home/super/sidechat-wake/wake-sidechats.json") WAKE_SIDECHATS_FILE = Path("/home/super/sidechat-wake/wake-sidechats.json")
SWARM_FILE = NETVM_ROOT / "swarms.json"
NUDGE_TRACKER_FILE = NETVM_ROOT / "conversation-nudge-tracker.json"
JOBS_DIR = NETVM_ROOT / "jobs" JOBS_DIR = NETVM_ROOT / "jobs"
DISPATCH_PY = BIN_DIR / "job-dispatch.py" DISPATCH_PY = BIN_DIR / "job-dispatch.py"
@@ -583,6 +585,8 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
tool_calls = parse_tool_calls(text) tool_calls = parse_tool_calls(text)
if tool_calls and not dry_run: if tool_calls and not dry_run:
for op, args in tool_calls: for op, args in tool_calls:
if isinstance(args, dict) and "agent" not in args:
args["agent"] = agent
print(f"[{agent}] Executing tool '{op}' from message {mid[:8]} in thread {thread_name or thread_id[:8]}") print(f"[{agent}] Executing tool '{op}' from message {mid[:8]} in thread {thread_name or thread_id[:8]}")
t_ok, t_res = execute_agent_tool(agent, op, args) t_ok, t_res = execute_agent_tool(agent, op, args)
tool_record = { tool_record = {
@@ -628,6 +632,18 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
} }
if not dry_run: if not dry_run:
append_jsonl(JOB_LOG, job_record) append_jsonl(JOB_LOG, job_record)
# Check if this is a swarm slot result: sw-YYYYMMDD-HHMMSS-xxxx/<slot>
if "/" in job_id and job_id.startswith("sw-"):
try:
s_sid, s_slot = job_id.split("/", 1)
s_proc = subprocess.Popen(
[sys.executable, str(BIN_DIR / "box-ctl.py"), "swarm-report", s_sid, s_slot],
stdin=subprocess.PIPE, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL
)
s_proc.communicate(input=json.dumps({"ok": not is_fail, "result": result_text}).encode("utf-8"))
except Exception as se:
sys.stderr.write(f"warning: failed to record swarm report: {se}\n")
else:
trigger_chain_next(job_id, result_text, success=not is_fail) trigger_chain_next(job_id, result_text, success=not is_fail)
if not is_fail: if not is_fail:
archive_ephemeral_thread(agent, thread_id, job_id=job_id) archive_ephemeral_thread(agent, thread_id, job_id=job_id)
@@ -640,6 +656,7 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
dry_run, job_id=job_id, verb=verb) dry_run, job_id=job_id, verb=verb)
else: else:
clear_matching_followups(followups, agent, thread_id, mid, text, dry_run) clear_matching_followups(followups, agent, thread_id, mid, text, dry_run)
maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=dry_run)
return new_messages, new_wm, job_results return new_messages, new_wm, job_results
@@ -782,6 +799,212 @@ def archive_ephemeral_thread(agent, thread_id, job_id=None):
sys.stderr.write(f"warning: archive_ephemeral_thread failed: {ae}\n") sys.stderr.write(f"warning: archive_ephemeral_thread failed: {ae}\n")
SWARM_WORKER_POOL = ["dev", "def", "muse"]
def reconcile_and_dispatch_swarms(dry_run=False):
"""
Autonomous Swarm Orchestrator:
1. Scans swarms.json for pending slots.
2. Dynamically allocates available auxiliary worker nodes (dev, def, muse).
3. Provisions ephemeral sidechat per slot and dispatches the task with [RESULT <swarm_id>/<slot>].
4. Upon completion of all slots, sends completion summary DM to originating coordinator.
"""
if not SWARM_FILE.exists() or dry_run:
return
try:
swarms = json.loads(SWARM_FILE.read_text(encoding="utf-8"))
except Exception:
return
modified = False
now = utcnow()
# Determine busy workers from running slots
busy_workers = set()
for sid, swarm in swarms.items():
if swarm.get("status") in ("pending", "running"):
for slot in swarm.get("slots", []):
if slot.get("status") == "running" and slot.get("agent_id"):
busy_workers.add(slot["agent_id"])
# Process each swarm
for sid, swarm in swarms.items():
st = swarm.get("status")
creator = swarm.get("created_by") or "646"
if creator not in ("646", "pip", "opm", "muse", "dev", "def"):
creator = "646"
# 1. Allocate & dispatch pending slots
if st in ("pending", "running"):
slots = swarm.get("slots", [])
for s in slots:
if s.get("status") == "pending":
# Find first available worker
worker = None
for w in SWARM_WORKER_POOL:
if w not in busy_workers:
worker = w
break
if not worker:
# Fallback round-robin across worker pool if all are busy
worker = SWARM_WORKER_POOL[s["slot"] % len(SWARM_WORKER_POOL)]
slot_idx = s["slot"]
task_text = swarm.get("task", "execute subagent task")
slot_target = f"{sid}-s{slot_idx}"
# Attach worker to slot
s["agent_id"] = worker
s["status"] = "running"
s["updated_ts"] = now
busy_workers.add(worker)
modified = True
# Prepare task directive prompt
prompt = (
f"[JOB {sid}/{slot_idx}] Task for swarm slot {slot_idx}:\n"
f"{task_text}\n\n"
f"Reply with [RESULT {sid}/{slot_idx}] OK <summary> or FAIL <reason>."
)
# Send to worker via dm.py (which handles sidechat creation & tracking)
try:
dm_cmd = [
sys.executable, str(BIN_DIR / "dm.py"), "send",
"--agent", creator,
"--to", worker,
"--target", slot_target,
prompt,
]
subprocess.Popen(dm_cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
print(f"[swarm] Dispatched slot {slot_idx} of {sid} to {worker} in sidechat {slot_target}")
except Exception as de:
sys.stderr.write(f"warning: failed to dispatch swarm slot: {de}\n")
# Update swarm rollup status
counts = {"pending": 0, "running": 0, "done": 0, "failed": 0, "killed": 0}
for slot in slots:
counts[slot.get("status", "pending")] = counts.get(slot.get("status", "pending"), 0) + 1
if counts["pending"] + counts["running"] == 0:
swarm["status"] = "completed" if counts["failed"] == 0 else "partial"
swarm["updated_ts"] = now
modified = True
elif counts["running"] or counts["done"]:
swarm["status"] = "running"
swarm["updated_ts"] = now
modified = True
# 2. Check if newly completed/partial and notify coordinator
if swarm.get("status") in ("completed", "partial") and not swarm.get("notified_coordinator"):
swarm["notified_coordinator"] = True
modified = True
done_cnt = sum(1 for sl in swarm.get("slots", []) if sl.get("status") == "done")
total_cnt = len(swarm.get("slots", []))
summary_msg = (
f"[Swarm Report] Swarm {sid} ({swarm.get('label') or 'task'}) finished: "
f"{done_cnt}/{total_cnt} slots successful. "
f"Console: https://box.muse-dev.online/#dashboard"
)
try:
coord_dm = [
sys.executable, str(BIN_DIR / "dm.py"), "send",
"--agent", "box",
"--to", creator,
"--target", "main",
"--allow-main-chat",
summary_msg,
]
subprocess.Popen(coord_dm, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
print(f"[swarm] Delivered completion report for {sid} to coordinator {creator}")
except Exception as ce:
sys.stderr.write(f"warning: failed to notify swarm coordinator: {ce}\n")
if modified:
tmp_swarms = str(SWARM_FILE) + f".tmp.{os.getpid()}"
with open(tmp_swarms, "w", encoding="utf-8") as f:
json.dump(swarms, f, indent=2)
os.replace(tmp_swarms, SWARM_FILE)
def check_and_archive_terminal_swarms():
"""Check swarms.json and auto-archive ephemeral sidechats for completed or terminal swarms."""
if not SWARM_FILE.exists() or not JOB_SIDECHATS_FILE.exists():
return
try:
swarms_data = json.loads(SWARM_FILE.read_text(encoding="utf-8"))
sc_data = json.loads(JOB_SIDECHATS_FILE.read_text(encoding="utf-8"))
except Exception:
return
for sid, swarm in swarms_data.items():
st = swarm.get("status")
if st in ("completed", "partial", "killed"):
# Check slots or matching sidechats
for key, val in sc_data.items():
if isinstance(val, dict) and not val.get("archived") and val.get("type") != "persistent":
if sid in key or (swarm.get("label") and swarm.get("label") in key):
tu = val.get("thread_uuid")
ag = val.get("agent", "opm")
if tu:
archive_ephemeral_thread(ag, tu, job_id=sid)
def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=False):
"""
If an agent replies conversationally in a sidechat backed by a job or follow-up
without providing [RESULT <id>] or tool directives, deliver a terse 1-turn nudge footer.
"""
if dry_run or not thread_id or thread_id == "main":
return
if not re.fullmatch(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", thread_id.lower()):
return
# Look up active follow-ups or job association for this thread
followups = load_json_file(FOLLOWUPS_FILE)
matching_job_id = None
for f_id, f_rec in followups.items():
if f_rec.get("status") in ("pending", "acknowledged") and f_rec.get("thread_uuid") == thread_id:
matching_job_id = f_rec.get("job_id") or f_id
break
if not matching_job_id and JOB_SIDECHATS_FILE.exists():
try:
sc_state = json.loads(JOB_SIDECHATS_FILE.read_text(encoding="utf-8"))
for k, v in sc_state.items():
if isinstance(v, dict) and v.get("thread_uuid") == thread_id:
if k.startswith(("box-", "job-", "pipe-")):
matching_job_id = k
break
except Exception:
pass
if not matching_job_id:
return
tracker = load_json_file(NUDGE_TRACKER_FILE)
rec = tracker.get(thread_id, {})
if rec.get("nudged_for_mid") == mid or rec.get("nudge_count", 0) >= 1:
return
# Inject terse nudge footer
nudge_msg = f"[Nudge: To advance the workflow, reply with [RESULT {matching_job_id}] <outcome>]"
try:
import muse_hybrid
print(f"[{agent}] Injecting 1-turn nudge into {thread_name or thread_id[:8]} for job {matching_job_id}")
muse_hybrid.send_message(agent, nudge_msg, thread_id=thread_id, wait=0)
tracker[thread_id] = {
"ts": utcnow(),
"job_id": matching_job_id,
"nudged_for_mid": mid,
"nudge_count": rec.get("nudge_count", 0) + 1,
}
save_json_file(NUDGE_TRACKER_FILE, tracker)
except Exception as ne:
sys.stderr.write(f"warning: failed to deliver conversational nudge: {ne}\n")
# Chain deduplication # Chain deduplication
CHAINED_JOBS_FILE = Path(__file__).parent / "chained-jobs.json" CHAINED_JOBS_FILE = Path(__file__).parent / "chained-jobs.json"
@@ -1081,6 +1304,8 @@ def harvest_cycle(target_agent=None, dry_run=False, output_json=False):
if not dry_run: if not dry_run:
save_json_file(WATERMARKS_FILE, watermarks) save_json_file(WATERMARKS_FILE, watermarks)
reconcile_and_dispatch_swarms()
check_and_archive_terminal_swarms()
# Output formatting # Output formatting
if output_json: if output_json:
+140 -2
View File
@@ -276,9 +276,14 @@ def make_digest_id(agent):
# In-band response-contract footer for ACTIONABLE digests. The verbs are matched # In-band response-contract footer for ACTIONABLE digests. The verbs are matched
# by response-harvester.py to resolve followups: ACK/CLAIM acknowledge (nudge # by response-harvester.py to resolve followups: ACK/CLAIM acknowledge (nudge
# suppression), RESULT/DECLINE/NO-ACTION close. ~94 chars, well under budget. # suppression), RESULT/DECLINE/NO-ACTION close. The closing line enforces the
# recursive box->agent->box discipline: post the RESULT back in this thread
# (box records it and dispatches the next chained step); never DM the next
# agent directly. ~166 chars, within the 600-char digest budget.
CONTRACT_FOOTER = ("Reply: [ACK id] seen | [CLAIM id] mine | " CONTRACT_FOOTER = ("Reply: [ACK id] seen | [CLAIM id] mine | "
"[RESULT id] done | [DECLINE id] | [NO-ACTION id]") "[RESULT id] done | [DECLINE id] | [NO-ACTION id]. "
"Report back here. Box dispatches the next step; "
"do not DM the next agent directly.")
def send_prompt(sender, agent, sidechat, digest): def send_prompt(sender, agent, sidechat, digest):
@@ -552,6 +557,63 @@ def check_subagents(agent, cfg, threads_meta):
return alerts return alerts
# ---------------------------------------------------------------------------
# Brain workspace — the main loop operates its thinking in the "main-loop
# brain" sidechat (opm's account). See bin/brain.py.
# User directive 2026-10-04: "we need main loop to operate its brains in
# side chat". Safety: brain posts carry [BRAIN], never [JOB]; never
# --expect-reply; own messages skipped on read; !loop only from authorized
# senders. All enforced in brain.py.
# ---------------------------------------------------------------------------
_BRAIN_LAST_IDS = {} # (agent, sidechat_name) -> last message_id seen
def read_brain_messages(agent, sidechat_name, since_ts):
"""read_fn adapter for brain.BrainWorkspace.
Resolves sidechat_name -> UUID via dm (never hardcoded; UUIDs rotate),
reads via the same muse_hybrid primitive the loop uses, returns
[{"sender", "text", "ts"}]. Tracks last message_id per (agent, name);
first run anchors at newest with no backfill (same policy as
new_messages for main chat).
"""
try:
import dm
uuid = dm.resolve_sidechat_target(sidechat_name, agent)
except Exception:
return []
if not uuid:
return []
try:
import muse_hybrid
msgs, err = muse_hybrid.get_history(agent, thread_id=uuid, limit=10)
except Exception:
return []
if err or not msgs:
return []
key = (agent, sidechat_name)
last_id = _BRAIN_LAST_IDS.get(key)
ids = [m.get("message_id") or "msg-%s" % m.get("seq") for m in msgs]
if last_id is None:
# First run: anchor at newest, no backfill (old !loop commands
# must not fire on deploy).
_BRAIN_LAST_IDS[key] = ids[-1] if ids else None
return []
if last_id in ids:
new_msgs = msgs[ids.index(last_id) + 1:]
else:
# Watermark fell out of the read window: treat all as new.
# Safe: brain intake skips own messages and only authorized
# !loop senders can act.
new_msgs = msgs
_BRAIN_LAST_IDS[key] = ids[-1] if ids else last_id
now = time.time()
return [{"sender": m.get("role", "unknown"),
"text": m.get("text", ""),
"ts": now} for m in new_msgs]
def do_check(only_agent=None): def do_check(only_agent=None):
st = load_state() st = load_state()
cfg = get_config(st) cfg = get_config(st)
@@ -564,6 +626,23 @@ def do_check(only_agent=None):
results = {} results = {}
prompted = 0 prompted = 0
errors = 0 errors = 0
quiet_held = 0
# Brain workspace: the loop thinks in the sidechat, takes !loop
# instructions there. Non-fatal: a brain failure must never break
# the tick.
brain = None
try:
from brain import BrainWorkspace
brain = BrainWorkspace(STATE_FILE, read_fn=read_brain_messages)
intake = brain.intake()
if intake.get("commands"):
log("brain: %d commands, %d acks, %d ignored-senders" % (
intake["commands"], intake.get("acks", 0),
intake.get("ignored_senders", 0)))
except Exception as e:
log("brain init/intake failed (non-fatal): %r" % e)
brain = None
import muse_hybrid import muse_hybrid
@@ -581,6 +660,10 @@ def do_check(only_agent=None):
if not enabled.get(agent, True): if not enabled.get(agent, True):
results[agent] = {"ok": True, "new": 0, "disabled": True} results[agent] = {"ok": True, "new": 0, "disabled": True}
continue continue
if brain is not None and brain.ignored(agent):
log("%s: skipped (brain !loop ignore active)" % agent)
results[agent] = {"ok": True, "new": 0, "ignored": True}
continue
wm = agents_state.get(agent) or {} wm = agents_state.get(agent) or {}
# 1. Main chat check # 1. Main chat check
@@ -640,6 +723,18 @@ def do_check(only_agent=None):
errors += 1 errors += 1
continue continue
# Brain quiet mode: hold actionable escalations. Reads continue,
# digests are composed, but nothing escalates to the agent — the
# thinking note in the brain carries the activity instead.
if brain is not None and brain.quiet():
is_act, _urg = classify_digest(digest)
if is_act:
log("%s: quiet mode - digest held (not escalated)" % agent)
results[agent] = {"ok": True, "new": total_new,
"quiet_held": True}
quiet_held += 1
continue
sent, detail, digest_id, actionable = send_prompt(cfg["sender"], agent, sidechat, digest) sent, detail, digest_id, actionable = send_prompt(cfg["sender"], agent, sidechat, digest)
if sent: if sent:
prompted += 1 prompted += 1
@@ -653,6 +748,49 @@ def do_check(only_agent=None):
results[agent] = {"ok": False, "error": detail, "new": total_new} results[agent] = {"ok": False, "error": detail, "new": total_new}
errors += 1 errors += 1
# Brain: post the tick's thinking to the sidechat workspace, then
# persist brain state. Non-fatal on failure.
if brain is not None:
try:
seen = {}
escalated = []
for a in agents:
r = results.get(a, {})
seen[a] = (r.get("new", 0), 0, 0)
if r.get("prompted"):
wm_a = agents_state.get(a) or {}
did = wm_a.get("last_digest_id")
if did and wm_a.get("last_digest_actionable"):
escalated.append(did)
closure_rate = None
health = None
try:
health = digest_health()
closure_rate = (health or {}).get("closure_rate")
except Exception:
pass
wm_epochs = {}
for a in agents:
ts_s = (agents_state.get(a) or {}).get("last_ts")
try:
if ts_s:
dt = datetime.fromisoformat(
ts_s.replace("Z", "+00:00"))
wm_epochs[a] = dt.timestamp()
except Exception:
pass
brain.post_thinking(
{"seen": seen, "escalated": escalated,
"skipped_info": quiet_held, "errors": errors,
"closure_rate": closure_rate},
cfg={a: bool(enabled.get(a, True)) for a in agents},
watermarks=wm_epochs,
health=health,
)
brain.save()
except Exception as e:
log("brain post_thinking failed (non-fatal): %r" % e)
# Save under the state lock with a fresh reload: an enable/disable may # Save under the state lock with a fresh reload: an enable/disable may
# have landed during the slow chat reads; preserve its config changes # have landed during the slow chat reads; preserve its config changes
# and only update the keys this run owns (watermarks, last_run/result). # and only update the keys this run owns (watermarks, last_run/result).