From 7444b52763908b9a042699e792fbad4e630b3955 Mon Sep 17 00:00:00 2001 From: operator Date: Tue, 6 Oct 2026 03:14:02 +0000 Subject: [PATCH] Fix split-brain swarm dispatch: narrow harvester SWARM_WORKER_POOL to [muse] The 2026-10-05 fix note claimed dev/def were removed from the pool but the code still listed [dev, def, muse]. The harvester's every-minute systemd dispatch raced the pool daemon claiming slots as dev/def, which refuse on attribution grounds (dev FAILs fast, def freezes until the 60-min reaper) - the fleet-wide dev-FAIL-fast + def-frozen pattern on every swarm. Pool now matches swarm_worker/daemon.py WORKER_POOL exactly: muse only, the single fully authenticated auxiliary worker. Reporting/harvest paths untouched. --- bin/response-harvester.py | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/bin/response-harvester.py b/bin/response-harvester.py index 3cb25e1..17109d4 100755 --- a/bin/response-harvester.py +++ b/bin/response-harvester.py @@ -960,6 +960,13 @@ def archive_ephemeral_thread(agent, thread_id, job_id=None): except Exception as e: sys.stderr.write(f"warning: failed to update job-sidechats state: {e}\n") + # Complete subagent tracker record if this thread corresponds to an ephemeral subagent session + try: + import subagent_tracker + subagent_tracker.complete_session(thread_id, note=f"Harvested verdict for job {job_id}") + except Exception: + pass + # Call muse-threads.py archive via subprocess (runs in node's netns) try: helper = BIN_DIR / "muse-threads.py" @@ -975,7 +982,13 @@ def archive_ephemeral_thread(agent, thread_id, job_id=None): # NOTE 2026-10-05 (Fix Agent 2/5): dev/def removed from the pool. No worker # agents exist on dev/def (no Meta sessions provisioned), so slots dispatched # to them froze with null results. Re-add only after real dev/def workers exist. -SWARM_WORKER_POOL = ["dev", "def", "muse"] +# NOTE 2026-10-06 (split-brain dispatch fix): the removal above was comment-only; +# the code still listed dev/def, and the harvester's every-minute dispatch raced +# the pool daemon claiming slots as dev/def, which refuse on attribution grounds +# (dev FAILs fast, def freezes until the 60-min reaper). Pool now matches +# swarm_worker/daemon.py WORKER_POOL exactly: only muse, the single fully +# authenticated auxiliary worker. +SWARM_WORKER_POOL = ["muse"] # Stuck-slot reaper: a slot that stays "running" with no result longer than # this is treated as wedged (worker died / dispatch lost). Healthy slots @@ -1003,7 +1016,8 @@ def reconcile_and_dispatch_swarms(dry_run=False): """ Autonomous Swarm Orchestrator: 1. Scans swarms.json for pending slots. - 2. Dynamically allocates available auxiliary worker nodes (dev, def, muse). + 2. Dynamically allocates the auxiliary worker pool (muse only; dev/def have + no provisioned worker agents and refuse on attribution grounds). 3. Provisions ephemeral sidechat per slot and dispatches the task with [RESULT /]. 4. Upon completion of all slots, sends completion summary DM to originating coordinator. """