feat: completion-enforcement loop (fallback, proof, emit-model, auditor)
Close the loop so dispatched work actually completes on bl: - on_no_result fallback in followup-sweeper (op + job forms via exec-constrained registry / job-dispatch), seeded on the three autonomy-pulse jobs; fallback_due() dedupes the gravity path - gravity.py: add __main__ entry (loop-remediator.timer was a no-op), 300s re-arm budget, fallback firing + stamp/skip logic - harvester: proof-of-result followups (result_has_evidence), acted-variant NACK, emit-model tool-hint wording - envelope: RESPONSE RULE states the emit model (agents EMIT directives verbatim; runtime executes; works from bare containers) - completion-audit.py + systemd 15-min timer: per-family funnel, swarm drain, followup backlog; digest DM when degraded, 6h heartbeat - tests/test_completion.py (29 tests), JOB-SPEC.md docs Tests: 67/67 focused green (completion + tool_calls).
This commit is contained in:
+86
-11
@@ -301,6 +301,61 @@ def _scan_bracket_calls(text):
|
||||
return out
|
||||
|
||||
|
||||
_PROOF_EVIDENCE_RE = re.compile(
|
||||
r"sw-\d{8}-\d{6}-[0-9a-f]{4}"
|
||||
r"|[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}"
|
||||
r"|/(?:[\w.-]+/)+[\w.-]+"
|
||||
r"|\b(?:swarm|timer|cron|job|thread|sidechat|slot)[-_ ]?(?:id|name|uuid)?\s*[:=]"
|
||||
r"|\b\d+/\d+\s*(?:slots?|checks?|workers?)",
|
||||
re.IGNORECASE)
|
||||
_UUID_RE = re.compile(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}")
|
||||
|
||||
|
||||
def result_has_evidence(result_text):
|
||||
"""True when a RESULT verdict carries checkable artifacts (IDs, paths, counts)."""
|
||||
return bool(_PROOF_EVIDENCE_RE.search(result_text or ""))
|
||||
|
||||
|
||||
def maybe_request_proof(agent, thread_id, job_id, result_text, dry_run=False):
|
||||
"""Ask for checkable evidence when a success RESULT has none.
|
||||
|
||||
One-shot per (thread, job) via the nudge tracker. Returns True when a
|
||||
proof followup was scheduled.
|
||||
"""
|
||||
if dry_run or not thread_id or not _UUID_RE.fullmatch(thread_id.lower()):
|
||||
return False
|
||||
if result_has_evidence(result_text):
|
||||
return False
|
||||
tracker = load_json_file(NUDGE_TRACKER_FILE)
|
||||
rec = tracker.get(thread_id, {})
|
||||
done = rec.get("proof_jobs", [])
|
||||
if job_id in done:
|
||||
return False
|
||||
ok, res = execute_agent_tool(agent, "followup.create", {
|
||||
"agent": agent,
|
||||
"in_m": 30,
|
||||
"thread": thread_id,
|
||||
"prompt": (
|
||||
f"[PROOF] Your [RESULT {job_id}] has no checkable evidence. "
|
||||
f"Reply in this thread with the swarm/timer IDs, paths, or command output "
|
||||
f"that prove the outcome — or say what is still missing."),
|
||||
})
|
||||
if ok:
|
||||
rec["proof_jobs"] = (done + [job_id])[-50:]
|
||||
tracker[thread_id] = rec
|
||||
save_json_file(NUDGE_TRACKER_FILE, tracker)
|
||||
append_jsonl(JOB_LOG, {
|
||||
"ts": utcnow(),
|
||||
"type": "proof_requested",
|
||||
"job_id": job_id,
|
||||
"agent": agent,
|
||||
"thread_id": thread_id,
|
||||
})
|
||||
return True
|
||||
sys.stderr.write(f"warning: proof followup failed for {job_id}: {res}\n")
|
||||
return False
|
||||
|
||||
|
||||
def parse_tool_calls(text):
|
||||
"""
|
||||
Extract structured tool/exec calls from assistant messages.
|
||||
@@ -893,7 +948,9 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
||||
thread_url = f"https://box.muse-dev.online/thread/{thread_id}"
|
||||
tool_hint = (
|
||||
f"[Runtime Context: {thread_url}]\n"
|
||||
f"Tools: [TOOL <op> <args>] or curl -sk -X POST https://exec.muse-dev.online/exec\n"
|
||||
f"Tools: EMIT one [TOOL <op> <args>] line per action (you do not run it;"
|
||||
f" the runtime executes it and replies here). curl -sk -X POST"
|
||||
f" https://exec.muse-dev.online/exec works too.\n"
|
||||
f" • [TOOL tools.list {{}}] — discover every op dynamically\n"
|
||||
f" • [TOOL swarm.spawn {{\"count\": 1, \"task\": \"<task>\"}}] — spawn subagents\n"
|
||||
f" • [DM {{\"to\": \"<agent>\", \"target\": \"<sidechat>\", \"message\": \"<text>\"}}] — send a DM\n"
|
||||
@@ -941,6 +998,11 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
||||
sys.stderr.write(f"warning: failed to record swarm report: {se}\n")
|
||||
else:
|
||||
trigger_chain_next(job_id, result_text, success=not is_fail)
|
||||
if not is_fail:
|
||||
try:
|
||||
maybe_request_proof(agent, thread_id, job_id, result_text)
|
||||
except Exception as pe:
|
||||
sys.stderr.write(f"warning: proof check failed: {pe}\n")
|
||||
archive_ephemeral_thread(agent, thread_id, job_id=job_id)
|
||||
clear_matching_followups(followups, agent, thread_id, mid, text,
|
||||
dry_run, job_id=job_id, verb="RESULT")
|
||||
@@ -951,7 +1013,8 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
||||
dry_run, job_id=job_id, verb=verb)
|
||||
else:
|
||||
clear_matching_followups(followups, agent, thread_id, mid, text, dry_run)
|
||||
maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=dry_run)
|
||||
maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text,
|
||||
dry_run=dry_run, acted=bool(tool_calls))
|
||||
|
||||
return new_messages, new_wm, job_results
|
||||
|
||||
@@ -1392,10 +1455,13 @@ def check_and_archive_terminal_swarms():
|
||||
|
||||
|
||||
|
||||
def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=False):
|
||||
def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=False,
|
||||
acted=False):
|
||||
"""
|
||||
If an agent replies conversationally in a sidechat backed by a job or follow-up
|
||||
without providing [RESULT <id>] or tool directives, deliver a terse 1-turn nudge footer.
|
||||
When acted=True the agent DID emit directives but never closed: remind to close
|
||||
with [RESULT] instead of rejecting the (good) action.
|
||||
"""
|
||||
if dry_run or not thread_id or thread_id == "main":
|
||||
return
|
||||
@@ -1440,14 +1506,23 @@ def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_
|
||||
matching_job_id, matching_job_id, prompt_envelope.pick_profile(matching_job_id))
|
||||
except Exception:
|
||||
_spawn = '[TOOL swarm.spawn {"count": 2, "task": "continue the job work"}]'
|
||||
nudge_msg = (
|
||||
f"{_spawn}\n"
|
||||
f"[STRICT ENFORCEMENT: Conversational commentary is rejected. Work requires active execution.]\n"
|
||||
f"Thread Console: {thread_url}\n"
|
||||
f"Emit executable tool calls now: [TOOL <op> <args>] or curl against https://exec.muse-dev.online/exec\n"
|
||||
f"When all operations are finished, close strictly with [RESULT {matching_job_id}] <outcome>.\n"
|
||||
f"{_spawn}"
|
||||
)
|
||||
if acted:
|
||||
nudge_msg = (
|
||||
f"Action received — now close the loop: reply with [RESULT {matching_job_id}] <outcome>.\n"
|
||||
f"Outcome needs checkable evidence (swarm/timer IDs, paths, or command output), not prose alone.\n"
|
||||
f"Thread Console: {thread_url}"
|
||||
)
|
||||
else:
|
||||
nudge_msg = (
|
||||
f"{_spawn}\n"
|
||||
f"[STRICT ENFORCEMENT: Conversational commentary is rejected. Work requires active execution.]\n"
|
||||
f"Thread Console: {thread_url}\n"
|
||||
f"EMIT tool calls verbatim in your reply — you do not run them yourself;"
|
||||
f" the Box runtime on bl executes each directive and posts the result back here"
|
||||
f" (works from containers with no box CLI). Or curl against https://exec.muse-dev.online/exec\n"
|
||||
f"When all operations are finished, close strictly with [RESULT {matching_job_id}] <outcome>.\n"
|
||||
f"{_spawn}"
|
||||
)
|
||||
try:
|
||||
import muse_hybrid
|
||||
print(f"[{agent}] Injecting 1-turn strict nudge into {thread_name or thread_id[:8]} for job {matching_job_id}")
|
||||
|
||||
Reference in New Issue
Block a user