feat: completion-enforcement loop (fallback, proof, emit-model, auditor)

Close the loop so dispatched work actually completes on bl:

- on_no_result fallback in followup-sweeper (op + job forms via
  exec-constrained registry / job-dispatch), seeded on the three
  autonomy-pulse jobs; fallback_due() dedupes the gravity path
- gravity.py: add __main__ entry (loop-remediator.timer was a no-op),
  300s re-arm budget, fallback firing + stamp/skip logic
- harvester: proof-of-result followups (result_has_evidence),
  acted-variant NACK, emit-model tool-hint wording
- envelope: RESPONSE RULE states the emit model (agents EMIT
  directives verbatim; runtime executes; works from bare containers)
- completion-audit.py + systemd 15-min timer: per-family funnel,
  swarm drain, followup backlog; digest DM when degraded, 6h heartbeat
- tests/test_completion.py (29 tests), JOB-SPEC.md docs

Tests: 67/67 focused green (completion + tool_calls).
This commit is contained in:
Muse Sidechat
2026-10-06 08:20:14 +00:00
parent b7e45010c3
commit a9f014f9fa
12 changed files with 1173 additions and 16 deletions
+312
View File
@@ -0,0 +1,312 @@
#!/usr/bin/env python3
"""Completion auditor: prove work gets done, or say exactly where it stalls.
Runs on a 15-minute systemd timer (systemd/completion-audit.*). Reads
job-log.jsonl, swarms.json, and followups.json; computes the completion
funnel per job family plus swarm drain and followup backlog; writes a JSON
report under logs/ and posts a compact digest to the ops heartbeat sidechat
when degraded (or a heartbeat summary every 6h when green).
Read-only except the digest DM and its own log/state files. Exit 0 always
on a completed audit; tracebacks (real errors) fail the timer visibly.
"""
import argparse
import json
import os
import re
import subprocess
import sys
from collections import Counter, defaultdict
from datetime import datetime, timezone, timedelta
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
JOB_LOG = REPO_ROOT / "job-log.jsonl"
SWARMS_FILE = REPO_ROOT / "swarms.json"
FOLLOWUPS_FILE = REPO_ROOT / "followups.json"
LOGS_DIR = REPO_ROOT / "logs"
STATE_FILE = LOGS_DIR / "completion-audit-state.json"
_JOB_ID_RE = re.compile(r"^(.+)-(\d{8})-(\d{6})-([0-9a-f]{8})$")
HEARTBEAT_INTERVAL_H = 6
STALE_RUNNING_MIN = 90
SILENT_MIN_SENT = 3
def utcnow():
return datetime.now(timezone.utc)
def family_of(job_id):
"""Strip the dispatch suffix (<name>-YYYYMMDD-HHMMSS-<hex8>) to the family."""
m = _JOB_ID_RE.match(job_id or "")
return m.group(1) if m else (job_id or "?")
def parse_ts(ts):
try:
t = datetime.fromisoformat(str(ts))
except Exception:
return None
if t.tzinfo is None:
t = t.replace(tzinfo=timezone.utc)
return t
def compute_funnel(events, cutoff):
"""Aggregate job-log events since cutoff.
Returns (families, tools) where families maps family -> counters and
tools holds global tool_exec stats. Pure over the event list.
"""
families = defaultdict(lambda: Counter())
tools = Counter()
tool_errs = Counter()
for e in events:
t = parse_ts(e.get("ts"))
if t is None or t < cutoff:
continue
ty = e.get("type")
if ty == "job_sent":
families[family_of(e.get("job_id"))]["sent"] += 1
elif ty == "job_dispatched":
families[family_of(e.get("job_id"))]["dispatched"] += 1
elif ty == "tool_exec":
tools["total"] += 1
if e.get("success"):
tools["ok"] += 1
else:
tools["fail"] += 1
tool_errs[e.get("op", "?")] += 1
elif ty == "job_result":
fam = family_of(e.get("job_id"))
families[fam]["results"] += 1
families[fam]["ok" if e.get("success") else "fail"] += 1
elif ty == "job_failed":
families[family_of(e.get("job_id"))]["failed"] += 1
elif ty == "fallback_executed":
families[family_of(e.get("job_id"))]["fallback_ok"] += 1
elif ty == "fallback_failed":
families[family_of(e.get("job_id"))]["fallback_fail"] += 1
elif ty == "proof_requested":
families[family_of(e.get("job_id"))]["proofs"] += 1
return families, {"tools": tools, "tool_errs": tool_errs}
def swarm_drain(now):
"""Status counts + stale-running slots from swarms.json."""
try:
data = json.load(open(SWARMS_FILE))
except Exception:
return {"error": "swarms.json unreadable"}, []
values = data.values() if isinstance(data, dict) else data
status = Counter()
stale = []
for s in values:
if not isinstance(s, dict):
continue
for sl in s.get("slots", []) or []:
status[sl.get("status", "?")] += 1
if sl.get("status") == "running":
upd = parse_ts(sl.get("updated_ts"))
if upd and (now - upd) > timedelta(minutes=STALE_RUNNING_MIN):
stale.append({
"swarm": s.get("swarm_id"),
"slot": sl.get("slot"),
"agent": sl.get("agent_id"),
"idle_min": int((now - upd).total_seconds() // 60),
})
return {"slots": dict(status)}, stale
def followup_backlog(now):
"""Pending/overdue/escalated counts from followups.json."""
try:
data = json.load(open(FOLLOWUPS_FILE))
except Exception:
return {"error": "followups.json unreadable"}
values = data.values() if isinstance(data, dict) else data
out = Counter()
for r in values:
if not isinstance(r, dict):
continue
st = r.get("status", "?")
out[st] += 1
if st == "pending":
dl = parse_ts(r.get("deadline"))
if dl and dl < now:
out["overdue"] += 1
return dict(out)
def build_report(hours):
now = utcnow()
cutoff = now - timedelta(hours=hours)
events = []
try:
with open(JOB_LOG) as f:
for line in f:
line = line.strip()
if not line:
continue
try:
events.append(json.loads(line))
except Exception:
continue
except FileNotFoundError:
pass
families, tools = compute_funnel(events, cutoff)
fam = {k: dict(v) for k, v in sorted(families.items())}
swarm, stale = swarm_drain(now)
backlog = followup_backlog(now)
totals = Counter()
for v in fam.values():
for k, n in v.items():
totals[k] += n
silent = sorted(
k for k, v in fam.items()
if v.get("sent", 0) >= SILENT_MIN_SENT and v.get("results", 0) == 0)
degraded_reasons = []
if silent:
degraded_reasons.append(f"{len(silent)} silent families: {', '.join(silent[:5])}")
if totals.get("failed"):
degraded_reasons.append(f"{totals['failed']} job_failed")
if tools["tools"].get("fail"):
top = tools["tool_errs"].most_common(3)
degraded_reasons.append(
"tool errors: " + ", ".join(f"{op}x{n}" for op, n in top))
if totals.get("fallback_fail"):
degraded_reasons.append(f"{totals['fallback_fail']} fallback_failed")
if stale:
degraded_reasons.append(f"{len(stale)} running slots idle >{STALE_RUNNING_MIN}m")
if backlog.get("overdue"):
degraded_reasons.append(f"{backlog['overdue']} overdue followups")
if backlog.get("escalated"):
degraded_reasons.append(f"{backlog['escalated']} escalated followups")
return {
"ts": now.isoformat(),
"window_h": hours,
"totals": dict(totals),
"tools": {k: dict(v) if isinstance(v, Counter) else v
for k, v in tools.items()},
"families": fam,
"silent_families": silent,
"swarms": swarm,
"stale_running": stale[:10],
"followups": backlog,
"degraded": bool(degraded_reasons),
"reasons": degraded_reasons,
}
def render_digest(rep):
t = rep["totals"]
tools = rep["tools"].get("tools", {})
lines = [
f"Completion audit ({rep['window_h']}h, {rep['ts'][:16]}Z)",
f"funnel: {t.get('sent', 0)} sent / {t.get('dispatched', 0)} dispatched / "
f"{tools.get('total', 0)} tool_exec / {t.get('results', 0)} results "
f"({t.get('ok', 0)} ok)",
]
if rep["silent_families"]:
lines.append("silent: " + ", ".join(rep["silent_families"][:6]))
bits = []
if t.get("failed"):
bits.append(f"{t['failed']} job_failed")
if tools.get("fail"):
bits.append(f"{tools['fail']} tool errors")
if t.get("fallback_ok") or t.get("fallback_fail"):
bits.append(f"fallback {t.get('fallback_ok', 0)} ok / {t.get('fallback_fail', 0)} fail")
if t.get("proofs"):
bits.append(f"{t['proofs']} proof reqs")
if bits:
lines.append("flags: " + ", ".join(bits))
sw = rep["swarms"].get("slots", {})
if sw:
lines.append("swarms now: " + " / ".join(f"{v} {k}" for k, v in sorted(sw.items())))
if rep["stale_running"]:
lines.append(f"stale running: {len(rep['stale_running'])} slots (see report)")
fb = rep["followups"]
if fb and "error" not in fb:
lines.append(
f"followups now: {fb.get('pending', 0)} pending / {fb.get('overdue', 0)} "
f"overdue / {fb.get('escalated', 0)} escalated")
if rep["degraded"]:
lines.append("verdict: DEGRADED — " + "; ".join(rep["reasons"][:3]))
else:
lines.append("verdict: HEALTHY — work flowing, results landing")
return "\n".join(lines)
def should_post(report):
"""Post on degraded, else heartbeat at most every HEARTBEAT_INTERVAL_H."""
if report["degraded"]:
return True, "degraded"
try:
state = json.load(open(STATE_FILE))
last = parse_ts(state.get("last_heartbeat"))
except Exception:
last = None
if last is None or (utcnow() - last) > timedelta(hours=HEARTBEAT_INTERVAL_H):
return True, "heartbeat"
return False, "green-quiet"
def post_digest(digest):
argv = [sys.executable, str(REPO_ROOT / "bin" / "dm.py"), "send",
"--agent", "super", "--to", "opm", "--target", "heartbeat", digest]
p = subprocess.run(argv, capture_output=True, text=True, timeout=120)
return p.returncode == 0, (p.stdout or p.stderr or "").strip()[:300]
def save_report(report):
LOGS_DIR.mkdir(parents=True, exist_ok=True)
stamp = report["ts"].replace("+00:00", "Z").replace(":", "")
dated = LOGS_DIR / f"completion-audit-{stamp[:15]}.json"
body = json.dumps(report, indent=2)
dated.write_text(body, encoding="utf-8")
latest = LOGS_DIR / "completion-audit-latest.json"
tmp = LOGS_DIR / f".completion-audit-latest.tmp.{os.getpid()}"
tmp.write_text(body, encoding="utf-8")
os.replace(tmp, latest)
return dated
def main():
ap = argparse.ArgumentParser(description="Completion funnel auditor")
ap.add_argument("--hours", type=int, default=24)
ap.add_argument("--post", dest="post", action="store_true", default=True)
ap.add_argument("--no-post", dest="post", action="store_false")
ap.add_argument("--json", action="store_true", help="Print raw report JSON")
args = ap.parse_args()
report = build_report(args.hours)
path = save_report(report)
if args.json:
print(json.dumps(report, indent=2))
else:
print(render_digest(report))
print(f"\nreport: {path}")
if not args.post:
print("post: skipped (--no-post)")
return 0
do_post, why = should_post(report)
if not do_post:
print(f"post: skipped ({why})")
return 0
ok, detail = post_digest(render_digest(report))
print(f"post: {'delivered' if ok else 'FAILED'} ({why}) {detail[:120]}")
if ok and why == "heartbeat":
try:
state = {}
if STATE_FILE.exists():
state = json.loads(STATE_FILE.read_text(encoding="utf-8"))
state["last_heartbeat"] = report["ts"]
STATE_FILE.write_text(json.dumps(state, indent=2), encoding="utf-8")
except Exception as e:
print(f"warning: state save failed: {e}")
return 0
if __name__ == "__main__":
sys.exit(main())
+185
View File
@@ -144,6 +144,164 @@ def _resolve_nudge_thread_uuid(nudge_output):
return fallback
_JOB_ID_RE = re.compile(r"^(.+)-(\d{8})-(\d{6})-([0-9a-f]{8})$")
_OP_NAME_RE = re.compile(r"^[a-z][a-z0-9_.]{0,63}$")
_JOB_NAME_RE = re.compile(r"^[a-z0-9-]{1,64}$")
_FALLBACK_RETRY_S = 3600
def fallback_due(rec, now=None):
"""True when a terminal followup should (re)attempt its on_no_result fallback.
Fires once per record; a failed attempt may retry after _FALLBACK_RETRY_S.
Shared by the sweeper terminal path and gravity remediate so the two
firing paths can never double-execute.
"""
fb = rec.get("fallback") or {}
if fb.get("ran"):
return False
ts = fb.get("ts")
if not ts:
return True
try:
last = datetime.fromisoformat(str(ts).replace("Z", "+00:00"))
except Exception:
return True
if last.tzinfo is None:
last = last.replace(tzinfo=timezone.utc)
base = now or utcnow_dt()
return (base - last).total_seconds() >= _FALLBACK_RETRY_S
_EXEC_OPS_MOD = None
def derive_job_name(job_id):
"""Extract the job name from a dispatched job_id (<name>-YYYYMMDD-HHMMSS-<hex8>)."""
m = _JOB_ID_RE.match(job_id or "")
if not m:
return None
name = m.group(1)
if not (JOBS_DIR / f"{name}.json").exists():
return None
return name
def load_job_fallback(job_name):
"""Return (spec, error) for a job's on_no_result fallback.
spec is None when the job declares none. Shape:
{"job": "<job-name>"} -> dispatch a fallback job, or
{"op": "<exec-op>", "args": {...}} -> run one exec-constrained op.
"""
try:
with open(JOBS_DIR / f"{job_name}.json", "r", encoding="utf-8") as f:
cfg = json.load(f)
except Exception as e:
return None, f"unreadable job {job_name}: {e}"
spec = cfg.get("on_no_result")
if spec is None:
return None, None
if not isinstance(spec, dict) or set(spec) - {"job", "op", "args"}:
return None, "on_no_result must be an object with job|op (+args)"
if bool(spec.get("job")) == bool(spec.get("op")):
return None, "on_no_result needs exactly one of job|op"
if spec.get("job"):
jn = spec["job"]
if not isinstance(jn, str) or not _JOB_NAME_RE.fullmatch(jn):
return None, "on_no_result.job must be a valid job name"
if not (JOBS_DIR / f"{jn}.json").exists():
return None, f"on_no_result.job {jn!r} does not exist"
else:
if not isinstance(spec.get("op"), str) or not _OP_NAME_RE.fullmatch(spec["op"]):
return None, "on_no_result.op must be a valid op name"
if "args" in spec and not isinstance(spec["args"], dict):
return None, "on_no_result.args must be an object"
return spec, None
def _load_exec_ops():
global _EXEC_OPS_MOD
if _EXEC_OPS_MOD is None:
import importlib.util
mod_spec = importlib.util.spec_from_file_location(
"exec_constrained_sweeper", str(BIN_DIR / "exec-constrained.py"))
mod = importlib.util.module_from_spec(mod_spec)
mod_spec.loader.exec_module(mod)
_EXEC_OPS_MOD = mod
return _EXEC_OPS_MOD
def run_no_result_fallback(rec, dry_run=False):
"""Execute a job's on_no_result fallback at terminal followup expiry.
Returns an outcome dict; never raises (failures are outcome data so
one bad spec can't break the sweep).
"""
dm_id = rec.get("dm_id", "?")
outcome = {"dm_id": dm_id, "ran": False, "mode": None,
"configured": False, "detail": "no job fallback"}
try:
job_name = derive_job_name(rec.get("job_id"))
if not job_name:
outcome["detail"] = "no resolvable job_id"
return outcome
spec, err = load_job_fallback(job_name)
if err:
outcome.update(configured=True, detail=err)
return outcome
if spec is None:
return outcome
outcome["configured"] = True
if dry_run:
outcome.update(mode="dry_run", detail=json.dumps(spec)[:200])
return outcome
if spec.get("job"):
env = os.environ.copy()
env["CHAIN_PREV_JOB_ID"] = rec.get("job_id", "")
env["CHAIN_PREV_RESULT"] = (
f"TIMEOUT: Agent {rec.get('recipient')} gave no result; "
f"on_no_result fallback for job {job_name}")
cmd = [sys.executable, str(DISPATCH_PY), spec["job"]]
try:
p = subprocess.run(cmd, capture_output=True, text=True,
timeout=180, env=env)
except Exception as e:
outcome.update(mode="job", detail=f"dispatch exception: {e}")
return outcome
ok = p.returncode == 0
outcome.update(ran=ok, mode="job",
detail=(f"dispatched {spec['job']}" if ok
else f"dispatch failed: {(p.stderr or p.stdout).strip()[:200]}"))
else:
mod = _load_exec_ops()
op = spec["op"]
op_spec = mod.OPS.get(op)
if op_spec is None:
outcome["detail"] = f"unknown op: {op}"
return outcome
args = dict(spec.get("args") or {})
try:
clean = op_spec["validate"](args)
except Exception as e:
outcome["detail"] = f"op validation failed: {e}"
return outcome
argv = op_spec["build"](clean)
try:
p = subprocess.run(argv, capture_output=True, text=True,
timeout=op_spec.get("timeout", 120))
except Exception as e:
outcome.update(mode="op", detail=f"op exception: {e}")
return outcome
ok = p.returncode == 0
out = (p.stdout or p.stderr or "").strip()
outcome.update(ran=ok, mode="op",
detail=(f"{op} ok: {out[:200]}" if ok
else f"{op} failed rc={p.returncode}: {out[:200]}"))
except Exception as e:
outcome["detail"] = f"fallback exception: {e}"
return outcome
def sweep_cycle(dry_run=False):
followups = load_followups()
if not followups:
@@ -152,6 +310,7 @@ def sweep_cycle(dry_run=False):
now = utcnow_dt()
nudges_count = 0
escalations_count = 0
fallbacks_count = 0
modified = False
for dm_id, rec in list(followups.items()):
@@ -334,6 +493,31 @@ def sweep_cycle(dry_run=False):
pipeline_engine.fail_pipeline(run_entry.get("run_id"), "step_timed_out_without_fallback")
except Exception:
pass
# on_no_result fallback: the agent never replied, so run
# the job's declared server-side effect now (if any).
# fallback_due() dedupes against the gravity firing path.
fb = (run_no_result_fallback(rec) if fallback_due(rec)
else {"configured": False, "ran": False, "mode": None,
"detail": "fallback already ran"})
if fb["configured"]:
rec["fallback"] = {"ran": fb["ran"], "mode": fb["mode"],
"detail": fb["detail"][:200],
"ts": utcnow_str()}
modified = True
append_job_log({
"ts": utcnow_str(),
"type": ("fallback_executed" if fb["ran"]
else "fallback_failed"),
"dm_id": dm_id,
"recipient": recipient,
"job_id": rec.get("job_id"),
"mode": fb["mode"],
"detail": fb["detail"][:300],
})
print(f"Sweeper: on_no_result fallback for {dm_id}: "
f"ran={fb['ran']} {fb['detail'][:120]}")
if fb["ran"]:
fallbacks_count += 1
else:
escalations_count += 1
@@ -346,6 +530,7 @@ def sweep_cycle(dry_run=False):
"pending": pending_count,
"nudges_sent": nudges_count,
"escalations": escalations_count,
"fallbacks": fallbacks_count,
}
+94 -1
View File
@@ -657,6 +657,62 @@ def diagnose_breaks() -> list:
return breaks
def _load_sweeper_module():
import importlib.util
mod_spec = importlib.util.spec_from_file_location(
"followup_sweeper_gravity",
str(Path(__file__).resolve().parent / "followup-sweeper.py"))
mod = importlib.util.module_from_spec(mod_spec)
mod_spec.loader.exec_module(mod)
return mod
def maybe_run_terminal_fallback(rec, now_iso, dry_run=False):
"""Run a job's on_no_result fallback once at terminal followup expiry.
Returns an outcome dict, or None when the record has no resolvable
fallback. Never raises. The sweeper terminal path shares the
fallback_due() guard, so the two firing paths can't double-execute.
"""
try:
sw = _load_sweeper_module()
except Exception as e:
return {"ran": False, "mode": None,
"detail": f"sweeper import failed: {e}"}
try:
if not sw.fallback_due(rec):
return None
job_name = sw.derive_job_name(rec.get("job_id"))
if not job_name:
return None
spec, err = sw.load_job_fallback(job_name)
if err or spec is None:
return None
if dry_run:
return {"ran": False, "mode": "dry_run",
"detail": json.dumps(spec)[:200]}
out = sw.run_no_result_fallback(rec)
rec["fallback"] = {"ran": out["ran"], "mode": out["mode"],
"detail": out["detail"][:200], "ts": now_iso}
try:
sw.append_job_log({
"ts": now_iso,
"type": ("fallback_executed" if out["ran"]
else "fallback_failed"),
"dm_id": rec.get("dm_id"),
"recipient": rec.get("recipient"),
"job_id": rec.get("job_id"),
"mode": out["mode"],
"detail": out["detail"][:300],
})
except Exception:
pass
return out
except Exception as e:
return {"ran": False, "mode": None,
"detail": f"fallback exception: {e}"}
def remediate_breaks(dry_run=False) -> dict:
"""Progressively auto-remediate soft loop breakages while escalating hard breakages.
@@ -746,6 +802,20 @@ def remediate_breaks(dry_run=False) -> dict:
f_modified = True
rearm_sweeper = True
# Terminal: nudges exhausted and still no reply. Run the
# job's on_no_result fallback (server-side guarantee).
if is_expired and nudges_sent >= nudges_allowed:
fb = maybe_run_terminal_fallback(rec, now_iso, dry_run)
if fb is not None:
remediated.append({
"action": "terminal_fallback",
"loop_id": dm_id,
"agent": rec.get("recipient"),
"detail": f"on_no_result ran={fb.get('ran')}: {fb.get('detail', '')[:160]}",
})
if not dry_run and "fallback" in rec:
f_modified = True
if f_modified and not dry_run:
tmp = f"{f_path}.tmp.{os.getpid()}"
with open(tmp, "w") as f:
@@ -756,7 +826,10 @@ def remediate_breaks(dry_run=False) -> dict:
sweeper_py = Path("/home/super/Projects/NetVM/bin/followup-sweeper.py")
if sweeper_py.exists():
try:
subprocess.run([sys.executable, str(sweeper_py), "--once"], timeout=10)
# A single nudge send takes ~10s median; give the sweep
# room to finish or it dies mid-first-send every time.
subprocess.run([sys.executable, str(sweeper_py), "--once"],
timeout=300)
except Exception:
pass
@@ -819,4 +892,24 @@ def remediate_breaks(dry_run=False) -> dict:
}
def main(argv=None):
import argparse
ap = argparse.ArgumentParser(description="Loop gravity: reconcile and remediate followup loops")
ap.add_argument("--remediate", action="store_true",
help="Run remediate_breaks once (what loop-remediator.timer invokes)")
ap.add_argument("--dry-run", action="store_true",
help="Report actions without writing state or sending anything")
args = ap.parse_args(argv)
if not args.remediate:
ap.print_help()
return 2
result = remediate_breaks(dry_run=args.dry_run)
print(json.dumps(result, indent=2))
return 0
if __name__ == "__main__":
sys.exit(main())
+4 -1
View File
@@ -17,7 +17,10 @@ BOX_API = "https://box.muse-dev.online/api/box"
def _response_rule():
return (
"\nRESPONSE RULE: Execute your steps using [TOOL ...] directives or background tmux commands."
"\nRESPONSE RULE: Act by EMITTING [TOOL ...] / [DM ...] directive lines verbatim in your reply"
" — you do not run them yourself. The Box runtime on bl executes each directive"
" (this works from containers with no box CLI or tmux socket) and posts the result back here."
" Background tmux commands work too when you have a shell."
" When complete, conclude your output with the [RESULT ...] line so the harvester records it.\n"
)
+86 -11
View File
@@ -301,6 +301,61 @@ def _scan_bracket_calls(text):
return out
_PROOF_EVIDENCE_RE = re.compile(
r"sw-\d{8}-\d{6}-[0-9a-f]{4}"
r"|[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}"
r"|/(?:[\w.-]+/)+[\w.-]+"
r"|\b(?:swarm|timer|cron|job|thread|sidechat|slot)[-_ ]?(?:id|name|uuid)?\s*[:=]"
r"|\b\d+/\d+\s*(?:slots?|checks?|workers?)",
re.IGNORECASE)
_UUID_RE = re.compile(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}")
def result_has_evidence(result_text):
"""True when a RESULT verdict carries checkable artifacts (IDs, paths, counts)."""
return bool(_PROOF_EVIDENCE_RE.search(result_text or ""))
def maybe_request_proof(agent, thread_id, job_id, result_text, dry_run=False):
"""Ask for checkable evidence when a success RESULT has none.
One-shot per (thread, job) via the nudge tracker. Returns True when a
proof followup was scheduled.
"""
if dry_run or not thread_id or not _UUID_RE.fullmatch(thread_id.lower()):
return False
if result_has_evidence(result_text):
return False
tracker = load_json_file(NUDGE_TRACKER_FILE)
rec = tracker.get(thread_id, {})
done = rec.get("proof_jobs", [])
if job_id in done:
return False
ok, res = execute_agent_tool(agent, "followup.create", {
"agent": agent,
"in_m": 30,
"thread": thread_id,
"prompt": (
f"[PROOF] Your [RESULT {job_id}] has no checkable evidence. "
f"Reply in this thread with the swarm/timer IDs, paths, or command output "
f"that prove the outcome — or say what is still missing."),
})
if ok:
rec["proof_jobs"] = (done + [job_id])[-50:]
tracker[thread_id] = rec
save_json_file(NUDGE_TRACKER_FILE, tracker)
append_jsonl(JOB_LOG, {
"ts": utcnow(),
"type": "proof_requested",
"job_id": job_id,
"agent": agent,
"thread_id": thread_id,
})
return True
sys.stderr.write(f"warning: proof followup failed for {job_id}: {res}\n")
return False
def parse_tool_calls(text):
"""
Extract structured tool/exec calls from assistant messages.
@@ -893,7 +948,9 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
thread_url = f"https://box.muse-dev.online/thread/{thread_id}"
tool_hint = (
f"[Runtime Context: {thread_url}]\n"
f"Tools: [TOOL <op> <args>] or curl -sk -X POST https://exec.muse-dev.online/exec\n"
f"Tools: EMIT one [TOOL <op> <args>] line per action (you do not run it;"
f" the runtime executes it and replies here). curl -sk -X POST"
f" https://exec.muse-dev.online/exec works too.\n"
f" • [TOOL tools.list {{}}] — discover every op dynamically\n"
f" • [TOOL swarm.spawn {{\"count\": 1, \"task\": \"<task>\"}}] — spawn subagents\n"
f" • [DM {{\"to\": \"<agent>\", \"target\": \"<sidechat>\", \"message\": \"<text>\"}}] — send a DM\n"
@@ -941,6 +998,11 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
sys.stderr.write(f"warning: failed to record swarm report: {se}\n")
else:
trigger_chain_next(job_id, result_text, success=not is_fail)
if not is_fail:
try:
maybe_request_proof(agent, thread_id, job_id, result_text)
except Exception as pe:
sys.stderr.write(f"warning: proof check failed: {pe}\n")
archive_ephemeral_thread(agent, thread_id, job_id=job_id)
clear_matching_followups(followups, agent, thread_id, mid, text,
dry_run, job_id=job_id, verb="RESULT")
@@ -951,7 +1013,8 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
dry_run, job_id=job_id, verb=verb)
else:
clear_matching_followups(followups, agent, thread_id, mid, text, dry_run)
maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=dry_run)
maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text,
dry_run=dry_run, acted=bool(tool_calls))
return new_messages, new_wm, job_results
@@ -1392,10 +1455,13 @@ def check_and_archive_terminal_swarms():
def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=False):
def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_run=False,
acted=False):
"""
If an agent replies conversationally in a sidechat backed by a job or follow-up
without providing [RESULT <id>] or tool directives, deliver a terse 1-turn nudge footer.
When acted=True the agent DID emit directives but never closed: remind to close
with [RESULT] instead of rejecting the (good) action.
"""
if dry_run or not thread_id or thread_id == "main":
return
@@ -1440,14 +1506,23 @@ def maybe_nudge_untagged_sidechat(agent, thread_id, thread_name, mid, text, dry_
matching_job_id, matching_job_id, prompt_envelope.pick_profile(matching_job_id))
except Exception:
_spawn = '[TOOL swarm.spawn {"count": 2, "task": "continue the job work"}]'
nudge_msg = (
f"{_spawn}\n"
f"[STRICT ENFORCEMENT: Conversational commentary is rejected. Work requires active execution.]\n"
f"Thread Console: {thread_url}\n"
f"Emit executable tool calls now: [TOOL <op> <args>] or curl against https://exec.muse-dev.online/exec\n"
f"When all operations are finished, close strictly with [RESULT {matching_job_id}] <outcome>.\n"
f"{_spawn}"
)
if acted:
nudge_msg = (
f"Action received — now close the loop: reply with [RESULT {matching_job_id}] <outcome>.\n"
f"Outcome needs checkable evidence (swarm/timer IDs, paths, or command output), not prose alone.\n"
f"Thread Console: {thread_url}"
)
else:
nudge_msg = (
f"{_spawn}\n"
f"[STRICT ENFORCEMENT: Conversational commentary is rejected. Work requires active execution.]\n"
f"Thread Console: {thread_url}\n"
f"EMIT tool calls verbatim in your reply — you do not run them yourself;"
f" the Box runtime on bl executes each directive and posts the result back here"
f" (works from containers with no box CLI). Or curl against https://exec.muse-dev.online/exec\n"
f"When all operations are finished, close strictly with [RESULT {matching_job_id}] <outcome>.\n"
f"{_spawn}"
)
try:
import muse_hybrid
print(f"[{agent}] Injecting 1-turn strict nudge into {thread_name or thread_id[:8]} for job {matching_job_id}")