feat: completion-enforcement loop (fallback, proof, emit-model, auditor)

Close the loop so dispatched work actually completes on bl:

- on_no_result fallback in followup-sweeper (op + job forms via
  exec-constrained registry / job-dispatch), seeded on the three
  autonomy-pulse jobs; fallback_due() dedupes the gravity path
- gravity.py: add __main__ entry (loop-remediator.timer was a no-op),
  300s re-arm budget, fallback firing + stamp/skip logic
- harvester: proof-of-result followups (result_has_evidence),
  acted-variant NACK, emit-model tool-hint wording
- envelope: RESPONSE RULE states the emit model (agents EMIT
  directives verbatim; runtime executes; works from bare containers)
- completion-audit.py + systemd 15-min timer: per-family funnel,
  swarm drain, followup backlog; digest DM when degraded, 6h heartbeat
- tests/test_completion.py (29 tests), JOB-SPEC.md docs

Tests: 67/67 focused green (completion + tool_calls).
This commit is contained in:
Muse Sidechat
2026-10-06 08:20:14 +00:00
parent b7e45010c3
commit a9f014f9fa
12 changed files with 1173 additions and 16 deletions
+185
View File
@@ -144,6 +144,164 @@ def _resolve_nudge_thread_uuid(nudge_output):
return fallback
_JOB_ID_RE = re.compile(r"^(.+)-(\d{8})-(\d{6})-([0-9a-f]{8})$")
_OP_NAME_RE = re.compile(r"^[a-z][a-z0-9_.]{0,63}$")
_JOB_NAME_RE = re.compile(r"^[a-z0-9-]{1,64}$")
_FALLBACK_RETRY_S = 3600
def fallback_due(rec, now=None):
"""True when a terminal followup should (re)attempt its on_no_result fallback.
Fires once per record; a failed attempt may retry after _FALLBACK_RETRY_S.
Shared by the sweeper terminal path and gravity remediate so the two
firing paths can never double-execute.
"""
fb = rec.get("fallback") or {}
if fb.get("ran"):
return False
ts = fb.get("ts")
if not ts:
return True
try:
last = datetime.fromisoformat(str(ts).replace("Z", "+00:00"))
except Exception:
return True
if last.tzinfo is None:
last = last.replace(tzinfo=timezone.utc)
base = now or utcnow_dt()
return (base - last).total_seconds() >= _FALLBACK_RETRY_S
_EXEC_OPS_MOD = None
def derive_job_name(job_id):
"""Extract the job name from a dispatched job_id (<name>-YYYYMMDD-HHMMSS-<hex8>)."""
m = _JOB_ID_RE.match(job_id or "")
if not m:
return None
name = m.group(1)
if not (JOBS_DIR / f"{name}.json").exists():
return None
return name
def load_job_fallback(job_name):
"""Return (spec, error) for a job's on_no_result fallback.
spec is None when the job declares none. Shape:
{"job": "<job-name>"} -> dispatch a fallback job, or
{"op": "<exec-op>", "args": {...}} -> run one exec-constrained op.
"""
try:
with open(JOBS_DIR / f"{job_name}.json", "r", encoding="utf-8") as f:
cfg = json.load(f)
except Exception as e:
return None, f"unreadable job {job_name}: {e}"
spec = cfg.get("on_no_result")
if spec is None:
return None, None
if not isinstance(spec, dict) or set(spec) - {"job", "op", "args"}:
return None, "on_no_result must be an object with job|op (+args)"
if bool(spec.get("job")) == bool(spec.get("op")):
return None, "on_no_result needs exactly one of job|op"
if spec.get("job"):
jn = spec["job"]
if not isinstance(jn, str) or not _JOB_NAME_RE.fullmatch(jn):
return None, "on_no_result.job must be a valid job name"
if not (JOBS_DIR / f"{jn}.json").exists():
return None, f"on_no_result.job {jn!r} does not exist"
else:
if not isinstance(spec.get("op"), str) or not _OP_NAME_RE.fullmatch(spec["op"]):
return None, "on_no_result.op must be a valid op name"
if "args" in spec and not isinstance(spec["args"], dict):
return None, "on_no_result.args must be an object"
return spec, None
def _load_exec_ops():
global _EXEC_OPS_MOD
if _EXEC_OPS_MOD is None:
import importlib.util
mod_spec = importlib.util.spec_from_file_location(
"exec_constrained_sweeper", str(BIN_DIR / "exec-constrained.py"))
mod = importlib.util.module_from_spec(mod_spec)
mod_spec.loader.exec_module(mod)
_EXEC_OPS_MOD = mod
return _EXEC_OPS_MOD
def run_no_result_fallback(rec, dry_run=False):
"""Execute a job's on_no_result fallback at terminal followup expiry.
Returns an outcome dict; never raises (failures are outcome data so
one bad spec can't break the sweep).
"""
dm_id = rec.get("dm_id", "?")
outcome = {"dm_id": dm_id, "ran": False, "mode": None,
"configured": False, "detail": "no job fallback"}
try:
job_name = derive_job_name(rec.get("job_id"))
if not job_name:
outcome["detail"] = "no resolvable job_id"
return outcome
spec, err = load_job_fallback(job_name)
if err:
outcome.update(configured=True, detail=err)
return outcome
if spec is None:
return outcome
outcome["configured"] = True
if dry_run:
outcome.update(mode="dry_run", detail=json.dumps(spec)[:200])
return outcome
if spec.get("job"):
env = os.environ.copy()
env["CHAIN_PREV_JOB_ID"] = rec.get("job_id", "")
env["CHAIN_PREV_RESULT"] = (
f"TIMEOUT: Agent {rec.get('recipient')} gave no result; "
f"on_no_result fallback for job {job_name}")
cmd = [sys.executable, str(DISPATCH_PY), spec["job"]]
try:
p = subprocess.run(cmd, capture_output=True, text=True,
timeout=180, env=env)
except Exception as e:
outcome.update(mode="job", detail=f"dispatch exception: {e}")
return outcome
ok = p.returncode == 0
outcome.update(ran=ok, mode="job",
detail=(f"dispatched {spec['job']}" if ok
else f"dispatch failed: {(p.stderr or p.stdout).strip()[:200]}"))
else:
mod = _load_exec_ops()
op = spec["op"]
op_spec = mod.OPS.get(op)
if op_spec is None:
outcome["detail"] = f"unknown op: {op}"
return outcome
args = dict(spec.get("args") or {})
try:
clean = op_spec["validate"](args)
except Exception as e:
outcome["detail"] = f"op validation failed: {e}"
return outcome
argv = op_spec["build"](clean)
try:
p = subprocess.run(argv, capture_output=True, text=True,
timeout=op_spec.get("timeout", 120))
except Exception as e:
outcome.update(mode="op", detail=f"op exception: {e}")
return outcome
ok = p.returncode == 0
out = (p.stdout or p.stderr or "").strip()
outcome.update(ran=ok, mode="op",
detail=(f"{op} ok: {out[:200]}" if ok
else f"{op} failed rc={p.returncode}: {out[:200]}"))
except Exception as e:
outcome["detail"] = f"fallback exception: {e}"
return outcome
def sweep_cycle(dry_run=False):
followups = load_followups()
if not followups:
@@ -152,6 +310,7 @@ def sweep_cycle(dry_run=False):
now = utcnow_dt()
nudges_count = 0
escalations_count = 0
fallbacks_count = 0
modified = False
for dm_id, rec in list(followups.items()):
@@ -334,6 +493,31 @@ def sweep_cycle(dry_run=False):
pipeline_engine.fail_pipeline(run_entry.get("run_id"), "step_timed_out_without_fallback")
except Exception:
pass
# on_no_result fallback: the agent never replied, so run
# the job's declared server-side effect now (if any).
# fallback_due() dedupes against the gravity firing path.
fb = (run_no_result_fallback(rec) if fallback_due(rec)
else {"configured": False, "ran": False, "mode": None,
"detail": "fallback already ran"})
if fb["configured"]:
rec["fallback"] = {"ran": fb["ran"], "mode": fb["mode"],
"detail": fb["detail"][:200],
"ts": utcnow_str()}
modified = True
append_job_log({
"ts": utcnow_str(),
"type": ("fallback_executed" if fb["ran"]
else "fallback_failed"),
"dm_id": dm_id,
"recipient": recipient,
"job_id": rec.get("job_id"),
"mode": fb["mode"],
"detail": fb["detail"][:300],
})
print(f"Sweeper: on_no_result fallback for {dm_id}: "
f"ran={fb['ran']} {fb['detail'][:120]}")
if fb["ran"]:
fallbacks_count += 1
else:
escalations_count += 1
@@ -346,6 +530,7 @@ def sweep_cycle(dry_run=False):
"pending": pending_count,
"nudges_sent": nudges_count,
"escalations": escalations_count,
"fallbacks": fallbacks_count,
}