fix: placement verification and sidechat policy enforcement

- dm.py: placement-aware verify (verify_placement), checked_uuid logging, placement_mismatch events
- main-chat-watchdog.py: P0 alerts on placement_mismatch
- box-ctl.py: notify routes to sidechat, box policy command
- jobs: ops-audit and pipe-demo use sidechats
- response-harvester.py: chain deduplication

Session: sidechat/ops-restore
This commit is contained in:
operator-main
2026-10-04 18:29:13 +00:00
parent ae1bf17de5
commit 550e30901b
14 changed files with 1122 additions and 47 deletions
+70 -1
View File
@@ -82,7 +82,9 @@ def append_job_log(entry):
def send_dm(sender, recipient, target, text):
"""Dispatch a DM via dm.py."""
"""Dispatch a DM via dm.py. The sweeper's delivery target is explicit
followup state (or explicit in-code escalation routing), so pass the
main-chat opt-in when the target is main (sidechat-first policy)."""
cmd = [
sys.executable,
str(DM_PY),
@@ -90,6 +92,7 @@ def send_dm(sender, recipient, target, text):
"--agent", sender,
"--to", recipient,
"--target", target,
] + (["--allow-main-chat"] if target == "main" else []) + [
text,
]
try:
@@ -125,6 +128,34 @@ def sweep_cycle(dry_run=False):
orig_target = rec.get("target", "main")
thread_uuid = rec.get("thread_uuid")
# Ghost-followup fail-fast (2026-10-04, operator-main): a null
# thread_uuid means the sidechat was never provisioned, so the
# harvester can never match a reply. Nudging is pointless -- flag
# for manual triage once instead of burning the nudge budget and
# escalating a ghost. Main-chat followups are unaffected (the
# harvester matches those by target).
if thread_uuid is None and orig_target != "main" and not rec.get("needs_review"):
rec["needs_review"] = True
rec["status"] = "needs_review"
rec["review_reason"] = (
"ghost: unresolvable, manual triage "
f"(thread_uuid null, target={orig_target}; "
"reply can never auto-resolve)"
)
modified = True
append_job_log({
"ts": utcnow_str(),
"type": "followup_ghost_suppressed",
"dm_id": dm_id,
"recipient": recipient,
"target": orig_target,
"reason": "ghost: unresolvable, manual triage",
})
print(f"Sweeper: suppressing ghost followup {dm_id} "
f"(thread_uuid null, target={orig_target}) -> needs_review",
file=sys.stderr)
continue
if nudges_sent < nudges_allowed:
# Deliver next nudge
nudge_num = nudges_sent + 1
@@ -161,6 +192,44 @@ def sweep_cycle(dry_run=False):
})
else:
print(f"Sweeper WARNING: nudge send failed: {out}", file=sys.stderr)
# Failure-mode fix (2026-10-04): advance state on send
# failure so a failing nudge is never re-fired every
# timer tick. failed_sends is tracked separately from
# nudges_sent so a delivery failure does not consume a
# real nudge. Backoff: 5 min base, doubling per failure.
failed = rec.get("failed_sends", 0) + 1
rec["failed_sends"] = failed
rec["last_failure_at"] = utcnow_str()
backoff_s = 300 * (2 ** min(failed - 1, 4)) # 5m,10m,20m,40m,80m cap
rec["deadline"] = (now + timedelta(seconds=backoff_s)).isoformat()
modified = True
append_job_log({
"ts": utcnow_str(),
"type": "followup_nudge_failed",
"dm_id": dm_id,
"nudge_num": nudge_num,
"recipient": recipient,
"target": delivery_target,
"failed_sends": failed,
"backoff_s": backoff_s,
"error": str(out)[:200],
})
# After 3 consecutive failures, stop retrying blindly and
# flag for manual review (delivery may be uncertain or the
# target may be permanently broken).
if failed >= 3:
rec["needs_review"] = True
rec["review_reason"] = (
f"nudge send failed {failed} times consecutively "
f"(last: {str(out)[:120]})"
)
append_job_log({
"ts": utcnow_str(),
"type": "followup_needs_review",
"dm_id": dm_id,
"recipient": recipient,
"failed_sends": failed,
})
else:
nudges_count += 1