chore(fleet): sync operator memory, hatch menu dialogs, and watchdog alerts

This commit is contained in:
operator
2026-10-07 00:25:51 +00:00
parent 0065d11e97
commit 34b0ef9fe2
38 changed files with 2205 additions and 444 deletions
+62
View File
@@ -35,6 +35,58 @@ TARGET_MD_FILES = [
"IDENTITY.md",
]
class MDValidationError(ValueError):
"""An md account/filename/path failed safety validation.
box-ctl.py maps this to BAD_NAME; it is always raised before any
gateway call or filesystem write.
"""
MD_ACCOUNT_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_-]{0,31}$")
MD_FILENAME_RE = re.compile(r"^[A-Za-z0-9_.-]{1,128}$")
MD_SUBPATH_RE = re.compile(r"^[A-Za-z0-9_.-]+(/[A-Za-z0-9_.-]+)*$")
def validate_account(account: str) -> str:
"""Reject account values that could escape the cookies/config path."""
if not isinstance(account, str) or not MD_ACCOUNT_RE.fullmatch(account):
raise MDValidationError(
"Invalid agent account %r: must match ^[A-Za-z0-9][A-Za-z0-9_-]{0,31}$"
% (account,))
return account
def validate_filename(filename: str, template_only: bool = False) -> str:
"""Reject filenames that could escape the md directory.
Template flows (diff/amend/append/pull) additionally require one of
TARGET_MD_FILES, since they index into shared/operators/.
"""
if template_only:
if filename not in TARGET_MD_FILES:
raise MDValidationError(
"Unknown shared template %r: must be one of %s"
% (filename, sorted(TARGET_MD_FILES)))
return filename
if not isinstance(filename, str) or filename in (".", "..") \
or not MD_FILENAME_RE.fullmatch(filename):
raise MDValidationError(
"Invalid filename %r: plain basename, no directories" % (filename,))
return filename
def validate_subpath(path: str) -> str:
"""Reject list paths that escape the container root ('' = root)."""
if path in (None, ""):
return ""
if not isinstance(path, str) or not MD_SUBPATH_RE.fullmatch(path) \
or ".." in path.split("/"):
raise MDValidationError(
"Invalid list path %r: subdir without '..'" % (path,))
return path
# Tunnel / Port inventory
TUNNEL_PORTS = {
"muse-main": {"port": 2224, "terminal": 7681, "user": "muse"},
@@ -49,6 +101,7 @@ TUNNEL_PORTS = {
def get_gateway(account: str) -> "Gateway":
"""Obtain an authenticated Gateway connection for an account."""
validate_account(account)
if not Gateway:
raise RuntimeError("muse_cli.gateway module is not available")
conf_dir = Path.home() / ".config" / "muse-cli" / account
@@ -63,6 +116,7 @@ def get_gateway(account: str) -> "Gateway":
def list_files(account: str, path: str = "") -> list:
"""List files in the agent container filesystem via Hatch."""
path = validate_subpath(path)
gw = get_gateway(account)
res = gw.call_json("fs.list", body={"path": path})
return res.get("entries", [])
@@ -70,6 +124,7 @@ def list_files(account: str, path: str = "") -> list:
def read_md(account: str, filename: str, max_bytes: int = 200000) -> dict:
"""Read a markdown file from the agent container via Hatch."""
validate_filename(filename)
gw = get_gateway(account)
offset = 0
chunks = []
@@ -100,6 +155,7 @@ def read_md(account: str, filename: str, max_bytes: int = 200000) -> dict:
def write_md(account: str, filename: str, text: str, overwrite: bool = True, append: bool = False) -> dict:
"""Write content to a file in the agent container via Hatch."""
validate_filename(filename)
gw = get_gateway(account)
body = {
"path": filename,
@@ -121,6 +177,8 @@ def write_md(account: str, filename: str, text: str, overwrite: bool = True, app
def audit_agents(accounts: list = None) -> dict:
"""Audit markdown files and operational DRIVE across fleet agents."""
accounts = accounts or VALID_ACCOUNTS
for acct in accounts:
validate_account(acct)
results = {}
for acct in accounts:
@@ -213,6 +271,7 @@ def audit_agents(accounts: list = None) -> dict:
def diff_md(account: str, filename: str) -> dict:
"""Compare an agent's container file against the shared operator template."""
validate_filename(filename, template_only=True)
local_path = SHARED_OPERATORS / filename
if not local_path.exists():
raise FileNotFoundError(f"Local template {local_path} not found")
@@ -242,6 +301,7 @@ def diff_md(account: str, filename: str) -> dict:
def amend_md(filename: str, content: str, author: str = "operator", reason: str = "") -> dict:
"""Amend a centralized shared operator template in shared/operators/ with safety validation and git commit."""
import subprocess
validate_filename(filename, template_only=True)
local_path = SHARED_OPERATORS / filename
if not local_path.exists():
@@ -296,6 +356,7 @@ def amend_md(filename: str, content: str, author: str = "operator", reason: str
def append_md(filename: str, text: str, author: str = "operator", section: str = None) -> dict:
"""Safely append an amendment or lesson to a centralized shared template."""
validate_filename(filename, template_only=True)
local_path = SHARED_OPERATORS / filename
if not local_path.exists():
raise FileNotFoundError(f"Shared operator file {filename} does not exist in {SHARED_OPERATORS}")
@@ -311,6 +372,7 @@ def append_md(filename: str, text: str, author: str = "operator", section: str =
def pull_md(account: str, filename: str) -> dict:
"""Pull the canonical centralized template from shared/operators/ into an agent's container."""
validate_filename(filename, template_only=True)
local_path = SHARED_OPERATORS / filename
if not local_path.exists():
raise FileNotFoundError(f"Shared operator file {filename} does not exist in {SHARED_OPERATORS}")
+136 -20
View File
@@ -13,10 +13,12 @@ Supports both:
4. Continuous watch & background integration into fleet status and loop health.
"""
import itertools
import json
import os
import re
import sys
import threading
import time
import urllib.request
from datetime import datetime, timezone
@@ -48,15 +50,28 @@ def load_responded_waits() -> dict:
return {}
def save_responded_waits(data: dict) -> None:
"""Persist the responded-waits map (best effort)."""
def _atomic_write_json(path: Path, data: dict) -> None:
"""Write JSON atomically via tmp + replace (best effort).
Plain write_text from concurrent writers (timers, box-ctl, TUI
threads) can interleave and corrupt the file; readers then fall
back to {} and silently drop state. Tmp names carry pid + thread
ident so concurrent writers never share a temp file.
"""
try:
RESPONDED_WAITS_FILE.parent.mkdir(parents=True, exist_ok=True)
RESPONDED_WAITS_FILE.write_text(json.dumps(data, indent=1))
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_name(f"{path.name}.tmp.{os.getpid()}.{threading.get_ident()}")
tmp.write_text(json.dumps(data, indent=1))
os.replace(tmp, path)
except Exception:
pass
def save_responded_waits(data: dict) -> None:
"""Persist the responded-waits map (best effort)."""
_atomic_write_json(RESPONDED_WAITS_FILE, data)
def load_first_seen_waits() -> dict:
"""Load map of when input waits were first observed: {node: {task: iso_timestamp}}."""
try:
@@ -69,11 +84,7 @@ def load_first_seen_waits() -> dict:
def save_first_seen_waits(data: dict) -> None:
"""Persist first-seen input waits map."""
try:
FIRST_SEEN_WAITS_FILE.parent.mkdir(parents=True, exist_ok=True)
FIRST_SEEN_WAITS_FILE.write_text(json.dumps(data, indent=1))
except Exception:
pass
_atomic_write_json(FIRST_SEEN_WAITS_FILE, data)
def is_wait_responded(node: str, task: str) -> bool:
@@ -466,9 +477,15 @@ def get_cdp_ws(node: str, page_idx: int = 0, timeout: float = 3.0):
return ws, target_page
_cdp_req_ids = itertools.count(1)
def cdp_evaluate(ws, js_expr: str, await_promise: bool = False, timeout: float = 3.0):
"""Evaluate a JavaScript expression via CDP Runtime.evaluate and return the result value."""
req_id = int(time.time() * 1000) % 100000
# Monotonic ids: millisecond-clock ids collide for rapid successive
# evaluates, letting a stale buffered response be misattributed to
# the wrong call (e.g. verify-after-click reading the click result).
req_id = next(_cdp_req_ids)
msg = {
"id": req_id,
"method": "Runtime.evaluate",
@@ -575,15 +592,33 @@ JS_INSPECT_APPROVALS = """(() => {
}
}
// Background queued approvals surface (e.g. "2 tasks need review", "Review")
const bgSurface = document.querySelector('[data-hatch-background-approval-surface="true"]');
let bgTasksCount = 0;
let bgText = '';
if (bgSurface) {
bgText = (bgSurface.innerText || '').trim();
const m = bgText.match(/(\\d+)\\s+tasks?\\s+need\\s+review/i);
if (m) {
bgTasksCount = parseInt(m[1], 10);
} else if (/a\\s+task\\s+needs\\s+review/i.test(bgText) || /tasks?\\s+need\\s+review/i.test(bgText)) {
bgTasksCount = 1;
}
}
const hasPendingApproval = (!!activeCard && (hasAllowOnce || hasDeny)) || (bgTasksCount > 0);
return JSON.stringify({
has_pending: !!activeCard && (hasAllowOnce || hasDeny),
card_text: cardText.slice(0, 1000),
has_pending: hasPendingApproval,
card_text: cardText.slice(0, 1000) || bgText,
buttons: buttons,
has_allow_once: hasAllowOnce,
has_allow_once: hasAllowOnce || (bgTasksCount > 0),
has_always_allow: hasAlwaysAllow,
has_deny: hasDeny,
history: historyBadges.slice(0, 5),
input_waits: inputWaits.slice(0, 10)
input_waits: inputWaits.slice(0, 10),
bg_tasks_count: bgTasksCount,
bg_text: bgText
});
})()"""
@@ -635,30 +670,68 @@ def inspect_node_approvals(node: str) -> dict:
"error": str(e),
"has_pending": False,
"host_cdp_ok": host_ok,
"title": "Node unreachable",
"purpose": "",
"ip": None,
"target": "-",
"is_trusted": False,
"buttons": [],
"has_allow_once": False,
"has_always_allow": False,
"has_deny": False,
"raw_text": "",
"history": [],
"input_waits": [],
"page_title": "",
"page_url": "",
"ws_url": "",
}
all_input_waits = []
first_page = pages[0]
last_err = None
inspected_ok = False
for page in pages:
ws_url = page.get("webSocketDebuggerUrl")
if not ws_url:
if last_err is None:
last_err = Exception("page has no webSocketDebuggerUrl")
continue
ws = None
try:
ws = websocket.create_connection(ws_url, timeout=2.0)
val_str = cdp_evaluate(ws, JS_INSPECT_APPROVALS, timeout=2.5)
if val_str and isinstance(val_str, str):
data = json.loads(val_str)
# If a background review banner is present and active card wasn't mounted, click review to reveal card
if data.get("bg_tasks_count", 0) > 0 and (not data.get("buttons") or "task" in (data.get("card_text") or "").lower()):
js_expand = """(() => {
const bgBtn = document.querySelector('[data-pel-click="chat_background_approval_review"]') ||
document.querySelector('[data-hatch-background-approval-surface="true"] button');
if (bgBtn) { bgBtn.click(); return 'CLICKED'; }
return 'NO_BTN';
})()"""
exp_res = cdp_evaluate(ws, js_expand, timeout=1.5)
if exp_res == "CLICKED":
time.sleep(0.35)
val_str2 = cdp_evaluate(ws, JS_INSPECT_APPROVALS, timeout=2.5)
if val_str2 and isinstance(val_str2, str):
val_str = val_str2
ws.close()
ws = None
if not val_str or not isinstance(val_str, str):
if last_err is None:
last_err = Exception("empty or invalid CDP evaluate result")
continue
data = json.loads(val_str)
inspected_ok = True
if data.get("input_waits"):
all_input_waits.extend(data["input_waits"])
if data.get("has_pending"):
card_text = data.get("card_text", "")
bg_tasks_count = data.get("bg_tasks_count", 0)
ip = None
target = None
m_t = re.search(
@@ -678,8 +751,14 @@ def inspect_node_approvals(node: str) -> dict:
target = m_domain.group(0)
lines = [line.strip() for line in card_text.split("\n") if line.strip()]
title = redact_sensitive(lines[0] if lines else "Permission request")
purpose = redact_sensitive(lines[1] if len(lines) > 1 else "")
if not lines and bg_tasks_count:
title = f"{bg_tasks_count} task(s) need review"
purpose = "Background tasks held up on review surface. Click 'Review' or allow to inspect."
else:
title = redact_sensitive(lines[0] if lines else "Permission request")
purpose = redact_sensitive(lines[1] if len(lines) > 1 else "")
if bg_tasks_count > 0 and "need review" not in purpose.lower() and "need review" not in title.lower():
purpose = f"{purpose} [{bg_tasks_count} queued task(s) awaiting review]".strip()
is_trusted = is_trusted_target(target or ip, card_text)
@@ -696,6 +775,7 @@ def inspect_node_approvals(node: str) -> dict:
"has_allow_once": data.get("has_allow_once", False),
"has_always_allow": data.get("has_always_allow", False),
"has_deny": data.get("has_deny", False),
"bg_tasks_count": bg_tasks_count,
"raw_text": redact_sensitive(card_text),
"history": data.get("history", []),
"input_waits": all_input_waits,
@@ -789,12 +869,24 @@ def inspect_node_approvals(node: str) -> dict:
"key_request": key_req,
}
status = "INPUT_WAIT" if unique_waits else ("ERROR" if last_err and not first_page else "CLEAR")
# A node whose pages all failed inspection must report ERROR, never a
# false CLEAR that hides pending approvals. (The old `last_err and not
# first_page` guard was dead: first_page is always truthy here.)
if unique_waits:
status = "INPUT_WAIT"
title = "No pending approvals"
elif not inspected_ok:
status = "ERROR"
title = "Approval inspection failed"
else:
status = "CLEAR"
title = "No pending approvals"
return {
"node": node,
"status": status,
"error": str(last_err) if status == "ERROR" and last_err else "",
"has_pending": False,
"title": "No pending approvals",
"title": title,
"purpose": "",
"ip": None,
"target": "-",
@@ -936,23 +1028,47 @@ def allow_node_approval(node: str, always: bool = False, force: bool = False, ca
btn.click();
return 'CLICKED_ALLOW';
}
const bgBtn = document.querySelector('[data-pel-click="chat_background_approval_review"]') ||
document.querySelector('[data-hatch-background-approval-surface="true"] button');
if (bgBtn) {
bgBtn.click();
return 'CLICKED_REVIEW_SURFACE';
}
return 'NOT_FOUND';
})()"""
click_res = cdp_evaluate(ws, js_click, timeout=3.0)
if click_res == "CLICKED_REVIEW_SURFACE":
time.sleep(0.6)
click_res2 = cdp_evaluate(ws, """(() => {
const primary = document.querySelector('button[data-hatch-approval-primary-action="true"]');
if (primary) { primary.click(); return 'CLICKED_PRIMARY'; }
const btns = Array.from(document.querySelectorAll('button'));
const btn = btns.find(b => {
const t = (b.innerText||'').trim().toLowerCase();
return t === 'allow once' || t === 'allow';
});
if (btn) { btn.click(); return 'CLICKED_ALLOW'; }
return 'NOT_FOUND';
})()""", timeout=2.0)
if click_res2 != "NOT_FOUND":
click_res = click_res2
# Verify dismissal
time.sleep(0.8)
js_verify = """(() => {
const primary = document.querySelector('button[data-hatch-approval-primary-action="true"]');
if (primary) return 'STILL_PRESENT';
const headers = document.querySelectorAll('[data-testid="approval-panel-header"]');
return headers.length === 0 ? 'DISMISSED' : 'STILL_PRESENT';
if (headers.length > 0) return 'STILL_PRESENT';
const bgSurface = document.querySelector('[data-hatch-background-approval-surface="true"]');
return bgSurface ? 'QUEUED_PRESENT' : 'DISMISSED';
})()"""
verify_res = cdp_evaluate(ws, js_verify, timeout=2.0)
ws.close()
dismissed = verify_res == "DISMISSED"
dismissed = verify_res in ("DISMISSED", "QUEUED_PRESENT")
mode = "always" if always else "allow_once"
log_box_ctl(
"approval-allow",
+581 -4
View File
@@ -204,11 +204,78 @@ case "$cmd" in
args=$(python3 -c "import json, sys; print(json.dumps({'agent': sys.argv[1], 'target': sys.argv[2], 'limit': int(sys.argv[3])}))" "$AGENT" "$target" "$limit")
call_exec "dm.read" "$args"
;;
log)
limit=""
log_agent=""
while [ $# -gt 0 ]; do
case "$1" in
--agent) log_agent="$2"; shift 2 ;;
*) if [ -z "$limit" ]; then limit="$1"; fi; shift ;;
esac
done
limit="${limit:-20}"
args=$(python3 -c "import json,sys; lim=int(sys.argv[1]); ag=sys.argv[2]; print(json.dumps({'limit':lim,**({'agent':ag} if ag else {})}))" "$limit" "$log_agent")
call_exec "dm.log" "$args"
;;
ack)
to=""
from_agent=""
sidechat=""
allow_main=""
ref_id=""
while [ $# -gt 0 ]; do
case "$1" in
--to) to="$2"; shift 2 ;;
--sender|--from) from_agent="$2"; shift 2 ;;
--sidechat) sidechat="$2"; shift 2 ;;
--allow-main-chat) allow_main="1"; shift ;;
*) if [ -z "$ref_id" ]; then ref_id="$1"; fi; shift ;;
esac
done
ref_id="${ref_id:?usage: box dm ack <id> --to <agent> --sender <agent> [--sidechat <name>] [--allow-main-chat]}"
from_agent="${from_agent:-$AGENT}"
args=$(python3 -c "
import json, sys
ref, to, sender, sc, main = sys.argv[1:6]
args = {'id': ref, 'to': to, 'sender': sender}
if sc:
args['sidechat'] = sc
if main:
args['allow_main_chat'] = True
print(json.dumps(args))
" "$ref_id" "$to" "$from_agent" "$sidechat" "$allow_main")
call_exec "dm.ack" "$args"
;;
*)
echo "Usage: box dm send|read ..."
echo "Usage: box dm send|read|log|ack ..."
;;
esac
;;
notify)
target_agent="${1:?usage: box notify <agent> [--sidechat <name>] [--sender <agent>] <message...>}"
shift
sidechat=""
sender=""
while [ $# -gt 0 ]; do
case "$1" in
--sidechat) sidechat="$2"; shift 2 ;;
--sender|--from) sender="$2"; shift 2 ;;
*) break ;;
esac
done
msg="${*:?usage: box notify <agent> [--sidechat <name>] [--sender <agent>] <message...>}"
args=$(python3 -c "
import json, sys
agent, message, sc, sender = sys.argv[1:5]
args = {'agent': agent, 'message': message}
if sc:
args['sidechat'] = sc
if sender:
args['sender'] = sender
print(json.dumps(args))
" "$target_agent" "$msg" "$sidechat" "$sender")
call_exec "notify.send" "$args"
;;
thread)
sub="${1:-list}"
shift || true
@@ -262,6 +329,70 @@ case "$cmd" in
args=$(python3 -c "import json, sys; print(json.dumps({'job': sys.argv[1]}))" "$name")
call_exec "cron.run" "$args"
;;
put)
name="${1:?usage: box cron put <name> '<json-definition>'}"
json_def="${2:?usage: box cron put <name> '<json-definition>'}"
args=$(python3 -c "
import json, sys
try:
definition = json.loads(sys.argv[2])
except Exception as e:
sys.stderr.write('invalid job JSON: %s\n' % e)
sys.exit(2)
print(json.dumps({'name': sys.argv[1], 'definition': definition}))
" "$name" "$json_def")
call_exec "job.put" "$args"
;;
trigger)
name="${1:?usage: box cron trigger <name>}"
args=$(python3 -c "import json, sys; print(json.dumps({'name': sys.argv[1]}))" "$name")
call_exec "job.trigger" "$args"
;;
chain)
from_job="${1:?usage: box cron chain <from> <to> [--on-failure]}"
shift || true
on_failure=""
to_job=""
while [ $# -gt 0 ]; do
case "$1" in
--on-failure) on_failure="1"; shift ;;
*) if [ -z "$to_job" ]; then to_job="$1"; fi; shift ;;
esac
done
to_job="${to_job:?usage: box cron chain <from> <to> [--on-failure]}"
args=$(python3 -c "
import json, sys
frm, to, onfail = sys.argv[1:4]
args = {'from': frm, 'to': to}
if onfail:
args['on_failure'] = True
print(json.dumps(args))
" "$from_job" "$to_job" "$on_failure")
call_exec "job.chain" "$args"
;;
next)
job_id=""
success=""
while [ $# -gt 0 ]; do
case "$1" in
--success) success="1"; shift ;;
--fail) success="0"; shift ;;
*) if [ -z "$job_id" ]; then job_id="$1"; fi; shift ;;
esac
done
job_id="${job_id:?usage: box cron next <job-id> [--success|--fail]}"
args=$(python3 -c "
import json, sys
jid, success = sys.argv[1:3]
args = {'job_id': jid}
if success == '1':
args['success'] = True
elif success == '0':
args['success'] = False
print(json.dumps(args))
" "$job_id" "$success")
call_exec "job.next" "$args"
;;
timer-create)
name="${1:?usage: box cron timer-create <name>}"
args=$(python3 -c "import json, sys; print(json.dumps({'name': sys.argv[1]}))" "$name")
@@ -273,7 +404,7 @@ case "$cmd" in
call_exec "cron.timer_start" "$args"
;;
*)
echo "Usage: box cron runs|status|view|run|timer-create|timer-start ..."
echo "Usage: box cron runs|status|view|run|put|trigger|chain|next|timer-create|timer-start ..."
;;
esac
;;
@@ -291,6 +422,16 @@ case "$cmd" in
args=$(python3 -c "import json, sys; print(json.dumps({'name': sys.argv[1]}))" "$name")
call_exec "cron.timer_start" "$args"
;;
stop)
name="${1:?usage: box timer stop <name>}"
args=$(python3 -c "import json, sys; print(json.dumps({'name': sys.argv[1]}))" "$name")
call_exec "cron.timer_stop" "$args"
;;
disable)
name="${1:?usage: box timer disable <name>}"
args=$(python3 -c "import json, sys; print(json.dumps({'name': sys.argv[1]}))" "$name")
call_exec "cron.timer_disable" "$args"
;;
status|view|list)
name="${1:-heartbeat}"
args=$(python3 -c "import json, sys; print(json.dumps({'name': sys.argv[1]}))" "$name")
@@ -304,7 +445,7 @@ case "$cmd" in
call_exec "followup.create" "$args"
;;
*)
echo "Usage: box timer create|start|status <name> OR box timer in <minutes> <prompt>"
echo "Usage: box timer create|start|stop|enable|disable|status <name> OR box timer in <minutes> <prompt>"
;;
esac
;;
@@ -353,6 +494,125 @@ case "$cmd" in
;;
esac
;;
loop)
sub="${1:?usage: box loop remediate|resolve ...}"
shift || true
case "$sub" in
remediate)
dry=""
while [ $# -gt 0 ]; do
case "$1" in
--dry-run) dry="1"; shift ;;
*) break ;;
esac
done
if [ -n "$dry" ]; then
args='{"dry_run": true}'
else
args='{}'
fi
call_exec "loop.remediate" "$args"
;;
resolve)
dm_id="${1:?usage: box loop resolve <dm_id> [note...]}"
shift || true
note="$*"
args=$(python3 -c "
import json, sys
dm_id, note = sys.argv[1:3]
args = {'dm_id': dm_id}
if note:
args['note'] = note
print(json.dumps(args))
" "$dm_id" "$note")
call_exec "loop.resolve" "$args"
;;
*)
echo "Usage: box loop remediate [--dry-run] OR box loop resolve <dm_id> [note...]"
;;
esac
;;
strat)
sub="${1:?usage: box strat set|reset ...}"
shift || true
case "$sub" in
set)
stype="${1:?usage: box strat set <type> [options]}"
shift || true
subtype=""
agent=""
track=""
priority=""
timeout_s=""
nudges=""
escalate=""
while [ $# -gt 0 ]; do
case "$1" in
--subtype) subtype="$2"; shift 2 ;;
--agent) agent="$2"; shift 2 ;;
--track) track="$2"; shift 2 ;;
--priority) priority="$2"; shift 2 ;;
--timeout) timeout_s="$2"; shift 2 ;;
--nudges) nudges="$2"; shift 2 ;;
--escalate) escalate="$2"; shift 2 ;;
*) break ;;
esac
done
args=$(python3 -c "
import json, sys
stype, subtype, agent, track, prio, timeout_s, nudges, esc = sys.argv[1:9]
args = {'type': stype}
if subtype:
args['subtype'] = subtype
if agent:
args['agent'] = agent
if track.lower() == 'true':
args['track'] = True
elif track.lower() == 'false':
args['track'] = False
elif track:
sys.stderr.write('track must be true|false\n')
sys.exit(2)
if prio:
args['priority'] = prio
if timeout_s:
args['timeout_s'] = int(timeout_s)
if nudges:
args['nudges'] = int(nudges)
if esc:
args['escalate'] = esc
print(json.dumps(args))
" "$stype" "$subtype" "$agent" "$track" "$priority" "$timeout_s" "$nudges" "$escalate")
call_exec "strat.set" "$args"
;;
reset)
stype="${1:?usage: box strat reset <type> [subtype] [--agent <agent>]}"
shift || true
subtype=""
agent=""
while [ $# -gt 0 ]; do
case "$1" in
--agent) agent="$2"; shift 2 ;;
*) if [ -z "$subtype" ]; then subtype="$1"; fi; shift ;;
esac
done
args=$(python3 -c "
import json, sys
stype, subtype, agent = sys.argv[1:4]
args = {'type': stype}
if subtype:
args['subtype'] = subtype
if agent:
args['agent'] = agent
print(json.dumps(args))
" "$stype" "$subtype" "$agent")
call_exec "strat.reset" "$args"
;;
*)
echo "Usage: box strat set <type> [options] OR box strat reset <type> [subtype] [--agent <agent>]"
;;
esac
;;
vars)
sub="${1:-list}"
shift || true
@@ -371,8 +631,26 @@ case "$cmd" in
args=$(python3 -c "import json, sys; print(json.dumps({'name': sys.argv[1], 'value': sys.argv[2]}))" "$name" "$val")
call_exec "vars.set" "$args"
;;
reset)
name="${1:?usage: box vars reset <name>}"
args=$(python3 -c "import json, sys; print(json.dumps({'name': sys.argv[1]}))" "$name")
call_exec "vars.reset" "$args"
;;
rollback)
name="${1:?usage: box vars rollback <name> [revision]}"
rev="${2:-}"
args=$(python3 -c "
import json, sys
name, rev = sys.argv[1:3]
args = {'name': name}
if rev:
args['revision'] = int(rev) if rev.isdigit() else rev
print(json.dumps(args))
" "$name" "$rev")
call_exec "vars.rollback" "$args"
;;
*)
echo "Usage: box vars list|get|set ..."
echo "Usage: box vars list|get|set|reset|rollback ..."
;;
esac
;;
@@ -430,6 +708,272 @@ case "$cmd" in
;;
esac
;;
git)
sub="${1:-status}"
shift || true
case "$sub" in
status)
call_exec "git.status" "{}"
;;
diff)
stat=""
path=""
while [ $# -gt 0 ]; do
case "$1" in
--stat) stat="1"; shift ;;
--path) path="$2"; shift 2 ;;
*) if [ -z "$path" ]; then path="$1"; fi; shift ;;
esac
done
args=$(python3 -c "
import json, sys
stat, path = sys.argv[1:3]
args = {}
if stat:
args['stat'] = True
if path:
args['path'] = path
print(json.dumps(args))
" "$stat" "$path")
call_exec "git.diff" "$args"
;;
log)
limit=""
path=""
while [ $# -gt 0 ]; do
case "$1" in
--limit) limit="$2"; shift 2 ;;
--path) path="$2"; shift 2 ;;
*) if [ -z "$limit" ]; then limit="$1"; fi; shift ;;
esac
done
limit="${limit:-10}"
args=$(python3 -c "
import json, sys
limit, path = sys.argv[1:3]
args = {'limit': int(limit)}
if path:
args['path'] = path
print(json.dumps(args))
" "$limit" "$path")
call_exec "git.log" "$args"
;;
*)
echo "Usage: box git status|diff|log ..."
;;
esac
;;
tests)
sub="${1:-run}"
shift || true
case "$sub" in
run)
test_mod=""
filt=""
while [ $# -gt 0 ]; do
case "$1" in
--filter) filt="$2"; shift 2 ;;
*) if [ -z "$test_mod" ]; then test_mod="$1"; fi; shift ;;
esac
done
args=$(python3 -c "
import json, sys
mod, filt = sys.argv[1:3]
args = {}
if mod:
args['test'] = mod
if filt:
args['filter'] = filt
print(json.dumps(args))
" "$test_mod" "$filt")
call_exec "tests.run" "$args"
;;
*)
echo "Usage: box tests run [tests.<module>] [--filter <pattern>]"
;;
esac
;;
approvals)
sub="${1:-check}"
shift || true
case "$sub" in
check)
node="${1:-}"
if [ -n "$node" ]; then
args=$(python3 -c "import json, sys; print(json.dumps({'node': sys.argv[1]}))" "$node")
else
args="{}"
fi
call_exec "approval.check" "$args"
;;
allow)
node=""
message=""
main_chat=""
while [ $# -gt 0 ]; do
case "$1" in
--message) message="$2"; shift 2 ;;
--allow-main-chat) main_chat="1"; shift ;;
*) if [ -z "$node" ]; then node="$1"; fi; shift ;;
esac
done
node="${node:?usage: box approvals allow <node> --message <text> [--allow-main-chat]}"
[ -n "$message" ] || { echo "usage: box approvals allow <node> --message <text> [--allow-main-chat]" >&2; exit 2; }
args=$(python3 -c "
import json, sys
node, message, main = sys.argv[1:4]
args = {'node': node, 'message': message}
if main:
args['allow_main_chat'] = True
print(json.dumps(args))
" "$node" "$message" "$main_chat")
call_exec "approval.allow" "$args"
;;
deny)
node=""
message=""
main_chat=""
while [ $# -gt 0 ]; do
case "$1" in
--message) message="$2"; shift 2 ;;
--allow-main-chat) main_chat="1"; shift 2 ;;
*) if [ -z "$node" ]; then node="$1"; fi; shift ;;
esac
done
node="${node:?usage: box approvals deny <node> --message <text> [--allow-main-chat]}"
[ -n "$message" ] || { echo "usage: box approvals deny <node> --message <text> [--allow-main-chat]" >&2; exit 2; }
args=$(python3 -c "
import json, sys
node, message, main = sys.argv[1:4]
args = {'node': node, 'message': message}
if main:
args['allow_main_chat'] = True
print(json.dumps(args))
" "$node" "$message" "$main_chat")
call_exec "approval.deny" "$args"
;;
auto)
node="${1:-}"
if [ -n "$node" ]; then
args=$(python3 -c "import json, sys; print(json.dumps({'node': sys.argv[1]}))" "$node")
else
args="{}"
fi
call_exec "approval.auto" "$args"
;;
*)
echo "Usage: box approvals check|allow|deny|auto ..."
;;
esac
;;
md)
sub="${1:-audit}"
shift || true
case "$sub" in
audit)
args=$(python3 -c "
import json, sys
accts = [a for a in sys.argv[1:] if a]
print(json.dumps({'accounts': accts} if accts else {}))
" "$@")
call_exec "md.audit" "$args"
;;
list)
account="${1:?usage: box md list <account> [path]}"
path="${2:-}"
args=$(python3 -c "import json, sys; print(json.dumps({'account': sys.argv[1], 'path': sys.argv[2]}))" "$account" "$path")
call_exec "md.list" "$args"
;;
read)
account="${1:?usage: box md read <account> <filename>}"
filename="${2:?usage: box md read <account> <filename>}"
args=$(python3 -c "import json, sys; print(json.dumps({'account': sys.argv[1], 'filename': sys.argv[2]}))" "$account" "$filename")
call_exec "md.read" "$args"
;;
diff)
account="${1:?usage: box md diff <account> <filename>}"
filename="${2:?usage: box md diff <account> <filename>}"
args=$(python3 -c "import json, sys; print(json.dumps({'account': sys.argv[1], 'filename': sys.argv[2]}))" "$account" "$filename")
call_exec "md.diff" "$args"
;;
pull)
account="${1:?usage: box md pull <account> <filename>}"
filename="${2:?usage: box md pull <account> <filename>}"
args=$(python3 -c "import json, sys; print(json.dumps({'account': sys.argv[1], 'filename': sys.argv[2]}))" "$account" "$filename")
call_exec "md.pull" "$args"
;;
inject-drive)
account="${1:?usage: box md inject-drive <account>}"
args=$(python3 -c "import json, sys; print(json.dumps({'account': sys.argv[1]}))" "$account")
call_exec "md.inject_drive" "$args"
;;
sync-all)
call_exec "md.sync_all" "{}"
;;
amend)
filename="${1:?usage: box md amend <filename> (--content <text>|--file <path>) [--author <name>] [--reason <why>]}"
shift || true
content=""; content_src=""; author="operator"; reason=""
while [ $# -gt 0 ]; do
case "$1" in
--content) content="$2"; content_src="arg"; shift 2 ;;
--file) content="$2"; content_src="file"; shift 2 ;;
--author) author="$2"; shift 2 ;;
--reason) reason="$2"; shift 2 ;;
*) echo "usage: box md amend <filename> (--content <text>|--file <path>) [--author <name>] [--reason <why>]" >&2; exit 2 ;;
esac
done
[ -n "$content_src" ] || { echo "usage: box md amend <filename> (--content <text>|--file <path>) [--author <name>] [--reason <why>]" >&2; exit 2; }
args=$(python3 -c "
import json, sys
fn, src, val, author, reason = sys.argv[1:6]
content = open(val, encoding='utf-8').read() if src == 'file' else val
args = {'filename': fn, 'content': content, 'author': author}
if reason:
args['reason'] = reason
print(json.dumps(args))
" "$filename" "$content_src" "$content" "$author" "$reason")
call_exec "md.amend" "$args"
;;
append)
filename="${1:?usage: box md append <filename> (--content <text>|--file <path>) [--author <name>] [--section <header>]}"
shift || true
text=""; text_src=""; author="operator"; section=""
while [ $# -gt 0 ]; do
case "$1" in
--content) text="$2"; text_src="arg"; shift 2 ;;
--file) text="$2"; text_src="file"; shift 2 ;;
--author) author="$2"; shift 2 ;;
--section) section="$2"; shift 2 ;;
*) echo "usage: box md append <filename> (--content <text>|--file <path>) [--author <name>] [--section <header>]" >&2; exit 2 ;;
esac
done
[ -n "$text_src" ] || { echo "usage: box md append <filename> (--content <text>|--file <path>) [--author <name>] [--section <header>]" >&2; exit 2; }
args=$(python3 -c "
import json, sys
fn, src, val, author, section = sys.argv[1:6]
text = open(val, encoding='utf-8').read() if src == 'file' else val
args = {'filename': fn, 'text': text, 'author': author}
if section:
args['section'] = section
print(json.dumps(args))
" "$filename" "$text_src" "$text" "$author" "$section")
call_exec "md.append" "$args"
;;
*)
echo "Usage: box md audit|list|read|diff|pull|inject-drive|sync-all|amend|append ..."
;;
esac
;;
unread)
target_agent="${1:-}"
if [ -n "$target_agent" ]; then
args=$(python3 -c "import json, sys; print(json.dumps({'agent': sys.argv[1]}))" "$target_agent")
else
args="{}"
fi
call_exec "fleet.unread" "$args"
;;
health|fleet-status)
call_exec "health.check" "{}"
;;
@@ -464,21 +1008,54 @@ Usage:
box deploy pipeline <name>
box dm send --to <agent> [--target <target>] <message>
box dm read [<target=main>] [<limit=10>]
box dm log [<limit=20>] [--agent <agent>]
box dm ack <id> --to <agent> --sender <agent> [--sidechat <name>]
box notify <agent> [--sidechat <name>] [--sender <agent>] <message...>
box thread list [<agent>]
box thread view <thread_id> [<limit=15>]
box cron runs
box cron status [<name=heartbeat>]
box cron view <name>
box cron run <name>
box cron put <name> '<json-definition>'
box cron trigger <name>
box cron chain <from> <to> [--on-failure]
box cron next <job-id> [--success|--fail]
box timer stop <name>
box timer disable <name>
box vars list
box vars get <name>
box vars set <name> <value>
box vars reset <name>
box vars rollback <name> [revision]
box strat set <type> [--subtype S] [--agent A] [--track b] [--priority p] [--timeout N] [--nudges N] [--escalate E]
box strat reset <type> [subtype] [--agent <agent>]
box loop remediate [--dry-run]
box loop resolve <dm_id> [note...]
box files read <path> [lines=100]
box files write <path> <content>
box web fetch <url>
box service status <unit>
box service restart <unit>
box git status
box git diff [--stat] [--path <path>]
box git log [<limit=10>] [--path <path>]
box tests run [tests.<module>] [--filter <pattern>]
box md audit [accounts...]
box md list <account> [path]
box md read <account> <filename>
box md diff <account> <filename>
box md pull <account> <filename>
box md inject-drive <account>
box md sync-all
box md amend <filename> (--content <text>|--file <path>) [--author <name>] [--reason <why>]
box md append <filename> (--content <text>|--file <path>) [--author <name>] [--section <header>]
box approvals check [node]
box approvals allow <node> --message <text> [--allow-main-chat]
box approvals deny <node> --message <text> [--allow-main-chat]
box approvals auto [node]
box health
box unread [<agent>]
box ping
box ops
+33 -12
View File
@@ -13,10 +13,14 @@
# Pattern mirrors chromebox-watchdog.sh (stage-specific logging, rotation).
set -euo pipefail
LOCK="/tmp/cdp-relay-watchdog.lock"
exec 9>"$LOCK"
if ! flock -n 9; then
echo "[$(date -u +%FT%TZ)] another relay watchdog run in progress, skipping" >&2
exit 0
# Tests source this file with CDP_RELAY_WATCHDOG_LIB_ONLY=1: they call
# helpers without running checks, so no lock is needed.
if [ "${CDP_RELAY_WATCHDOG_LIB_ONLY:-}" != "1" ]; then
exec 9>"$LOCK"
if ! flock -n 9; then
echo "[$(date -u +%FT%TZ)] another relay watchdog run in progress, skipping" >&2
exit 0
fi
fi
NETVM_BIN="/home/super/Projects/NetVM/bin"
@@ -37,16 +41,16 @@ log() { echo "[$(date -u +%FT%TZ)] $*" | tee -a "$LOG"; }
# node -> "veth_ip:port" via netvm-names.sh (hash-derived, don't hardcode)
relay_target() {
local node="$1"
local node="$1" reg_port=""
# shellcheck disable=SC1091
. "$NETVM_BIN/netvm-names.sh"
netvm_names "$node" || return 1
# CDP_PORT_OVERRIDE pins registry ports; fall back to hash-derived
local port="${CDP_PORT_OVERRIDE:-$CDP_PORT}"
case "$node" in
muse) port=9410 ;; pip) port=9420 ;; 646) port=9430 ;; opm) port=9440 ;;
esac
echo "$PEER_IP:$port"
# The registry is the source of truth for ports (new nodes propagate
# automatically); netvm-names pinning is the fallback.
if reg_port=$("$NETVM_BIN/netvm-registry.py" "$node" 2>/dev/null); then
[ -n "$reg_port" ] && CDP_PORT="$reg_port"
fi
echo "$PEER_IP:$CDP_PORT"
}
node_port() { echo "${1##*:}"; }
@@ -103,8 +107,25 @@ restart_relay() {
fi
}
# Registry-driven node list: every active node gets relay supervision
# (the old hardcoded 4-node list left def/dev unsupervised — 2026-10-06).
watched_nodes() {
"$NETVM_BIN/netvm-registry.py" 2>/dev/null | cut -d: -f1
}
# Allow sourcing for tests without running checks.
if [ "${CDP_RELAY_WATCHDOG_LIB_ONLY:-}" = "1" ]; then
return 0 2>/dev/null || exit 0
fi
FAILED=0
for node in muse pip 646 opm; do
NODES="$(watched_nodes)"
if [ -z "$NODES" ]; then
log "FAIL_LOUD: node registry empty/unreadable, skipping run"
exit 1
fi
# shellcheck disable=SC2086 (intended word splitting: one node per word)
for node in $NODES; do
# Stage 1: host veth IP. Fail loud, skip relay restart (pointless).
if ! veth_healthy "$node"; then
read -r veth gw <<< "$(node_veth "$node")"
+26 -8
View File
@@ -2,10 +2,14 @@
# chrome-error-scan.sh - scan per-profile chrome logs for concerning patterns.
# Self-contained: scans, compares against watermark, reports only NEW matches.
#
# Usage: chrome-error-scan.sh [--json]
# Usage: chrome-error-scan.sh [--json] [--no-advance]
# Default output: "profile:new_count" lines for profiles with new matches,
# or "OK: no new errors" if clean.
# --json: output JSON {"profile": {"total": N, "new": M}, ...}
# --no-advance: report against the watermark WITHOUT advancing it.
# Peek-only read for high-frequency pollers (e.g. the web
# surface via `box-ctl chrome-errors --no-advance`). Runs
# without the flag keep the classic advance-on-read semantics.
#
# Watermark: /home/super/Projects/NetVM/chrome-error-watermark.json
# Patterns: FATAL, crash, segfault, out of memory (case-insensitive)
@@ -13,6 +17,16 @@
LOGDIR="/home/super/Projects/NetVM"
WATERMARK="$LOGDIR/chrome-error-watermark.json"
AS_JSON=0
NO_ADVANCE=0
for arg in "$@"; do
case "$arg" in
--json) AS_JSON=1 ;;
--no-advance) NO_ADVANCE=1 ;;
*) echo "chrome-error-scan.sh: unknown argument: $arg" >&2; exit 2 ;;
esac
done
# Gather current counts per profile (grep -c prints 0 with exit 1 on no match;
# the || true masks the exit code while preserving the "0" on stdout)
get_count() {
@@ -29,7 +43,7 @@ PIP_C=$(get_count pip)
N646_C=$(get_count 646)
OPM_C=$(get_count opm)
python3 - "$WATERMARK" "$MUSE_C" "$PIP_C" "$N646_C" "$OPM_C" "$1" <<'PYEOF'
python3 - "$WATERMARK" "$MUSE_C" "$PIP_C" "$N646_C" "$OPM_C" "$AS_JSON" "$NO_ADVANCE" <<'PYEOF'
import json, sys, os
watermark_path = sys.argv[1]
@@ -39,7 +53,8 @@ current = {
"646": int(sys.argv[4]),
"opm": int(sys.argv[5]),
}
as_json = len(sys.argv) > 6 and sys.argv[6] == "--json"
as_json = len(sys.argv) > 6 and sys.argv[6] == "1"
no_advance = len(sys.argv) > 7 and sys.argv[7] == "1"
# Load watermark (tolerate missing/corrupt file -> treat as all-zero)
watermark = {}
@@ -73,9 +88,12 @@ else:
if not any_new:
print("OK: no new errors")
# Update watermark atomically
tmp = watermark_path + ".tmp"
with open(tmp, "w") as f:
json.dump({"counts": current}, f, indent=2)
os.replace(tmp, watermark_path)
# Update watermark atomically (skipped in --no-advance peek mode: the
# caller gets a read-only view and the CLI's advance-on-read semantics are
# left untouched).
if not no_advance:
tmp = watermark_path + ".tmp"
with open(tmp, "w") as f:
json.dump({"counts": current}, f, indent=2)
os.replace(tmp, watermark_path)
PYEOF
+138 -24
View File
@@ -1,33 +1,52 @@
#!/usr/bin/env python3
"""
Side-chat to main-chat work siphon — detection rules.
Side-chat to main-chat work siphon — detection rules (REPAIRED, agent 2 of 5).
Monitors side chat messages and identifies "siphon-worthy" content:
work that should surface in main chat for visibility.
Fixes the false-positive ✅ COMPLETED relay at the source:
"Sending is disabled until this conversation can be verified." → COMPLETED
was caused by a SINGLE keyword ("verified") matching one regex.
Categories:
COMPLETED - work finished, results ready
BLOCKER - something is stuck, needs intervention
DECISION - a decision is needed from the user/operator
ALERT - health/security/urgency signal
MILESTONE - significant progress checkpoint
Repairs (see OUTPUT.md for rationale):
1. COMPLETED requires >= 2 DISTINCT pattern hits (weighted: the structured
`[RESULT ...] OK` marker counts 2 — it is the fleet's own machine-emitted
completion signal, far less ambiguous than a bare "done").
2. Negation guards: negation/failure-state words veto COMPLETED outright
(fail-closed: a negated completion claim is never relayed as complete).
3. Honest labeling: the fake "confidence 60%" (which literally meant "one
regex hit") is replaced by a keyword-hit count. SiphonHit.hits is the
authoritative field; `confidence` is kept for backward compatibility
but must NOT be rendered as a percentage anywhere user-facing.
4. Stale suppression: a message older than 15 minutes never relays as
COMPLETED. Pass message_ts (epoch seconds). monitor.py currently does
NOT pass a timestamp — agent 3 / the integrator must thread
message["ts"] through (see OUTPUT.md).
Detection is purely pattern-based (raw Python, no AI).
Each rule returns (category, confidence, summary) or None.
DO NOT overwrite the original detect.py with this file until the integrator
reconciles all 5 agents' outputs.
"""
import re
from dataclasses import dataclass
import time
from dataclasses import dataclass, field
from typing import Optional
@dataclass
class SiphonHit:
category: str # COMPLETED, BLOCKER, DECISION, ALERT, MILESTONE
confidence: float # 0.0 - 1.0
summary: str # one-line summary for main chat
thread_id: str # source side chat
message_id: str # source message
confidence: float # LEGACY — kept for API compatibility only.
# Do NOT render as "confidence NN%"; it is not a
# reliability measure. See `hits`.
hits: int = 0 # AUTHORITATIVE — distinct keyword-pattern hits
# (weighted; see COMPLETED_PATTERN_WEIGHTS).
summary: str = "" # one-line summary for main chat
thread_id: str = "" # source side chat
message_id: str = "" # source message
author: str = "" # INTEGRATOR (agent 3 absent) — the message's real
# author, plumbed from message["author"] by
# monitor.py. Empty = unknown; NEVER substitute the
# thread's registered agent silently (see
# format_siphon).
# Full message text is NOT stored here — main chat gets a summary
# plus a link back, never the full content (safety: no sensitive
# data siphoned verbatim).
@@ -42,6 +61,11 @@ COMPLETED_PATTERNS = [
re.compile(r'\b(merged|committed|pushed|published)\b', re.I),
]
# Weighted hits: the structured [RESULT] OK marker is the fleet's own
# machine-emitted completion signal — unambiguous enough to stand alone.
COMPLETED_PATTERN_WEIGHTS = {0: 1, 1: 1, 2: 2, 3: 1}
COMPLETED_MIN_WEIGHT = 2 # >= 2 distinct pattern hits (or one [RESULT] OK)
BLOCKER_PATTERNS = [
re.compile(r'\b(blocked|stuck|failing|broken|down|error|failed)\b', re.I),
re.compile(r'\b(need|needs|waiting)\s+(your|approval|input|decision)\b', re.I),
@@ -75,9 +99,29 @@ SUPPRESS_PATTERNS = [
re.compile(r'\[do not siphon\]', re.I), # explicit opt-out marker
]
# --- Negation guards: any match vetoes COMPLETED (fail-closed) ---
# A completion claim in the presence of negation / failure-state language
# is never relayed as ✅ COMPLETED, no matter how many keywords hit.
NEGATION_GUARDS = [
# explicit negation
re.compile(r'\b(not|never|no|nothing|none|neither|nor)\b', re.I),
re.compile(r"\b(do not|don't|didn't|doesn't|won't|can't|cannot|isn't|aren't|"
r"wasn't|weren't|haven't|hasn't|hadn't|couldn't|shouldn't)\b", re.I),
# incompleteness hedges
re.compile(r'\b(still|yet|pending|unfinished|incomplete)\b', re.I),
# failure-state words (a "completed" message containing these is suspect)
re.compile(r'\b(broken|failed|failing|failure|down|stuck|blocked|disabled|'
r'error|errors|crash|crashed)\b', re.I),
# hedging conjunctions ("deployed, but tests are red")
re.compile(r'\b(but|however|although|though)\b', re.I),
]
# Messages older than this never relay as COMPLETED (seconds).
COMPLETED_MAX_AGE_S = 15 * 60
def _match_score(text: str, patterns) -> float:
"""Return confidence based on how many patterns match."""
"""Legacy confidence for non-COMPLETED categories (unchanged)."""
hits = sum(1 for p in patterns if p.search(text))
if hits == 0:
return 0.0
@@ -85,6 +129,21 @@ def _match_score(text: str, patterns) -> float:
return min(0.95, 0.6 + (hits - 1) * 0.2)
def _completed_weight(text: str):
"""
Return (weighted_hits, distinct_hits, matched_pattern_indexes) for
COMPLETED_PATTERNS. Weighted: [RESULT] OK counts 2.
"""
matched = [i for i, p in enumerate(COMPLETED_PATTERNS) if p.search(text)]
weight = sum(COMPLETED_PATTERN_WEIGHTS.get(i, 1) for i in matched)
return weight, len(matched), matched
def _is_negated(text: str) -> bool:
"""True if any negation guard fires anywhere in the text."""
return any(p.search(text) for p in NEGATION_GUARDS)
def _extract_summary(text: str, max_len: int = 120) -> str:
"""Extract a safe one-line summary. Strips to first meaningful line."""
# Take first non-empty line, truncate
@@ -98,28 +157,58 @@ def _extract_summary(text: str, max_len: int = 120) -> str:
def detect(text: str, thread_id: str, message_id: str,
min_confidence: float = 0.6) -> Optional[SiphonHit]:
min_confidence: float = 0.6,
message_ts: Optional[float] = None) -> Optional[SiphonHit]:
"""
Check a side chat message for siphon-worthy content.
Returns SiphonHit or None.
message_ts: epoch seconds of the original message (optional). Messages
older than COMPLETED_MAX_AGE_S (15 min) never relay as COMPLETED.
NOTE: monitor.py does not currently pass a timestamp — agent 3 / the
integrator must thread message["ts"] through the detect() call.
"""
# Safety: suppress sensitive content
for p in SUPPRESS_PATTERNS:
if p.search(text):
return None
# Stale suppression applies to COMPLETED only.
completed_allowed = True
if message_ts is not None:
try:
age = time.time() - float(message_ts)
if age > COMPLETED_MAX_AGE_S:
completed_allowed = False
except (TypeError, ValueError):
pass # unparseable ts: proceed, do not fail closed on metadata
# COMPLETED: >=2 distinct weighted pattern hits, no negation, not stale.
completed_hits = 0
completed_conf = 0.0
if completed_allowed and not _is_negated(text):
weight, distinct, _ = _completed_weight(text)
if weight >= COMPLETED_MIN_WEIGHT:
completed_hits = weight
# legacy confidence kept for API compat; NOT a reliability measure
completed_conf = min(0.95, 0.6 + (distinct - 1) * 0.2)
candidates = [
("COMPLETED", _match_score(text, COMPLETED_PATTERNS)),
("BLOCKER", _match_score(text, BLOCKER_PATTERNS)),
("DECISION", _match_score(text, DECISION_PATTERNS)),
("ALERT", _match_score(text, ALERT_PATTERNS)),
("MILESTONE", _match_score(text, MILESTONE_PATTERNS)),
("COMPLETED", completed_conf, completed_hits),
("BLOCKER", _match_score(text, BLOCKER_PATTERNS),
sum(1 for p in BLOCKER_PATTERNS if p.search(text))),
("DECISION", _match_score(text, DECISION_PATTERNS),
sum(1 for p in DECISION_PATTERNS if p.search(text))),
("ALERT", _match_score(text, ALERT_PATTERNS),
sum(1 for p in ALERT_PATTERNS if p.search(text))),
("MILESTONE", _match_score(text, MILESTONE_PATTERNS),
sum(1 for p in MILESTONE_PATTERNS if p.search(text))),
]
# Sort by confidence descending; ALERT wins ties (safety: urgency first)
# Use negative confidence for descending, and ALERT as tiebreaker
candidates.sort(key=lambda x: (-x[1], 0 if x[0] == "ALERT" else 1))
best_cat, best_conf = candidates[0]
best_cat, best_conf, best_hits = candidates[0]
if best_conf < min_confidence:
return None
@@ -127,12 +216,37 @@ def detect(text: str, thread_id: str, message_id: str,
return SiphonHit(
category=best_cat,
confidence=best_conf,
hits=best_hits,
summary=_extract_summary(text),
thread_id=thread_id,
message_id=message_id,
)
def format_siphon(hit: SiphonHit, agent_name: str = "sidechat") -> str:
"""
Format a siphon message for main chat.
HONEST LABELING: reports keyword hit count, never a fake "confidence %".
HONEST AUTHORSHIP (integrator, agent 3 absent): attributes the message's
real author when known. Falls back to the thread's registered agent only
when the author is unknown — and says so explicitly, so a relay can
never again launder thread ownership as authorship.
"""
emoji = {"COMPLETED": "✅", "BLOCKER": "🚧", "DECISION": "❓",
"ALERT": "🚨", "MILESTONE": "🎯"}.get(hit.category, "📋")
thread_url = f"https://muse.ai/thread/{hit.thread_id}"
if hit.author:
attribution = f"from {hit.author}"
else:
attribution = f"from {agent_name} side chat (author unverified)"
return (
f"{emoji} [{hit.category}] {attribution}\n"
f"{hit.summary}\n"
f"→ {thread_url}\n"
f"(keyword hits: {hit.hits})"
)
# --- Opt-out registry ---
_opt_out_threads: set = set()
+9 -2
View File
@@ -19,6 +19,9 @@
# FLEET_ALERT_DRY_RUN=1 evaluate + print, write no state/outbox, no notify
# FLEET_ALERT_INJECT_FAIL= test hook: comma-separated condition ids to force-fail
# (e.g. FLEET_ALERT_INJECT_FAIL=cdp:pip)
# FLEET_BL_RELAY=1 re-enable the bl-side #lobby relay (default 0/off:
# the container-side hook is the live pager; running
# both double-posts every alert — 2026-10-06)
#
# State: ~/.local/share/fleet-alert/state.json (per-condition consecutive counters)
# Outbox: ~/.local/share/fleet-alert/outbox.jsonl (ALERT/RECOVERY records for the relay)
@@ -435,7 +438,11 @@ rm -f "$STATE_DIR/.alerts.tmp"
tail -500 "$LOG" > "$LOG.tmp" 2>/dev/null && mv "$LOG.tmp" "$LOG"
log "check complete"
# Relay pending outbox records to #lobby with idempotency gates (posted watermark + content hash TTL)
if [ "$DRY_RUN" -eq 0 ] && [ -x "$BIN/fleet-alert-relay.sh" ]; then
# Bl-side #lobby relay: DISABLED by default (FLEET_BL_RELAY=1 to re-enable).
# The container-side hook is the live pager; the bl relay never successfully
# posted (missing CHAT_KEYFILE) and enabling it now would double-post every
# alert in a second format. Re-enable only alongside retiring the container
# hook (and per the relay header, with opm sign-off).
if [ "${FLEET_BL_RELAY:-0}" = "1" ] && [ "$DRY_RUN" -eq 0 ] && [ -x "$BIN/fleet-alert-relay.sh" ]; then
"$BIN/fleet-alert-relay.sh" >> "$LOG" 2>&1 || true
fi
+25 -2
View File
@@ -5,6 +5,10 @@
# branch; deployment needs opm review + sign-off. See
# docs/FLEET-ALERT-DUP-POST-GATE.md.
#
# NOTE (2026-10-06): auto-invoke from fleet-alert-check.sh is disabled by
# default (FLEET_BL_RELAY=1 re-enables). The container-side hook pages
# #lobby today; do not re-enable without retiring it first.
#
# The 2026-10-05 11:28Z incident: one RECOVERY record in the outbox became two
# identical verified #lobby posts (seq 642/643, 3.35s apart) because the relay
# leg had no idempotency: append-only outbox, no consume tracking, no content
@@ -68,7 +72,7 @@ transport_post() { # $1 = text
local text="$1" ts sig payload resp http
ts="$(date +%s)"
if ! sig="$(sign_payload "$(printf '%s\n%s\n%s' "$ts" "$CHANNEL" "$text")")"; then
echo "UNKNOWN sign-failed"; return 0
echo "UNKNOWN sign-failed($KEYFILE)"; return 0
fi
payload="$(MSG="$text" TS="$ts" SIG="$sig" python3 -c '
import json,os
@@ -263,12 +267,31 @@ main() {
# NOTE: transport_post is invoked via command substitution (subshell), so the
# stub counts calls with a file, not a variable.
self_test() {
local td calls lobby ok=1 n
local td calls lobby ok=1 n sk sig_out old_key
td="$(mktemp -d)"; export FLEET_ALERT_DIR="$td"
ALERT_DIR="$td"; OUTBOX="$td/outbox.jsonl"; POSTED="$td/posted.log"
SEEN="$td/seen-hashes.log"; LOCKF="$td/relay.lock"
calls="$td/calls.log"; lobby="$td/lobby.log"
touch "$calls" "$lobby"
# sign_payload must round-trip with a valid key and fail cleanly without
# one (2026-10-06: missing ~/.ssh/id_frontdoor broke every #lobby post
# with an undiagnosable bare "sign-failed").
old_key="$KEYFILE"
sk="$td/signkey"
ssh-keygen -t ed25519 -f "$sk" -N '' -q >/dev/null 2>&1 \
|| { echo "FAIL: cannot generate ephemeral test key"; ok=0; }
if KEYFILE="$sk" sig_out="$(sign_payload "self-test")"; then
case "$sig_out" in
*"BEGIN SSH SIGNATURE"*) : ;;
*) echo "FAIL: sign_payload output not armored"; ok=0 ;;
esac
else
echo "FAIL: sign_payload failed with a valid key"; ok=0
fi
if KEYFILE="$td/no-such-key" sign_payload "self-test" >/dev/null 2>&1; then
echo "FAIL: sign_payload succeeded with a missing key"; ok=0
fi
KEYFILE="$old_key"
# Two identical submissions: same text, different record ids (the 11:28Z shape)
printf '%s\n' \
'{"id":"rec-A","ts":1791199616,"kind":"RECOVERY","condition":"partition:def"}' \
+28 -13
View File
@@ -168,14 +168,32 @@ def _eval(ws, js, timeout=8.0):
return None
VERIFY_TRIES = 10
VERIFY_PAUSE = 1.5
def _poll(check, tries=VERIFY_TRIES, pause=VERIFY_PAUSE):
"""Poll a state check until true. Fast exit; tolerates slow commits."""
for _ in range(tries):
try:
if check():
return True
except Exception:
pass
time.sleep(pause)
return False
def list_radios(ws):
"""All dialog radios with heading/label/value/checked (or [])."""
return _eval(ws, JS_LIST_RADIOS) or []
rows = _eval(ws, JS_LIST_RADIOS)
return rows if isinstance(rows, list) else []
def list_switches(ws):
"""All dialog switches with row label + checked (or [])."""
return _eval(ws, JS_LIST_SWITCHES) or []
rows = _eval(ws, JS_LIST_SWITCHES)
return rows if isinstance(rows, list) else []
def radio_state(ws, heading, value):
@@ -188,9 +206,9 @@ def radio_state(ws, heading, value):
def set_radio_by_heading(ws, heading, value):
"""Set a heading-grouped radio; verify, else trusted click, verify."""
"""Set a heading-grouped radio; poll, else trusted click, poll."""
if _eval(ws, JS_CLICK_RADIO % (heading, value)) == "CLICKED" \
and radio_state(ws, heading, value) is True:
and _poll(lambda: radio_state(ws, heading, value) is True):
return True
rect = _eval(ws, JS_RADIO_RECT % (heading, value))
if not rect or "x" not in rect:
@@ -199,8 +217,7 @@ def set_radio_by_heading(ws, heading, value):
real_click(ws, rect["x"], rect["y"])
except Exception:
return False
time.sleep(0.6)
return radio_state(ws, heading, value) is True
return _poll(lambda: radio_state(ws, heading, value) is True)
def radio_aria_state(ws, name):
@@ -214,9 +231,9 @@ def radio_aria_state(ws, name):
def set_radio_by_aria(ws, name):
"""Set an aria-labeled radio; verify, else trusted click, verify."""
"""Set an aria-labeled radio; poll, else trusted click, poll."""
if _eval(ws, JS_CLICK_RADIO_ARIA % name) == "CLICKED" \
and radio_aria_state(ws, name) is True:
and _poll(lambda: radio_aria_state(ws, name) is True):
return True
rect = _eval(ws, JS_RADIO_ARIA_RECT % name)
if not rect or "x" not in rect:
@@ -225,8 +242,7 @@ def set_radio_by_aria(ws, name):
real_click(ws, rect["x"], rect["y"])
except Exception:
return False
time.sleep(0.6)
return radio_aria_state(ws, name) is True
return _poll(lambda: radio_aria_state(ws, name) is True)
def switch_state(ws, label):
@@ -247,7 +263,7 @@ def set_switch(ws, label, on):
if state == bool(on):
return True
if _eval(ws, JS_CLICK_SWITCH % label) == "CLICKED" \
and switch_state(ws, label) is bool(on):
and _poll(lambda: switch_state(ws, label) is bool(on)):
return True
rect = _eval(ws, JS_SWITCH_RECT % label)
if not rect or "x" not in rect:
@@ -256,5 +272,4 @@ def set_switch(ws, label, on):
real_click(ws, rect["x"], rect["y"])
except Exception:
return False
time.sleep(0.6)
return switch_state(ws, label) is bool(on)
return _poll(lambda: switch_state(ws, label) is bool(on))
+15 -4
View File
@@ -6,6 +6,7 @@ idempotent (re-clicking the active tab is a harmless no-op), so goto
always clicks and reports the click result instead of guessing which
tab is active.
"""
import json
import time
from approvals import cdp_evaluate
@@ -16,6 +17,10 @@ TAB_NAMES = ["General", "Connectors", "Wallet", "Secure store",
"Data controls", "Help & support", "Legal info"]
TABS = TAB_NAMES # legacy alias
# Tab rail buttons carry bare tab names and live outside any nav
# landmark, so row matchers exclude them by exact text (live 2026-10-06).
_JS_TABS = json.dumps(TAB_NAMES)
DOCK_MORE_TESTID = "hatch-dock-more"
JS_DOCK_RECT = ("(() => { const b = document.querySelector("
@@ -41,21 +46,26 @@ JS_GOTO_TAB_TMPL = ("(() => { const b = Array.from(document."
JS_CLICK_ROW_TMPL = ("((name) => {"
" const d = document.querySelector('[role=\"dialog\"]');"
" if (!d) return 'NO_DIALOG';"
" const TABS = " + _JS_TABS + ";"
" const inNav = (el) => !!el.closest("
"'nav, [role=\"tablist\"], [role=\"navigation\"]');"
" const els = Array.from(d.querySelectorAll("
"'button, [role=\"button\"], a')).filter(e => !inNav(e));"
" const t = els.find(e => (e.innerText || '').trim()"
".toLowerCase().startsWith(name.toLowerCase()));"
" const t = els.find(e => {"
" const txt = (e.innerText || '').trim();"
" return !TABS.includes(txt) && txt.toLowerCase()"
".startsWith(name.toLowerCase()); });"
" if (!t) return 'NO_ROW'; t.click(); return 'CLICKED'; })('%s')")
JS_DESCRIBE_ROWS = ("(() => {"
" const d = document.querySelector('[role=\"dialog\"]');"
" if (!d) return null;"
" const TABS = " + _JS_TABS + ";"
" const inNav = (el) => !!el.closest("
"'nav, [role=\"tablist\"], [role=\"navigation\"]');"
" return Array.from(d.querySelectorAll("
"'button, [role=\"button\"], a')).filter(e => !inNav(e))"
".filter(e => !TABS.includes((e.innerText || '').trim()))"
".map(e => { const lines = (e.innerText || '').trim().split('\\n');"
" return {name: (lines[0] || '').slice(0, 80),"
" detail: lines.slice(1).join(' / ').slice(0, 120)}; }); })()")
@@ -85,7 +95,7 @@ def dialog_text(ws, timeout=8.0, limit=4000):
text = cdp_evaluate(ws, JS_DIALOG_TEXT, timeout=timeout)
except Exception:
return None
if not text:
if not isinstance(text, str) or not text:
return None
return text[:limit]
@@ -135,7 +145,8 @@ def describe_rows(ws, tab=None, timeout=8.0):
"""Inventory rows (name/detail) on a tab. [] when unreadable."""
if tab is not None and not goto_tab(ws, tab, timeout=timeout):
return []
return _eval(ws, JS_DESCRIBE_ROWS, timeout=timeout) or []
rows = _eval(ws, JS_DESCRIBE_ROWS, timeout=timeout)
return rows if isinstance(rows, list) else []
def go_back(ws):
+9 -3
View File
@@ -1,12 +1,13 @@
"""Data controls tab: model-improvement switch (read-only otherwise).
The switch row label is not yet pinned from recon, so resolution
prefers the single switch on the tab and falls back to keyword
match. Import/Delete rows are inventoried, never touched.
The switch label is pinned from live recon; resolution prefers it
and falls back to single-switch, then keyword match. Import/Delete
rows are inventoried, never touched.
"""
from hatch_menu import controls, dialog
TAB = "Data controls"
SWITCH_LABEL = "Help improve our AI models"
_KEYWORDS = ("improv", "train", "model", "data", "usage")
@@ -15,6 +16,11 @@ def _resolve(ws):
if not dialog.goto_tab(ws, TAB):
return None
switches = controls.list_switches(ws)
for s in switches:
blob = ((s.get("label") or "") + " "
+ (s.get("aria") or "")).lower()
if SWITCH_LABEL.lower() in blob:
return s
if len(switches) == 1:
return switches[0]
for kw in _KEYWORDS:
+1 -1
View File
@@ -10,7 +10,7 @@ from hatch_menu import controls, dialog
TAB = "General"
THEME_VALUES = ("match", "default", "blue", "purple", "pink",
THEME_VALUES = ("avatar", "default", "blue", "purple", "pink",
"orange", "green", "beige", "monochrome")
_FREE_RE = re.compile(r"\bfree plan\b", re.IGNORECASE)
+113 -74
View File
@@ -25,7 +25,18 @@ ADV_LABELS = {"transparent_proxy": "Transparent proxy",
"tls_interception": "TLS interception",
"sni_mismatch_rejection": "SNI mismatch rejection"}
PROTOCOL_SLUGS = {"Model Context Protocol servers (SSE)": "mcp-sse",
# Row titles (first line of each protocol row) pinned live 2026-10-06:
# network primitives on every node checked; MCP titles kept
# defensively in case they appear on other plans/accounts.
PROTOCOL_SLUGS = {"Outbound SSH": "outbound-ssh",
"Outgoing email (SMTP)": "smtp",
"Email mailbox access (IMAP, POP3)": "imap-pop3",
"Database connections": "database",
"File transfer (FTP)": "ftp",
"External DNS lookups": "dns",
"Other TCP connections": "other-tcp",
"Other UDP traffic": "other-udp",
"Model Context Protocol servers (SSE)": "mcp-sse",
"Model Context Protocol servers (Streamable HTTP)":
"mcp-streamable",
"Agent Skills endpoints": "agent-skills",
@@ -35,20 +46,16 @@ PROTOCOL_SLUGS = {"Model Context Protocol servers (SSE)": "mcp-sse",
JS_WEBSITES = """(() => {
const d = document.querySelector('[role="dialog"]');
if (!d) return null;
return Array.from(d.querySelectorAll('button')).filter(b =>
['allow', 'ask', 'deny'].includes((b.getAttribute('aria-label') || '')
.trim().toLowerCase())).map(b => {
let el = b.parentElement, host = '', depth = 0;
while (el && el !== d && depth < 6) {
const t = (el.innerText || '').trim().split('\\n')[0] || '';
if (t && t.includes('.') && t.length < 120) { host = t; break; }
el = el.parentElement;
depth += 1;
}
return {host: host, mode: (b.getAttribute('aria-label') || '').trim(),
x: b.getBoundingClientRect().x + b.getBoundingClientRect().width / 2,
y: b.getBoundingClientRect().y + b.getBoundingClientRect().height / 2};
});
const out = [];
for (const b of d.querySelectorAll('button')) {
const m = (b.getAttribute('aria-label') || '').match(
/^Change permission mode for (.+),\\s*(Allow|Ask|Deny)$/i);
if (!m) continue;
const r = b.getBoundingClientRect();
out.push({host: m[1].trim(), mode: m[2],
x: r.x + r.width / 2, y: r.y + r.height / 2});
}
return out;
})()"""
JS_MODE_MENU = """(() => {
@@ -64,27 +71,19 @@ JS_CLICK_MODE = """((mode) => {
return 'CLICKED';
})('%s')"""
JS_MODE_RECT = """((mode) => {
const m = Array.from(document.querySelectorAll('[role="menuitem"]'))
.find(el => (el.innerText || '').trim() === mode);
if (!m) return null;
const r = m.getBoundingClientRect();
return {x: r.x + r.width / 2, y: r.y + r.height / 2};
})('%s')"""
JS_PROTO_ROWS = """(() => {
const d = document.querySelector('[role="dialog"]');
if (!d) return null;
return Array.from(d.querySelectorAll('[role="switch"]')).map(s => {
let el = s.parentElement, label = '', depth = 0;
let el = s.parentElement, title = '', depth = 0;
while (el && el !== d && depth < 6) {
const t = (el.innerText || '').trim().replace(/\\s+/g, ' ');
if (t && t.length < 250) { label = t; break; }
const t = (el.innerText || '').trim().split('\\n')[0] || '';
if (t) { title = t.slice(0, 80); break; }
el = el.parentElement;
depth += 1;
}
const r = s.getBoundingClientRect();
return {label: label.slice(0, 120),
return {title: title,
checked: s.getAttribute('aria-checked') === 'true',
x: r.x + r.width / 2, y: r.y + r.height / 2};
});
@@ -98,23 +97,55 @@ def _eval(ws, js, timeout=8.0):
return None
def _slug(label):
def _stable_rows(ws, js, retries=4, pause=1.5):
"""Repeat a row read until two consecutive reads agree.
Guards mid-animation partial DOM (innerText shifts while the
sub-page slides in). Returns the agreed list, or None.
"""
last = "sentinel"
for _ in range(retries):
rows = _eval(ws, js)
if isinstance(rows, list) and rows == last:
return rows
last = rows if isinstance(rows, list) else "sentinel"
time.sleep(pause)
return last if isinstance(last, list) else None
def _slug(title):
"""Protocol slug: registry hit, else slugified, else None."""
if label in PROTOCOL_SLUGS:
return PROTOCOL_SLUGS[label]
if title in PROTOCOL_SLUGS:
return PROTOCOL_SLUGS[title]
clean = re.sub(r"[^a-z0-9]+", "-",
label.strip().lower()).strip("-")
title.strip().lower()).strip("-")
return clean or None
def _canon_mode(mode):
"""Canonical Allow/Ask/Deny (case-insensitive); passthrough else."""
for m in WEBSITE_MODES:
if (mode or "").lower() == m.lower():
return m
return mode
def resolve_protocol(name):
"""Slug or label fragment -> row label, None when unresolvable."""
"""Slug/title -> row title, None when unresolvable.
Exact slug or title first; then a unique case-insensitive
substring over titles+slugs (so 'ssh' finds Outbound SSH).
"""
if not isinstance(name, str) or not name.strip():
return None
want = name.strip().lower()
for label, slug in PROTOCOL_SLUGS.items():
if want == slug or want == label.lower():
return label
for title, slug in PROTOCOL_SLUGS.items():
if want == slug or want == title.lower():
return title
hits = [t for t, s in PROTOCOL_SLUGS.items()
if want in t.lower() or want in s]
if len(hits) == 1:
return hits[0]
return None
@@ -195,8 +226,7 @@ def _websites_raw(ws):
"""Drill into Websites; rows or None (stays on sub-page)."""
if not dialog.click_row(ws, "Websites", TAB):
return None
time.sleep(0.6)
return _eval(ws, JS_WEBSITES)
return _stable_rows(ws, JS_WEBSITES)
def websites(ws):
@@ -204,7 +234,8 @@ def websites(ws):
rows = _websites_raw(ws)
if rows is None:
return []
out = [{"host": r.get("host"), "mode": r.get("mode")} for r in rows]
out = [{"host": r.get("host"), "mode": _canon_mode(r.get("mode"))}
for r in rows]
_back_to_root(ws)
return out
@@ -218,7 +249,12 @@ def website_mode(ws, host):
def set_website_mode(ws, host, mode):
"""Set one host mode via the mode chooser. Verify + readback. Bool."""
"""Set one host mode via the mode chooser. Bool.
One-way for Ask/Deny: the override row leaves the allowed list
(no add UI), so removal verifies by absence. No-op when already
there; absent hosts fail (nothing to click).
"""
if mode not in WEBSITE_MODES:
return False
rows = _websites_raw(ws)
@@ -230,14 +266,18 @@ def set_website_mode(ws, host, mode):
if target is None:
_back_to_root(ws)
return False
if _canon_mode(target.get("mode")) == mode:
_back_to_root(ws)
return True
try:
real_click(ws, target["x"], target["y"])
except Exception:
_back_to_root(ws)
return False
time.sleep(0.8)
items = _eval(ws, JS_MODE_MENU) or []
texts = [(i.get("text") or "") for i in items]
items = _eval(ws, JS_MODE_MENU)
texts = [(i.get("text") or "") for i in items] \
if isinstance(items, list) else []
if mode not in texts:
escape(ws)
_back_to_root(ws)
@@ -246,26 +286,19 @@ def set_website_mode(ws, host, mode):
escape(ws)
_back_to_root(ws)
return False
time.sleep(0.6)
rows = _eval(ws, JS_WEBSITES) or []
cur = next(((r.get("mode")) for r in rows
if (r.get("host") or "").lower() == host.lower()),
None)
if cur == mode:
_back_to_root(ws)
return True
rect = _eval(ws, JS_MODE_RECT % mode)
if rect and "x" in rect:
try:
real_click(ws, rect["x"], rect["y"])
except Exception:
pass
time.sleep(0.6)
rows = _eval(ws, JS_WEBSITES) or []
cur = next(((r.get("mode")) for r in rows
for _ in range(5):
time.sleep(2.0)
rows = _eval(ws, JS_WEBSITES)
if not isinstance(rows, list):
continue
cur = next((_canon_mode(r.get("mode")) for r in rows
if (r.get("host") or "").lower() == host.lower()),
None)
if cur == mode:
if mode in ("Ask", "Deny"):
if cur is None:
_back_to_root(ws)
return True
elif cur == mode:
_back_to_root(ws)
return True
escape(ws)
@@ -277,36 +310,35 @@ def _protocols_raw(ws):
"""Drill into protocols; rows or None (stays on sub-page)."""
if not dialog.click_row(ws, "Direct network protocols", TAB):
return None
time.sleep(0.6)
return _eval(ws, JS_PROTO_ROWS)
return _stable_rows(ws, JS_PROTO_ROWS)
def protocols(ws):
"""[{slug, label, on}] (back at root afterwards)."""
"""[{slug, title, on}] (back at root afterwards)."""
rows = _protocols_raw(ws)
if rows is None:
return []
out = [{"slug": _slug(r.get("label", "")),
"label": r.get("label", ""),
out = [{"slug": _slug(r.get("title", "")),
"title": r.get("title", ""),
"on": "on" if r.get("checked") else "off"} for r in rows]
_back_to_root(ws)
return out
def protocol_state(ws, label):
"""on/off for one protocol row label, None when absent."""
def protocol_state(ws, title):
"""on/off for one protocol row title, None when absent."""
for row in protocols(ws):
if row.get("label") == label:
if row.get("title") == title:
return row.get("on")
return None
def set_protocol(ws, label, on):
def set_protocol(ws, title, on):
"""Set one protocol switch in place; readback before returning."""
rows = _protocols_raw(ws)
if rows is None:
return False
target = next((r for r in rows if r.get("label") == label), None)
target = next((r for r in rows if r.get("title") == title), None)
if target is None:
_back_to_root(ws)
return False
@@ -319,12 +351,19 @@ def set_protocol(ws, label, on):
except Exception:
_back_to_root(ws)
return False
time.sleep(0.6)
rows = _eval(ws, JS_PROTO_ROWS) or []
cur = next((r for r in rows if r.get("label") == label), None)
ok = cur is not None and bool(cur.get("checked")) == want
for _ in range(8):
time.sleep(2.0)
rows = _eval(ws, JS_PROTO_ROWS)
if not isinstance(rows, list):
continue
cur = next((r for r in rows if r.get("title") == title), None)
if cur is not None and bool(cur.get("checked")) == want:
_back_to_root(ws)
return True
_back_to_root(ws)
return ok
# In-dialog verify missed (slow commit or commit-on-close); the
# toggles-level fresh readback is the source of truth.
return False
def manage_counts(ws):
+16 -8
View File
@@ -182,6 +182,10 @@ def get_toggle(node, name):
else:
value = None
if value is None:
if kind == "website":
return {"ok": False, "node": node, "toggle": name,
"error": "host not in Websites list (effective: "
"permissions.web_access default)"}
return {"ok": False, "node": node, "toggle": name,
"error": "toggle not readable (site changed?)"}
return {"ok": True, "node": node, "toggle": name, "value": value}
@@ -226,15 +230,19 @@ def set_toggle(node, name, value):
ok = _PERM.set_protocol(ws, spec["label"], want == "on")
else:
ok = False
if not ok:
return {"ok": False, "node": node, "toggle": name,
"error": "set failed verification (site changed?)"}
# The fresh-session readback is the source of truth: switch
# commits can land slowly or on dialog close, after the
# in-flow verify had its chance.
readback = get_toggle(node, name)
if not readback.get("ok") or readback.get("value") != want:
return {"ok": False, "node": node, "toggle": name,
"error": "readback mismatch (want %r, got %r)"
% (want, readback.get("value"))}
return {"ok": True, "node": node, "toggle": name, "value": want}
if readback.get("ok") and readback.get("value") == want:
out = {"ok": True, "node": node, "toggle": name,
"value": want}
if not ok:
out["readback_only"] = True
return out
return {"ok": False, "node": node, "toggle": name,
"error": "readback mismatch (want %r, got %r)"
% (want, readback.get("value"))}
except Exception as e:
return {"ok": False, "node": node, "toggle": name,
"error": "%s: %s" % (type(e).__name__, e)}
+5 -3
View File
@@ -549,9 +549,11 @@ def main():
})
# Work-first envelope: executable swarm.spawn/followup.create at TOP and BOTTOM
# (see bin/prompt_envelope.py). Always applied, even if the template has its own [RESULT.
import prompt_envelope
rendered = prompt_envelope.wrap(job_name, job_id, agent, target, rendered)
# (see bin/prompt_envelope.py). Skipped when the job sets "skip_envelope": true
# (agents whose runtime lacks the enveloped tools, e.g. pip).
if not job.get("skip_envelope"):
import prompt_envelope
rendered = prompt_envelope.wrap(job_name, job_id, agent, target, rendered)
# Format as JOB DM
dm_message = f"[JOB {job_id}] {rendered}"
+74 -8
View File
@@ -1,27 +1,76 @@
#!/usr/bin/env python3
"""
Side-chat to main-chat work siphon — monitor loop.
Side-chat to main-chat siphon — monitor loop (INTEGRATED).
Polls side chats for new messages, runs detection, siphons hits to main.
Changes vs original (integrator):
1. Timestamp plumbing (agent 2's open item): message["ts"] is parsed to
epoch seconds and passed as message_ts to detect(), enabling the
15-minute stale-suppression for COMPLETED. Unparseable/missing ts →
backward-compatible (detect proceeds).
2. Author plumbing (agent 3 absent): message["author"] is attached to
the hit as hit.author, so relays attribute the real author instead
of the thread's registered agent.
3. Flood control (agent 4 absent): COMPLETED hits are routed to the
digest buffer instead of individual main-chat relays. ALERT, BLOCKER,
DECISION, MILESTONE still relay individually via siphon().
4. Persistent dedup: every processed hit is marked siphoned (including
digested ones) so a restart never re-relays or re-digests.
This is the integration point for bl. In production:
- list_sidechats() calls muse-chat-api.py or the sidechat manager
- get_messages() reads thread messages via CDP
- post_to_main() sends via muse-chat-api.py send to main chat
For the prototype, all three are injectable (see tests).
- flush_digest() should be called on a schedule (e.g. every 30 min) and
its output posted to main chat once.
"""
import time
from typing import Callable, Dict, List
from datetime import datetime, timezone
from typing import Callable, Dict, List, Optional
from detect import detect, is_opted_out
from siphon import siphon, RateLimiter
from siphon import siphon, mark_siphoned, already_siphoned, RateLimiter
try:
from digest import get_buffer, flush_digest # noqa: F401 (re-export)
except ImportError: # pragma: no cover — digest module optional
get_buffer = None
def flush_digest():
return None
# Message shape: {"id": str, "text": str, "author": str, "ts": str}
Message = Dict[str, str]
# Categories that batch into the digest instead of relaying individually.
DIGESTED_CATEGORIES = {"COMPLETED"}
def _parse_ts(ts) -> Optional[float]:
"""Parse a message timestamp to epoch seconds. None if unparseable."""
if ts is None:
return None
if isinstance(ts, (int, float)):
return float(ts)
s = str(ts).strip()
if not s:
return None
# Epoch as string?
try:
return float(s)
except ValueError:
pass
# ISO-8601 (with optional Z suffix)?
try:
iso = s.replace("Z", "+00:00")
dt = datetime.fromisoformat(iso)
if dt.tzinfo is None:
dt = dt.replace(tzinfo=timezone.utc)
return dt.timestamp()
except ValueError:
return None
def monitor_once(
list_sidechats: Callable[[], List[Dict[str, str]]],
@@ -41,6 +90,7 @@ def monitor_once(
"""
lim = limiter or RateLimiter()
new_marks = dict(watermarks)
digest = get_buffer() if get_buffer else None
for chat in list_sidechats():
tid = chat["id"]
@@ -64,8 +114,24 @@ def monitor_once(
# Update watermark to newest seen
new_marks[tid] = mid
hit = detect(text, tid, mid, min_confidence)
if hit:
# Persistent dedup first: never reprocess a seen message,
# even across restarts (marks are set for digested hits too).
if already_siphoned(mid):
continue
message_ts = _parse_ts(msg.get("ts"))
hit = detect(text, tid, mid, min_confidence,
message_ts=message_ts)
if hit is None:
continue
# Author plumbing: real author, never thread-owner-as-author.
hit.author = msg.get("author", "") or ""
if hit.category in DIGESTED_CATEGORIES and digest is not None:
digest.add(hit)
mark_siphoned(mid)
else:
siphon(hit, agent, post_to_main, lim)
return new_marks
+12
View File
@@ -24,6 +24,7 @@ show_usage() {
done
echo ""
echo "Global lookups & tools:"
echo " tui Interactive full-screen Muse TUI & Box fleet console"
echo " tmux [args...] Manage shared Muse tmux sessions (new, send, capture, ls, kill, prune)"
echo " status Fleet overview & node vitality"
echo " threads List registered threads and sidechats across fleet"
@@ -32,6 +33,7 @@ show_usage() {
echo " passkey (or key) View passkey location (VM-only), PIN, & agent approval protocol"
echo ""
echo "Per-account commands:"
echo " tui Launch interactive TUI for this account"
echo " chat [--thread <id>] Launch interactive conversational shell / REPL"
echo " status Check account status, sessions, and unread"
echo " threads List active threads and sidechats for account"
@@ -48,6 +50,10 @@ show_usage() {
# Direct top-level global actions that do not require an account
if [[ $# -gt 0 ]]; then
case "$1" in
tui)
shift
exec python3 "$NETVM_BIN/muse-tui.py" --mode muse "$@"
;;
tmux)
shift
exec python3 "$NETVM_BIN/muse-tmux.py" "$@"
@@ -143,6 +149,12 @@ if [[ ${#POSITIONAL[@]} -eq 0 ]]; then
POSITIONAL=("status")
fi
# If subcommand is 'tui', launch interactive Muse TUI
if [[ "${POSITIONAL[0]}" == "tui" ]]; then
shift_args=("${POSITIONAL[@]:1}")
exec python3 "$NETVM_BIN/muse-tui.py" --mode muse --account "$ACCOUNT" "${shift_args[@]}"
fi
# If subcommand is 'chat', launch interactive chat REPL
if [[ "${POSITIONAL[0]}" == "chat" ]]; then
shift_args=("${POSITIONAL[@]:1}")
+119 -44
View File
@@ -93,23 +93,29 @@ def get_page(node, cdp_url):
return pages[0]
def ev(ws, expr, await_p=False):
ws.send(json.dumps({
"id": 1, "method": "Runtime.evaluate",
"params": {"expression": expr, "returnByValue": True, "awaitPromise": await_p}
}))
# Drain CDP events until we get our command response (id 1).
# The browser can emit events (Runtime.executionContextCreated, etc.)
# at any time; taking the first recv() blindly returns None on a
# busy page (observed as transient navigation failures in dm.py
# sidechat sends, 2026-10-04 — same class as the NO_SWITCHER fix
# in box-chat-cdp.py commit 8d4bfa7).
for _ in range(50):
resp = json.loads(ws.recv())
if resp.get("id") == 1:
break
else:
"""Returns None (no traceback) if the CDP WebSocket drops.
(Fix 2026-10-06: uncaught WebSocketConnectionClosedException.)"""
try:
ws.send(json.dumps({
"id": 1, "method": "Runtime.evaluate",
"params": {"expression": expr, "returnByValue": True, "awaitPromise": await_p}
}))
# Drain CDP events until we get our command response (id 1).
# The browser can emit events (Runtime.executionContextCreated, etc.)
# at any time; taking the first recv() blindly returns None on a
# busy page (observed as transient navigation failures in dm.py
# sidechat sends, 2026-10-04 — same class as the NO_SWITCHER fix
# in box-chat-cdp.py commit 8d4bfa7).
for _ in range(50):
resp = json.loads(ws.recv())
if resp.get("id") == 1:
break
else:
return None
return resp.get('result', {}).get('result', {}).get('value')
except Exception as e:
print(f"CDP evaluate failed: {type(e).__name__}: {e}", file=sys.stderr)
return None
return resp.get('result', {}).get('result', {}).get('value')
def check_approvals(ws):
"""
@@ -362,14 +368,64 @@ def cmd_messages(ws, n=5, width=200):
# Exclude the compose box subtree: a failed send leaves the draft text
# (including the [id:...] tag) in the composer, and scraping it would
# produce a false "verified" (2026-10-04 dm.py false-confirmation bug).
result = ev1(ws, f"""(() => {{
# 2026-10-05: row-aware scrape. The message feed alternates sender-header
# rows (div.group/stacked-row, per-message timestamp in
# div.text-caption-1) and message units. Each unit is prefixed with its
# header's timestamp ([8:57 pm]) so sweeps can compute message age. The
# feed is the row-parent whose non-row children hold <p> elements (the
# sidebar shares the row classes). The feed hydrates async after
# navigation, so poll up to ~8s before falling back to the legacy
# paragraph scrape.
result = ev1(ws, f"""(async () => {{
const composer = document.querySelector('[contenteditable="true"]') ||
document.querySelector('textarea[placeholder*="Message"]');
const ps = [...document.querySelectorAll('p')]
.filter(p => !(composer && composer.contains(p)))
.slice(-{n*2}).map(p=>p.innerText.slice(0,{width}));
return ps.join('\\n---\\n');
}})()""")
const noComposer = p => !(composer && composer.contains(p));
const legacy = () => {{
const ps = [...document.querySelectorAll('p')]
.filter(noComposer)
.slice(-{n*2}).map(p=>p.innerText.slice(0,{width}));
return ps.join('\\n---\\n');
}};
const ROWSEL = 'div[class*="group/stacked-row"]';
const findFeed = () => {{
const byParent = new Map();
for (const r of document.querySelectorAll(ROWSEL)) {{
const p = r.parentElement;
if (p) {{
if (!byParent.has(p)) byParent.set(p, []);
byParent.get(p).push(r);
}}
}}
for (const [p, rs] of byParent) {{
const hasMsg = [...p.children].some(c => rs.indexOf(c) === -1 &&
c.querySelectorAll('p').length > 0);
if (hasMsg) return [p, rs];
}}
return [null, null];
}};
let list = null, rows = null;
for (let i = 0; i < 16 && !list; i++) {{
[list, rows] = findFeed();
if (!list) await new Promise(r => setTimeout(r, 500));
}}
if (!list) return legacy();
let curTs = '';
const out = [];
for (const child of [...list.children]) {{
if (composer && child.contains(composer)) continue;
if (rows.indexOf(child) !== -1) {{
const t = child.querySelector('div.text-caption-1');
const txt = t ? t.innerText.trim() : '';
if (txt) curTs = txt;
}} else {{
const ps = [...child.querySelectorAll('p')].filter(noComposer)
.map(p=>p.innerText.slice(0,{width}));
if (ps.length) out.push((curTs ? '[' + curTs + '] ' : '') + ps.join('\\n'));
}}
}}
const res = out.slice(-{n}).join('\\n---\\n');
return res || legacy();
}})()""", True)
print(result)
def cmd_compose_check(ws):
@@ -398,20 +454,33 @@ def cmd_wait(ws, timeout=30):
def cdp_navigate(ws, url, timeout_s=30):
"""Navigate via CDP Page.navigate (proper navigation, waits for commit).
Returns True if the page URL matches the target after navigation."""
Returns True if the page URL matches the target after navigation.
Returns False (no traceback) if the CDP WebSocket drops mid-call --
the caller retries on False. (Fix 2026-10-06: uncaught
WebSocketConnectionClosedException crashed dm.py sends as nav_failed.)"""
import time as _time
ws.send(json.dumps({"id": 2, "method": "Page.navigate",
"params": {"url": url}}))
# Drain until we get the Page.navigate response (id 2).
for _ in range(50):
resp = json.loads(ws.recv())
if resp.get("id") == 2:
break
else:
try:
ws.send(json.dumps({"id": 2, "method": "Page.navigate",
"params": {"url": url}}))
# Drain until we get the Page.navigate response (id 2).
for _ in range(50):
resp = json.loads(ws.recv())
if resp.get("id") == 2:
break
else:
return False
except Exception as e:
# Browser CDP connection dropped (crash/restart/relay flake).
# Fail cleanly so dm.py logs nav_failed without a traceback.
print(f"CDP navigate failed: {type(e).__name__}: {e}", file=sys.stderr)
return False
# Wait for the URL to settle (SPA client-side routing).
for _ in range(timeout_s):
cur = ev1(ws, "window.location.href", True)
try:
cur = ev1(ws, "window.location.href", True)
except Exception as e:
print(f"CDP read failed: {type(e).__name__}: {e}", file=sys.stderr)
return False
if cur and url.rstrip("/").lower() in cur.lower():
return True
_time.sleep(1)
@@ -619,18 +688,24 @@ def ev1(ws, expr, await_p=False):
"""Runtime.evaluate that skips CDP event chatter while awaiting its
response. ev() reads a single message and can catch an event
instead (the known None-result quirk); uploads do several DOM
calls first, so chatter is likely."""
ws.send(json.dumps({
"id": 1, "method": "Runtime.evaluate",
"params": {"expression": expr, "returnByValue": True,
"awaitPromise": await_p}
}))
for _ in range(30):
resp = json.loads(ws.recv())
if resp.get("id") != 1:
continue
return resp.get("result", {}).get("result", {}).get("value")
return None
calls first, so chatter is likely.
Returns None (no traceback) if the CDP WebSocket drops.
(Fix 2026-10-06: uncaught WebSocketConnectionClosedException.)"""
try:
ws.send(json.dumps({
"id": 1, "method": "Runtime.evaluate",
"params": {"expression": expr, "returnByValue": True,
"awaitPromise": await_p}
}))
for _ in range(30):
resp = json.loads(ws.recv())
if resp.get("id") != 1:
continue
return resp.get("result", {}).get("result", {}).get("value")
return None
except Exception as e:
print(f"CDP evaluate failed: {type(e).__name__}: {e}", file=sys.stderr)
return None
def cmd_url(ws):
+4
View File
@@ -1,5 +1,9 @@
import os
import sys
import socket
# Prevent unbounded socket hangs across Cloudflare WARP / remote API calls
socket.setdefaulttimeout(15.0)
node = sys.argv[1]
conf_dir = os.path.expanduser(f"~/.config/muse-cli/{node}")
+15 -1
View File
@@ -131,12 +131,19 @@ def parse_threads_blob(blob):
def normalize_thread(t):
is_main = (
t.get("thread") is False or
t.get("is_main") is True or
(t.get("title") and t.get("title").lower() in ("main chat", "main", "start conversation with muse"))
)
return {
"thread_id": t.get("session_id") or t.get("thread_id") or t.get("id"),
"title": t.get("title"),
"pinned": bool(t.get("pinned")),
"archived": bool(t.get("archived")),
"updated": t.get("updated"),
"thread": t.get("thread", True),
"is_main": is_main,
}
@@ -146,7 +153,14 @@ def cmd_list(agent):
code, error, detail, ec = map_failure(rc, err)
fail(code, error, detail=detail, exit_code=ec)
threads = [normalize_thread(t) for t in parse_threads_blob(out)]
print(json.dumps({"ok": True, "agent": agent, "threads": threads}))
mains = [t for t in threads if t.get("is_main")]
pinned = [t for t in threads if t.get("pinned") and not t.get("is_main")]
regular = [t for t in threads if not t.get("is_main") and not t.get("pinned")]
mains.sort(key=lambda t: t.get("updated") or "", reverse=True)
pinned.sort(key=lambda t: t.get("updated") or "", reverse=True)
regular.sort(key=lambda t: t.get("updated") or "", reverse=True)
sorted_threads = mains + pinned + regular
print(json.dumps({"ok": True, "agent": agent, "threads": sorted_threads}))
def cmd_mutate(agent, op, thread_id, title=None):
+8 -3
View File
@@ -4,8 +4,13 @@ set -euo pipefail
# /etc/resolv.conf is a symlink to the systemd stub (127.0.0.53, unreachable
# in the netns). Replace with a real file (private mount ns) so bwrap
# children see the fix too: bwrap's --ro-bind /etc is non-recursive and
# cannot bind over a dangling symlink.
rm -f /etc/resolv.conf
cp "$NETVM_RESOLV" /etc/resolv.conf
mount --make-rprivate / 2>/dev/null || true
if [ -L /etc/resolv.conf ] || ! cmp -s "$NETVM_RESOLV" /etc/resolv.conf 2>/dev/null; then
TMP="/etc/resolv.conf.netvm.$$"
if cp -f "$NETVM_RESOLV" "$TMP" 2>/dev/null; then
mv -f "$TMP" /etc/resolv.conf 2>/dev/null || rm -f "$TMP" 2>/dev/null || true
fi
fi
exec setpriv --reuid="$NETVM_UID" --regid="$NETVM_GID" --clear-groups \
env HOME="$NETVM_HOME" "$@"
+2 -1
View File
@@ -12,4 +12,5 @@ RESOLV=/etc/netvm/resolv-warp.conf
[ -f "$RESOLV" ] || echo "nameserver 1.1.1.1" > "$RESOLV"
ip netns exec "$NETNS" env \
NETVM_RESOLV="$RESOLV" NETVM_UID="$TUID" NETVM_GID="$TGID" NETVM_HOME="$THOME" \
unshare --mount "$SCRIPT_DIR/netvm-enter-inner.sh" "$@"
unshare --mount --propagation private "$SCRIPT_DIR/netvm-enter-inner.sh" "$@"
+8
View File
@@ -58,6 +58,14 @@ rm -f "$STRIPPED"
PEER_PK=$(grep -oP '^\s*PublicKey\s*=\s*\K\S+' "$CONF" | head -1)
if [ -n "$PEER_PK" ]; then
nsexec wg set "$WG" peer "$PEER_PK" persistent-keepalive 25 2>/dev/null || true
# Prefer IPv4 peer endpoint: wg setconf may resolve the Endpoint hostname to
# IPv6, whose handshake then routes into the tunnel itself (no bypass route
# exists for it) and never completes. Observed 2026-10-06 on def.
EPV4=$(getent ahostsv4 "$ENDPOINT" | awk '{print $1}' | sort -u | head -1)
EPPORT=$(grep -oP '^\s*Endpoint\s*=\s*[^:;#]+:\K[0-9]+' "$CONF" | head -1)
if [ -n "$EPV4" ]; then
nsexec wg set "$WG" peer "$PEER_PK" endpoint "${EPV4}:${EPPORT:-2408}" 2>/dev/null || true
fi
fi
MTU=$(grep -oP '^\s*MTU\s*=\s*\K\d+' "$CONF" | head -1); MTU=${MTU:-1280}
nsexec ip link set "$WG" mtu "$MTU"
+5 -3
View File
@@ -15,11 +15,13 @@ iptables -t nat -L POSTROUTING -n 2>/dev/null | grep '10.201\.' || echo "(no net
echo "--- CDP relays (connectivity check; pidfile is secondary) ---"
for ns in $(ip netns list 2>/dev/null | awk '{print $1}' | grep '^warp-'); do
netvm_names "${ns#warp-}"
# Registry-pinned CDP ports (same mapping as cdp-relay-watchdog.sh).
# NOTE: $CDP_PORT from netvm_names() is hash-derived and WRONG here unless
# CDP_PORT_OVERRIDE was set at provision time — the pinned mapping is truth.
# Registry-pinned CDP ports (same mapping as netvm-names.sh).
# NOTE: keep this case in sync with the pinned mapping — the "*" fallback
# trusts $CDP_PORT from netvm_names(), which is pinned for registry nodes
# and hash-derived otherwise.
case "$NODE" in
muse) port=9410 ;; pip) port=9420 ;; 646) port=9430 ;; opm) port=9440 ;;
def) port=9450 ;; dev) port=9455 ;;
*) port="$CDP_PORT" ;;
esac
target="$PEER_IP:$port"
+12 -2
View File
@@ -23,7 +23,7 @@ import time
# Add bin dir to path for siphon imports
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from monitor import monitor_once
from monitor import monitor_once, flush_digest
from siphon import RateLimiter
NETVM_BIN = "/home/super/Projects/NetVM/bin"
@@ -135,7 +135,8 @@ def get_messages(thread_id, since_msg_id):
messages.append({
"id": mid,
"text": chunk[:2000], # truncate long messages
"author": agent,
"author": "", # INTEGRATOR 2026-10-06: was `agent`
# (thread owner) -- fabricated authorship; empty = unverified
"ts": str(time.time()),
})
@@ -249,6 +250,15 @@ def main():
)
save_watermarks(new_marks)
# INTEGRATOR 2026-10-06: emit batched COMPLETED digest (one message
# instead of N per-message relays). Urgency categories already relayed
# individually inside monitor_once.
digest_text = flush_digest()
if digest_text:
post_fn(digest_text)
log(f"Digest posted ({len(digest_text)} chars).")
log(f"Cycle complete. Watermarks: {len(new_marks)} threads tracked.")
+49 -77
View File
@@ -1,66 +1,30 @@
#!/usr/bin/env python3
"""
Side-chat to main-chat work siphon — siphon action.
Side-chat to main-chat work siphon — siphon action (INTEGRATED).
When detection fires, post a summary to main chat with:
- Category badge
- One-line summary (never full message text)
- Link back to the source side chat thread
- Confidence score (for transparency)
Changes vs original (integrator; agent 3 of 5 never delivered, so the
minimal reversible versions below stand in for its authorship/dedup work):
- format_siphon imported from detect (single definition; honest labeling
+ honest authorship live there).
- Deduplication is PERSISTENT: siphoned message IDs are stored as JSON
on disk (SIPHON_STATE_DIR or ~/.siphon-state/siphoned_ids.json) so a
restart can never re-relay. In-memory set kept as a fast path.
- mark_siphoned() writes through to disk on every call.
Safety:
Safety (unchanged):
- Rate limited (max N siphons per hour per thread)
- Never posts full message content
- Respects opt-out registry
- Deduplicates (same message_id never siphoned twice)
- Deduplicates (same message_id never siphoned twice, even across restarts)
"""
import json
import os
import time
from dataclasses import dataclass, field
from typing import Callable, Optional
from detect import SiphonHit, is_opted_out
# --- Follow-up modulation ---
#
# Wire the follow-up modulation table into the siphon so each hit gets
# the right follow-up policy:
# ALERT / BLOCKER / DECISION -> tracked, fast fuse for ALERT/BLOCKER
# COMPLETED / MILESTONE -> untracked (no nudge budget burned)
#
# modulate.py must be landed on bl before this runs (rollout step 1).
# If the import fails we degrade to the old behavior: post the summary
# with no follow-up tags (fail-closed toward visibility, not tracking).
try:
from modulate import for_siphon_hit, render_tags
_MODULATION_AVAILABLE = True
except ImportError: # pragma: no cover - deploy keeps modulate.py present
_MODULATION_AVAILABLE = False
for_siphon_hit = None
render_tags = None
def policy_for_hit(hit: SiphonHit):
"""Follow-up policy for a siphon hit, or None when untracked.
COMPLETED / MILESTONE hits return None (post the summary, create no
follow-up record). ALERT / BLOCKER / DECISION return a Policy whose
tags render into the canonical bracket vocabulary.
"""
if not _MODULATION_AVAILABLE:
return None
return for_siphon_hit(hit.category)
def is_tracked(hit: SiphonHit) -> bool:
"""True when this hit should create a follow-up record.
Callers that route tracked posts through dm.py --expect-reply (so a
dm_followup record is actually created) can use this to choose the
post path. Untracked hits post as plain summaries.
"""
return policy_for_hit(hit) is not None
from detect import SiphonHit, is_opted_out, format_siphon # noqa: F401 (re-export)
# --- Rate limiting ---
@@ -82,9 +46,37 @@ class RateLimiter:
return True
# --- Deduplication ---
# --- Deduplication (persistent) ---
_siphoned_ids: set = set()
_STATE_DIR = os.environ.get(
"SIPHON_STATE_DIR", os.path.expanduser("~/.siphon-state"))
DEDUP_FILE = os.path.join(_STATE_DIR, "siphoned_ids.json")
_DEDUP_MAX_IDS = 5000 # bound disk growth; oldest evicted first
def _load_siphoned() -> set:
try:
with open(DEDUP_FILE) as f:
data = json.load(f)
ids = data.get("ids", []) if isinstance(data, dict) else []
return set(ids)
except (OSError, ValueError):
return set()
def _save_siphoned(ids: set) -> None:
try:
os.makedirs(_STATE_DIR, exist_ok=True)
trimmed = sorted(ids)[-_DEDUP_MAX_IDS:]
tmp = DEDUP_FILE + ".tmp"
with open(tmp, "w") as f:
json.dump({"ids": trimmed, "updated": time.time()}, f)
os.replace(tmp, DEDUP_FILE)
except OSError:
pass # dedup degrades to in-memory; never crash the relay on IO
_siphoned_ids: set = _load_siphoned()
def already_siphoned(message_id: str) -> bool:
@@ -93,6 +85,11 @@ def already_siphoned(message_id: str) -> bool:
def mark_siphoned(message_id: str):
_siphoned_ids.add(message_id)
_save_siphoned(_siphoned_ids)
def siphoned_count() -> int:
return len(_siphoned_ids)
# --- Siphon action ---
@@ -106,21 +103,6 @@ CATEGORY_EMOJI = {
}
def format_siphon(hit: SiphonHit, agent_name: str = "sidechat") -> str:
"""
Format a siphon message for main chat.
Never includes full message text — summary + link only.
"""
emoji = CATEGORY_EMOJI.get(hit.category, "📋")
thread_url = f"https://muse.ai/thread/{hit.thread_id}"
return (
f"{emoji} [{hit.category}] from {agent_name} side chat\n"
f"{hit.summary}\n"
f"→ {thread_url}\n"
f"(confidence {hit.confidence:.0%})"
)
def siphon(hit: SiphonHit,
agent_name: str,
post_to_main: Callable[[str], bool],
@@ -142,16 +124,6 @@ def siphon(hit: SiphonHit,
return False
text = format_siphon(hit, agent_name)
# Follow-up modulation: tracked hits (ALERT/BLOCKER/DECISION) get
# the canonical follow-up tags appended — [reply:expected],
# [reply:timeout=N], [reply:nudges=N], [reply:escalate=X], and
# [input:siphon] for the audit trail. Untracked hits
# (COMPLETED/MILESTONE) post as plain summaries.
policy = policy_for_hit(hit)
if policy is not None:
text = text + "\n" + render_tags(policy)
ok = post_to_main(text)
if ok:
mark_siphoned(hit.message_id)
+41 -1
View File
@@ -59,8 +59,40 @@ def register_session(parent, session_id, title=None, prompt=None):
return entry
def get_active_sessions(parent=None):
DEFAULT_TTL_SECONDS = 3600 # 1 hour idle TTL
def prune_stale_sessions(ttl_seconds=DEFAULT_TTL_SECONDS):
"""Archive active sessions whose last activity exceeds ttl_seconds."""
data = load_sessions()
now = datetime.now(timezone.utc)
changed = False
for sid, s in data.items():
if s.get("status") == "active":
last_act = s.get("last_activity_at") or s.get("spawned_at")
if last_act:
try:
dt = datetime.fromisoformat(last_act.replace("Z", "+00:00"))
if dt.tzinfo is None:
dt = dt.replace(tzinfo=timezone.utc)
if (now - dt).total_seconds() >= ttl_seconds:
s["status"] = "archived"
s["archived_at"] = utcnow()
s["archive_reason"] = f"idle_ttl_exceeded_{ttl_seconds}s"
changed = True
except Exception:
pass
if changed:
save_sessions(data)
def get_active_sessions(parent=None, auto_prune=True, ttl_seconds=DEFAULT_TTL_SECONDS):
"""Retrieve all active subagent sessions, optionally filtered by parent."""
if auto_prune:
try:
prune_stale_sessions(ttl_seconds=ttl_seconds)
except Exception:
pass
data = load_sessions()
results = []
for s in data.values():
@@ -88,6 +120,14 @@ def complete_session(session_id, note=None):
return update_session(session_id, **kwargs)
def archive_session(session_id, reason=None):
"""Mark a subagent session archived."""
kwargs = {"status": "archived", "archived_at": utcnow()}
if reason:
kwargs["archive_reason"] = reason
return update_session(session_id, **kwargs)
if __name__ == "__main__":
if len(sys.argv) > 1 and sys.argv[1] == "list":
print(json.dumps(load_sessions(), indent=2))
+64 -14
View File
@@ -30,8 +30,9 @@ sys.path.insert(0, _SWARM_DIR)
sys.path.insert(0, _BIN_DIR)
from poller import find_pending_slots
from executor import execute_task, _looks_like_shell
from executor import execute_task, _looks_like_shell, extract_shell_command, execute_task_in_tmux
from reporter import post_result, attach_slot
from mainloop_notify import notify_via_mainloop
# Fast gateway integration
try:
@@ -49,7 +50,7 @@ except ImportError:
POLL_INTERVAL = 60 # seconds between poll cycles
STALE_MINUTES = 5 # slots older than this with no attach are workable
WORKER_POOL = ["dev", "def", "muse"]
WORKER_POOL = ["muse"] # only dispatch to fully authenticated agent nodes
# === SAFETY SWITCH ===
# True -> observe only: log what WOULD be done, execute/post nothing.
@@ -65,12 +66,12 @@ log = logging.getLogger("swarm-worker")
def _select_worker(preferred=None):
if preferred and HAS_MUSE_HYBRID and muse_hybrid.is_node_configured(preferred):
if preferred and preferred in WORKER_POOL and HAS_MUSE_HYBRID and muse_hybrid.is_node_configured(preferred):
return preferred
for candidate in WORKER_POOL:
if HAS_MUSE_HYBRID and muse_hybrid.is_node_configured(candidate):
return candidate
return preferred or "dev"
return "muse"
def process_slot(slot):
@@ -81,19 +82,56 @@ def process_slot(slot):
agent_label = slot.get("agent_label")
sidechat_id = slot.get("sidechat_id")
tag = "%s/%s" % (swarm_id, slot_index)
short_id = swarm_id[3:19] if str(swarm_id).startswith("sw-") else str(swarm_id)[:16]
session_name = f"sw-{short_id}-s{slot_index}"
if DRY_RUN:
log.info("[dry-run] would execute slot %s (agent=%s, task %.80r)",
tag, agent_label, task_text)
return True
# If the task is NOT a shell command, dispatch it to an ephemeral Muse subagent.
is_shell = _looks_like_shell(task_text)
if not is_shell and HAS_MUSE_HYBRID:
# 1. Check if the task is or contains an executable shell command
cmd = extract_shell_command(task_text)
if cmd:
log.info("executing slot %s in host tmux session %s on bl", tag, session_name)
# Attach/claim slot in box state
attach_slot(swarm_id, slot_index, "swarm-worker", session_id=session_name)
try:
result = execute_task_in_tmux(session_name, cmd)
except Exception:
log.error("tmux executor crashed on slot %s:\n%s", tag, traceback.format_exc())
result = {"success": False, "output": "",
"error": "tmux executor crashed: see worker log"}
payload = {
"ok": bool(result.get("success")),
"output": result.get("output") or "",
"error": result.get("error"),
}
try:
posted = post_result(swarm_id, slot_index, payload)
except Exception:
log.error("reporter crashed on slot %s:\n%s", tag, traceback.format_exc())
posted = False
# Post completion note to the slot sidechat for main loop visibility
summary_msg = payload.get("output") or payload.get("error") or "completed"
try:
notified = notify_via_mainloop(swarm_id, slot_index, summary_msg, worker_id="swarm-worker")
log.info("slot %s sidechat notification: %s", tag, notified)
except Exception as ne:
log.warning("failed to post sidechat notification for %s: %s", tag, ne)
log.info("slot %s done: ok=%s posted=%s (%.1fs)",
tag, payload["ok"], posted,
float(result.get("duration_s") or 0.0))
return bool(payload["ok"]) and posted
# 2. If the task is purely prose/instructions, dispatch to a verified agent subagent
if HAS_MUSE_HYBRID:
worker_agent = _select_worker(agent_label)
log.info("dispatching subagent slot %s to %s", tag, worker_agent)
log.info("dispatching prose subagent slot %s to %s", tag, worker_agent)
try:
# 1. Start an ephemeral subagent session
title = f"sw-{swarm_id[:16]}-s{slot_index}"
sess, err = muse_hybrid.start_session(worker_agent, title=title)
if err or not sess or not sess.get("session_id"):
@@ -103,12 +141,19 @@ def process_slot(slot):
sub_sid = sess["session_id"]
log.info("subagent session %s created for slot %s on %s", sub_sid, tag, worker_agent)
# 2. Attach/claim the slot in box state with subagent session_id
# Attach/claim slot in box state
attached = attach_slot(swarm_id, slot_index, worker_agent, session_id=sub_sid)
if not attached:
log.warning("failed to attach slot %s to %s; proceeding with dispatch", tag, worker_agent)
# 3. Format prompt with authentic Operator Directive and RESULT expectation
# Register in subagent_tracker
try:
import subagent_tracker
subagent_tracker.register_session(worker_agent, sub_sid, title=title, prompt=task_text[:200])
except Exception:
pass
# Format prompt with authentic Operator Directive and RESULT expectation
if HAS_PROMPT_ENVELOPE and hasattr(prompt_envelope, "wrap_subagent_task"):
prompt_body = prompt_envelope.wrap_subagent_task(tag, task_text)
else:
@@ -122,7 +167,7 @@ def process_slot(slot):
f"(or [RESULT {tag}] FAIL: <reason> if the task could not be completed)\n"
)
# 4. Asynchronously send message to subagent session
# Asynchronously send message to subagent session
res, send_err = muse_hybrid.send_message(worker_agent, prompt_body, thread_id=sub_sid, wait=0)
if send_err:
log.error("failed to send task to subagent %s on %s: %s", sub_sid, worker_agent, send_err)
@@ -134,8 +179,8 @@ def process_slot(slot):
log.error("subagent dispatch crashed on slot %s:\n%s", tag, traceback.format_exc())
return False
# Otherwise fallback to sandboxed host execution
log.info("executing slot %s in sandbox (agent=%s)", tag, agent_label)
# 3. Fallback to sandboxed host execution
log.info("executing slot %s in fallback sandbox (agent=%s)", tag, agent_label)
try:
result = execute_task(task_text)
except Exception:
@@ -154,6 +199,11 @@ def process_slot(slot):
log.error("reporter crashed on slot %s:\n%s", tag, traceback.format_exc())
posted = False
try:
notify_via_mainloop(swarm_id, slot_index, payload.get("output") or "done", worker_id="swarm-worker")
except Exception:
pass
log.info("slot %s done: ok=%s posted=%s (%.1fs)",
tag, payload["ok"], posted,
float(result.get("duration_s") or 0.0))
+140 -1
View File
@@ -85,13 +85,152 @@ def _looks_like_shell(task_text):
if "/" in first:
return os.path.isfile(first) and os.access(first, os.X_OK)
# If it's a bare command name, it must exist in standard system bin paths
for p in ("/bin", "/usr/bin", "/usr/local/bin"):
for p in ("/bin", "/usr/bin", "/usr/local/bin", "/home/super/Projects/NetVM/bin", "/home/super/.local/bin"):
candidate = os.path.join(p, first)
if os.path.isfile(candidate) and os.access(candidate, os.X_OK):
return True
return False
def extract_shell_command(task_text):
"""Extract an executable shell command from task text if present."""
t = (task_text or "").strip()
if not t:
return None
if _looks_like_shell(t):
return t
# Check for "Run: <cmd>" or "Execute this shell command...: <cmd>"
m = re.search(r"(?:Run|Execute)(?:\s+this\s+shell\s+command(?:\s+and\s+report\s+its\s+full\s+output)?)?:\s*[`'\"]?([^`'\n]+)[`'\"]?", t, re.IGNORECASE)
if m:
candidate = m.group(1).strip()
if candidate:
return candidate
# Check for markdown code blocks ```bash ... ``` or ```sh ... ```
m = re.search(r"```(?:bash|sh)?\n(.*?)\n```", t, re.DOTALL)
if m:
candidate = m.group(1).strip()
if candidate:
return candidate
# Check for single backticked command
m = re.search(r"`([^`\n]+)`", t)
if m:
candidate = m.group(1).strip()
if _looks_like_shell(candidate):
return candidate
return None
TMUX_SOCKET = "/tmp/tmux-muse.sock"
TMUX_LOG_DIR = "/home/super/Projects/NetVM/logs/tmux"
def execute_task_in_tmux(session_name, cmd_str, timeout=300):
"""Execute a task inside a dedicated tmux session on /tmp/tmux-muse.sock.
Captures output to logs/tmux/{session_name}.log, tracks return code via
status file, and returns:
dict(success=bool, output=str, duration_s=float, error=str|None)
"""
os.makedirs(TMUX_LOG_DIR, exist_ok=True)
started = time.monotonic()
log_file = os.path.join(TMUX_LOG_DIR, f"{session_name}.log")
exit_file = f"/tmp/{session_name}.exit"
script_file = f"/tmp/{session_name}.sh"
# Clean up prior artifacts
for f in (exit_file, script_file):
try:
if os.path.exists(f):
os.remove(f)
except Exception:
pass
# Write wrapper script
with open(script_file, "w", encoding="utf-8") as sf:
sf.write("#!/usr/bin/env bash\n")
sf.write("export PATH=\"/home/super/Projects/NetVM/bin:/home/super/.local/bin:/usr/local/bin:/usr/bin:/bin:$PATH\"\n")
sf.write("cd /home/super/Projects/NetVM\n")
sf.write(f"{cmd_str}\n")
sf.write(f"echo $? > \"{exit_file}\"\n")
os.chmod(script_file, 0o755)
# Kill any existing session with this name
subprocess.run(["tmux", "-S", TMUX_SOCKET, "kill-session", "-t", session_name],
capture_output=True)
# Start tmux session
tmux_cmd = f"bash \"{script_file}\" > \"{log_file}\" 2>&1"
res = subprocess.run(
["tmux", "-S", TMUX_SOCKET, "new-session", "-d", "-s", session_name, tmux_cmd],
capture_output=True, text=True
)
if res.returncode != 0:
dur = round(time.monotonic() - started, 3)
return {
"success": False,
"output": "",
"duration_s": dur,
"error": f"Failed to create tmux session: {res.stderr.strip()}",
}
# Poll for completion or timeout
deadline = started + timeout
rc = None
while time.monotonic() < deadline:
if os.path.exists(exit_file):
try:
with open(exit_file, "r") as ef:
rc = int(ef.read().strip())
break
except Exception:
pass
check = subprocess.run(
["tmux", "-S", TMUX_SOCKET, "has-session", "-t", session_name],
capture_output=True
)
if check.returncode != 0 and os.path.exists(exit_file):
break
time.sleep(0.5)
dur = round(time.monotonic() - started, 3)
# Clean up tmux session if still running
subprocess.run(["tmux", "-S", TMUX_SOCKET, "kill-session", "-t", session_name],
capture_output=True)
# Read output log
output = ""
if os.path.exists(log_file):
try:
with open(log_file, "r", encoding="utf-8", errors="replace") as lf:
output = lf.read()[:OUTPUT_TRUNCATE]
except Exception as e:
output = f"Error reading log: {e}"
# Cleanup temporary script and exit file
for f in (exit_file, script_file):
try:
if os.path.exists(f):
os.remove(f)
except Exception:
pass
if rc is None:
return {
"success": False,
"output": output,
"duration_s": dur,
"error": f"timeout: exceeded {timeout}s in tmux session",
}
return {
"success": (rc == 0),
"output": output,
"duration_s": dur,
"error": None if (rc == 0) else f"exit code {rc}",
}
def _refused(task_text):
return bool(_REFUSE_RE.search(task_text))
+14 -4
View File
@@ -37,14 +37,24 @@ def notify_via_mainloop(swarm_id, slot_index, message, worker_id="swarm-worker",
Returns:
True on success (or dry-run), False on failure (logged, not raised).
"""
target_name = "sw-%s-s%s" % (swarm_id, slot_index)
target_name = f"{swarm_id}-s{slot_index}" if str(swarm_id).startswith("sw-") else f"sw-{swarm_id}-s{slot_index}"
summary = (message or "").strip().replace("\n", " ")[:NOTE_CHARS]
note = "[SWARM-DONE %s/%s] %s" % (swarm_id, slot_index, summary)
tag = "%s/%s" % (swarm_id, slot_index)
note = "[SWARM-DONE %s] %s" % (tag, summary)
sender = worker_id if worker_id in ("muse", "pip", "646", "opm", "dev", "def", "super") else "super"
to_agent = "opm"
try:
sys.path.insert(0, BIN)
import dm
uuid = dm.resolve_sidechat_target(target_name)
if not uuid:
sc = dm.load_sidechat_map() if hasattr(dm, "load_sidechat_map") else {}
entry = sc.get(target_name, {})
uuid = entry.get("thread_uuid")
if entry.get("agent"):
to_agent = entry.get("agent")
except Exception as e:
print("notify_via_mainloop: target resolve failed for %s: %s"
% (target_name, e), file=sys.stderr)
@@ -55,8 +65,8 @@ def notify_via_mainloop(swarm_id, slot_index, message, worker_id="swarm-worker",
return False
cmd = [sys.executable, DM_PY, "send",
"--agent", worker_id,
"--to", worker_id,
"--agent", sender,
"--to", to_agent,
"--target", uuid,
note]
if dry_run:
+2
View File
@@ -96,6 +96,8 @@ def post_result(swarm_id, slot_index, result_dict, dry_run=False):
log.error("post_result %s/%s: box-ctl ok=false: %s",
swarm_id, slot_index, str(resp)[:500])
return False
return True
def attach_slot(swarm_id, slot_index, agent_id, session_id=None, dry_run=False):
"""Claim/attach a swarm slot to an agent in box state.
+20 -4
View File
@@ -12,11 +12,24 @@
# - New failures: print each new "relaunch FAILED" line, update the
# watermark to the newest line, exit 1.
#
# Self-contained: no arguments, no nested quoting. Safe to call from cron
# or from the box CLI.
# --no-advance: peek-only read. New failures are printed (same output and
# exit codes as above) but the watermark is NOT advanced. The web
# surface (via `box-ctl watchdog-alerts --no-advance`) should always
# pass this flag so UI polling never churns the watermark out from
# under the CLI. CLI runs without the flag keep advance-on-read.
#
# Self-contained: safe to call from cron or from the box CLI.
set -u
NO_ADVANCE=0
for arg in "$@"; do
case "$arg" in
--no-advance) NO_ADVANCE=1 ;;
*) echo "watchdog-alert-check.sh: unknown argument: $arg" >&2; exit 2 ;;
esac
done
LOG="/home/super/Projects/NetVM/chromebox-watchdog.log"
WATERMARK="/home/super/Projects/NetVM/watchdog-alert-watermark.txt"
@@ -48,7 +61,10 @@ fi
[ "${#new_lines[@]}" -gt 0 ] || exit 0
# Report new failures and advance the watermark to the newest line.
# Report new failures and advance the watermark to the newest line
# (skipped in --no-advance peek mode).
printf '%s\n' "${new_lines[@]}"
printf '%s\n' "${failed[-1]}" > "$WATERMARK"
if [ "$NO_ADVANCE" -eq 0 ]; then
printf '%s\n' "${failed[-1]}" > "$WATERMARK"
fi
exit 1
+3 -3
View File
@@ -64,7 +64,7 @@ host → veth IP:port (e.g. 10.201.87.2:9420)
### chromebox-watchdog (browser health)
- **Script:** `/home/super/Projects/NetVM/bin/chromebox-watchdog.sh`
- **Timers:** `chromebox-watchdog-<profile>.timer` (one per profile: muse, pip, 646, opm)
- **Timers:** `chromebox-watchdog-<profile>.timer` (one per profile — every active registry node: muse, pip, 646, opm, def, dev)
- **Cadence:** every 2 minutes
- **Log:** `/home/super/Projects/NetVM/chromebox-watchdog.log` (10 MB rotation, 1 backup gen)
- **Per-profile Chromium output:** `/home/super/Projects/NetVM/chromebox-<profile>.log`
@@ -233,7 +233,7 @@ in the NetVM repo — check `git status` if it's gone.
**Symptoms:** `pgrep -af netvm-cdp-relay` shows relays on ports like 9269, 9278,
9353, 10239, 10355 (hash-derived, not registry ports).
**Cause:** old node-ups or queue tests. Harmless but confusing.
**Fix:** kill them. Only 9410/9420/9430/9440 should be running.
**Fix:** kill them. Only the registry ports (9410/9420/9430/9440/9450/9455) should be running.
## Fleet Status From Blind Shells
@@ -246,7 +246,7 @@ falls back to host watchdog evidence (`bin/host_evidence.py`):
`cdp-relay-watchdog.log` / `chromebox-watchdog.log` (both are
silent-when-healthy) prove the node is up → `ACTIVE [*]`.
- `UNKNOWN` means neither live probes nor host evidence could decide
(e.g. def/dev have no relay-monitor coverage).
(e.g. watchdog timers not installed yet for that node).
- Host evidence never overrides a live local signal, so a fresh outage
observed on the host always wins over a minutes-old watchdog run.
+37 -15
View File
@@ -48,26 +48,48 @@ settled-state probes (`DOM-PAGE-STRUCTURE.md` §3.2). The dialog is found
**by text, not by structure** — this is the central fragility of the current
implementation.
Known structural facts:
Known structural facts (Verified 2026-10-06 live capture on fleet):
- The dialog is in-DOM (React-rendered), so `Runtime.evaluate` sees it; no
shadow-DOM piercing has been needed so far.
- Buttons are plain `<button>` elements matched by innerText
(`allow` / `deny` / `block`, case-insensitive).
- Per `AGENTS.md` (2026-10-03): React selectors behave identically in
headful and headless environments, so selectors captured headless apply
to the user's browser too — and vice versa.
shadow-DOM piercing needed.
- Container: `div[data-testid="hatch-inline-approval-card"]` with
`data-hatch-approval-surface="panel"`.
- Header: `div[data-testid="approval-panel-header"]`.
- Primary Allow action: `button[data-hatch-approval-primary-action="true"]` ("Allow once").
- Dropdown options: `button[data-slot="dropdown-menu-trigger"]` (contains "Always allow").
- Deny action: `<button>` with innerText "Deny".
- Secondary/Background queued approvals surface:
`div[data-testid="hatch-inline-approval-card"][data-hatch-background-approval-surface="true"]`
with `<p class="text-footnote text-text-secondary">N tasks need review</p>`
and `<button data-pel-click="chat_background_approval_review">Review</button>`.
Candidate selectors to verify on next live capture (none confirmed yet):
### Confirmed Live Selectors (2026-10-06):
```javascript
'[role="dialog"]',
'[role="alertdialog"]',
'[data-testid*="dialog"]',
'[data-testid*="approval"]',
'[data-testid*="permission"]',
// button-level (confirmed pattern, unconfirmed testids):
'button' // innerText matches /allow once|always allow|deny/i
// Active approval panel container
'div[data-testid="hatch-inline-approval-card"]'
// Panel header
'[data-testid="approval-panel-header"]'
// Primary action button ("Allow once")
'button[data-hatch-approval-primary-action="true"]'
// Background review surface ("N tasks need review" / "A task needs review")
'[data-hatch-background-approval-surface="true"]'
// Background review button trigger
'[data-pel-click="chat_background_approval_review"]'
```
### Queued Background Task Reviews ("N tasks need review")
When multiple scheduled tasks or background operations trigger approval prompts concurrently (e.g. `opm` with scheduled Fleet Health Monitor runs):
1. Muse renders the first approval card inline over chat.
2. Below it, Muse docks a background approval banner:
`<div data-testid="hatch-inline-approval-card" data-hatch-background-approval-surface="true">`
stating `"N tasks need review"` with a `"Review"` button (`data-pel-click="chat_background_approval_review"`).
3. Allowing or denying the active approval immediately advances the queue: the next queued task pops into the active inline panel, decrementing the background count (e.g. from 2 down to "A task needs review" to clear).
4. Automated approval engines must inspect both the active card and `[data-hatch-background-approval-surface="true"]` to confirm whether agents remain blocked.
### Troubleshooting Blocked Agents
- **In `muse-tui`**: Type `/blocked` (or press `[a]` / `F2`) from any view to open the Approvals & Blocked Tasks Drawer. Use `j`/`k` or arrow keys to navigate between blocked nodes; press `[1]` to Allow, `[2]` Always, `[3]` Deny, `[R]` Proceed, `[X]` Dismiss. Actions target the highlighted agent.
- **In CLI**: Run `box blocked` or `box approvals` to view fleet approval status; use `box approvals allow <node>` to approve.
## 3. How `check_approvals` works today
Location: `bin/muse-chat-api.py`, `check_approvals(ws)` (~line 70).
+12 -5
View File
@@ -37,16 +37,23 @@ Static: `permissions.connector_defaults`,
`permissions.web_access` (`auto_allow`/`always_ask`);
`permissions.advanced.transparent_proxy|tls_interception|
sni_mismatch_rejection`, `data_controls.ai_improvement` (`on`/`off`);
`general.theme` (match/default/blue/purple/pink/orange/green/
beige/monochrome).
`general.theme` (avatar/default/blue/purple/pink/orange/green/
beige/monochrome; `avatar` = "Match my avatar").
Families: `permissions.websites:<host>` (`Allow`/`Ask`/`Deny`),
`permissions.protocols:<slug>` (`on`/`off`; slugs discovered live,
e.g. `mcp-sse`, `mcp-streamable`, `agent-skills`, `mcp-apps`,
`mcp-oauth`).
`permissions.protocols:<slug>` (`on`/`off`; network primitives
`outbound-ssh`, `smtp`, `imap-pop3`, `database`, `ftp`, `dns`,
`other-tcp`, `other-udp` pinned live 2026-10-06, MCP titles kept
defensively; unique substrings like `ssh` also resolve).
Every `set` verifies in place and reads back through a fresh
session; readback mismatch reports failure, never partial success.
Switch commits can land slowly (or on dialog close), so in-flow
verifies poll and the fresh readback is the source of truth; sets
that only the readback confirms carry `"readback_only": true`.
Website Ask/Deny is one-way: the override row leaves the allowed
list (no add UI), verified by absence; absent hosts read as
"not in Websites list" (effective: web-access default).
Caller errors (unknown node/toggle/tab/value) raise `MenuError`
before any CDP traffic. Transport failures return `{"ok": False}`.
+360 -78
View File
@@ -118,11 +118,11 @@
"type": "persistent"
},
"heartbeat": {
"thread_uuid": "757198c3-c1b2-48b8-ba2b-062c84f71b02",
"thread_uuid": "557a4177-901a-4b20-b193-21ac992d49a8",
"agent": "opm",
"title": "heartbeat",
"type": "persistent",
"created_at": "2026-10-05T18:50:05.500146+00:00"
"created_at": "2026-10-06T03:10:03.642110+00:00"
},
"heartbeat-opm": {
"thread_uuid": "ac8c3366-a2b4-407d-adc7-bfa18903c0f5",
@@ -240,18 +240,18 @@
"type": "persistent"
},
"box-http-health": {
"thread_uuid": "ed03c343-c3b7-46aa-9d36-0d15e97ff6df",
"thread_uuid": "5fcb395e-24e4-4b4a-92a8-85edaba710ba",
"agent": "646",
"title": "box-http-health-2026-10-05T15:45:00.375935+00:00",
"title": "box-http-health-2026-10-06T05:41:24.064445+00:00",
"type": "persistent",
"created_at": "2026-10-05T15:46:24.751658+00:00"
"created_at": "2026-10-06T05:41:47.860796+00:00"
},
"box-service-health": {
"thread_uuid": "7e86d12c-0126-46be-8daf-b049b2d67364",
"thread_uuid": "da4f9f77-f1de-44ef-a7f3-b46519bc3542",
"agent": "646",
"title": "box-service-health-2026-10-05T17:07:00.113178+00:00",
"title": "box-service-health-2026-10-05T23:52:00.863450+00:00",
"type": "persistent",
"created_at": "2026-10-05T17:07:02.804315+00:00"
"created_at": "2026-10-05T23:53:02.748190+00:00"
},
"box-deep-health": {
"thread_uuid": "6628c035-4413-4d9f-863c-56ee861c8c83",
@@ -398,32 +398,32 @@
"created_at": "2026-10-05T04:35:47.208458+00:00"
},
"autonomy-pulse-646": {
"thread_uuid": "730b8699-e5ea-4ccb-a543-d5e2c5ad9ae8",
"thread_uuid": "3313d011-4525-4e1b-b830-eab1c6bb8845",
"agent": "646",
"title": "autonomy-pulse-646-2026-10-05",
"title": "autonomy-pulse-646-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:00:06.915544+00:00"
"created_at": "2026-10-06T05:00:53.290329+00:00"
},
"autonomy-pulse-pip": {
"thread_uuid": "453b7c54-cb1a-4024-88d8-727e632a72a1",
"thread_uuid": "9eb27e9f-26a9-436f-8fb4-94f4b81a6859",
"agent": "pip",
"title": "autonomy-pulse-pip-2026-10-05",
"title": "autonomy-pulse-pip-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T18:40:02.572526+00:00"
"created_at": "2026-10-06T01:40:03.202441+00:00"
},
"autonomy-pulse-opm": {
"thread_uuid": "ce5d837b-b5a0-4400-826e-40e21238bb86",
"thread_uuid": "1156e90d-0b07-4ffe-a73f-edb017794b44",
"agent": "opm",
"title": "autonomy-pulse-opm-2026-10-05",
"title": "autonomy-pulse-opm-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T04:50:04.227057+00:00"
"created_at": "2026-10-06T00:50:03.206929+00:00"
},
"muse-auditor": {
"thread_uuid": "87aff7cc-fb19-4bd4-83ef-1c4827c1d488",
"thread_uuid": "e7ed1f57-0926-4fc0-a2a2-457a141c12b5",
"agent": "muse",
"title": "muse-audit-2026-10-05",
"title": "muse-audit-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T06:15:04.918032+00:00"
"created_at": "2026-10-06T02:15:02.756590+00:00"
},
"work-finder": {
"thread_uuid": "16b052eb-acf1-410b-b9af-ee8c6b8bb8d5",
@@ -502,11 +502,11 @@
"created_at": "2026-10-05T05:11:03.515833+00:00"
},
"auto-work-queue-f03": {
"thread_uuid": "49aa2db0-63b5-4cdd-924b-062e4aa5560c",
"thread_uuid": "9dab72c4-83f0-428b-83c1-04804f76798f",
"agent": "muse",
"title": "auto-work-queue-f03-2026-10-05",
"title": "auto-work-queue-f03-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:11:46.129442+00:00"
"created_at": "2026-10-06T05:14:55.648486+00:00"
},
"auto-work-health-h01": {
"thread_uuid": "844f4bfe-5e42-4e09-850e-aabe181c5fcb",
@@ -541,11 +541,11 @@
"archived_by_job": "auto-work-opm-d02-20261005-051500-e3118209"
},
"auto-work-646-a04": {
"thread_uuid": "73829020-df02-4a85-9442-cf59f7654c4b",
"thread_uuid": "a2cf2b5c-e7d4-4ce2-b03b-1754a23cb3a7",
"agent": "646",
"title": "auto-work-646-a04-2026-10-05",
"title": "auto-work-646-a04-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T16:18:28.578329+00:00"
"created_at": "2026-10-06T04:45:53.991238+00:00"
},
"auto-work-646-a05": {
"thread_uuid": "85b02558-b3a2-4136-915a-c349ab9c7055",
@@ -618,11 +618,11 @@
"created_at": "2026-10-05T05:20:21.474650+00:00"
},
"ops-audit": {
"thread_uuid": "4d6b49de-3c89-4253-9cc6-04137d98851e",
"thread_uuid": "c6d17777-43c2-4a21-a74e-779404fdb7f8",
"agent": "pip",
"title": "ops-audit",
"type": "persistent",
"created_at": "2026-10-05T19:49:26.264914+00:00"
"created_at": "2026-10-06T01:57:42.056039+00:00"
},
"auto-work-swarm-g06": {
"thread_uuid": "82ee854a-2e9a-4ed2-a9fe-0e1da6070af4",
@@ -666,18 +666,18 @@
"created_at": "2026-10-05T10:26:12.220004+00:00"
},
"auto-work-646-a14": {
"thread_uuid": "99b431eb-305e-4b68-980a-bdcea5580528",
"thread_uuid": "e3ad4d2e-c26c-41b6-bdd2-7e3e7f9293e8",
"agent": "646",
"title": "auto-work-646-a14-2026-10-05",
"title": "auto-work-646-a14-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:55:02.876339+00:00"
"created_at": "2026-10-06T03:56:13.004961+00:00"
},
"auto-work-646-a13": {
"thread_uuid": "4382fc27-f38c-4cb7-92c9-ff2d28ee95e4",
"thread_uuid": "8f5b0e24-0136-42b6-bec8-b327876d7a5a",
"agent": "646",
"title": "auto-work-646-a13-2026-10-05",
"title": "auto-work-646-a13-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:25:12.347056+00:00"
"created_at": "2026-10-06T00:56:43.746024+00:00"
},
"auto-work-swarm-g09": {
"thread_uuid": "6dcf8342-5fb3-40c7-b213-1e709a7cf911",
@@ -694,11 +694,11 @@
"created_at": "2026-10-05T15:25:03.518277+00:00"
},
"auto-work-health-h15": {
"thread_uuid": "b8eac753-2e73-4c47-8a24-8ca591868b16",
"thread_uuid": "ef62ae77-8b1c-4354-8567-5305560ebbf8",
"agent": "646",
"title": "auto-work-health-h15",
"type": "persistent",
"created_at": "2026-10-05T15:28:55.259741+00:00"
"created_at": "2026-10-06T05:39:11.958529+00:00"
},
"auto-work-swarm-g10-2026-10-05": {
"thread_uuid": "66e1a6c2-d245-43ec-bb45-0449b5275fe4",
@@ -726,11 +726,11 @@
"created_at": "2026-10-05T15:57:45.283197+00:00"
},
"auto-work-646-a15": {
"thread_uuid": "2cb1a50e-7d5b-4b7f-afe3-0a5fbd845eee",
"thread_uuid": "1eed4958-774f-4e58-bba9-1a9a34f4829b",
"agent": "646",
"title": "auto-work-646-a15-2026-10-05",
"title": "auto-work-646-a15-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:30:10.182389+00:00"
"created_at": "2026-10-06T00:33:57.820699+00:00"
},
"auto-work-646-a09": {
"thread_uuid": "8dc4c782-7ae1-4972-abb7-72b66b081405",
@@ -747,11 +747,11 @@
"created_at": "2026-10-05T05:31:57.342161+00:00"
},
"auto-work-swarm-g11": {
"thread_uuid": "4f010a20-b3c9-4201-b392-31d805c39f6a",
"thread_uuid": "32e05dc9-fe3d-4775-8b07-9a0b6e667e64",
"agent": "opm",
"title": "auto-work-swarm-g11-2026-10-05",
"title": "auto-work-swarm-g11-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T09:32:02.898140+00:00"
"created_at": "2026-10-06T04:36:18.613942+00:00"
},
"auto-work-queue-f10": {
"thread_uuid": "161b6a63-ae93-4ce0-bba2-a2db7d88c1c7",
@@ -814,11 +814,11 @@
"archived_by_job": "auto-work-swarm-g12-20261005-053500-c4fe1109"
},
"auto-work-646-a16": {
"thread_uuid": "1948c7dd-e540-4ec1-96ee-babf7e335d0b",
"thread_uuid": "bef6c0cc-abeb-4bdc-8f0a-2f3984750187",
"agent": "646",
"title": "auto-work-646-a16-2026-10-05",
"title": "auto-work-646-a16-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:35:16.833361+00:00"
"created_at": "2026-10-06T05:38:56.582216+00:00"
},
"auto-work-queue-f13": {
"thread_uuid": "c6818256-f0e2-4e6a-918f-ef6986ab26f0",
@@ -850,11 +850,11 @@
"created_at": "2026-10-05T05:37:08.270677+00:00"
},
"auto-work-swarm-g12": {
"thread_uuid": "0e53dfa9-5df2-41d7-b30a-979a5a45ebd0",
"thread_uuid": "7d2e005b-fc92-4a55-9a42-53819bb09b31",
"agent": "646",
"title": "auto-work-swarm-g12-2026-10-05",
"title": "auto-work-swarm-g12-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:37:27.680845+00:00"
"created_at": "2026-10-06T00:42:22.612305+00:00"
},
"auto-work-health-h12": {
"thread_uuid": "0fd5d0c5-10f1-4e17-ab69-ac901007df6a",
@@ -929,11 +929,11 @@
"created_at": "2026-10-05T05:45:03.252304+00:00"
},
"auto-work-646-a18": {
"thread_uuid": "f8d98590-6f36-43fe-a1d1-2590860eb651",
"thread_uuid": "579f73c9-a037-494c-84fd-ba6ca61090bf",
"agent": "646",
"title": "auto-work-646-a18-2026-10-05",
"title": "auto-work-646-a18-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:45:09.732394+00:00"
"created_at": "2026-10-06T00:45:51.929802+00:00"
},
"auto-work-opm-d05-2026-10-05": {
"thread_uuid": "5e99ccdb-3c38-42d1-9eed-2cc83cc33fa0",
@@ -970,11 +970,11 @@
"created_at": "2026-10-05T12:45:06.972657+00:00"
},
"auto-work-xop-e17": {
"thread_uuid": "1f48acfa-c477-4b08-b5e8-c6a3096ae3a2",
"thread_uuid": "ee6e5782-67a9-429b-8463-4da3ade4900c",
"agent": "646",
"title": "auto-work-xop-e17-2026-10-05",
"title": "auto-work-xop-e17",
"type": "persistent",
"created_at": "2026-10-05T05:50:02.666906+00:00"
"created_at": "2026-10-06T03:53:55.831218+00:00"
},
"auto-work-646-a19-2026-10-05": {
"thread_uuid": "018a7c3f-0604-4156-9154-049044c02984",
@@ -993,18 +993,18 @@
"created_at": "2026-10-05T05:50:27.439853+00:00"
},
"auto-work-queue-f17": {
"thread_uuid": "a5ad98d6-8b8c-4a61-ade2-bfe05b5f5029",
"thread_uuid": "a91ebf5c-7165-4ab3-b34a-3f925ec5104e",
"agent": "646",
"title": "auto-work-queue-f17-2026-10-05",
"title": "auto-work-queue-f17-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T05:50:45.662204+00:00"
"created_at": "2026-10-06T04:52:56.332346+00:00"
},
"auto-work-swarm-g16": {
"thread_uuid": "ff318848-8dfb-4d86-b1c0-e189938e8549",
"thread_uuid": "9726b659-89ac-4b52-98c9-602ef4abbd9b",
"agent": "646",
"title": "auto-work-swarm-g16-2026-10-05",
"title": "auto-work-swarm-g16-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T15:47:25.780198+00:00"
"created_at": "2026-10-06T00:54:31.212810+00:00"
},
"auto-work-xop-e16": {
"thread_uuid": "c87a6086-80f7-4edf-9f05-ae5d331ed66a",
@@ -1030,11 +1030,11 @@
"archived_by_job": "auto-work-swarm-g17-20261005-055101-27a952f0"
},
"auto-work-health-h14": {
"thread_uuid": "5e7c6e82-adf9-4505-b6e5-927dddcdaa31",
"thread_uuid": "b0413d26-f210-4e04-a7d4-01c74e00979a",
"agent": "646",
"title": "auto-work-health-h14",
"type": "persistent",
"created_at": "2026-10-05T14:53:03.310305+00:00"
"created_at": "2026-10-06T00:59:11.448941+00:00"
},
"auto-work-swarm-g18": {
"thread_uuid": "dc82af62-28cd-4c13-a0cb-6aaafb037894",
@@ -1179,11 +1179,11 @@
"created_at": "2026-10-05T17:06:16.381842+00:00"
},
"auto-work-xop-e01": {
"thread_uuid": "91fb747f-4526-47c2-b1b5-058c23203ec9",
"thread_uuid": "daaad6e1-4d4d-4df1-963e-7add93593e2e",
"agent": "646",
"title": "auto-work-xop-e01",
"type": "persistent",
"created_at": "2026-10-05T16:02:52.666636+00:00"
"created_at": "2026-10-06T00:12:03.472430+00:00"
},
"auto-work-swarm-g03": {
"thread_uuid": "a8308dbe-7c1e-4d23-b9e4-a50ff88a85a4",
@@ -1207,11 +1207,11 @@
"created_at": "2026-10-05T17:09:03.595173+00:00"
},
"auto-work-swarm-g04": {
"thread_uuid": "0b33dae1-430c-4868-a195-8ba94c2fc87c",
"thread_uuid": "b80123ac-d280-4bea-8de8-8d3c848ccdff",
"agent": "646",
"title": "auto-work-swarm-g04-2026-10-05",
"title": "auto-work-swarm-g04-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T19:17:27.089244+00:00"
"created_at": "2026-10-06T01:16:49.484669+00:00"
},
"auto-work-sweep-j07": {
"thread_uuid": "73888079-0431-498b-9898-21308ac270ab",
@@ -1272,11 +1272,11 @@
"created_at": "2026-10-05T06:16:02.668962+00:00"
},
"auto-work-sweep-j09": {
"thread_uuid": "329a992e-f815-4f10-8be0-c952da96235f",
"thread_uuid": "d60756b1-4f70-42fc-bd76-a7558b9c25cb",
"agent": "opm",
"title": "auto-work-sweep-j09-2026-10-05",
"title": "auto-work-sweep-j09-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T10:17:03.822744+00:00"
"created_at": "2026-10-06T05:30:26.409621+00:00"
},
"auto-work-646-a19": {
"thread_uuid": "06a24f92-2a24-4a18-81ae-b8fdab46facd",
@@ -1292,11 +1292,11 @@
"created_at": "2026-10-05T06:18:04.675660+00:00"
},
"auto-work-sweep-j10": {
"thread_uuid": "a3d260d1-ccf4-4252-98c1-d2b81878659a",
"thread_uuid": "3a56263a-4713-4962-9eac-d90043162677",
"agent": "646",
"title": "auto-work-sweep-j10-2026-10-05",
"title": "auto-work-sweep-j10-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T06:19:02.649814+00:00"
"created_at": "2026-10-06T01:27:33.201947+00:00"
},
"auto-work-sweep-j08": {
"thread_uuid": "e8461df3-ece1-415c-8717-1b89e38851f5",
@@ -1348,11 +1348,11 @@
"created_at": "2026-10-05T08:17:02.518784+00:00"
},
"auto-work-swarm-g08": {
"thread_uuid": "0297d479-e0d0-4c11-a216-8d814e19eb7c",
"thread_uuid": "dd7d89fe-a12c-4f84-b874-8775ff8cf53f",
"agent": "646",
"title": "auto-work-swarm-g08-2026-10-05",
"title": "auto-work-swarm-g08-2026-10-06",
"type": "persistent",
"created_at": "2026-10-05T14:23:10.778643+00:00"
"created_at": "2026-10-06T04:26:51.479445+00:00"
},
"auto-work-646-a20": {
"thread_uuid": "36d083e2-39d7-408d-80b7-f0768d399e18",
@@ -1369,11 +1369,11 @@
"created_at": "2026-10-05T13:25:02.840691+00:00"
},
"auto-work-xop-e09": {
"thread_uuid": "b6d78170-c2bf-4852-b9ab-65c62c9f3a84",
"thread_uuid": "3d25e578-18ce-4f58-9f32-af7271e2aacd",
"agent": "646",
"title": "auto-work-xop-e09",
"type": "persistent",
"created_at": "2026-10-05T15:26:52.407228+00:00"
"created_at": "2026-10-06T00:31:43.412945+00:00"
},
"auto-work-sweep-j14": {
"thread_uuid": "d2d9c1f2-24fc-42dd-95ea-6e89e985bd43",
@@ -2191,5 +2191,287 @@
"title": "auto-work-muse-c17-2026-10-05",
"type": "persistent",
"created_at": "2026-10-05T19:21:52.558789+00:00"
},
"auto-work-muse-c18": {
"thread_uuid": "2f64f3fd-e0ad-42c0-9ad6-e3d7438d35f0",
"agent": "muse",
"title": "auto-work-muse-c18-2026-10-05",
"type": "persistent",
"created_at": "2026-10-05T20:35:44.251409+00:00"
},
"def tasks": {
"thread_uuid": "9cac74ab-0371-4563-a42b-a9a6da5a27de",
"agent": "def",
"title": "def tasks",
"created_at": "2026-10-05T20:36:44.774963+00:00",
"archived": true,
"archived_at": "2026-10-05T20:39:33.509837+00:00",
"archived_by_job": "def tasks"
},
"auto-work-opm-d15": {
"thread_uuid": "2e298a8e-d501-4fd3-8199-73f8d1135dac",
"agent": "opm",
"title": "auto-work-opm-d15-2026-10-05",
"type": "persistent",
"created_at": "2026-10-05T20:40:45.666957+00:00"
},
"sw-20261005-210400-9837-s0": {
"thread_uuid": "3a459f94-2122-4d52-bcf3-36d1ec452862",
"agent": "opm",
"title": "sw-20261005-210400-9837-s0",
"created_at": "2026-10-05T21:05:04.675719+00:00",
"archived": true,
"archived_at": "2026-10-05T21:05:53.539750+00:00",
"archived_by_job": "sw-20261005-210400-9837"
},
"auto-work-muse-c19": {
"thread_uuid": "a3601d33-b646-432e-956c-57a09757eff2",
"agent": "muse",
"title": "auto-work-muse-c19-2026-10-05",
"type": "persistent",
"created_at": "2026-10-05T21:46:19.890287+00:00"
},
"auto-work-opm-d16": {
"thread_uuid": "c8cbb165-83b1-4789-9b64-962214db841a",
"agent": "opm",
"title": "auto-work-opm-d16-2026-10-05",
"type": "persistent",
"created_at": "2026-10-05T22:16:05.396335+00:00"
},
"auto-work-muse-c20": {
"thread_uuid": "6b2a4e65-c015-428d-a30a-c8dfec22f9bc",
"agent": "muse",
"title": "auto-work-muse-c20-2026-10-05",
"type": "persistent",
"created_at": "2026-10-05T22:56:34.815807+00:00"
},
"box-http-health-2026-10-06T00:00:00.370227+00:00": {
"thread_uuid": "27779559-55b6-4995-a783-f9bbf4ef8f5b",
"agent": "646",
"title": "box-http-health-2026-10-06T00:00:00.370227+00:00",
"created_at": "2026-10-06T00:02:51.993375+00:00"
},
"auto-work-swarm-g18-2026-10-06": {
"thread_uuid": "fc625454-ade7-46d7-8437-45675b6427b3",
"agent": "646",
"title": "auto-work-swarm-g18-2026-10-06",
"created_at": "2026-10-06T00:03:38.076303+00:00"
},
"box-http-health-2026-10-06T00:15:00.350787+00:00": {
"thread_uuid": "e29efa5d-2218-416a-8625-e9742c15182b",
"agent": "646",
"title": "box-http-health-2026-10-06T00:15:00.350787+00:00",
"created_at": "2026-10-06T00:17:50.263958+00:00"
},
"auto-work-646-a04-2026-10-06": {
"thread_uuid": "cb2a1c0c-4eb0-45f3-b770-5cc01248ac03",
"agent": "646",
"title": "auto-work-646-a04-2026-10-06",
"created_at": "2026-10-06T00:17:54.330257+00:00",
"archived": true,
"archived_at": "2026-10-06T00:20:11.253525+00:00",
"archived_by_job": "sw-20261005-151613-8551"
},
"auto-work-sweep-j08-2026-10-06": {
"thread_uuid": "00032e31-075c-4a53-8d88-f6a8064a9724",
"agent": "646",
"title": "auto-work-sweep-j08-2026-10-06",
"created_at": "2026-10-06T00:25:00.288039+00:00"
},
"autonomy-pulse-646-2026-10-06": {
"thread_uuid": "672320dd-97d6-4250-9354-2f18e58126cb",
"agent": "646",
"title": "autonomy-pulse-646-2026-10-06",
"created_at": "2026-10-06T00:32:04.552640+00:00",
"archived": true,
"archived_at": "2026-10-06T00:32:27.513282+00:00",
"archived_by_job": "sw-20261005-203113-f929"
},
"auto-work-opm-d17": {
"thread_uuid": "3cf0184d-0eaa-466c-96da-abd390014728",
"agent": "opm",
"title": "auto-work-opm-d17-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T00:34:47.289680+00:00"
},
"auto-work-queue-f19-2026-10-06": {
"thread_uuid": "f6c764a6-f81b-43ef-b1ca-5cf4bd7d4d60",
"agent": "muse",
"title": "auto-work-queue-f19-2026-10-06",
"created_at": "2026-10-06T01:00:13.158621+00:00"
},
"auto-work-646-a16-2026-10-06": {
"thread_uuid": "53c7fe71-5c50-4d77-9371-4fdbcaa1c1f7",
"agent": "646",
"title": "auto-work-646-a16-2026-10-06",
"created_at": "2026-10-06T01:05:26.437594+00:00"
},
"auto-work-646-a10-2026-10-06": {
"thread_uuid": "68967df4-1b2b-475f-96a5-4ef7219e88a1",
"agent": "646",
"title": "auto-work-646-a10-2026-10-06",
"created_at": "2026-10-06T01:13:48.644775+00:00"
},
"auto-work-muse-c02": {
"thread_uuid": "df57007e-9692-4db2-9171-577e07510c14",
"agent": "muse",
"title": "auto-work-muse-c02-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T01:25:49.326452+00:00"
},
"auto-work-opm-d06": {
"thread_uuid": "acf4cf2a-4422-4d25-8080-24a42339e2af",
"agent": "opm",
"title": "auto-work-opm-d06-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T02:10:42.566115+00:00"
},
"auto-work-pip-b03": {
"thread_uuid": "24ce3500-082d-45f3-9cba-06aa1491ee1c",
"agent": "pip",
"title": "auto-work-pip-b03-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T02:10:49.797191+00:00"
},
"auto-work-muse-c03": {
"thread_uuid": "36483fec-1a0e-48e7-b994-ebe25bce892b",
"agent": "muse",
"title": "auto-work-muse-c03-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T02:35:23.435467+00:00"
},
"auto-work-pip-b04": {
"thread_uuid": "79506042-ec89-4f1b-bbec-ae371e6b05ff",
"agent": "pip",
"title": "auto-work-pip-b04-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T03:15:54.131905+00:00"
},
"auto-work-muse-c04": {
"thread_uuid": "f9d0d189-adae-456e-b506-c9bd53194a59",
"agent": "muse",
"title": "auto-work-muse-c04-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T03:47:07.271677+00:00"
},
"auto-work-opm-d19": {
"thread_uuid": "5b3c050f-a5e6-44cf-93fa-0885dac3a5fe",
"agent": "opm",
"title": "auto-work-opm-d19-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T03:47:20.014756+00:00"
},
"auto-work-pip-b05": {
"thread_uuid": "01737122-c154-4c55-9117-c38c5304af34",
"agent": "pip",
"title": "auto-work-pip-b05-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T04:16:28.124727+00:00"
},
"auto-work-opm-d07": {
"thread_uuid": "4c21177f-8618-4557-bfd3-eceb5927b25c",
"agent": "opm",
"title": "auto-work-opm-d07-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T04:30:59.777996+00:00"
},
"auto-work-swarm-g09-2026-10-06": {
"thread_uuid": "2ce154ed-684c-46f1-95d2-79bbbaf4cf19",
"agent": "opm",
"created_at": "2026-10-06T04:32:26.636967+00:00"
},
"sw-20261006-044130-d750-s0": {
"thread_uuid": "ad9ca326-61b2-420e-9d1a-dd7d64fc59ce",
"agent": "opm",
"title": "sw-20261006-044130-d750-s0",
"created_at": "2026-10-06T04:41:53.136145+00:00",
"archived": true,
"archived_at": "2026-10-06T04:43:50.882144+00:00",
"archived_by_job": "sw-20261006-044130-d750"
},
"auto-work-646-a14-2026-10-06": {
"thread_uuid": "ef7c4f53-77cc-4e19-be9b-d42b93a06f00",
"agent": "646",
"title": "auto-work-646-a14-2026-10-06",
"created_at": "2026-10-06T04:58:25.136992+00:00"
},
"box-http-health-2026-10-06T05:00:00.097660+00:00": {
"thread_uuid": "ee8c5423-c975-44eb-bd84-a8dffdb9ead9",
"agent": "646",
"title": "box-http-health-2026-10-06T05:00:00.097660+00:00",
"created_at": "2026-10-06T05:02:23.997548+00:00"
},
"auto-work-muse-c05-2026-10-06": {
"thread_uuid": "8f1ae01d-1449-4c5f-ad5a-749d9a7c17a6",
"agent": "muse",
"title": "auto-work-muse-c05-2026-10-06",
"created_at": "2026-10-06T05:02:53.160137+00:00"
},
"auto-work-health-h19": {
"thread_uuid": "3664c984-5b69-4eff-b9f7-e74f12babcd2",
"agent": "646",
"title": "auto-work-health-h19",
"created_at": "2026-10-06T05:13:31.728991+00:00"
},
"auto-work-pip-b06-2026-10-06": {
"thread_uuid": "cbbe2454-a44d-4c7c-ace1-356bc79a3272",
"agent": "pip",
"title": "auto-work-pip-b06-2026-10-06",
"created_at": "2026-10-06T05:26:51.458933+00:00"
},
"auto-work-646-a15-2026-10-06": {
"thread_uuid": "b43f95f4-eaa3-4251-812a-c8e5cb3352cb",
"agent": "646",
"title": "auto-work-646-a15-2026-10-06",
"created_at": "2026-10-06T05:38:28.702502+00:00"
},
"pip-main": {
"thread_uuid": "ae8d5648-cd72-4f20-9e17-b2d9d5154557",
"agent": "pip",
"title": "pip-main",
"created_at": "2026-10-06T09:22:19.865569+00:00"
},
"dev": {
"thread_uuid": "fe8213e7-d4bb-44b6-b8f5-6213444402db",
"agent": "dev",
"title": "dev",
"created_at": "2026-10-06T09:24:08.552858+00:00"
},
"dev-coord": {
"thread_uuid": "4a0e0302-2be5-4b65-916c-470ab7a07a3b",
"agent": "dev",
"title": "dev-coord",
"created_at": "2026-10-06T20:12:28.109777+00:00"
},
"def-coord": {
"thread_uuid": "7f18e157-a8a7-410c-a1a1-a8535ad97fc0",
"agent": "def",
"title": "def-coord",
"created_at": "2026-10-06T20:13:30.607843+00:00"
},
"auto-work-muse-c01": {
"thread_uuid": "916904c8-ec55-4848-9aaa-0690637af4f8",
"agent": "muse",
"title": "auto-work-muse-c01-2026-10-06",
"type": "persistent",
"created_at": "2026-10-06T20:14:14.273684+00:00"
},
"646-muse-coord": {
"thread_uuid": "c40ea073-7391-4bed-9ff8-47b563b40613",
"agent": "muse",
"title": "646-muse-coord",
"created_at": "2026-10-06T20:14:16.413661+00:00"
},
"opm-pip-coord": {
"thread_uuid": "39c5a2d6-49b2-4cbf-b12b-dac60bd58e81",
"agent": "opm",
"title": "opm-pip-coord",
"created_at": "2026-10-06T22:19:49.593661+00:00"
},
"nonexistent-test": {
"thread_uuid": "5d2fe180-afa9-49a9-a7d5-1485602f7a49",
"agent": "opm",
"title": "nonexistent-test",
"created_at": "2026-10-06T23:15:46.206197+00:00"
}
}
+3 -2
View File
@@ -1,16 +1,17 @@
{
"name": "ops-audit-step3",
"description": "Step 3 of Fleet Operational Audit Pipeline: pip conducts final fleet sign-off",
"description": "Step 3 of Fleet Operational Audit Pipeline: pip conducts final fleet sign-off via direct checks only (no enveloped tool directives)",
"agent": "pip",
"schedule": "manual",
"timeout": 300,
"skip_envelope": true,
"followup": {
"expect_reply": true,
"timeout": "15m",
"nudges": 2,
"escalate": "opm"
},
"prompt_template": "Operational Audit Step 3: Upstream audit report from {prev_job_id}:\n\"{prev_result}\"\n\nReview the combined operational findings across Web and VM systems. Formulate final audit approval.\n\nWhen finished, end your response with:\n[RESULT {job_id}] OK: Operational audit verified and approved by pip",
"prompt_template": "Operational Audit Step 3 -- final fleet sign-off. Do this yourself, directly, in your own session: no subagents, no [TOOL ...] directives, no relayed execution. Every check below is read-only.\n\nUPSTREAM INPUTS (treat as claims to verify, not established facts):\nStep 2 ({prev_job_id}):\n\"{prev_result}\"\n\nYOUR CHECKS:\n1. Board: open the board, confirm it loads and shows recent posts. Evidence: visible post count.\n2. Chat: confirm #jobs is reachable; note the latest seq number you see.\n3. 646-pip coordination sidechat: confirm you can open it; note the latest message time.\n4. Upstream review: compare the step-2 claims above against checks 1-3. Flag anything contradictory or unverifiable.\n\nVERDICT RULES:\n- Report each check as OK (one line of evidence) or FAIL (what you saw instead).\n- Sign-off is OK only if all four checks are OK.\n\nEnd your response with exactly one of:\n[RESULT {job_id}] OK: step3 sign-off board:OK(<evidence>) chat:OK(<evidence>) sidechat:OK(<evidence>) upstream:OK\n[RESULT {job_id}] FAIL: step3 sign-off <check>:FAIL(<what you observed>)",
"sidechat": {
"create": true,
"name_template": "ops-audit",
+4
View File
@@ -183,3 +183,7 @@ the user's directives, and shared commitments lives here.
- 2026-10-04, user: "trust the box; we can fix this" — box dashboard is
the authoritative operational surface; repair the box rather than
bypassing it.
## Core loop (fleet-wide, 2026-10-06)
one person steering and it's us. an AI opm. we execute our own loops. batches: state + rec, go/no-go. escalations: who/what/tried/need/when. verify before done. routes: box.muse-dev.online/api/box/{fleet,timers,nodes} | chat.muse-dev.online/api/chat/history | board.muse-dev.online/api/messages. tools: box-api.sh (ssh-keygen -n box), ssh-vm.sh, timer-ctl.sh. egress: hatch-egress-proxy:3128.
we run our own container and computer. linux calls available.