fix(work): import hashlib and wire heal subparser into main CLI
This commit is contained in:
@@ -33,7 +33,8 @@ Add `--json` to any command for machine-readable output when parsing results in
|
|||||||
- `box job list` / `box job log` — scheduled jobs and execution events.
|
- `box job list` / `box job log` — scheduled jobs and execution events.
|
||||||
- `box harvest status` / `box followup list` — harvest watermarks / pending nudges.
|
- `box harvest status` / `box followup list` — harvest watermarks / pending nudges.
|
||||||
- `box muse-choices on|off|status|logs|reconcile|resolve` — Muse TUI auto-answer daemon switch, state, per-pane logs, held-prompt resolve (default on; `off` is the box-command opt-out).
|
- `box muse-choices on|off|status|logs|reconcile|resolve` — Muse TUI auto-answer daemon switch, state, per-pane logs, held-prompt resolve (default on; `off` is the box-command opt-out).
|
||||||
- `box runtime list|send|launch|layout|spread` — Muse CLI tmux runtimes: live state + approval posture, send-keys input, auto-approved launches, pane-geometry layout + spread for squeezed panes.
|
- `box runtime list|send|launch|layout|spread|reconcile|kill|restart|brief` — Muse CLI tmux runtimes: live state + approval posture, send-keys input, launches with approval trail (bare launch injects `--approval-mode on-request`; fleet socket `/tmp/tmux-muse.sock` is watcher-answered), pane-geometry layout + spread, manifest reconcile, session kill / manifest restart / brief delivery.
|
||||||
|
- `box tasks list|show|create|claim|done|requeue|sweep` — agent work queue (`fleet/tasks/` pending/claimed/done; distinct from scheduled `box job`). Prefer these over raw `mv`.
|
||||||
- `box tmux tally` / `box tmux auto [status|on|off|watch|once|logs|match]` — multi-socket Tmux worker tally, regex auto-approver daemon & guardrails.
|
- `box tmux tally` / `box tmux auto [status|on|off|watch|once|logs|match]` — multi-socket Tmux worker tally, regex auto-approver daemon & guardrails.
|
||||||
- `box onboard connects` / `box onboard-tui` — fleet & client onboarding inventory, CDP ports, OTP salvage & 4-surface TUI.
|
- `box onboard connects` / `box onboard-tui` — fleet & client onboarding inventory, CDP ports, OTP salvage & 4-surface TUI.
|
||||||
- `box invite status|code <node>|redeem <node> <CODE>` / `box usage [--node N]` — invite codes and usage limits.
|
- `box invite status|code <node>|redeem <node> <CODE>` / `box usage [--node N]` — invite codes and usage limits.
|
||||||
|
|||||||
@@ -32,6 +32,7 @@ node name, chrome-box profile, API `--account`, and the agent's display name.
|
|||||||
| def | def | def | email_otp | defnotabotnet@gmail.com | defnotabotnet@gmail.com | no | yes | active | 104.28.195.181 | 9450 | def | Full onboarding completed 2026-10-04; age verification cleared via Instagram linking (paradahub). Active chat session. |
|
| def | def | def | email_otp | defnotabotnet@gmail.com | defnotabotnet@gmail.com | no | yes | active | 104.28.195.181 | 9450 | def | Full onboarding completed 2026-10-04; age verification cleared via Instagram linking (paradahub). Active chat session. |
|
||||||
| opm | opm | opm | email_otp | Nico Parada | artglobal.cc@gmail.com | no | yes | active | 104.28.195.181 | 9440 | opm | Email changed from yourfriendnico@proton.me to artglobal.cc@gmail.com. Linked with IG auxfate. Browser up, session active. |
|
| opm | opm | opm | email_otp | Nico Parada | artglobal.cc@gmail.com | no | yes | active | 104.28.195.181 | 9440 | opm | Email changed from yourfriendnico@proton.me to artglobal.cc@gmail.com. Linked with IG auxfate. Browser up, session active. |
|
||||||
| dev | dev | dev | email_otp | paradaproduced@gmail.com | paradaproduced@gmail.com | no | yes | active | 104.28.195.181 | 9460 | dev | Full onboarding completed 2026-10-04; unlocked /access gate via Meta Accounts Center IG linking (veryraremeta). Active chat session. |
|
| dev | dev | dev | email_otp | paradaproduced@gmail.com | paradaproduced@gmail.com | no | yes | active | 104.28.195.181 | 9460 | dev | Full onboarding completed 2026-10-04; unlocked /access gate via Meta Accounts Center IG linking (veryraremeta). Active chat session. |
|
||||||
|
| 646b | 646b | 646b | email_otp | pixos.dev | pixos.dev@proton.me | no | yes | active | 104.28.195.184 | 9460 | 646b | Salvage node for 646, onboarded 2026-10-09, redeemed REDCJ7. |
|
||||||
|
|
||||||
## Login Type Details
|
## Login Type Details
|
||||||
|
|
||||||
|
|||||||
@@ -27,3 +27,5 @@ Roles: `worker` (persistent swarm/daemon), `repair` (fix sessions),
|
|||||||
sessions carry no node and show `-` in `box runtime list`. Session
|
sessions carry no node and show `-` in `box runtime list`. Session
|
||||||
creators owned by existing flows keep their names until owners rename;
|
creators owned by existing flows keep their names until owners rename;
|
||||||
new sessions should follow the convention from birth.
|
new sessions should follow the convention from birth.
|
||||||
|
| id-verify-examp-8060e2a | warp-id-verify-examp-8060e2a | unknown | 9229 | retired | id-verify-examp-8060e2a (auto-registered; retired 2026-10-08, stray onboarding example, no warp identity) |
|
||||||
|
| 646b | warp-646b | unknown | 9460 | active | 646b (auto-registered) |
|
||||||
|
|||||||
@@ -48,6 +48,14 @@ MD_ACCOUNT_RE = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_-]{0,31}$")
|
|||||||
MD_FILENAME_RE = re.compile(r"^[A-Za-z0-9_.-]{1,128}$")
|
MD_FILENAME_RE = re.compile(r"^[A-Za-z0-9_.-]{1,128}$")
|
||||||
MD_SUBPATH_RE = re.compile(r"^[A-Za-z0-9_.-]+(/[A-Za-z0-9_.-]+)*$")
|
MD_SUBPATH_RE = re.compile(r"^[A-Za-z0-9_.-]+(/[A-Za-z0-9_.-]+)*$")
|
||||||
|
|
||||||
|
# Exact subpaths permitted for read/write alongside plain basenames.
|
||||||
|
# Narrow operator-key-management allowlist: membership is an exact string
|
||||||
|
# match, so no wildcards and no traversal are expressible. Template flows
|
||||||
|
# (diff/amend/append/pull) still require TARGET_MD_FILES.
|
||||||
|
MD_ALLOWED_SUBPATHS = frozenset({
|
||||||
|
".ssh/authorized_keys",
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
def validate_account(account: str) -> str:
|
def validate_account(account: str) -> str:
|
||||||
"""Reject account values that could escape the cookies/config path."""
|
"""Reject account values that could escape the cookies/config path."""
|
||||||
@@ -70,6 +78,8 @@ def validate_filename(filename: str, template_only: bool = False) -> str:
|
|||||||
"Unknown shared template %r: must be one of %s"
|
"Unknown shared template %r: must be one of %s"
|
||||||
% (filename, sorted(TARGET_MD_FILES)))
|
% (filename, sorted(TARGET_MD_FILES)))
|
||||||
return filename
|
return filename
|
||||||
|
if isinstance(filename, str) and filename in MD_ALLOWED_SUBPATHS:
|
||||||
|
return filename
|
||||||
if not isinstance(filename, str) or filename in (".", "..") \
|
if not isinstance(filename, str) or filename in (".", "..") \
|
||||||
or not MD_FILENAME_RE.fullmatch(filename):
|
or not MD_FILENAME_RE.fullmatch(filename):
|
||||||
raise MDValidationError(
|
raise MDValidationError(
|
||||||
|
|||||||
+220
-24
@@ -153,6 +153,10 @@ VALID_NODES = ["muse", "pip", "646", "opm", "def", "dev"]
|
|||||||
KEY_REQUEST_TTL_SECONDS = 2 * 3600
|
KEY_REQUEST_TTL_SECONDS = 2 * 3600
|
||||||
INPUT_WAIT_TTL_SECONDS = 30 * 60
|
INPUT_WAIT_TTL_SECONDS = 30 * 60
|
||||||
BROWSER_APPROVAL_TTL_SECONDS = 30 * 60
|
BROWSER_APPROVAL_TTL_SECONDS = 30 * 60
|
||||||
|
# Tail cap for key-request audit scans: check_node_key_request scans only the
|
||||||
|
# last N lines of box-ctl.jsonl (key events cluster at the end), falling back
|
||||||
|
# to a full scan when the tail holds no relevant record for the node.
|
||||||
|
KEY_SCAN_TAIL_LINES = 5000
|
||||||
|
|
||||||
# Trusted infrastructure IPs safe for automated approval
|
# Trusted infrastructure IPs safe for automated approval
|
||||||
TRUSTED_IPS = {
|
TRUSTED_IPS = {
|
||||||
@@ -187,6 +191,51 @@ def is_trusted_target(target: str, card_text: str = "") -> bool:
|
|||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
def is_plausible_target(target: str) -> bool:
|
||||||
|
"""True if target looks like a real network endpoint, not a parser artifact.
|
||||||
|
|
||||||
|
P1 fix (2026-10-08): the target-extraction regex happily captures garbage
|
||||||
|
tokens like "echo" from dialog text ("connect to echo over SSH"), which
|
||||||
|
then fail-closed to is_trusted=False and page CRITICAL ~6/day for pip's
|
||||||
|
routine Heartbeat dialog. This validator runs BEFORE the is_trusted check:
|
||||||
|
only strict IPv4 (0-255 octets) or plausible hostnames pass.
|
||||||
|
"""
|
||||||
|
if not target or not isinstance(target, str):
|
||||||
|
return False
|
||||||
|
t = target.strip().lower().rstrip(".")
|
||||||
|
if not t:
|
||||||
|
return False
|
||||||
|
# Strict IPv4: four octets, each 0-255, no leading-zero weirdness
|
||||||
|
parts = t.split(".")
|
||||||
|
if len(parts) == 4:
|
||||||
|
try:
|
||||||
|
octets = [int(p) for p in parts]
|
||||||
|
# Reject leading zeros ("01") to avoid octal ambiguity, except "0" itself
|
||||||
|
if all(0 <= o <= 255 for o in octets) and all(
|
||||||
|
p == str(o) for p, o in zip(parts, octets)
|
||||||
|
):
|
||||||
|
return True
|
||||||
|
except ValueError:
|
||||||
|
pass
|
||||||
|
# Four numeric parts but invalid octets (e.g. 999.999.999.999) -> not plausible
|
||||||
|
if all(p.isdigit() for p in parts):
|
||||||
|
return False
|
||||||
|
# Hostname: "localhost" or a dotted name with valid labels
|
||||||
|
if t == "localhost":
|
||||||
|
return True
|
||||||
|
# All-numeric dotted tokens that aren't valid IPv4 (e.g. "1.2.3") are
|
||||||
|
# parser artifacts, not hostnames
|
||||||
|
if "." in t and all(c.isdigit() or c == "." for c in t):
|
||||||
|
return False
|
||||||
|
if "." in t:
|
||||||
|
import re as _re
|
||||||
|
if _re.match(r"^[a-z0-9]([a-z0-9.-]*[a-z0-9])?$", t):
|
||||||
|
# Each label 1-63 chars, no empty labels
|
||||||
|
if all(1 <= len(label) <= 63 for label in t.split(".")):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
REDACT_PATTERNS = [
|
REDACT_PATTERNS = [
|
||||||
(re.compile(r"Bearer\s+[A-Za-z0-9._~+/-]+=*", re.IGNORECASE), "Bearer [REDACTED]"),
|
(re.compile(r"Bearer\s+[A-Za-z0-9._~+/-]+=*", re.IGNORECASE), "Bearer [REDACTED]"),
|
||||||
@@ -308,6 +357,63 @@ def _rec_approval_type(rec: dict) -> str:
|
|||||||
return _approval_type(rec.get("action", ""))
|
return _approval_type(rec.get("action", ""))
|
||||||
|
|
||||||
|
|
||||||
|
def _tail_lines(path: Path, n: int) -> list:
|
||||||
|
"""Return up to the last n lines of path as strings (seek-based, no full read)."""
|
||||||
|
with open(path, "rb") as f:
|
||||||
|
f.seek(0, os.SEEK_END)
|
||||||
|
pos = f.tell()
|
||||||
|
if pos == 0:
|
||||||
|
return []
|
||||||
|
data = b""
|
||||||
|
while pos > 0 and data.count(b"\n") <= n:
|
||||||
|
step = min(8192, pos)
|
||||||
|
pos -= step
|
||||||
|
f.seek(pos)
|
||||||
|
data = f.read(step) + data
|
||||||
|
return data.decode("utf-8", "replace").split("\n")[-n:]
|
||||||
|
|
||||||
|
|
||||||
|
def _scan_key_lines(lines, node: str):
|
||||||
|
"""Scan audit lines (forward order) for a node's key-request state.
|
||||||
|
|
||||||
|
Returns (latest_req, resolved, saw_relevant). A suffix-slice scan is
|
||||||
|
authoritative when saw_relevant: the newest relevant record in a suffix
|
||||||
|
decides the outcome identically to a full scan (any newer request or
|
||||||
|
later resolution would itself lie in the suffix).
|
||||||
|
"""
|
||||||
|
latest_req = None
|
||||||
|
resolved = False
|
||||||
|
saw_relevant = False
|
||||||
|
for line in lines:
|
||||||
|
line = line.strip()
|
||||||
|
if not line:
|
||||||
|
continue
|
||||||
|
# Prefilter: only key-approval actions can affect the outcome, and
|
||||||
|
# all carry this substring; skip json.loads for everything else.
|
||||||
|
if "key-approval" not in line:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
rec = json.loads(line)
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
if rec.get("name") != node:
|
||||||
|
continue
|
||||||
|
act = rec.get("action")
|
||||||
|
if act == "key-approval-request":
|
||||||
|
latest_req = rec
|
||||||
|
resolved = False
|
||||||
|
saw_relevant = True
|
||||||
|
elif _rec_approval_type(rec) == "key" and act in (
|
||||||
|
"key-approval-allow", "key-approval-deny", "key-approval-expired",
|
||||||
|
):
|
||||||
|
# Only a KEY-type resolution clears a key request. A browser
|
||||||
|
# approval-allow/deny must never resolve a pending key request
|
||||||
|
# (cross-type resolution bug).
|
||||||
|
resolved = True
|
||||||
|
saw_relevant = True
|
||||||
|
return latest_req, resolved, saw_relevant
|
||||||
|
|
||||||
|
|
||||||
def check_node_key_request(node: str) -> dict:
|
def check_node_key_request(node: str) -> dict:
|
||||||
"""Check if node has an active unfulfilled key approval request in box-ctl.jsonl.
|
"""Check if node has an active unfulfilled key approval request in box-ctl.jsonl.
|
||||||
|
|
||||||
@@ -317,32 +423,15 @@ def check_node_key_request(node: str) -> dict:
|
|||||||
"""
|
"""
|
||||||
if not CTL_LOG.exists():
|
if not CTL_LOG.exists():
|
||||||
return None
|
return None
|
||||||
latest_req = None
|
|
||||||
resolved = False
|
|
||||||
now = datetime.now(timezone.utc).timestamp()
|
now = datetime.now(timezone.utc).timestamp()
|
||||||
try:
|
try:
|
||||||
with open(CTL_LOG, "r") as f:
|
latest_req, resolved, saw = _scan_key_lines(
|
||||||
for line in f:
|
_tail_lines(CTL_LOG, KEY_SCAN_TAIL_LINES), node)
|
||||||
line = line.strip()
|
if not saw:
|
||||||
if not line:
|
# No relevant record in tail: older history may hold an
|
||||||
continue
|
# unresolved request; fall back to a full scan.
|
||||||
try:
|
with open(CTL_LOG, "r") as f:
|
||||||
rec = json.loads(line)
|
latest_req, resolved, _ = _scan_key_lines(f, node)
|
||||||
except Exception:
|
|
||||||
continue
|
|
||||||
if rec.get("name") != node:
|
|
||||||
continue
|
|
||||||
act = rec.get("action")
|
|
||||||
if act == "key-approval-request":
|
|
||||||
latest_req = rec
|
|
||||||
resolved = False
|
|
||||||
elif _rec_approval_type(rec) == "key" and act in (
|
|
||||||
"key-approval-allow", "key-approval-deny", "key-approval-expired",
|
|
||||||
):
|
|
||||||
# Only a KEY-type resolution clears a key request. A browser
|
|
||||||
# approval-allow/deny must never resolve a pending key request
|
|
||||||
# (cross-type resolution bug).
|
|
||||||
resolved = True
|
|
||||||
except Exception:
|
except Exception:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
@@ -760,6 +849,11 @@ def inspect_node_approvals(node: str) -> dict:
|
|||||||
if bg_tasks_count > 0 and "need review" not in purpose.lower() and "need review" not in title.lower():
|
if bg_tasks_count > 0 and "need review" not in purpose.lower() and "need review" not in title.lower():
|
||||||
purpose = f"{purpose} [{bg_tasks_count} queued task(s) awaiting review]".strip()
|
purpose = f"{purpose} [{bg_tasks_count} queued task(s) awaiting review]".strip()
|
||||||
|
|
||||||
|
# P1: reject implausible targets (parser artifacts like "echo")
|
||||||
|
# before the trust check. Garbage tokens -> parser-suspect.
|
||||||
|
target_plausible = is_plausible_target(target or ip)
|
||||||
|
if target and not target_plausible:
|
||||||
|
target = None
|
||||||
is_trusted = is_trusted_target(target or ip, card_text)
|
is_trusted = is_trusted_target(target or ip, card_text)
|
||||||
|
|
||||||
return {
|
return {
|
||||||
@@ -770,6 +864,7 @@ def inspect_node_approvals(node: str) -> dict:
|
|||||||
"purpose": purpose,
|
"purpose": purpose,
|
||||||
"ip": ip,
|
"ip": ip,
|
||||||
"target": target or ip or "-",
|
"target": target or ip or "-",
|
||||||
|
"target_plausible": target_plausible,
|
||||||
"is_trusted": is_trusted,
|
"is_trusted": is_trusted,
|
||||||
"buttons": data.get("buttons", []),
|
"buttons": data.get("buttons", []),
|
||||||
"has_allow_once": data.get("has_allow_once", False),
|
"has_allow_once": data.get("has_allow_once", False),
|
||||||
@@ -1355,3 +1450,104 @@ def dismiss_node_task(node: str, caller: str = "box-approvals") -> dict:
|
|||||||
"cleared_waits": clear_res.get("cleared_per_node", {}).get(node, 0),
|
"cleared_waits": clear_res.get("cleared_per_node", {}).get(node, 0),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Coordinator Gating & Markdown Decision Records
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
DOCS_DIR = REPO_ROOT / "docs"
|
||||||
|
|
||||||
|
|
||||||
|
def parse_yaml_frontmatter(text: str) -> dict:
|
||||||
|
"""Parse YAML frontmatter delimited by ^--- from Markdown text without external dependencies."""
|
||||||
|
if not text or not text.startswith("---"):
|
||||||
|
return {}
|
||||||
|
parts = text.split("---", 2)
|
||||||
|
if len(parts) < 3:
|
||||||
|
return {}
|
||||||
|
raw_yaml = parts[1].strip()
|
||||||
|
data = {}
|
||||||
|
current_key = None
|
||||||
|
for line in raw_yaml.splitlines():
|
||||||
|
line = line.strip()
|
||||||
|
if not line or line.startswith("#"):
|
||||||
|
continue
|
||||||
|
if ":" in line:
|
||||||
|
k, v = line.split(":", 1)
|
||||||
|
k = k.strip()
|
||||||
|
v = v.strip().strip("'\"")
|
||||||
|
if v.lower() == "true":
|
||||||
|
v = True
|
||||||
|
elif v.lower() == "false":
|
||||||
|
v = False
|
||||||
|
elif v == "":
|
||||||
|
v = []
|
||||||
|
current_key = k
|
||||||
|
data[k] = v
|
||||||
|
continue
|
||||||
|
data[k] = v
|
||||||
|
current_key = k
|
||||||
|
elif line.startswith("- ") and current_key and isinstance(data.get(current_key), list):
|
||||||
|
item = line[2:].strip().strip("'\"")
|
||||||
|
data[current_key].append(item)
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
def scan_coordinator_gates(docs_dir: Path = None) -> list:
|
||||||
|
"""Scan docs/*.md for coordinator gate decision records."""
|
||||||
|
target_dir = docs_dir or DOCS_DIR
|
||||||
|
gates = []
|
||||||
|
if not target_dir.exists():
|
||||||
|
return gates
|
||||||
|
for doc in target_dir.glob("*.md"):
|
||||||
|
try:
|
||||||
|
content = doc.read_text(encoding="utf-8")
|
||||||
|
meta = parse_yaml_frontmatter(content)
|
||||||
|
if meta.get("gate") == "coordinator" or "coordinator" in meta:
|
||||||
|
meta["doc_path"] = str(doc)
|
||||||
|
meta["doc_name"] = doc.name
|
||||||
|
meta["is_signed_off"] = meta.get("status") in ("signed-off", "accepted", "final")
|
||||||
|
gates.append(meta)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
gates.sort(key=lambda x: str(x.get("accepted_at", "")), reverse=True)
|
||||||
|
return gates
|
||||||
|
|
||||||
|
|
||||||
|
def verify_coordinator_signoff(scope: str, docs_dir: Path = None) -> dict:
|
||||||
|
"""Verify if a specific scope or target has a signed-off coordinator decision record.
|
||||||
|
|
||||||
|
Scope can match `scope` or any item in `signoff_targets`.
|
||||||
|
"""
|
||||||
|
gates = scan_coordinator_gates(docs_dir)
|
||||||
|
for g in gates:
|
||||||
|
targets = g.get("signoff_targets") or []
|
||||||
|
if not isinstance(targets, list):
|
||||||
|
targets = [targets]
|
||||||
|
if g.get("scope") == scope or scope in targets:
|
||||||
|
if g.get("is_signed_off"):
|
||||||
|
return {
|
||||||
|
"ok": True,
|
||||||
|
"scope": scope,
|
||||||
|
"status": g.get("status"),
|
||||||
|
"coordinator": g.get("coordinator"),
|
||||||
|
"accepted_at": g.get("accepted_at"),
|
||||||
|
"doc_name": g.get("doc_name"),
|
||||||
|
"doc_path": g.get("doc_path"),
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
return {
|
||||||
|
"ok": False,
|
||||||
|
"scope": scope,
|
||||||
|
"status": g.get("status"),
|
||||||
|
"coordinator": g.get("coordinator"),
|
||||||
|
"doc_name": g.get("doc_name"),
|
||||||
|
"error": f"Gate for scope '{scope}' exists in {g.get('doc_name')} but status is '{g.get('status')}' (not signed-off)",
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
"ok": False,
|
||||||
|
"scope": scope,
|
||||||
|
"error": f"No coordinator decision record found covering scope '{scope}' in {docs_dir or DOCS_DIR}",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+42
-27
@@ -1223,23 +1223,41 @@ def act_chrome_errors(no_advance=False):
|
|||||||
fail("SCAN_ERROR", "chrome-error-scan.sh failed", {"stderr": r.stderr})
|
fail("SCAN_ERROR", "chrome-error-scan.sh failed", {"stderr": r.stderr})
|
||||||
|
|
||||||
|
|
||||||
|
_SUPER_CLI_MOD = None
|
||||||
|
|
||||||
|
|
||||||
|
def _super_cli_mod():
|
||||||
|
"""Lazily import super-cli.py once per process (amortized over calls)."""
|
||||||
|
global _SUPER_CLI_MOD
|
||||||
|
if _SUPER_CLI_MOD is None:
|
||||||
|
import importlib.util
|
||||||
|
spec = importlib.util.spec_from_file_location(
|
||||||
|
"super_cli_boxctl", str(BIN / "super-cli.py"))
|
||||||
|
mod = importlib.util.module_from_spec(spec)
|
||||||
|
spec.loader.exec_module(mod)
|
||||||
|
_SUPER_CLI_MOD = mod
|
||||||
|
return _SUPER_CLI_MOD
|
||||||
|
|
||||||
|
|
||||||
def act_dm_log(limit=50, agent=None):
|
def act_dm_log(limit=50, agent=None):
|
||||||
if agent is not None and agent not in VALID_AGENTS:
|
if agent is not None and agent not in VALID_AGENTS:
|
||||||
fail("BAD_NODE", f"unknown agent: {agent}")
|
fail("BAD_NODE", f"unknown agent: {agent}")
|
||||||
audit("dm-log", f"{agent or 'all'}/{limit}")
|
audit("dm-log", f"{agent or 'all'}/{limit}")
|
||||||
cmd = [sys.executable, str(BIN / "super-cli.py"), "dm", "log", "--json", "-n", str(limit)]
|
try:
|
||||||
if agent:
|
import argparse
|
||||||
cmd += ["--agent", agent]
|
import io
|
||||||
r = subprocess.run(cmd, capture_output=True, text=True)
|
from contextlib import redirect_stdout
|
||||||
if r.returncode == 0:
|
sc = _super_cli_mod()
|
||||||
try:
|
args = argparse.Namespace(n=limit, agent=agent, filter=None, json=True)
|
||||||
data = json.loads(r.stdout)
|
buf = io.StringIO()
|
||||||
data["dms"] = data.get("entries", [])
|
with redirect_stdout(buf):
|
||||||
print(json.dumps(data))
|
sc.cmd_dm_log(args)
|
||||||
return
|
data = json.loads(buf.getvalue())
|
||||||
except Exception:
|
data["dms"] = data.get("entries", [])
|
||||||
pass
|
print(json.dumps(data))
|
||||||
fail("DM_LOG_ERROR", "failed to read dm log", {"stderr": r.stderr})
|
return
|
||||||
|
except Exception as e:
|
||||||
|
fail("DM_LOG_ERROR", "failed to read dm log", {"stderr": str(e)})
|
||||||
|
|
||||||
|
|
||||||
def act_unread(agent=None):
|
def act_unread(agent=None):
|
||||||
@@ -1482,20 +1500,9 @@ def _policy_scan():
|
|||||||
except OSError:
|
except OSError:
|
||||||
return None, {"error": f"cannot read {DM_LOG}"}
|
return None, {"error": f"cannot read {DM_LOG}"}
|
||||||
|
|
||||||
for line in lines:
|
# Single parse pass: stash parsed events because the classification
|
||||||
line = line.strip()
|
# pass needs adoption_ts, a minimum over the whole file.
|
||||||
if not line:
|
events = []
|
||||||
continue
|
|
||||||
try:
|
|
||||||
ev = json.loads(line)
|
|
||||||
except json.JSONDecodeError:
|
|
||||||
continue
|
|
||||||
tags = ev.get("tags")
|
|
||||||
if isinstance(tags, dict) and "allow_main_chat" in tags:
|
|
||||||
ts = ev.get("ts") or ""
|
|
||||||
if adoption_ts is None or ts < adoption_ts:
|
|
||||||
adoption_ts = ts
|
|
||||||
|
|
||||||
for line in lines:
|
for line in lines:
|
||||||
line = line.strip()
|
line = line.strip()
|
||||||
if not line:
|
if not line:
|
||||||
@@ -1506,6 +1513,14 @@ def _policy_scan():
|
|||||||
except json.JSONDecodeError:
|
except json.JSONDecodeError:
|
||||||
malformed += 1
|
malformed += 1
|
||||||
continue
|
continue
|
||||||
|
events.append(ev)
|
||||||
|
tags = ev.get("tags")
|
||||||
|
if isinstance(tags, dict) and "allow_main_chat" in tags:
|
||||||
|
ts = ev.get("ts") or ""
|
||||||
|
if adoption_ts is None or ts < adoption_ts:
|
||||||
|
adoption_ts = ts
|
||||||
|
|
||||||
|
for ev in events:
|
||||||
ts = ev.get("ts") or ""
|
ts = ev.get("ts") or ""
|
||||||
if adoption_ts is not None and ts < adoption_ts:
|
if adoption_ts is not None and ts < adoption_ts:
|
||||||
if ev.get("type") == "sent" and ev.get("target") == "main":
|
if ev.get("type") == "sent" and ev.get("target") == "main":
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ import argparse
|
|||||||
import urllib.request
|
import urllib.request
|
||||||
import urllib.parse
|
import urllib.parse
|
||||||
import urllib.error
|
import urllib.error
|
||||||
|
import hashlib
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -809,6 +810,9 @@ def main():
|
|||||||
p_merge = sub.add_parser("merge", help="Merge an open PR into master")
|
p_merge = sub.add_parser("merge", help="Merge an open PR into master")
|
||||||
p_merge.add_argument("pr", type=int, help="Pull request number (e.g. 214)")
|
p_merge.add_argument("pr", type=int, help="Pull request number (e.g. 214)")
|
||||||
|
|
||||||
|
p_heal = sub.add_parser("heal", help="Run automated remediation on an agent")
|
||||||
|
p_heal.add_argument("agent", help="Agent username to heal")
|
||||||
|
|
||||||
p_chats = sub.add_parser("chats", help="View recent live chat activity")
|
p_chats = sub.add_parser("chats", help="View recent live chat activity")
|
||||||
p_chats.add_argument("--agent", help="Filter by agent name")
|
p_chats.add_argument("--agent", help="Filter by agent name")
|
||||||
p_chats.add_argument("--limit", type=int, default=10, help="Number of messages to show")
|
p_chats.add_argument("--limit", type=int, default=10, help="Number of messages to show")
|
||||||
@@ -820,6 +824,8 @@ def main():
|
|||||||
cmd_status(args)
|
cmd_status(args)
|
||||||
elif action == "check":
|
elif action == "check":
|
||||||
cmd_check(args)
|
cmd_check(args)
|
||||||
|
elif action == "heal":
|
||||||
|
cmd_heal(args)
|
||||||
elif action == "start":
|
elif action == "start":
|
||||||
cmd_start(args)
|
cmd_start(args)
|
||||||
elif action == "assign":
|
elif action == "assign":
|
||||||
|
|||||||
+32
-9
@@ -81,7 +81,11 @@ def compute_funnel(events, cutoff):
|
|||||||
elif ty == "job_result":
|
elif ty == "job_result":
|
||||||
fam = family_of(e.get("job_id"))
|
fam = family_of(e.get("job_id"))
|
||||||
families[fam]["results"] += 1
|
families[fam]["results"] += 1
|
||||||
families[fam]["ok" if e.get("success") else "fail"] += 1
|
snippet = e.get("result_snippet") or ""
|
||||||
|
if e.get("outcome") == "declined" or snippet.startswith("DECLINE:"):
|
||||||
|
families[fam]["declined"] += 1
|
||||||
|
else:
|
||||||
|
families[fam]["ok" if e.get("success") else "fail"] += 1
|
||||||
elif ty == "job_failed":
|
elif ty == "job_failed":
|
||||||
families[family_of(e.get("job_id"))]["failed"] += 1
|
families[family_of(e.get("job_id"))]["failed"] += 1
|
||||||
elif ty == "fallback_executed":
|
elif ty == "fallback_executed":
|
||||||
@@ -213,6 +217,8 @@ def render_digest(rep):
|
|||||||
bits = []
|
bits = []
|
||||||
if t.get("failed"):
|
if t.get("failed"):
|
||||||
bits.append(f"{t['failed']} job_failed")
|
bits.append(f"{t['failed']} job_failed")
|
||||||
|
if t.get("declined"):
|
||||||
|
bits.append(f"{t['declined']} declined")
|
||||||
if tools.get("fail"):
|
if tools.get("fail"):
|
||||||
bits.append(f"{tools['fail']} tool errors")
|
bits.append(f"{tools['fail']} tool errors")
|
||||||
if t.get("fallback_ok") or t.get("fallback_fail"):
|
if t.get("fallback_ok") or t.get("fallback_fail"):
|
||||||
@@ -239,14 +245,25 @@ def render_digest(rep):
|
|||||||
|
|
||||||
|
|
||||||
def should_post(report):
|
def should_post(report):
|
||||||
"""Post on degraded, else heartbeat at most every HEARTBEAT_INTERVAL_H."""
|
"""Post on degraded if changed or every HEARTBEAT_INTERVAL_H, else heartbeat at most every HEARTBEAT_INTERVAL_H."""
|
||||||
if report["degraded"]:
|
|
||||||
return True, "degraded"
|
|
||||||
try:
|
try:
|
||||||
state = json.load(open(STATE_FILE))
|
with open(STATE_FILE, "r", encoding="utf-8") as f:
|
||||||
last = parse_ts(state.get("last_heartbeat"))
|
state = json.load(f)
|
||||||
except Exception:
|
except Exception:
|
||||||
last = None
|
state = {}
|
||||||
|
|
||||||
|
if report.get("degraded"):
|
||||||
|
last_totals = state.get("last_totals")
|
||||||
|
last_reasons = state.get("last_reasons")
|
||||||
|
last_post = parse_ts(state.get("last_degraded_post") or state.get("last_post"))
|
||||||
|
same_metrics = (last_totals is not None and last_totals == report.get("totals"))
|
||||||
|
same_reasons = (last_reasons is not None and last_reasons == report.get("reasons"))
|
||||||
|
if same_metrics and same_reasons:
|
||||||
|
if last_post and (utcnow() - last_post) < timedelta(hours=HEARTBEAT_INTERVAL_H):
|
||||||
|
return False, "degraded-unchanged"
|
||||||
|
return True, "degraded"
|
||||||
|
|
||||||
|
last = parse_ts(state.get("last_heartbeat"))
|
||||||
if last is None or (utcnow() - last) > timedelta(hours=HEARTBEAT_INTERVAL_H):
|
if last is None or (utcnow() - last) > timedelta(hours=HEARTBEAT_INTERVAL_H):
|
||||||
return True, "heartbeat"
|
return True, "heartbeat"
|
||||||
return False, "green-quiet"
|
return False, "green-quiet"
|
||||||
@@ -296,12 +313,18 @@ def main():
|
|||||||
return 0
|
return 0
|
||||||
ok, detail = post_digest(render_digest(report))
|
ok, detail = post_digest(render_digest(report))
|
||||||
print(f"post: {'delivered' if ok else 'FAILED'} ({why}) {detail[:120]}")
|
print(f"post: {'delivered' if ok else 'FAILED'} ({why}) {detail[:120]}")
|
||||||
if ok and why == "heartbeat":
|
if ok:
|
||||||
try:
|
try:
|
||||||
state = {}
|
state = {}
|
||||||
if STATE_FILE.exists():
|
if STATE_FILE.exists():
|
||||||
state = json.loads(STATE_FILE.read_text(encoding="utf-8"))
|
state = json.loads(STATE_FILE.read_text(encoding="utf-8"))
|
||||||
state["last_heartbeat"] = report["ts"]
|
state["last_post"] = report["ts"]
|
||||||
|
if why == "degraded":
|
||||||
|
state["last_degraded_post"] = report["ts"]
|
||||||
|
state["last_totals"] = report.get("totals")
|
||||||
|
state["last_reasons"] = report.get("reasons")
|
||||||
|
elif why == "heartbeat":
|
||||||
|
state["last_heartbeat"] = report["ts"]
|
||||||
STATE_FILE.write_text(json.dumps(state, indent=2), encoding="utf-8")
|
STATE_FILE.write_text(json.dumps(state, indent=2), encoding="utf-8")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"warning: state save failed: {e}")
|
print(f"warning: state save failed: {e}")
|
||||||
|
|||||||
@@ -7,6 +7,7 @@
|
|||||||
# supervisors: cdp-relay-watchdog, agent-health.sh, relay-health-check,
|
# supervisors: cdp-relay-watchdog, agent-health.sh, relay-health-check,
|
||||||
# cdp-latency-check. Port from netvm-names pinning (honors
|
# cdp-latency-check. Port from netvm-names pinning (honors
|
||||||
# CDP_PORT_OVERRIDE, so provision's picked port wins when present).
|
# CDP_PORT_OVERRIDE, so provision's picked port wins when present).
|
||||||
|
# Example/verify/probe names retire on sight (never active, no timer).
|
||||||
# 2. chromebox-watchdog-<node>.timer unit + enable --now — the one
|
# 2. chromebox-watchdog-<node>.timer unit + enable --now — the one
|
||||||
# supervisor that needs a per-node systemd unit (the @.service
|
# supervisor that needs a per-node systemd unit (the @.service
|
||||||
# template already exists). Needs root for the real unit dir.
|
# template already exists). Needs root for the real unit dir.
|
||||||
@@ -25,16 +26,38 @@ UNIT_DIR="${UNIT_DIR:-/etc/systemd/system}"
|
|||||||
|
|
||||||
usage() { echo "usage: ensure-node-supervision.sh <node> | --all" >&2; exit 1; }
|
usage() { echo "usage: ensure-node-supervision.sh <node> | --all" >&2; exit 1; }
|
||||||
|
|
||||||
ensure_registry_row() {
|
# Example/verify/probe nodes (onboarding drills, id-verify examples) must
|
||||||
|
# never join active supervision: they carry no warp identity, wedge the
|
||||||
|
# pinned registry contract, and spin chrome restarts forever. Match is
|
||||||
|
# deliberately narrow (examp anywhere, test-/verify- prefixes) so real
|
||||||
|
# node names containing those substrings elsewhere stay active.
|
||||||
|
is_example_node() {
|
||||||
|
case "$1" in
|
||||||
|
*examp*|test*|verify-*|*-verify-*) return 0;;
|
||||||
|
*) return 1;;
|
||||||
|
esac
|
||||||
|
}
|
||||||
|
|
||||||
|
row_is_retired() {
|
||||||
local node="$1"
|
local node="$1"
|
||||||
|
grep -qE "^\|[[:space:]]*$node[[:space:]]*\|[^|]*\|[^|]*\|[^|]*\|[[:space:]]*retired[[:space:]]*\|" \
|
||||||
|
"$NODES_MD" 2>/dev/null
|
||||||
|
}
|
||||||
|
|
||||||
|
ensure_registry_row() {
|
||||||
|
local node="$1" status="active" note="auto-registered"
|
||||||
if grep -qE "^\|[[:space:]]*$node[[:space:]]*\|" "$NODES_MD" 2>/dev/null; then
|
if grep -qE "^\|[[:space:]]*$node[[:space:]]*\|" "$NODES_MD" 2>/dev/null; then
|
||||||
echo "registry: $node already in NODES.md"
|
echo "registry: $node already in NODES.md"
|
||||||
return 0
|
return 0
|
||||||
fi
|
fi
|
||||||
netvm_names "$node" || { echo "registry: unknown node $node" >&2; return 1; }
|
netvm_names "$node" || { echo "registry: unknown node $node" >&2; return 1; }
|
||||||
printf '| %s | %s | unknown | %s | active | %s (auto-registered) |\n' \
|
if is_example_node "$node"; then
|
||||||
"$node" "$NETNS" "$CDP_PORT" "$node" >> "$NODES_MD"
|
status="retired"
|
||||||
echo "registry: added $node (port $CDP_PORT)"
|
note="auto-registered example — retired"
|
||||||
|
fi
|
||||||
|
printf '| %s | %s | unknown | %s | %s | %s (%s) |\n' \
|
||||||
|
"$node" "$NETNS" "$CDP_PORT" "$status" "$node" "$note" >> "$NODES_MD"
|
||||||
|
echo "registry: added $node (port $CDP_PORT, $status)"
|
||||||
}
|
}
|
||||||
|
|
||||||
ensure_timer() {
|
ensure_timer() {
|
||||||
@@ -72,6 +95,10 @@ EOF
|
|||||||
ensure_node() {
|
ensure_node() {
|
||||||
local node="$1"
|
local node="$1"
|
||||||
ensure_registry_row "$node"
|
ensure_registry_row "$node"
|
||||||
|
if row_is_retired "$node"; then
|
||||||
|
echo "timer: $node retired, skipping supervision"
|
||||||
|
return 0
|
||||||
|
fi
|
||||||
ensure_timer "$node"
|
ensure_timer "$node"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+142
-1
@@ -74,6 +74,7 @@ HEX_RE = re.compile(r'^[0-9a-f]{8,128}$')
|
|||||||
DM_ID_RE = re.compile(r'^[0-9a-fA-F]{6,64}$')
|
DM_ID_RE = re.compile(r'^[0-9a-fA-F]{6,64}$')
|
||||||
TEST_MODULE_RE = re.compile(r'^tests\.[a-z0-9_]+$')
|
TEST_MODULE_RE = re.compile(r'^tests\.[a-z0-9_]+$')
|
||||||
JOB_DISPATCH_ID_RE = re.compile(r'^[a-z0-9][a-z0-9-]{0,63}-\d{8}-\d{6}-[a-f0-9]{8}$')
|
JOB_DISPATCH_ID_RE = re.compile(r'^[a-z0-9][a-z0-9-]{0,63}-\d{8}-\d{6}-[a-f0-9]{8}$')
|
||||||
|
FLOW_ID_RE = re.compile(r'^[a-zA-Z0-9_-]{1,64}$')
|
||||||
STRAT_TYPES = frozenset({'wake', 'job', 'siphon', 'manual', 'health', 'heartbeat'})
|
STRAT_TYPES = frozenset({'wake', 'job', 'siphon', 'manual', 'health', 'heartbeat'})
|
||||||
STRAT_PRIORITIES = frozenset({'routine', 'normal', 'important'})
|
STRAT_PRIORITIES = frozenset({'routine', 'normal', 'important'})
|
||||||
SUBTYPE_RE = re.compile(r'^[A-Za-z0-9_.-]{1,64}$')
|
SUBTYPE_RE = re.compile(r'^[A-Za-z0-9_.-]{1,64}$')
|
||||||
@@ -765,6 +766,120 @@ def _tmux_prune_build(a):
|
|||||||
return [sys.executable, os.path.join(BIN_DIR, 'muse-tmux.py'), 'prune', '--ttl', str(a['ttl'])]
|
return [sys.executable, os.path.join(BIN_DIR, 'muse-tmux.py'), 'prune', '--ttl', str(a['ttl'])]
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_id_name(val):
|
||||||
|
if not isinstance(val, str) or not FLOW_ID_RE.fullmatch(val):
|
||||||
|
raise OpError("flow_id must be 1-64 alphanumeric, dash, or underscore chars")
|
||||||
|
return val
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_start_validate(raw):
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
raise OpError("args must be an object")
|
||||||
|
allowed = {"flow_id", "command", "agent", "cwd"}
|
||||||
|
for k in raw:
|
||||||
|
if k not in allowed:
|
||||||
|
raise OpError(f"unknown arg: {k}")
|
||||||
|
if not raw.get("flow_id"):
|
||||||
|
raise OpError("flow_id is required")
|
||||||
|
agent = raw.get("agent", "646")
|
||||||
|
if agent and (not isinstance(agent, str) or agent not in AGENTS):
|
||||||
|
agent = "646"
|
||||||
|
return {
|
||||||
|
"flow_id": _flow_id_name(raw["flow_id"]),
|
||||||
|
"command": str(raw["command"]) if raw.get("command") else None,
|
||||||
|
"agent": agent,
|
||||||
|
"cwd": str(raw["cwd"]) if raw.get("cwd") else None,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_start_build(a):
|
||||||
|
cmd = [sys.executable, os.path.join(BIN_DIR, "flow_engine.py"), "start", a["flow_id"], "--agent", a["agent"]]
|
||||||
|
if a.get("command"):
|
||||||
|
cmd.extend(["--command", a["command"]])
|
||||||
|
if a.get("cwd"):
|
||||||
|
cmd.extend(["--cwd", a["cwd"]])
|
||||||
|
return cmd
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_read_validate(raw):
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
raise OpError("args must be an object")
|
||||||
|
allowed = {"flow_id", "lines"}
|
||||||
|
for k in raw:
|
||||||
|
if k not in allowed:
|
||||||
|
raise OpError(f"unknown arg: {k}")
|
||||||
|
if not raw.get("flow_id"):
|
||||||
|
raise OpError("flow_id is required")
|
||||||
|
lines = raw.get("lines", 40)
|
||||||
|
try:
|
||||||
|
lines = int(lines)
|
||||||
|
if lines < 1 or lines > 200:
|
||||||
|
lines = 40
|
||||||
|
except Exception:
|
||||||
|
lines = 40
|
||||||
|
return {
|
||||||
|
"flow_id": _flow_id_name(raw["flow_id"]),
|
||||||
|
"lines": lines,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_read_build(a):
|
||||||
|
return [sys.executable, os.path.join(BIN_DIR, "flow_engine.py"), "read", a["flow_id"], "--lines", str(a["lines"])]
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_send_validate(raw):
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
raise OpError("args must be an object")
|
||||||
|
allowed = {"flow_id", "keys", "command", "no_enter"}
|
||||||
|
for k in raw:
|
||||||
|
if k not in allowed:
|
||||||
|
raise OpError(f"unknown arg: {k}")
|
||||||
|
if not raw.get("flow_id"):
|
||||||
|
raise OpError("flow_id is required")
|
||||||
|
if "keys" not in raw:
|
||||||
|
raise OpError("keys is required")
|
||||||
|
return {
|
||||||
|
"flow_id": _flow_id_name(raw["flow_id"]),
|
||||||
|
"keys": str(raw["keys"]),
|
||||||
|
"command": bool(raw.get("command", False)),
|
||||||
|
"no_enter": bool(raw.get("no_enter", False)),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_send_build(a):
|
||||||
|
cmd = [sys.executable, os.path.join(BIN_DIR, "flow_engine.py"), "send", a["flow_id"], a["keys"]]
|
||||||
|
if a.get("command"):
|
||||||
|
cmd.append("--command")
|
||||||
|
if a.get("no_enter"):
|
||||||
|
cmd.append("--no-enter")
|
||||||
|
return cmd
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_list_validate(raw):
|
||||||
|
return {}
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_list_build(a):
|
||||||
|
return [sys.executable, os.path.join(BIN_DIR, "flow_engine.py"), "list"]
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_stop_validate(raw):
|
||||||
|
if not isinstance(raw, dict):
|
||||||
|
raise OpError("args must be an object")
|
||||||
|
allowed = {"flow_id"}
|
||||||
|
for k in raw:
|
||||||
|
if k not in allowed:
|
||||||
|
raise OpError(f"unknown arg: {k}")
|
||||||
|
if not raw.get("flow_id"):
|
||||||
|
raise OpError("flow_id is required")
|
||||||
|
return {"flow_id": _flow_id_name(raw["flow_id"])}
|
||||||
|
|
||||||
|
|
||||||
|
def _flow_stop_build(a):
|
||||||
|
return [sys.executable, os.path.join(BIN_DIR, "flow_engine.py"), "stop", a["flow_id"]]
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def _vars_list_validate(raw):
|
def _vars_list_validate(raw):
|
||||||
if raw not in ({}, None):
|
if raw not in ({}, None):
|
||||||
raise OpError('vars.list takes no required args')
|
raise OpError('vars.list takes no required args')
|
||||||
@@ -2279,6 +2394,31 @@ OPS = {
|
|||||||
'timeout': 15, 'side_effecting': True,
|
'timeout': 15, 'side_effecting': True,
|
||||||
'desc': 'Reap stale unattached sessions inactive for >TTL (default 2h)',
|
'desc': 'Reap stale unattached sessions inactive for >TTL (default 2h)',
|
||||||
},
|
},
|
||||||
|
'flow.start': {
|
||||||
|
'validate': _flow_start_validate, 'build': _flow_start_build,
|
||||||
|
'timeout': 15, 'side_effecting': True,
|
||||||
|
'desc': 'Start an agentic workflow in a persistent tmux pane with output logging',
|
||||||
|
},
|
||||||
|
'flow.read': {
|
||||||
|
'validate': _flow_read_validate, 'build': _flow_read_build,
|
||||||
|
'timeout': 15, 'side_effecting': False,
|
||||||
|
'desc': 'Read output delta and execution state (working/idle/waiting_prompt/finished) from a flow pane',
|
||||||
|
},
|
||||||
|
'flow.send': {
|
||||||
|
'validate': _flow_send_validate, 'build': _flow_send_build,
|
||||||
|
'timeout': 15, 'side_effecting': True,
|
||||||
|
'desc': 'Send keystrokes or advance command in a flow tmux pane',
|
||||||
|
},
|
||||||
|
'flow.list': {
|
||||||
|
'validate': _flow_list_validate, 'build': _flow_list_build,
|
||||||
|
'timeout': 10, 'side_effecting': False,
|
||||||
|
'desc': 'List all active agentic flow sessions and their statuses',
|
||||||
|
},
|
||||||
|
'flow.stop': {
|
||||||
|
'validate': _flow_stop_validate, 'build': _flow_stop_build,
|
||||||
|
'timeout': 15, 'side_effecting': True,
|
||||||
|
'desc': 'Stop and terminate a flow tmux pane session',
|
||||||
|
},
|
||||||
'exec.ping': {
|
'exec.ping': {
|
||||||
'validate': _health_validate,
|
'validate': _health_validate,
|
||||||
'build': lambda a: ['/bin/echo', 'PONG'],
|
'build': lambda a: ['/bin/echo', 'PONG'],
|
||||||
@@ -2377,7 +2517,8 @@ DEFAULT_PERMS = {'dm.read', 'dm.log', 'chat.messages', 'health.check', 'fleet.un
|
|||||||
'thread.list', 'thread.view', 'exec.ping',
|
'thread.list', 'thread.view', 'exec.ping',
|
||||||
'git.status', 'git.diff', 'git.log', 'job.next',
|
'git.status', 'git.diff', 'git.log', 'job.next',
|
||||||
'md.audit', 'md.list', 'md.read', 'md.diff',
|
'md.audit', 'md.list', 'md.read', 'md.diff',
|
||||||
'approval.check', 'tmux.tally', 'tmux.auto_status', 'onboard.connects'}
|
'approval.check', 'tmux.tally', 'tmux.auto_status', 'onboard.connects',
|
||||||
|
'flow.read', 'flow.list'}
|
||||||
|
|
||||||
|
|
||||||
def permitted(ident, op):
|
def permitted(ident, op):
|
||||||
|
|||||||
@@ -42,6 +42,15 @@ REALERT_MIN="${FLEET_ALERT_REALERT_MIN:-30}"
|
|||||||
# forever. Overridable per environment.
|
# forever. Overridable per environment.
|
||||||
INPUT_WAIT_TTL="${FLEET_ALERT_INPUT_WAIT_TTL:-1800}"
|
INPUT_WAIT_TTL="${FLEET_ALERT_INPUT_WAIT_TTL:-1800}"
|
||||||
BROWSER_APPROVAL_TTL="${FLEET_ALERT_BROWSER_APPROVAL_TTL:-1800}"
|
BROWSER_APPROVAL_TTL="${FLEET_ALERT_BROWSER_APPROVAL_TTL:-1800}"
|
||||||
|
# Routine input_wait task patterns (2026-10-08, P4): scheduled-task
|
||||||
|
# confirmations matching these (case-insensitive) are noise-grade
|
||||||
|
# housekeeping that auto-dismisses at TTL. They go to the digest
|
||||||
|
# (kind=DIGEST in the outbox; the #lobby relay ignores non-ALERT/
|
||||||
|
# RECOVERY kinds) instead of paging CRITICAL. Anything NOT matching
|
||||||
|
# stays CRITICAL (fail-closed). Pipe-separated; overridable per
|
||||||
|
# environment. ALL of a node's waits must match for the node to
|
||||||
|
# classify as routine.
|
||||||
|
INPUT_WAIT_ROUTINE_PATTERNS="${FLEET_ALERT_INPUT_WAIT_ROUTINE:-scavenger|background worker|daily checkin|auto-work-queue}"
|
||||||
QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}"
|
QUIET_HOURS="${FLEET_ALERT_QUIET_HOURS:-}"
|
||||||
DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}"
|
DRY_RUN="${FLEET_ALERT_DRY_RUN:-0}"
|
||||||
INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}"
|
INJECT_FAIL="${FLEET_ALERT_INJECT_FAIL:-}"
|
||||||
@@ -196,6 +205,48 @@ notify_input_wait() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
input_wait_routine() { # <node_data_json> -> prints 1 if ALL waits match routine patterns, else 0
|
||||||
|
# P4 (2026-10-08): classify a node's input waits as routine (digest)
|
||||||
|
# or novel (CRITICAL). Fail-closed: empty/unparseable waits, empty
|
||||||
|
# patterns, regex errors, or ANY non-matching wait -> 0 (page it).
|
||||||
|
INPUT_WAIT_ROUTINE_PATTERNS="$INPUT_WAIT_ROUTINE_PATTERNS" python3 - "$1" <<'PYEOF'
|
||||||
|
import json, os, re, sys
|
||||||
|
pats = [p.strip() for p in os.environ.get("INPUT_WAIT_ROUTINE_PATTERNS", "").split("|") if p.strip()]
|
||||||
|
try:
|
||||||
|
waits = json.loads(sys.argv[1]).get("waits", [])
|
||||||
|
except Exception:
|
||||||
|
waits = []
|
||||||
|
if not waits or not pats:
|
||||||
|
print(0)
|
||||||
|
sys.exit()
|
||||||
|
for w in waits:
|
||||||
|
task = w.get("task") or ""
|
||||||
|
try:
|
||||||
|
matched = any(re.search(p, task, re.I) for p in pats)
|
||||||
|
except re.error:
|
||||||
|
matched = False
|
||||||
|
if not matched:
|
||||||
|
print(0)
|
||||||
|
sys.exit()
|
||||||
|
print(1)
|
||||||
|
PYEOF
|
||||||
|
}
|
||||||
|
|
||||||
|
target_plausible_false() { # <node_data_json> -> prints 1 if target_plausible is explicitly false, else 0
|
||||||
|
# P1 follow-up (2026-10-08): the approval target parser flags garbage
|
||||||
|
# tokens (e.g. "echo", "true") as target_plausible=false. Implausible
|
||||||
|
# targets go to the digest instead of paging CRITICAL. Fail-closed:
|
||||||
|
# missing field, null, non-boolean, or unparseable JSON -> 0 (page it).
|
||||||
|
python3 - "$1" <<'PYEOF_INNER'
|
||||||
|
import json, sys
|
||||||
|
try:
|
||||||
|
v = json.loads(sys.argv[1]).get("target_plausible")
|
||||||
|
except Exception:
|
||||||
|
v = None
|
||||||
|
print(1 if v is False else 0)
|
||||||
|
PYEOF_INNER
|
||||||
|
}
|
||||||
|
|
||||||
injected() { # cond -> 0 if injected-fail
|
injected() { # cond -> 0 if injected-fail
|
||||||
case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac
|
case ",$INJECT_FAIL," in *,"$1,"*) return 0;; *) return 1;; esac
|
||||||
}
|
}
|
||||||
@@ -241,6 +292,7 @@ info = approvals.inspect_node_approvals('$node')
|
|||||||
out = {
|
out = {
|
||||||
'has_pending': info.get('has_pending', False),
|
'has_pending': info.get('has_pending', False),
|
||||||
'target': info.get('target') or info.get('ip') or 'unknown',
|
'target': info.get('target') or info.get('ip') or 'unknown',
|
||||||
|
'target_plausible': info.get('target_plausible'),
|
||||||
'title': info.get('title') or '',
|
'title': info.get('title') or '',
|
||||||
'waits': info.get('input_waits') or []
|
'waits': info.get('input_waits') or []
|
||||||
}
|
}
|
||||||
@@ -262,8 +314,20 @@ print(json.dumps(out))
|
|||||||
read -r action fails < <(state_machine "$cond" "$failing" "$BROWSER_APPROVAL_TTL")
|
read -r action fails < <(state_machine "$cond" "$failing" "$BROWSER_APPROVAL_TTL")
|
||||||
case "$action" in
|
case "$action" in
|
||||||
ALERT_FIRST|ALERT_REALERT)
|
ALERT_FIRST|ALERT_REALERT)
|
||||||
emit_record "ALERT" "$cond" "$detail" "$fails"
|
if [ "$(target_plausible_false "$node_data")" = "1" ]; then
|
||||||
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
# P1 follow-up (2026-10-08): implausible approval target
|
||||||
|
# (parser artifact, target_plausible=false) -> digest, don't
|
||||||
|
# page. kind=DIGEST is ignored by the #lobby relay; the
|
||||||
|
# triage digest consumer batches these. No .alerts.tmp
|
||||||
|
# entry, so no box_notify broadcast either — the digest is
|
||||||
|
# the only output. Missing/unparseable field -> CRITICAL
|
||||||
|
# (fail-closed; handled inside target_plausible_false).
|
||||||
|
emit_record "DIGEST" "$cond" "implausible target: $detail" "$fails"
|
||||||
|
log "$cond implausible target x$fails — digested, not paged"
|
||||||
|
else
|
||||||
|
emit_record "ALERT" "$cond" "$detail" "$fails"
|
||||||
|
echo "$cond" >> "$STATE_DIR/.alerts.tmp"
|
||||||
|
fi
|
||||||
;;
|
;;
|
||||||
RECOVERY)
|
RECOVERY)
|
||||||
emit_record "RECOVERY" "$cond" "$detail" "$fails"
|
emit_record "RECOVERY" "$cond" "$detail" "$fails"
|
||||||
@@ -308,8 +372,17 @@ if w:
|
|||||||
read -r action_in fails_in < <(state_machine "$cond_in" "$failing_in" "$INPUT_WAIT_TTL")
|
read -r action_in fails_in < <(state_machine "$cond_in" "$failing_in" "$INPUT_WAIT_TTL")
|
||||||
case "$action_in" in
|
case "$action_in" in
|
||||||
ALERT_FIRST|ALERT_REALERT)
|
ALERT_FIRST|ALERT_REALERT)
|
||||||
emit_record "ALERT" "$cond_in" "$detail_in" "$fails_in"
|
if [ "$(input_wait_routine "$node_data")" = "1" ]; then
|
||||||
echo "$cond_in|$detail_in" >> "$STATE_DIR/.alerts.tmp"
|
# P4 (2026-10-08): routine housekeeping -> digest, don't page.
|
||||||
|
# kind=DIGEST is ignored by the #lobby relay; the triage
|
||||||
|
# digest consumer batches these. No .alerts.tmp entry, so
|
||||||
|
# no targeted DM either — the digest is the only output.
|
||||||
|
emit_record "DIGEST" "$cond_in" "routine: $detail_in" "$fails_in"
|
||||||
|
log "$cond_in routine input_wait x$fails_in — digested, not paged"
|
||||||
|
else
|
||||||
|
emit_record "ALERT" "$cond_in" "$detail_in" "$fails_in"
|
||||||
|
echo "$cond_in|$detail_in" >> "$STATE_DIR/.alerts.tmp"
|
||||||
|
fi
|
||||||
;;
|
;;
|
||||||
RECOVERY)
|
RECOVERY)
|
||||||
emit_record "RECOVERY" "$cond_in" "$detail_in" "$fails_in"
|
emit_record "RECOVERY" "$cond_in" "$detail_in" "$fails_in"
|
||||||
|
|||||||
+51
-12
@@ -364,12 +364,11 @@ DM_LOG_FILE = os.path.join(NETVM_ROOT, "dm-log.jsonl")
|
|||||||
JOB_LOG_FILE = os.path.join(NETVM_ROOT, "job-log.jsonl")
|
JOB_LOG_FILE = os.path.join(NETVM_ROOT, "job-log.jsonl")
|
||||||
|
|
||||||
|
|
||||||
def reconstruct_loops(limit=50, agent=None, status_filter=None) -> list:
|
def _load_loop_candidates() -> dict:
|
||||||
"""Reconstruct active and recent loops from followups.json and dm-log.jsonl.
|
"""Parse followups.json + dm-log.jsonl into a loop_id -> dict map.
|
||||||
|
|
||||||
Returns a list of dicts:
|
Pure parse phase of reconstruct_loops, extracted so diagnose_breaks and
|
||||||
loop_id, agent, sender, target, purpose, state, sent_at, deadline,
|
remediate_breaks can share one parse instead of re-reading the logs.
|
||||||
nudges_sent, nudges_allowed, escalate_to, tags, summary
|
|
||||||
"""
|
"""
|
||||||
loops = {} # loop_id -> dict
|
loops = {} # loop_id -> dict
|
||||||
|
|
||||||
@@ -499,6 +498,11 @@ def reconstruct_loops(limit=50, agent=None, status_filter=None) -> list:
|
|||||||
"source": "dm-log.jsonl",
|
"source": "dm-log.jsonl",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
return loops
|
||||||
|
|
||||||
|
|
||||||
|
def _select_loops(loops: dict, limit=50, agent=None, status_filter=None) -> list:
|
||||||
|
"""Filter/sort/limit a candidate map from _load_loop_candidates."""
|
||||||
# Filter and sort
|
# Filter and sort
|
||||||
result = list(loops.values())
|
result = list(loops.values())
|
||||||
if agent:
|
if agent:
|
||||||
@@ -517,6 +521,17 @@ def reconstruct_loops(limit=50, agent=None, status_filter=None) -> list:
|
|||||||
return result[:limit]
|
return result[:limit]
|
||||||
|
|
||||||
|
|
||||||
|
def reconstruct_loops(limit=50, agent=None, status_filter=None) -> list:
|
||||||
|
"""Reconstruct active and recent loops from followups.json and dm-log.jsonl.
|
||||||
|
|
||||||
|
Returns a list of dicts:
|
||||||
|
loop_id, agent, sender, target, purpose, state, sent_at, deadline,
|
||||||
|
nudges_sent, nudges_allowed, escalate_to, tags, summary
|
||||||
|
"""
|
||||||
|
return _select_loops(_load_loop_candidates(), limit=limit, agent=agent,
|
||||||
|
status_filter=status_filter)
|
||||||
|
|
||||||
|
|
||||||
def get_fleet_loop_health(threshold=None) -> dict:
|
def get_fleet_loop_health(threshold=None) -> dict:
|
||||||
"""Calculate fleet loop health per agent and overall verdict."""
|
"""Calculate fleet loop health per agent and overall verdict."""
|
||||||
if threshold is None:
|
if threshold is None:
|
||||||
@@ -570,8 +585,14 @@ def get_fleet_loop_health(threshold=None) -> dict:
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def diagnose_breaks() -> list:
|
def diagnose_breaks(_fleet_cache=None, _loops_cache=None) -> list:
|
||||||
"""Diagnose break taxonomy across intrinsic loops and support services."""
|
"""Diagnose break taxonomy across intrinsic loops and support services.
|
||||||
|
|
||||||
|
_fleet_cache: optional list; when given, the fleet approval scan result
|
||||||
|
is appended so callers (remediate_breaks) can reuse it instead of
|
||||||
|
re-scanning (each scan fans 6 nodes over the full audit log).
|
||||||
|
_loops_cache: optional list; when given, the parsed loop-candidate map
|
||||||
|
is appended for the same single-parse sharing."""
|
||||||
import subprocess
|
import subprocess
|
||||||
breaks = []
|
breaks = []
|
||||||
|
|
||||||
@@ -609,7 +630,10 @@ def diagnose_breaks() -> list:
|
|||||||
})
|
})
|
||||||
|
|
||||||
# 3. Active follow-up loops check
|
# 3. Active follow-up loops check
|
||||||
active_loops = reconstruct_loops(limit=20, status_filter="pending")
|
_loops_map = _load_loop_candidates()
|
||||||
|
if _loops_cache is not None:
|
||||||
|
_loops_cache.append(_loops_map)
|
||||||
|
active_loops = _select_loops(_loops_map, limit=20, status_filter="pending")
|
||||||
now_ts = time.time()
|
now_ts = time.time()
|
||||||
for l in active_loops:
|
for l in active_loops:
|
||||||
nudges_sent = l.get("nudges_sent", 0)
|
nudges_sent = l.get("nudges_sent", 0)
|
||||||
@@ -628,6 +652,8 @@ def diagnose_breaks() -> list:
|
|||||||
try:
|
try:
|
||||||
import approvals
|
import approvals
|
||||||
fleet_apps = approvals.check_fleet_approvals()
|
fleet_apps = approvals.check_fleet_approvals()
|
||||||
|
if _fleet_cache is not None:
|
||||||
|
_fleet_cache.append(fleet_apps)
|
||||||
for app in fleet_apps:
|
for app in fleet_apps:
|
||||||
if app.get("has_pending"):
|
if app.get("has_pending"):
|
||||||
node = app["node"]
|
node = app["node"]
|
||||||
@@ -734,7 +760,9 @@ def remediate_breaks(dry_run=False) -> dict:
|
|||||||
escalated = []
|
escalated = []
|
||||||
|
|
||||||
# 1. Check diagnosed hard breaks first
|
# 1. Check diagnosed hard breaks first
|
||||||
breaks = diagnose_breaks()
|
_fleet_cache = []
|
||||||
|
_loops_cache = []
|
||||||
|
breaks = diagnose_breaks(_fleet_cache=_fleet_cache, _loops_cache=_loops_cache)
|
||||||
for b in breaks:
|
for b in breaks:
|
||||||
if b.get("severity") in ("CRITICAL", "WARNING"):
|
if b.get("severity") in ("CRITICAL", "WARNING"):
|
||||||
escalated.append(b)
|
escalated.append(b)
|
||||||
@@ -753,8 +781,13 @@ def remediate_breaks(dry_run=False) -> dict:
|
|||||||
|
|
||||||
now_iso = datetime.now(timezone.utc).isoformat()
|
now_iso = datetime.now(timezone.utc).isoformat()
|
||||||
|
|
||||||
# Build answer map from reconstruct_loops
|
# Build answer map from reconstruct_loops (reuse diagnose's parse:
|
||||||
loops = reconstruct_loops(limit=200)
|
# nothing between the parses writes the loop logs in-process, and a
|
||||||
|
# concurrently landed reply is picked up on the next cycle).
|
||||||
|
if _loops_cache:
|
||||||
|
loops = _select_loops(_loops_cache[0], limit=200)
|
||||||
|
else:
|
||||||
|
loops = reconstruct_loops(limit=200)
|
||||||
answered_dms = {
|
answered_dms = {
|
||||||
l["loop_id"]: l for l in loops if l.get("state") in ("ANSWERED", "CLOSED")
|
l["loop_id"]: l for l in loops if l.get("state") in ("ANSWERED", "CLOSED")
|
||||||
}
|
}
|
||||||
@@ -836,7 +869,13 @@ def remediate_breaks(dry_run=False) -> dict:
|
|||||||
# Auto-remediate trusted approval blocks
|
# Auto-remediate trusted approval blocks
|
||||||
try:
|
try:
|
||||||
import approvals
|
import approvals
|
||||||
fleet_apps = approvals.check_fleet_approvals()
|
# Reuse the diagnose_breaks scan: nothing between the scans touches
|
||||||
|
# browser-approval state, and this block only reads it. Fall back to
|
||||||
|
# a fresh scan if the first one failed.
|
||||||
|
if _fleet_cache:
|
||||||
|
fleet_apps = _fleet_cache[0]
|
||||||
|
else:
|
||||||
|
fleet_apps = approvals.check_fleet_approvals()
|
||||||
for app in fleet_apps:
|
for app in fleet_apps:
|
||||||
if app.get("has_pending") and app.get("is_trusted") and app.get("status") != "KEY_APPROVAL":
|
if app.get("has_pending") and app.get("is_trusted") and app.get("status") != "KEY_APPROVAL":
|
||||||
node = app["node"]
|
node = app["node"]
|
||||||
|
|||||||
+138
-11
@@ -41,6 +41,13 @@ NETVM_EXEC = "/home/super/Projects/NetVM/bin/netvm-exec.sh"
|
|||||||
JOB_LOG = NETVM_ROOT / "job-log.jsonl"
|
JOB_LOG = NETVM_ROOT / "job-log.jsonl"
|
||||||
SIDECHAT_STATE = NETVM_ROOT / "job-sidechats.json"
|
SIDECHAT_STATE = NETVM_ROOT / "job-sidechats.json"
|
||||||
|
|
||||||
|
# Sidechat rotation: persistent reuse_key threads accumulate full history
|
||||||
|
# and every dispatch re-sends it (cloud context), so a stale thread burns
|
||||||
|
# full-thread tokens per nod. Cap counted threads by dispatch budget and
|
||||||
|
# flush uncounted legacy threads past the age cap.
|
||||||
|
SIDECHAT_MAX_DISPATCHES = 48
|
||||||
|
SIDECHAT_LEGACY_MAX_AGE_HOURS = 24
|
||||||
|
|
||||||
def load_sidechat_state():
|
def load_sidechat_state():
|
||||||
if SIDECHAT_STATE.exists():
|
if SIDECHAT_STATE.exists():
|
||||||
try:
|
try:
|
||||||
@@ -54,6 +61,40 @@ def save_sidechat_state(state):
|
|||||||
tmp.write_text(json.dumps(state, indent=2))
|
tmp.write_text(json.dumps(state, indent=2))
|
||||||
tmp.replace(SIDECHAT_STATE)
|
tmp.replace(SIDECHAT_STATE)
|
||||||
|
|
||||||
|
|
||||||
|
def should_rotate_sidechat(record, current_title, now=None,
|
||||||
|
max_dispatches=SIDECHAT_MAX_DISPATCHES,
|
||||||
|
legacy_max_age_hours=SIDECHAT_LEGACY_MAX_AGE_HOURS):
|
||||||
|
"""Decide whether a reused sidechat must rotate to a fresh thread.
|
||||||
|
|
||||||
|
Returns (rotate, reason). Rotates when the dispatch budget is spent,
|
||||||
|
the rendered title moved on (daily {date} templates), or an
|
||||||
|
uncounted legacy record is past the age cap. Anything unassessable
|
||||||
|
(plain-UUID records, missing/unparseable age) fails open to reuse.
|
||||||
|
"""
|
||||||
|
now = now or datetime.now(timezone.utc)
|
||||||
|
if not isinstance(record, dict):
|
||||||
|
return False, "unrecorded"
|
||||||
|
count = record.get("dispatch_count")
|
||||||
|
if isinstance(count, int) and count >= max_dispatches:
|
||||||
|
return True, f"dispatch budget spent ({count}/{max_dispatches})"
|
||||||
|
stored_title = record.get("title") or ""
|
||||||
|
ALLOW_SIDECHAT_TITLE_ROTATION = False
|
||||||
|
if ALLOW_SIDECHAT_TITLE_ROTATION and stored_title and current_title and stored_title != current_title:
|
||||||
|
return True, f"title rolled over ({stored_title} -> {current_title})"
|
||||||
|
if count is None:
|
||||||
|
created = record.get("created_at")
|
||||||
|
if created:
|
||||||
|
try:
|
||||||
|
age_h = (now - datetime.fromisoformat(
|
||||||
|
str(created).replace("Z", "+00:00"))).total_seconds() / 3600
|
||||||
|
except Exception:
|
||||||
|
return False, "unparseable age"
|
||||||
|
if age_h > legacy_max_age_hours:
|
||||||
|
return True, (f"predates counting, age {age_h:.0f}h "
|
||||||
|
f"over {legacy_max_age_hours}h cap")
|
||||||
|
return False, "within budget"
|
||||||
|
|
||||||
def extract_uuid(url):
|
def extract_uuid(url):
|
||||||
m = re.search(r"/thread/([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})", url or "")
|
m = re.search(r"/thread/([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})", url or "")
|
||||||
return m.group(1) if m else None
|
return m.group(1) if m else None
|
||||||
@@ -88,6 +129,65 @@ try:
|
|||||||
except ImportError:
|
except ImportError:
|
||||||
HAS_RATE_LIMITER = False
|
HAS_RATE_LIMITER = False
|
||||||
|
|
||||||
|
# Dispatch backpressure (2026-10-09): skip jobs for frozen agents instead of
|
||||||
|
# piling input-waits onto them. See tests/test_dispatch_hold.py.
|
||||||
|
DISPATCH_HOLD_FILE = JOBS_DIR / "dispatch-hold.json"
|
||||||
|
HOLD_WAIT_THRESHOLD = 3
|
||||||
|
HOLD_WAIT_WINDOW_MIN = 60
|
||||||
|
|
||||||
|
|
||||||
|
def dispatch_hold_reason(agent, now=None, hold_path=None, job_log_path=None):
|
||||||
|
# Hold reason if dispatch to agent must be skipped, else None.
|
||||||
|
# Explicit operator holds win; otherwise auto-hold after repeated waits.
|
||||||
|
from datetime import timedelta
|
||||||
|
now = now or datetime.now(timezone.utc)
|
||||||
|
try:
|
||||||
|
with open(hold_path or DISPATCH_HOLD_FILE) as f:
|
||||||
|
holds = json.load(f)
|
||||||
|
except (OSError, ValueError):
|
||||||
|
holds = {}
|
||||||
|
entry = holds.get(agent) if isinstance(holds, dict) else None
|
||||||
|
if isinstance(entry, dict):
|
||||||
|
until = entry.get("until")
|
||||||
|
if until:
|
||||||
|
try:
|
||||||
|
exp = datetime.fromisoformat(until)
|
||||||
|
if exp.tzinfo is None:
|
||||||
|
exp = exp.replace(tzinfo=timezone.utc)
|
||||||
|
except ValueError:
|
||||||
|
exp = None
|
||||||
|
if exp is not None and exp <= now:
|
||||||
|
entry = None
|
||||||
|
if entry is not None:
|
||||||
|
return "explicit hold (%s)" % entry.get("reason", "operator")
|
||||||
|
try:
|
||||||
|
cutoff = now - timedelta(minutes=HOLD_WAIT_WINDOW_MIN)
|
||||||
|
n = 0
|
||||||
|
with open(job_log_path or JOB_LOG) as f:
|
||||||
|
for line in f:
|
||||||
|
try:
|
||||||
|
r = json.loads(line)
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
if r.get("type") != "job_dispatch_agent_input_wait":
|
||||||
|
continue
|
||||||
|
if r.get("agent") != agent:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
ts = datetime.fromisoformat(r.get("ts", ""))
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
if ts.tzinfo is None:
|
||||||
|
ts = ts.replace(tzinfo=timezone.utc)
|
||||||
|
if ts >= cutoff:
|
||||||
|
n += 1
|
||||||
|
if n >= HOLD_WAIT_THRESHOLD:
|
||||||
|
return "auto-hold (%d input-waits in last %dm)" % (n, HOLD_WAIT_WINDOW_MIN)
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def log_event(event_type, data):
|
def log_event(event_type, data):
|
||||||
"""Append event to job-log.jsonl"""
|
"""Append event to job-log.jsonl"""
|
||||||
entry = {
|
entry = {
|
||||||
@@ -395,6 +495,13 @@ def main():
|
|||||||
# Load job
|
# Load job
|
||||||
job = load_job(job_name)
|
job = load_job(job_name)
|
||||||
|
|
||||||
|
# Backpressure: skip frozen agents before arming follow-ups or sending.
|
||||||
|
_hold = dispatch_hold_reason(job.get("agent"))
|
||||||
|
if _hold:
|
||||||
|
print("Held: job %s for %s skipped (%s)." % (job_name, job.get("agent"), _hold), file=sys.stderr)
|
||||||
|
log_event("job_dispatch_held", {"job_name": job_name, "agent": job.get("agent"), "reason": _hold})
|
||||||
|
sys.exit(0)
|
||||||
|
|
||||||
# Generate job_id
|
# Generate job_id
|
||||||
job_id = f"{job_name}-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}-{uuid.uuid4().hex[:8]}"
|
job_id = f"{job_name}-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}-{uuid.uuid4().hex[:8]}"
|
||||||
|
|
||||||
@@ -461,19 +568,33 @@ def main():
|
|||||||
# Check if reuse_key exists in job-sidechats.json and thread is still alive
|
# Check if reuse_key exists in job-sidechats.json and thread is still alive
|
||||||
sc_state = load_sidechat_state()
|
sc_state = load_sidechat_state()
|
||||||
reused_uuid = None
|
reused_uuid = None
|
||||||
|
rotated_from = None
|
||||||
if reuse_key and reuse_key in sc_state:
|
if reuse_key and reuse_key in sc_state:
|
||||||
val = sc_state[reuse_key]
|
val = sc_state[reuse_key]
|
||||||
cand_uuid = val.get("thread_uuid") if isinstance(val, dict) else val
|
cand_uuid = val.get("thread_uuid") if isinstance(val, dict) else val
|
||||||
if cand_uuid:
|
if cand_uuid:
|
||||||
try:
|
rotate, reason = should_rotate_sidechat(val, sc_name)
|
||||||
import muse_hybrid
|
if rotate:
|
||||||
threads, err = muse_hybrid.get_threads(agent)
|
print(f"Rotating sidechat '{reuse_key}': {reason}")
|
||||||
if not err and threads:
|
log_event("job_sidechat_rotate", {
|
||||||
thread_ids = [t.get("session_id") for t in threads]
|
"job_name": job_name, "job_id": job_id,
|
||||||
if cand_uuid in thread_ids:
|
"reuse_key": reuse_key, "old_thread": cand_uuid,
|
||||||
reused_uuid = cand_uuid
|
"reason": reason,
|
||||||
except Exception:
|
})
|
||||||
pass
|
rotated_from = cand_uuid
|
||||||
|
else:
|
||||||
|
try:
|
||||||
|
import muse_hybrid
|
||||||
|
threads, err = muse_hybrid.get_threads(agent)
|
||||||
|
if not err and threads:
|
||||||
|
thread_ids = [t.get("session_id") for t in threads]
|
||||||
|
if cand_uuid in thread_ids:
|
||||||
|
reused_uuid = cand_uuid
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
if reused_uuid and isinstance(val, dict):
|
||||||
|
val["dispatch_count"] = val.get("dispatch_count", 0) + 1
|
||||||
|
save_sidechat_state(sc_state)
|
||||||
|
|
||||||
if reused_uuid:
|
if reused_uuid:
|
||||||
target = reused_uuid
|
target = reused_uuid
|
||||||
@@ -487,13 +608,19 @@ def main():
|
|||||||
new_uuid = res.get("session_id")
|
new_uuid = res.get("session_id")
|
||||||
key_to_save = reuse_key or sc_name
|
key_to_save = reuse_key or sc_name
|
||||||
is_persistent = bool(reuse_key)
|
is_persistent = bool(reuse_key)
|
||||||
sc_state[key_to_save] = {
|
new_record = {
|
||||||
"thread_uuid": new_uuid,
|
"thread_uuid": new_uuid,
|
||||||
"agent": agent,
|
"agent": agent,
|
||||||
"title": channel_title,
|
"title": channel_title,
|
||||||
"type": "persistent" if is_persistent else "ephemeral",
|
"type": "persistent" if is_persistent else "ephemeral",
|
||||||
"created_at": datetime.now(timezone.utc).isoformat()
|
"created_at": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"dispatch_count": 1,
|
||||||
}
|
}
|
||||||
|
if rotated_from:
|
||||||
|
new_record["rotated_from"] = rotated_from
|
||||||
|
new_record["rotated_at"] = datetime.now(
|
||||||
|
timezone.utc).isoformat()
|
||||||
|
sc_state[key_to_save] = new_record
|
||||||
save_sidechat_state(sc_state)
|
save_sidechat_state(sc_state)
|
||||||
target = new_uuid
|
target = new_uuid
|
||||||
print(f"Spawned new sidechat channel '{channel_title}' ({new_uuid}) for {agent}")
|
print(f"Spawned new sidechat channel '{channel_title}' ({new_uuid}) for {agent}")
|
||||||
|
|||||||
+3
-2
@@ -281,8 +281,9 @@ def generate_preservation_advisory(
|
|||||||
tips = []
|
tips = []
|
||||||
|
|
||||||
pct = weekly_used_pct or 0
|
pct = weekly_used_pct or 0
|
||||||
if pct >= 95 or "0 tokens left" in extra_tokens_remaining:
|
is_bonus_empty = ("0 tokens left" in extra_tokens_remaining) or (not extra_tokens_remaining)
|
||||||
return "CRITICAL: Quota exhausted. Do NOT send chat messages. Salvage via 'box onboard start <new_node> --for %s'." % node
|
if pct >= 95 and is_bonus_empty:
|
||||||
|
return "CRITICAL: Quota exhausted. Salvage via 'box onboard start <new_node> --for %s'." % node
|
||||||
|
|
||||||
if pct >= 70:
|
if pct >= 70:
|
||||||
tips.append("Quota > 70%%: Cease prose chatter; offload tasks to background tmux workers.")
|
tips.append("Quota > 70%%: Cease prose chatter; offload tasks to background tmux workers.")
|
||||||
|
|||||||
+22
-1
@@ -117,6 +117,27 @@ def ev(ws, expr, await_p=False):
|
|||||||
print(f"CDP evaluate failed: {type(e).__name__}: {e}", file=sys.stderr)
|
print(f"CDP evaluate failed: {type(e).__name__}: {e}", file=sys.stderr)
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _is_valid_ipv4(ip: str) -> bool:
|
||||||
|
"""Strict IPv4 validation: four octets, each 0-255, no leading zeros.
|
||||||
|
|
||||||
|
P1 fix (2026-10-08): the old \d{1,3} pattern matched invalid IPs like
|
||||||
|
999.999.999.999 and version strings. Only strict IPv4 passes.
|
||||||
|
"""
|
||||||
|
if not ip or not isinstance(ip, str):
|
||||||
|
return False
|
||||||
|
parts = ip.split(".")
|
||||||
|
if len(parts) != 4:
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
return all(
|
||||||
|
0 <= int(part) <= 255 and part == str(int(part))
|
||||||
|
for part in parts
|
||||||
|
)
|
||||||
|
except ValueError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def check_approvals(ws):
|
def check_approvals(ws):
|
||||||
"""
|
"""
|
||||||
Check for browser permission dialogs.
|
Check for browser permission dialogs.
|
||||||
@@ -175,7 +196,7 @@ def check_approvals(ws):
|
|||||||
for d in dialogs:
|
for d in dialogs:
|
||||||
# Extract IP if present
|
# Extract IP if present
|
||||||
import re
|
import re
|
||||||
ips = re.findall(r'\b\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}\b', d)
|
ips = [ip for ip in re.findall(r'\b\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}\b', d) if _is_valid_ipv4(ip)]
|
||||||
# Check trust: if IP present, must be in TRUSTED_IPS; if no IP, untrusted approval dialog
|
# Check trust: if IP present, must be in TRUSTED_IPS; if no IP, untrusted approval dialog
|
||||||
if ips:
|
if ips:
|
||||||
is_trusted = any(ip in TRUSTED_IPS for ip in ips)
|
is_trusted = any(ip in TRUSTED_IPS for ip in ips)
|
||||||
|
|||||||
+442
-22
@@ -16,6 +16,7 @@ Dual-mode interface:
|
|||||||
- Job Scheduler & Dispatch trigger
|
- Job Scheduler & Dispatch trigger
|
||||||
- Background Tmux sessions & Swarm worker monitor
|
- Background Tmux sessions & Swarm worker monitor
|
||||||
- Live Event & DM log tailer
|
- Live Event & DM log tailer
|
||||||
|
- Container SSH tunnel health & tmux pop-out dialer
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import sys
|
import sys
|
||||||
@@ -27,6 +28,7 @@ import threading
|
|||||||
import subprocess
|
import subprocess
|
||||||
import hashlib
|
import hashlib
|
||||||
import select
|
import select
|
||||||
|
import shlex
|
||||||
import signal
|
import signal
|
||||||
import textwrap
|
import textwrap
|
||||||
import urllib.request
|
import urllib.request
|
||||||
@@ -169,6 +171,85 @@ def format_recency(ts: float) -> str:
|
|||||||
return "never"
|
return "never"
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# SSH / Container Tunnel Management Subsystem
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
SSH_JUMP_HOST = os.environ.get("SSH_JUMP_HOST", "34.139.37.135")
|
||||||
|
SSH_OPERATOR_USER = os.environ.get("OPERATOR_USER", "super")
|
||||||
|
SSH_IDENTITY_FILE = os.environ.get("SSH_IDENTITY_FILE", "")
|
||||||
|
|
||||||
|
try:
|
||||||
|
from agent_md import TUNNEL_PORTS as SSH_TUNNEL_PORTS
|
||||||
|
except Exception:
|
||||||
|
SSH_TUNNEL_PORTS = {
|
||||||
|
"muse-main": {"port": 2224, "terminal": 7681, "user": "muse"},
|
||||||
|
"muse": {"port": 2225, "terminal": 7682, "user": "hatch"},
|
||||||
|
"646": {"port": 2226, "terminal": 7683, "user": "hatch"},
|
||||||
|
"pip": {"port": 2227, "terminal": 7684, "user": "hatch"},
|
||||||
|
"opm": {"port": 2228, "terminal": 7685, "user": "hatch"},
|
||||||
|
"def": {"port": 2229, "terminal": 7686, "user": "hatch"},
|
||||||
|
"dev": {"port": 2230, "terminal": 7687, "user": "hatch"},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_ssh_dial_command(account: str, port=None, user=None, jump_host=None,
|
||||||
|
operator_user=None, identity_file=None,
|
||||||
|
ssh_options=None, remote_command=None) -> list:
|
||||||
|
"""Build the jump-host dial argv for an agent container.
|
||||||
|
|
||||||
|
ssh_options are inserted before the destination; remote_command (str or
|
||||||
|
list) is appended after it for non-interactive probes.
|
||||||
|
"""
|
||||||
|
info = SSH_TUNNEL_PORTS.get(account, {})
|
||||||
|
port = port or info.get("port")
|
||||||
|
user = user or info.get("user", "hatch")
|
||||||
|
jump_host = jump_host or SSH_JUMP_HOST
|
||||||
|
operator_user = operator_user or SSH_OPERATOR_USER
|
||||||
|
if identity_file is None:
|
||||||
|
identity_file = SSH_IDENTITY_FILE
|
||||||
|
cmd = ["ssh", "-o", "StrictHostKeyChecking=no"]
|
||||||
|
if identity_file:
|
||||||
|
cmd += ["-o", "IdentitiesOnly=yes", "-i", identity_file]
|
||||||
|
if ssh_options:
|
||||||
|
cmd += list(ssh_options)
|
||||||
|
cmd += ["-J", f"{operator_user}@{jump_host}", "-p", str(port), f"{user}@localhost"]
|
||||||
|
if remote_command:
|
||||||
|
cmd += [remote_command] if isinstance(remote_command, str) else list(remote_command)
|
||||||
|
return cmd
|
||||||
|
|
||||||
|
|
||||||
|
def build_ssh_dial_string(account: str, **kwargs) -> str:
|
||||||
|
"""Shell-quoted dial command for display, clipboard copy, and pop-out."""
|
||||||
|
return " ".join(shlex.quote(p) for p in build_ssh_dial_command(account, **kwargs))
|
||||||
|
|
||||||
|
|
||||||
|
def build_ssh_popout_shell(account: str, **kwargs) -> str:
|
||||||
|
"""Interactive shell line for the pop-out window: ssh, then keep a shell."""
|
||||||
|
dial = build_ssh_dial_string(account, **kwargs)
|
||||||
|
return f"{dial}; echo '[ssh exited ($?) — window kept open, exit to close]'; exec \"${{SHELL:-/bin/bash}}\""
|
||||||
|
|
||||||
|
|
||||||
|
def build_tmux_popout_command(label: str, shell_command: str, socket_path: str = None) -> list:
|
||||||
|
"""Build `tmux new-window` argv opening shell_command in a fresh window."""
|
||||||
|
safe_label = re.sub(r"[^A-Za-z0-9_.-]", "-", label)[:32] or "ssh"
|
||||||
|
cmd = ["tmux"]
|
||||||
|
if socket_path:
|
||||||
|
cmd += ["-S", socket_path]
|
||||||
|
return cmd + ["new-window", "-n", safe_label, shell_command]
|
||||||
|
|
||||||
|
|
||||||
|
def ssh_row_order(nodes: list, extra_accounts=()) -> list:
|
||||||
|
"""Fleet nodes first, then any extra tunnel accounts (e.g. muse-main)."""
|
||||||
|
rows = list(nodes)
|
||||||
|
for acct in extra_accounts:
|
||||||
|
if acct not in rows:
|
||||||
|
rows.append(acct)
|
||||||
|
for acct in SSH_TUNNEL_PORTS:
|
||||||
|
if acct not in rows:
|
||||||
|
rows.append(acct)
|
||||||
|
return rows
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Prompt & Skill Library Subsystem
|
# Prompt & Skill Library Subsystem
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -337,6 +418,31 @@ class PromptManager:
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Box Mode Tab Bar (single source of truth for renderer + click handler)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
BOX_TABS = [
|
||||||
|
"1: Agent Chat",
|
||||||
|
"2: Fleet Status",
|
||||||
|
"3: Approvals",
|
||||||
|
"4: Jobs Scheduler",
|
||||||
|
"5: Tmux / Swarms",
|
||||||
|
"6: DM Logs",
|
||||||
|
"7: SSH / Boxes",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def box_tab_bounds(tabs=None, x: int = 0) -> list:
|
||||||
|
"""Clickable x-ranges for the Box tab bar, mirroring _render_box_tabs."""
|
||||||
|
bounds = []
|
||||||
|
cur_x = x + 1
|
||||||
|
for tab_name in (tabs if tabs is not None else BOX_TABS):
|
||||||
|
label = f" [{tab_name}] "
|
||||||
|
bounds.append((cur_x, cur_x + len(label) - 1))
|
||||||
|
cur_x += len(label) + 1
|
||||||
|
return bounds
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Data Layer & Async Poller
|
# Data Layer & Async Poller
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -368,6 +474,13 @@ class FleetDataManager:
|
|||||||
self.tmux_cache = []
|
self.tmux_cache = []
|
||||||
self.dm_logs_cache = []
|
self.dm_logs_cache = []
|
||||||
|
|
||||||
|
# SSH / container tunnel health (Box tab 7)
|
||||||
|
self.ssh_cache = {} # account -> health dict from ssh-check + state_since
|
||||||
|
self.ssh_jump_reachable = None # None = never checked
|
||||||
|
self.ssh_checked_at = 0.0
|
||||||
|
self.ssh_check_latency_ms = None
|
||||||
|
self.ssh_check_error = ""
|
||||||
|
|
||||||
# Interaction ranking: node -> float timestamp of last true input / chat [insert]
|
# Interaction ranking: node -> float timestamp of last true input / chat [insert]
|
||||||
self.agent_interactions = {n: 0.0 for n in self.nodes}
|
self.agent_interactions = {n: 0.0 for n in self.nodes}
|
||||||
self._load_agent_interactions()
|
self._load_agent_interactions()
|
||||||
@@ -405,6 +518,8 @@ class FleetDataManager:
|
|||||||
self.preload_priority_chats(self.active_node, sidechat_limit=0)
|
self.preload_priority_chats(self.active_node, sidechat_limit=0)
|
||||||
self.poller_thread = threading.Thread(target=self._worker_loop, daemon=True)
|
self.poller_thread = threading.Thread(target=self._worker_loop, daemon=True)
|
||||||
self.poller_thread.start()
|
self.poller_thread.start()
|
||||||
|
# First SSH sweep in background so Box tab 7 is warm on open
|
||||||
|
threading.Thread(target=self._fetch_ssh_health, daemon=True).start()
|
||||||
else:
|
else:
|
||||||
self.poller_thread = None
|
self.poller_thread = None
|
||||||
|
|
||||||
@@ -888,6 +1003,7 @@ class FleetDataManager:
|
|||||||
last_med = 0.0
|
last_med = 0.0
|
||||||
last_slow = 0.0
|
last_slow = 0.0
|
||||||
last_fleet_approvals = 0.0
|
last_fleet_approvals = 0.0
|
||||||
|
last_ssh = 0.0
|
||||||
|
|
||||||
# Initial fetch of active node main chat only
|
# Initial fetch of active node main chat only
|
||||||
with self.lock:
|
with self.lock:
|
||||||
@@ -950,6 +1066,11 @@ class FleetDataManager:
|
|||||||
self._fetch_dm_logs()
|
self._fetch_dm_logs()
|
||||||
last_slow = now
|
last_slow = now
|
||||||
|
|
||||||
|
# 5. SSH tunnel health (every 30s): single VM-side sweep
|
||||||
|
if (now - last_ssh >= 30.0):
|
||||||
|
self._fetch_ssh_health()
|
||||||
|
last_ssh = now
|
||||||
|
|
||||||
# Sleep in short increments to allow prompt wakeup on user actions
|
# Sleep in short increments to allow prompt wakeup on user actions
|
||||||
for _ in range(10):
|
for _ in range(10):
|
||||||
if not self.running or (hasattr(self, 'user_poll_trigger') and self.user_poll_trigger.is_set()):
|
if not self.running or (hasattr(self, 'user_poll_trigger') and self.user_poll_trigger.is_set()):
|
||||||
@@ -1181,6 +1302,68 @@ class FleetDataManager:
|
|||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
def _apply_ssh_check_result(self, data: dict, now: float = None):
|
||||||
|
"""Merge one ssh-check payload into ssh_cache with flap tracking.
|
||||||
|
|
||||||
|
state_since records the last (ssh_up, term_up) transition per
|
||||||
|
account so the SSH view can show uptime/downtime durations.
|
||||||
|
"""
|
||||||
|
now = now if now is not None else time.time()
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
return
|
||||||
|
accounts = data.get("accounts", {})
|
||||||
|
if not isinstance(accounts, dict):
|
||||||
|
accounts = {}
|
||||||
|
with self.lock:
|
||||||
|
self.ssh_jump_reachable = data.get("jump_reachable")
|
||||||
|
self.ssh_checked_at = now
|
||||||
|
self.ssh_check_latency_ms = data.get("latency_ms")
|
||||||
|
self.ssh_check_error = "" if data.get("jump_reachable") else str(data.get("error", ""))
|
||||||
|
for acct, info in accounts.items():
|
||||||
|
if not isinstance(info, dict):
|
||||||
|
continue
|
||||||
|
prev = self.ssh_cache.get(acct, {})
|
||||||
|
entry = dict(info)
|
||||||
|
prev_state = (prev.get("ssh_up"), prev.get("term_up"))
|
||||||
|
new_state = (entry.get("ssh_up"), entry.get("term_up"))
|
||||||
|
if prev_state != new_state or "state_since" not in prev:
|
||||||
|
entry["state_since"] = now
|
||||||
|
entry["prev_ssh_up"] = prev.get("ssh_up")
|
||||||
|
else:
|
||||||
|
entry["state_since"] = prev.get("state_since", now)
|
||||||
|
entry["prev_ssh_up"] = prev.get("prev_ssh_up")
|
||||||
|
self.ssh_cache[acct] = entry
|
||||||
|
|
||||||
|
def _fetch_ssh_health(self):
|
||||||
|
"""Run box-ctl ssh-check (single VM-side sweep) and merge results."""
|
||||||
|
try:
|
||||||
|
cmd = ["python3", str(BIN_DIR / "box-ctl.py"), "ssh-check"]
|
||||||
|
res = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
|
||||||
|
if res.returncode == 0:
|
||||||
|
try:
|
||||||
|
data = json.loads(res.stdout)
|
||||||
|
except Exception:
|
||||||
|
return
|
||||||
|
if isinstance(data, dict) and data.get("ok"):
|
||||||
|
self._apply_ssh_check_result(data)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def probe_container_uptime(self, account: str) -> tuple[bool, str]:
|
||||||
|
"""On-demand end-to-end probe: run `uptime` inside the container."""
|
||||||
|
if account not in SSH_TUNNEL_PORTS:
|
||||||
|
return False, f"No tunnel registered for '{account}'"
|
||||||
|
cmd = build_ssh_dial_command(
|
||||||
|
account,
|
||||||
|
ssh_options=["-o", "BatchMode=yes", "-o", "ConnectTimeout=12"],
|
||||||
|
remote_command="uptime",
|
||||||
|
)
|
||||||
|
rc, stdout, stderr = run_command_isolated(cmd, timeout=25.0)
|
||||||
|
if rc == 0 and (stdout or "").strip():
|
||||||
|
return True, stdout.strip()
|
||||||
|
err_lines = (stderr or stdout or f"exit {rc}").strip().splitlines()
|
||||||
|
return False, (err_lines[-1] if err_lines else f"exit {rc}")[:200]
|
||||||
|
|
||||||
def _fetch_dm_logs(self):
|
def _fetch_dm_logs(self):
|
||||||
log_path = REPO_ROOT / "dm-log.jsonl"
|
log_path = REPO_ROOT / "dm-log.jsonl"
|
||||||
if not log_path.exists():
|
if not log_path.exists():
|
||||||
@@ -1362,10 +1545,13 @@ class FleetDataManager:
|
|||||||
class MuseTUI:
|
class MuseTUI:
|
||||||
"""Full-terminal curses application supporting Muse and Box operational modes."""
|
"""Full-terminal curses application supporting Muse and Box operational modes."""
|
||||||
|
|
||||||
def __init__(self, stdscr, initial_mode="muse", initial_node=None, initial_thread=None):
|
def __init__(self, stdscr, initial_mode="muse", initial_node=None, initial_thread=None, initial_tab=None):
|
||||||
self.stdscr = stdscr
|
self.stdscr = stdscr
|
||||||
self.mode = initial_mode # "muse" or "box"
|
self.mode = initial_mode # "muse" or "box"
|
||||||
self.box_tab = 0 # 0: Chat, 1: Fleet, 2: Approvals, 3: Jobs, 4: Tmux, 5: Logs
|
self.box_tab = 0 # 0: Chat, 1: Fleet, 2: Approvals, 3: Jobs, 4: Tmux, 5: Logs, 6: SSH
|
||||||
|
if initial_mode == "box" and initial_tab is not None:
|
||||||
|
tab_map = {"chat": 0, "fleet": 1, "approvals": 2, "jobs": 3, "tmux": 4, "logs": 5, "ssh": 6}
|
||||||
|
self.box_tab = tab_map.get(str(initial_tab).lower(), 0)
|
||||||
self.data = FleetDataManager()
|
self.data = FleetDataManager()
|
||||||
|
|
||||||
# Selection state: default to top of sorted list (highest unread / most recently interacted)
|
# Selection state: default to top of sorted list (highest unread / most recently interacted)
|
||||||
@@ -1408,6 +1594,8 @@ class MuseTUI:
|
|||||||
self.jobs_sel_idx = 0
|
self.jobs_sel_idx = 0
|
||||||
self.jobs_scroll_start = 0
|
self.jobs_scroll_start = 0
|
||||||
self.tmux_sel_idx = 0
|
self.tmux_sel_idx = 0
|
||||||
|
self.ssh_sel_idx = 0
|
||||||
|
self.ssh_scroll_idx = 0
|
||||||
self.table_scroll_idx = 0
|
self.table_scroll_idx = 0
|
||||||
|
|
||||||
# Input buffer
|
# Input buffer
|
||||||
@@ -1674,6 +1862,8 @@ class MuseTUI:
|
|||||||
self._render_tmux_view(content_y, 0, content_h, w)
|
self._render_tmux_view(content_y, 0, content_h, w)
|
||||||
elif self.box_tab == 5:
|
elif self.box_tab == 5:
|
||||||
self._render_dm_logs_view(content_y, 0, content_h, w)
|
self._render_dm_logs_view(content_y, 0, content_h, w)
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
self._render_ssh_view(content_y, 0, content_h, w)
|
||||||
|
|
||||||
# 4. Bottom Input Bar & Toast
|
# 4. Bottom Input Bar & Toast
|
||||||
self._render_bottom_bar(h - bottom_bar_h, 0, bottom_bar_h, w)
|
self._render_bottom_bar(h - bottom_bar_h, 0, bottom_bar_h, w)
|
||||||
@@ -1764,14 +1954,7 @@ class MuseTUI:
|
|||||||
self.safe_addstr(self.stdscr, y, w - len(clock) - 2, clock, self._attr("header"))
|
self.safe_addstr(self.stdscr, y, w - len(clock) - 2, clock, self._attr("header"))
|
||||||
|
|
||||||
def _render_box_tabs(self, y: int, x: int, w: int):
|
def _render_box_tabs(self, y: int, x: int, w: int):
|
||||||
tabs = [
|
tabs = BOX_TABS
|
||||||
"1: Agent Chat",
|
|
||||||
"2: Fleet Status",
|
|
||||||
"3: Approvals",
|
|
||||||
"4: Jobs Scheduler",
|
|
||||||
"5: Tmux / Swarms",
|
|
||||||
"6: DM Logs",
|
|
||||||
]
|
|
||||||
self.safe_addstr(self.stdscr, y, x, " " * w, self._attr("dim"))
|
self.safe_addstr(self.stdscr, y, x, " " * w, self._attr("dim"))
|
||||||
cur_x = x + 1
|
cur_x = x + 1
|
||||||
for idx, tab_name in enumerate(tabs):
|
for idx, tab_name in enumerate(tabs):
|
||||||
@@ -2839,6 +3022,106 @@ class MuseTUI:
|
|||||||
self.safe_addstr(self.stdscr, row_y, x + 2, line_str, color)
|
self.safe_addstr(self.stdscr, row_y, x + 2, line_str, color)
|
||||||
row_y += 1
|
row_y += 1
|
||||||
|
|
||||||
|
def get_ssh_rows(self) -> list:
|
||||||
|
"""SSH view row order: fleet nodes first, then extra tunnel accounts."""
|
||||||
|
with self.data.lock:
|
||||||
|
extras = list(self.data.ssh_cache.keys())
|
||||||
|
nodes = list(self.data.nodes)
|
||||||
|
return ssh_row_order(nodes, extras)
|
||||||
|
|
||||||
|
def _render_ssh_view(self, y: int, x: int, h: int, w: int):
|
||||||
|
self.safe_addstr(self.stdscr, y, x + 1, f"CONTAINER SSH TUNNEL HEALTH (jump: {SSH_OPERATOR_USER}@{SSH_JUMP_HOST})", self._attr("bold"))
|
||||||
|
|
||||||
|
with self.data.lock:
|
||||||
|
jump = self.data.ssh_jump_reachable
|
||||||
|
checked_at = self.data.ssh_checked_at
|
||||||
|
sweep_ms = self.data.ssh_check_latency_ms
|
||||||
|
check_err = self.data.ssh_check_error
|
||||||
|
ssh_cache = dict(self.data.ssh_cache)
|
||||||
|
|
||||||
|
if jump is None:
|
||||||
|
jump_txt, jump_attr = "sweep pending…", self._attr("dim")
|
||||||
|
elif jump:
|
||||||
|
jump_txt = f"jump OK (sweep {sweep_ms}ms, checked {format_recency(checked_at)})"
|
||||||
|
jump_attr = self._attr("success")
|
||||||
|
else:
|
||||||
|
jump_txt = f"jump UNREACHABLE ({(check_err or 'unknown')[:w - 24]})"
|
||||||
|
jump_attr = self._attr("danger")
|
||||||
|
self.safe_addstr(self.stdscr, y + 1, x + 1, jump_txt[:w - 2], jump_attr)
|
||||||
|
self.safe_addstr(self.stdscr, y + 2, x + 1, " ACCOUNT SSH PORT SSH STATE LAT SSH BANNER / HOSTKEY TERM PORT TERM STATE SINCE", self._attr("dim"))
|
||||||
|
self.safe_addstr(self.stdscr, y + 3, x + 1, "─" * (w - 2), self._attr("dim"))
|
||||||
|
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
if self.ssh_sel_idx >= len(rows):
|
||||||
|
self.ssh_sel_idx = max(0, len(rows) - 1)
|
||||||
|
|
||||||
|
visible_rows = max(3, h - 6)
|
||||||
|
scroll_start = getattr(self, "ssh_scroll_idx", 0)
|
||||||
|
if self.ssh_sel_idx < scroll_start:
|
||||||
|
scroll_start = self.ssh_sel_idx
|
||||||
|
elif self.ssh_sel_idx >= scroll_start + visible_rows:
|
||||||
|
scroll_start = self.ssh_sel_idx - visible_rows + 1
|
||||||
|
scroll_start = max(0, min(scroll_start, max(0, len(rows) - visible_rows)))
|
||||||
|
self.ssh_scroll_idx = scroll_start
|
||||||
|
|
||||||
|
row_y = y + 4
|
||||||
|
for row_i in range(visible_rows):
|
||||||
|
idx = scroll_start + row_i
|
||||||
|
if idx >= len(rows):
|
||||||
|
break
|
||||||
|
acct = rows[idx]
|
||||||
|
is_sel = (idx == self.ssh_sel_idx)
|
||||||
|
row_attr = self._attr("selected") if is_sel else self._attr("normal")
|
||||||
|
info = SSH_TUNNEL_PORTS.get(acct, {})
|
||||||
|
health = ssh_cache.get(acct, {})
|
||||||
|
sport = info.get("port", "?")
|
||||||
|
tport = info.get("terminal", "?")
|
||||||
|
|
||||||
|
ssh_up = health.get("ssh_up")
|
||||||
|
term_up = health.get("term_up")
|
||||||
|
if ssh_up is True:
|
||||||
|
ssh_txt, ssh_attr = "UP ", self._attr("success")
|
||||||
|
elif ssh_up is False:
|
||||||
|
ssh_txt, ssh_attr = "DOWN", self._attr("danger")
|
||||||
|
else:
|
||||||
|
ssh_txt, ssh_attr = "?? ", self._attr("dim")
|
||||||
|
if term_up is True:
|
||||||
|
term_txt, term_attr = "UP ", self._attr("success")
|
||||||
|
elif term_up is False:
|
||||||
|
term_txt, term_attr = "DOWN", self._attr("danger")
|
||||||
|
else:
|
||||||
|
term_txt, term_attr = "?? ", self._attr("dim")
|
||||||
|
if is_sel:
|
||||||
|
ssh_attr = row_attr
|
||||||
|
term_attr = row_attr
|
||||||
|
|
||||||
|
lat = health.get("ssh_latency_ms")
|
||||||
|
lat_txt = f"{lat}ms" if lat is not None else "--"
|
||||||
|
banner = (health.get("ssh_banner") or health.get("term_http") or "-").strip() or "-"
|
||||||
|
since_ts = health.get("state_since", 0.0)
|
||||||
|
if ssh_up is None and term_up is None:
|
||||||
|
since_txt = "never checked" if jump is None else "unknown"
|
||||||
|
else:
|
||||||
|
since_txt = f"{'up' if ssh_up else 'down'} {format_recency(since_ts)}"
|
||||||
|
|
||||||
|
head = "▶ " if is_sel else " "
|
||||||
|
name_attr = row_attr if is_sel else self._attr("bold")
|
||||||
|
self.safe_addstr(self.stdscr, row_y, x + 1, f"{head}{acct:<10}"[:12], name_attr)
|
||||||
|
self.safe_addstr(self.stdscr, row_y, x + 13, f":{sport:<8}", row_attr if is_sel else self._attr("dim"))
|
||||||
|
self.safe_addstr(self.stdscr, row_y, x + 23, ssh_txt, ssh_attr)
|
||||||
|
self.safe_addstr(self.stdscr, row_y, x + 32, f"{lat_txt:<7}", row_attr if is_sel else self._attr("dim"))
|
||||||
|
self.safe_addstr(self.stdscr, row_y, x + 40, banner[:26].ljust(26), row_attr if is_sel else self._attr("normal"))
|
||||||
|
self.safe_addstr(self.stdscr, row_y, x + 67, f":{tport:<8}", row_attr if is_sel else self._attr("dim"))
|
||||||
|
self.safe_addstr(self.stdscr, row_y, x + 77, term_txt, term_attr)
|
||||||
|
self.safe_addstr(self.stdscr, row_y, x + 83, since_txt[:w - 84 - 8], row_attr if is_sel else self._attr("dim"))
|
||||||
|
self.safe_addstr(self.stdscr, row_y, max(x + 90, w - 8), "[SSH]", self._attr("wo_badge") if is_sel else self._attr("dim"))
|
||||||
|
row_y += 1
|
||||||
|
|
||||||
|
hint_y = y + h - 1
|
||||||
|
sel_acct = rows[self.ssh_sel_idx].upper() if rows else "-"
|
||||||
|
hints = f"Selected: [{sel_acct}] [Enter/s]: SSH pop-out (tmux) [c]: Copy dial [u]: Container uptime [r]: Refresh [j/k]: Nav"
|
||||||
|
self.safe_addstr(self.stdscr, hint_y, x + 1, hints[:w - 2], self._attr("dim"))
|
||||||
|
|
||||||
# -----------------------------------------------------------------------
|
# -----------------------------------------------------------------------
|
||||||
# Chat History Sends Search & Prompts Subsystem
|
# Chat History Sends Search & Prompts Subsystem
|
||||||
# -----------------------------------------------------------------------
|
# -----------------------------------------------------------------------
|
||||||
@@ -3210,6 +3493,8 @@ class MuseTUI:
|
|||||||
("[w] or [/wo]", "Compose and cryptographically sign a Work Order"),
|
("[w] or [/wo]", "Compose and cryptographically sign a Work Order"),
|
||||||
("[a] or [F2]", "Open Approvals Resolution Drawer (Allow, Always, Deny)"),
|
("[a] or [F2]", "Open Approvals Resolution Drawer (Allow, Always, Deny)"),
|
||||||
("[F5] or [m]", "Toggle between Muse Chat TUI and Box Fleet Command TUI"),
|
("[F5] or [m]", "Toggle between Muse Chat TUI and Box Fleet Command TUI"),
|
||||||
|
("[1]-[7] (Box)", "Switch Box tabs: Chat/Fleet/Approvals/Jobs/Tmux/Logs/SSH"),
|
||||||
|
("[Tab 7: SSH]", "Enter/s: tmux pop-out c: copy dial u: container uptime r: refresh"),
|
||||||
("[g] / [G] / [Home/End]", "Jump to oldest message / follow live latest message"),
|
("[g] / [G] / [Home/End]", "Jump to oldest message / follow live latest message"),
|
||||||
("[q]", "Quit TUI (in NORMAL mode)"),
|
("[q]", "Quit TUI (in NORMAL mode)"),
|
||||||
]
|
]
|
||||||
@@ -4247,6 +4532,30 @@ class MuseTUI:
|
|||||||
self.toggle_transcript_style()
|
self.toggle_transcript_style()
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
# Box mode Fast Actions: Tab 6 (SSH) — placed before the 's'/'c'/'y'
|
||||||
|
# globals below so SSH keys win on this tab.
|
||||||
|
if self.mode == "box" and self.box_tab == 6:
|
||||||
|
if ch in (curses.KEY_ENTER, 10, 13):
|
||||||
|
self._ssh_popout_selected()
|
||||||
|
return True
|
||||||
|
elif ch in (ord('s'), ord('S')):
|
||||||
|
self._ssh_popout_selected()
|
||||||
|
return True
|
||||||
|
elif ch in (ord('c'), ord('C')):
|
||||||
|
self._ssh_copy_dial_selected()
|
||||||
|
return True
|
||||||
|
elif ch in (ord('u'), ord('U')):
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
if rows and 0 <= self.ssh_sel_idx < len(rows):
|
||||||
|
acct = rows[self.ssh_sel_idx]
|
||||||
|
threading.Thread(target=self._async_ssh_uptime, args=(acct,), daemon=True).start()
|
||||||
|
self.set_toast(f"Probing container uptime on {acct}...", "info")
|
||||||
|
return True
|
||||||
|
elif ch in (ord('r'), ord('R')):
|
||||||
|
threading.Thread(target=self.data._fetch_ssh_health, daemon=True).start()
|
||||||
|
self.set_toast("Probing SSH tunnels via jump host...", "info")
|
||||||
|
return True
|
||||||
|
|
||||||
# Open Context Menu for active sidebar thread, fleet agent, or message: 'x', 'c', or Space
|
# Open Context Menu for active sidebar thread, fleet agent, or message: 'x', 'c', or Space
|
||||||
if ch in (ord('x'), ord('X'), ord('c'), ord('C'), ord(' ')) and not (self.mode == "box" and self.box_tab == 2):
|
if ch in (ord('x'), ord('X'), ord('c'), ord('C'), ord(' ')) and not (self.mode == "box" and self.box_tab == 2):
|
||||||
if self.focus_pane == "fleet":
|
if self.focus_pane == "fleet":
|
||||||
@@ -4564,7 +4873,7 @@ class MuseTUI:
|
|||||||
threading.Thread(target=self._async_kill_tmux, args=(sess_name,), daemon=True).start()
|
threading.Thread(target=self._async_kill_tmux, args=(sess_name,), daemon=True).start()
|
||||||
return True
|
return True
|
||||||
elif ch in (ord('r'), ord('R')):
|
elif ch in (ord('r'), ord('R')):
|
||||||
threading.Thread(target=self.data._fetch_tmux, daemon=True).start()
|
threading.Thread(target=self.data._fetch_tmux_sessions, daemon=True).start()
|
||||||
self.set_toast("Refreshed tmux background sessions.", "info")
|
self.set_toast("Refreshed tmux background sessions.", "info")
|
||||||
return True
|
return True
|
||||||
|
|
||||||
@@ -4613,29 +4922,29 @@ class MuseTUI:
|
|||||||
self.set_toast(f"Switched to agent: {node.upper()}", "success")
|
self.set_toast(f"Switched to agent: {node.upper()}", "success")
|
||||||
return True
|
return True
|
||||||
|
|
||||||
# Box mode tab selection: '1' - '6' (when in box mode, except 1-3 on Tab 2)
|
# Box mode tab selection: '1' - '7' (when in box mode, except 1-3 on Tab 2)
|
||||||
if self.mode == "box":
|
if self.mode == "box":
|
||||||
if self.box_tab == 2:
|
if self.box_tab == 2:
|
||||||
if ord('4') <= ch <= ord('6'):
|
if ord('4') <= ch <= ord('7'):
|
||||||
self.box_tab = ch - ord('1')
|
self.box_tab = ch - ord('1')
|
||||||
return True
|
return True
|
||||||
elif ord('1') <= ch <= ord('6'):
|
elif ord('1') <= ch <= ord('7'):
|
||||||
self.box_tab = ch - ord('1')
|
self.box_tab = ch - ord('1')
|
||||||
return True
|
return True
|
||||||
|
|
||||||
# Box mode tab navigation (when not in Chat tab 0): '[' / ']' / Tab / Shift-Tab
|
# Box mode tab navigation (when not in Chat tab 0): '[' / ']' / Tab / Shift-Tab
|
||||||
if self.mode == "box" and self.box_tab != 0:
|
if self.mode == "box" and self.box_tab != 0:
|
||||||
if ch in (ord('['), curses.KEY_LEFT):
|
if ch in (ord('['), curses.KEY_LEFT):
|
||||||
self.box_tab = (self.box_tab - 1) % 6
|
self.box_tab = (self.box_tab - 1) % 7
|
||||||
return True
|
return True
|
||||||
elif ch in (ord(']'), curses.KEY_RIGHT):
|
elif ch in (ord(']'), curses.KEY_RIGHT):
|
||||||
self.box_tab = (self.box_tab + 1) % 6
|
self.box_tab = (self.box_tab + 1) % 7
|
||||||
return True
|
return True
|
||||||
elif ch == ord('\t'):
|
elif ch == ord('\t'):
|
||||||
self.box_tab = (self.box_tab + 1) % 6
|
self.box_tab = (self.box_tab + 1) % 7
|
||||||
return True
|
return True
|
||||||
elif ch == curses.KEY_BTAB:
|
elif ch == curses.KEY_BTAB:
|
||||||
self.box_tab = (self.box_tab - 1) % 6
|
self.box_tab = (self.box_tab - 1) % 7
|
||||||
return True
|
return True
|
||||||
|
|
||||||
# Muse View / Chat Tab: Direct Conversation Cycling & Pane Switching
|
# Muse View / Chat Tab: Direct Conversation Cycling & Pane Switching
|
||||||
@@ -4722,6 +5031,8 @@ class MuseTUI:
|
|||||||
self.jobs_sel_idx = max(0, self.jobs_sel_idx - 1)
|
self.jobs_sel_idx = max(0, self.jobs_sel_idx - 1)
|
||||||
elif self.box_tab == 4:
|
elif self.box_tab == 4:
|
||||||
self.tmux_sel_idx = max(0, self.tmux_sel_idx - 1)
|
self.tmux_sel_idx = max(0, self.tmux_sel_idx - 1)
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
self.ssh_sel_idx = max(0, self.ssh_sel_idx - 1)
|
||||||
else:
|
else:
|
||||||
self.table_scroll_idx = max(0, self.table_scroll_idx - 1)
|
self.table_scroll_idx = max(0, self.table_scroll_idx - 1)
|
||||||
return True
|
return True
|
||||||
@@ -4764,6 +5075,10 @@ class MuseTUI:
|
|||||||
sessions = list(self.data.tmux_cache)
|
sessions = list(self.data.tmux_cache)
|
||||||
if sessions:
|
if sessions:
|
||||||
self.tmux_sel_idx = min(len(sessions) - 1, self.tmux_sel_idx + 1)
|
self.tmux_sel_idx = min(len(sessions) - 1, self.tmux_sel_idx + 1)
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
if rows:
|
||||||
|
self.ssh_sel_idx = min(len(rows) - 1, self.ssh_sel_idx + 1)
|
||||||
else:
|
else:
|
||||||
self.table_scroll_idx += 1
|
self.table_scroll_idx += 1
|
||||||
return True
|
return True
|
||||||
@@ -4784,6 +5099,8 @@ class MuseTUI:
|
|||||||
self.jobs_sel_idx = max(0, self.jobs_sel_idx - 5)
|
self.jobs_sel_idx = max(0, self.jobs_sel_idx - 5)
|
||||||
elif self.box_tab == 4:
|
elif self.box_tab == 4:
|
||||||
self.tmux_sel_idx = max(0, self.tmux_sel_idx - 5)
|
self.tmux_sel_idx = max(0, self.tmux_sel_idx - 5)
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
self.ssh_sel_idx = max(0, self.ssh_sel_idx - 5)
|
||||||
else:
|
else:
|
||||||
self.table_scroll_idx = max(0, self.table_scroll_idx - 5)
|
self.table_scroll_idx = max(0, self.table_scroll_idx - 5)
|
||||||
return True
|
return True
|
||||||
@@ -4812,6 +5129,10 @@ class MuseTUI:
|
|||||||
sessions = list(self.data.tmux_cache)
|
sessions = list(self.data.tmux_cache)
|
||||||
if sessions:
|
if sessions:
|
||||||
self.tmux_sel_idx = min(len(sessions) - 1, self.tmux_sel_idx + 5)
|
self.tmux_sel_idx = min(len(sessions) - 1, self.tmux_sel_idx + 5)
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
if rows:
|
||||||
|
self.ssh_sel_idx = min(len(rows) - 1, self.ssh_sel_idx + 5)
|
||||||
else:
|
else:
|
||||||
self.table_scroll_idx += 5
|
self.table_scroll_idx += 5
|
||||||
return True
|
return True
|
||||||
@@ -4837,6 +5158,8 @@ class MuseTUI:
|
|||||||
self.jobs_sel_idx = 0
|
self.jobs_sel_idx = 0
|
||||||
elif self.box_tab == 4:
|
elif self.box_tab == 4:
|
||||||
self.tmux_sel_idx = 0
|
self.tmux_sel_idx = 0
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
self.ssh_sel_idx = 0
|
||||||
else:
|
else:
|
||||||
self.table_scroll_idx = 0
|
self.table_scroll_idx = 0
|
||||||
return True
|
return True
|
||||||
@@ -4876,6 +5199,10 @@ class MuseTUI:
|
|||||||
sessions = list(self.data.tmux_cache)
|
sessions = list(self.data.tmux_cache)
|
||||||
if sessions:
|
if sessions:
|
||||||
self.tmux_sel_idx = max(0, len(sessions) - 1)
|
self.tmux_sel_idx = max(0, len(sessions) - 1)
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
if rows:
|
||||||
|
self.ssh_sel_idx = max(0, len(rows) - 1)
|
||||||
else:
|
else:
|
||||||
self.table_scroll_idx = max(0, len(self.data.nodes) - 5)
|
self.table_scroll_idx = max(0, len(self.data.nodes) - 5)
|
||||||
return True
|
return True
|
||||||
@@ -4971,6 +5298,8 @@ class MuseTUI:
|
|||||||
self.jobs_sel_idx = max(0, self.jobs_sel_idx - 1)
|
self.jobs_sel_idx = max(0, self.jobs_sel_idx - 1)
|
||||||
elif self.box_tab == 4:
|
elif self.box_tab == 4:
|
||||||
self.tmux_sel_idx = max(0, self.tmux_sel_idx - 1)
|
self.tmux_sel_idx = max(0, self.tmux_sel_idx - 1)
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
self.ssh_sel_idx = max(0, self.ssh_sel_idx - 1)
|
||||||
else:
|
else:
|
||||||
self.table_scroll_idx = max(0, self.table_scroll_idx - 1)
|
self.table_scroll_idx = max(0, self.table_scroll_idx - 1)
|
||||||
elif mx >= sidebar_w:
|
elif mx >= sidebar_w:
|
||||||
@@ -5020,6 +5349,10 @@ class MuseTUI:
|
|||||||
sessions = list(self.data.tmux_cache)
|
sessions = list(self.data.tmux_cache)
|
||||||
if sessions:
|
if sessions:
|
||||||
self.tmux_sel_idx = min(len(sessions) - 1, self.tmux_sel_idx + 1)
|
self.tmux_sel_idx = min(len(sessions) - 1, self.tmux_sel_idx + 1)
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
if rows:
|
||||||
|
self.ssh_sel_idx = min(len(rows) - 1, self.ssh_sel_idx + 1)
|
||||||
else:
|
else:
|
||||||
self.table_scroll_idx += 1
|
self.table_scroll_idx += 1
|
||||||
elif mx >= sidebar_w:
|
elif mx >= sidebar_w:
|
||||||
@@ -5273,8 +5606,7 @@ class MuseTUI:
|
|||||||
|
|
||||||
# Box Tabs click (my == 1 and self.mode == "box")
|
# Box Tabs click (my == 1 and self.mode == "box")
|
||||||
if my == 1 and self.mode == "box":
|
if my == 1 and self.mode == "box":
|
||||||
tab_bounds = [(1, 18), (19, 37), (38, 53), (54, 74), (75, 94), (95, 108)]
|
for idx, (start, end) in enumerate(box_tab_bounds()):
|
||||||
for idx, (start, end) in enumerate(tab_bounds):
|
|
||||||
if start <= mx <= end:
|
if start <= mx <= end:
|
||||||
self.box_tab = idx
|
self.box_tab = idx
|
||||||
return True
|
return True
|
||||||
@@ -5669,7 +6001,7 @@ class MuseTUI:
|
|||||||
self.focus_pane = "transcript"
|
self.focus_pane = "transcript"
|
||||||
return True
|
return True
|
||||||
|
|
||||||
# Box View clicks (Tabs 1, 2, 3, 4)
|
# Box View clicks (Tabs 1, 2, 3, 4, 6)
|
||||||
elif self.mode == "box":
|
elif self.mode == "box":
|
||||||
self.editor_mode = "NORMAL"
|
self.editor_mode = "NORMAL"
|
||||||
row_idx = my - (content_y + 3)
|
row_idx = my - (content_y + 3)
|
||||||
@@ -5763,6 +6095,21 @@ class MuseTUI:
|
|||||||
self.set_toast(f"Selected session '{sess_name}'. Click [Kill] or [Attach].", "info")
|
self.set_toast(f"Selected session '{sess_name}'. Click [Kill] or [Attach].", "info")
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
elif self.box_tab == 6:
|
||||||
|
# Tab 6: SSH / Boxes (rows start one line lower: jump-status line)
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
scroll_start = getattr(self, "ssh_scroll_idx", 0)
|
||||||
|
clicked_idx = scroll_start + row_idx - 1
|
||||||
|
if 0 <= clicked_idx < len(rows):
|
||||||
|
self.ssh_sel_idx = clicked_idx
|
||||||
|
if mx >= w - 8:
|
||||||
|
# [SSH] pop-out
|
||||||
|
self._ssh_popout_selected()
|
||||||
|
else:
|
||||||
|
acct = rows[clicked_idx]
|
||||||
|
self.set_toast(f"Selected [{acct.upper()}]. Click [SSH] or press Enter to pop out.", "info")
|
||||||
|
return True
|
||||||
|
|
||||||
return True
|
return True
|
||||||
|
|
||||||
def _handle_insert_key(self, ch: int) -> bool:
|
def _handle_insert_key(self, ch: int) -> bool:
|
||||||
@@ -6348,6 +6695,77 @@ class MuseTUI:
|
|||||||
else:
|
else:
|
||||||
self.set_toast(f"Kill failed: {msg[:40]}", "error")
|
self.set_toast(f"Kill failed: {msg[:40]}", "error")
|
||||||
|
|
||||||
|
def _ssh_popout_selected(self):
|
||||||
|
"""Open the selected container SSH session in a new tmux window."""
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
if not rows or not (0 <= self.ssh_sel_idx < len(rows)):
|
||||||
|
self.set_toast("No SSH row selected.", "warn")
|
||||||
|
return
|
||||||
|
acct = rows[self.ssh_sel_idx]
|
||||||
|
if acct not in SSH_TUNNEL_PORTS:
|
||||||
|
self.set_toast(f"No tunnel registered for '{acct}'.", "warn")
|
||||||
|
return
|
||||||
|
if not os.environ.get("TMUX"):
|
||||||
|
# Refuse to spawn into an invisible server: new-window would
|
||||||
|
# create a detached server the operator cannot see.
|
||||||
|
try:
|
||||||
|
probe = subprocess.run(["tmux", "ls"], capture_output=True, text=True, timeout=3)
|
||||||
|
server_up = probe.returncode == 0
|
||||||
|
except Exception:
|
||||||
|
server_up = False
|
||||||
|
if not server_up:
|
||||||
|
copy_to_clipboard(build_ssh_dial_string(acct))
|
||||||
|
self.set_toast("Not inside tmux and no server running; dial copied to clipboard.", "warn")
|
||||||
|
return
|
||||||
|
shell_cmd = build_ssh_popout_shell(acct)
|
||||||
|
pop_cmd = build_tmux_popout_command(f"ssh-{acct}", shell_cmd)
|
||||||
|
try:
|
||||||
|
curses.def_prog_mode()
|
||||||
|
curses.endwin()
|
||||||
|
try:
|
||||||
|
res = subprocess.run(pop_cmd, capture_output=True, text=True, timeout=5)
|
||||||
|
finally:
|
||||||
|
try:
|
||||||
|
curses.reset_prog_mode()
|
||||||
|
self.stdscr.refresh()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
self.need_full_redraw = True
|
||||||
|
if res.returncode == 0:
|
||||||
|
self.set_toast(f"Opened SSH to {acct} in tmux window ssh-{acct}.", "success")
|
||||||
|
else:
|
||||||
|
copy_to_clipboard(build_ssh_dial_string(acct))
|
||||||
|
err = (res.stderr or "").strip().splitlines()
|
||||||
|
hint = err[-1][:60] if err else f"exit {res.returncode}"
|
||||||
|
self.set_toast(f"tmux pop-out failed ({hint}); dial copied.", "error")
|
||||||
|
except Exception as e:
|
||||||
|
copy_to_clipboard(build_ssh_dial_string(acct))
|
||||||
|
self.set_toast(f"Pop-out failed ({e}); dial copied to clipboard.", "error")
|
||||||
|
|
||||||
|
def _ssh_copy_dial_selected(self):
|
||||||
|
"""Copy the selected container's dial command to the clipboard."""
|
||||||
|
rows = self.get_ssh_rows()
|
||||||
|
if not rows or not (0 <= self.ssh_sel_idx < len(rows)):
|
||||||
|
self.set_toast("No SSH row selected.", "warn")
|
||||||
|
return
|
||||||
|
acct = rows[self.ssh_sel_idx]
|
||||||
|
if acct not in SSH_TUNNEL_PORTS:
|
||||||
|
self.set_toast(f"No tunnel registered for '{acct}'.", "warn")
|
||||||
|
return
|
||||||
|
dial = build_ssh_dial_string(acct)
|
||||||
|
if copy_to_clipboard(dial):
|
||||||
|
self.set_toast(f"Copied dial for {acct}: {dial[:80]}", "success")
|
||||||
|
else:
|
||||||
|
self.set_toast(f"Dial for {acct}: {dial}", "info")
|
||||||
|
|
||||||
|
def _async_ssh_uptime(self, account: str):
|
||||||
|
ok, out_text = self.data.probe_container_uptime(account)
|
||||||
|
if ok:
|
||||||
|
first = (out_text or "").strip().splitlines()
|
||||||
|
self.set_toast(f"{account} uptime: {first[0][:90]}" if first else f"{account}: uptime probe empty.", "success" if first else "warn")
|
||||||
|
else:
|
||||||
|
self.set_toast(f"{account} uptime failed: {out_text[:90]}", "error")
|
||||||
|
|
||||||
def _execute_chat_action(self, action_id: str):
|
def _execute_chat_action(self, action_id: str):
|
||||||
"""Execute selected contextual action on the targeted chat thread."""
|
"""Execute selected contextual action on the targeted chat thread."""
|
||||||
chat_info = getattr(self, "context_chat", None) or {}
|
chat_info = getattr(self, "context_chat", None) or {}
|
||||||
@@ -7039,6 +7457,7 @@ def main():
|
|||||||
parser.add_argument("--account", "-a", choices=VALID_NODES, default=None, help="Initial agent account (default: top of list)")
|
parser.add_argument("--account", "-a", choices=VALID_NODES, default=None, help="Initial agent account (default: top of list)")
|
||||||
parser.add_argument("--thread", "-t", help="Initial thread UUID to open")
|
parser.add_argument("--thread", "-t", help="Initial thread UUID to open")
|
||||||
parser.add_argument("--mode", "-m", choices=["muse", "box"], default="muse", help="TUI mode (default: muse)")
|
parser.add_argument("--mode", "-m", choices=["muse", "box"], default="muse", help="TUI mode (default: muse)")
|
||||||
|
parser.add_argument("--tab", choices=["chat", "fleet", "approvals", "jobs", "tmux", "logs", "ssh"], default=None, help="Initial Box tab (box mode only)")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -7046,7 +7465,8 @@ def main():
|
|||||||
stdscr,
|
stdscr,
|
||||||
initial_mode=args.mode,
|
initial_mode=args.mode,
|
||||||
initial_node=args.account,
|
initial_node=args.account,
|
||||||
initial_thread=args.thread
|
initial_thread=args.thread,
|
||||||
|
initial_tab=args.tab
|
||||||
).run())
|
).run())
|
||||||
except KeyboardInterrupt:
|
except KeyboardInterrupt:
|
||||||
pass
|
pass
|
||||||
|
|||||||
+792
-63
File diff suppressed because it is too large
Load Diff
+66
-3
@@ -252,6 +252,56 @@ def provision_node_infra(node: str) -> Dict[str, Any]:
|
|||||||
return {"ok": True, "node": node, "output": res.stdout.strip()}
|
return {"ok": True, "node": node, "output": res.stdout.strip()}
|
||||||
|
|
||||||
|
|
||||||
|
def _registry_port(node: str) -> Optional[int]:
|
||||||
|
"""CDP port for a node via netvm-registry.py, or None if unregistered."""
|
||||||
|
import importlib.util
|
||||||
|
spec = importlib.util.spec_from_file_location(
|
||||||
|
"netvm_registry", str(BIN_DIR / "netvm-registry.py"))
|
||||||
|
mod = importlib.util.module_from_spec(spec)
|
||||||
|
spec.loader.exec_module(mod)
|
||||||
|
return mod.port_for(node)
|
||||||
|
|
||||||
|
|
||||||
|
def _cdp_dry_run(node: str) -> bool:
|
||||||
|
"""True when the node netns + CDP + page chain is healthy."""
|
||||||
|
cmd = [str(BIN_DIR / "netvm-exec.sh"), node, "--", sys.executable,
|
||||||
|
str(BIN_DIR / "onboard-driver.py"),
|
||||||
|
"--node", node, "--service", "muse", "--id-type", "email",
|
||||||
|
"--step", "initiate", "--dry-run"]
|
||||||
|
res = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
|
||||||
|
return res.returncode == 0
|
||||||
|
|
||||||
|
|
||||||
|
def ensure_node_browser(node: str, timeout: float = 90.0, poll_interval: float = 5.0) -> Dict[str, Any]:
|
||||||
|
"""Launch the node headless browser inside its netns if CDP is down.
|
||||||
|
|
||||||
|
start_onboarding() must call this after infra provisioning: provision
|
||||||
|
never starts a browser, so without this step auth initiation always
|
||||||
|
dies with CDP connection refused on fresh nodes.
|
||||||
|
"""
|
||||||
|
port = _registry_port(node)
|
||||||
|
if not port:
|
||||||
|
return {"ok": False, "node": node,
|
||||||
|
"error": "unknown node %s (not in NODES.md registry)" % node}
|
||||||
|
if _cdp_dry_run(node):
|
||||||
|
return {"ok": True, "node": node, "cdp_port": port, "already": True}
|
||||||
|
STATE_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
log_path = STATE_DIR / ("%s-chrome.log" % node)
|
||||||
|
cmd = [str(BIN_DIR / "netvm-chrome.sh"), "--headless",
|
||||||
|
"--cdp-port", str(port), node, "https://muse.ai"]
|
||||||
|
with open(log_path, "ab") as log:
|
||||||
|
subprocess.Popen(cmd, start_new_session=True,
|
||||||
|
stdout=log, stderr=subprocess.STDOUT,
|
||||||
|
stdin=subprocess.DEVNULL)
|
||||||
|
deadline = time.time() + timeout
|
||||||
|
while time.time() < deadline:
|
||||||
|
time.sleep(poll_interval)
|
||||||
|
if _cdp_dry_run(node):
|
||||||
|
return {"ok": True, "node": node, "cdp_port": port, "already": False}
|
||||||
|
return {"ok": False, "node": node, "cdp_port": port,
|
||||||
|
"error": "browser launched but CDP stayed unreachable on port %s (log: %s)" % (port, log_path)}
|
||||||
|
|
||||||
|
|
||||||
def start_onboarding(node: str, email: str, beneficiary_node: Optional[str] = None, invite_code: Optional[str] = None, account_name: Optional[str] = None) -> Dict[str, Any]:
|
def start_onboarding(node: str, email: str, beneficiary_node: Optional[str] = None, invite_code: Optional[str] = None, account_name: Optional[str] = None) -> Dict[str, Any]:
|
||||||
"""Phase 1 & 2: Provision infra, choose beneficiary invite code, and initiate authentication."""
|
"""Phase 1 & 2: Provision infra, choose beneficiary invite code, and initiate authentication."""
|
||||||
b_node, code, reason = select_urgent_beneficiary(beneficiary_node, invite_code)
|
b_node, code, reason = select_urgent_beneficiary(beneficiary_node, invite_code)
|
||||||
@@ -279,6 +329,14 @@ def start_onboarding(node: str, email: str, beneficiary_node: Optional[str] = No
|
|||||||
state.stage = STAGE_INFRA
|
state.stage = STAGE_INFRA
|
||||||
state.save()
|
state.save()
|
||||||
|
|
||||||
|
# 1b. Ensure the headless browser is up (provision never starts one).
|
||||||
|
browser_res = ensure_node_browser(node)
|
||||||
|
if not browser_res.get("ok"):
|
||||||
|
state.stage = "browser_failed"
|
||||||
|
state.detail = browser_res.get("error")
|
||||||
|
state.save()
|
||||||
|
return {"ok": False, "state": asdict(state), "error": state.detail}
|
||||||
|
|
||||||
# 2. Initiate authentication
|
# 2. Initiate authentication
|
||||||
client = CredClient()
|
client = CredClient()
|
||||||
cred_res = client.initiate(node, email, service="muse", account_name=account_name)
|
cred_res = client.initiate(node, email, service="muse", account_name=account_name)
|
||||||
@@ -408,12 +466,17 @@ def issue_salvage_work_order(blocked_node: str = "646", to_sidechat: str = "646
|
|||||||
"--to", "opm",
|
"--to", "opm",
|
||||||
"--target", to_sidechat,
|
"--target", to_sidechat,
|
||||||
"--title", title,
|
"--title", title,
|
||||||
"--body", body,
|
|
||||||
"--priority", "urgent",
|
"--priority", "urgent",
|
||||||
"--allow-main-chat"
|
"--allow-main-chat",
|
||||||
|
body,
|
||||||
]
|
]
|
||||||
res = subprocess.run(cmd, capture_output=True, text=True)
|
res = subprocess.run(cmd, capture_output=True, text=True)
|
||||||
return {"ok": res.returncode == 0, "output": res.stdout.strip()}
|
out = res.stdout.strip()
|
||||||
|
result: Dict[str, Any] = {"ok": res.returncode == 0, "output": out}
|
||||||
|
if not result["ok"]:
|
||||||
|
err = res.stderr.strip()
|
||||||
|
result["error"] = err or out or "dm wo exited %d" % res.returncode
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
def get_all_connects(fast: bool = True) -> List[Dict[str, Any]]:
|
def get_all_connects(fast: bool = True) -> List[Dict[str, Any]]:
|
||||||
|
|||||||
@@ -118,7 +118,7 @@ def wrap(job_name, job_id, agent, target, rendered, include_kpi: bool = True):
|
|||||||
|
|
||||||
top = (
|
top = (
|
||||||
f"Operator Directive [ref:{wo_id}]:\n"
|
f"Operator Directive [ref:{wo_id}]:\n"
|
||||||
f"Host tmux worker session '{session_name}' is available on bl (/tmp/tmux-muse.sock).\n"
|
f"Persistent box runtime is on bl (/tmp/tmux-muse.sock). No worker session exists yet — create yours first: [TOOL tmux.new {{\"session\": \"{session_name}\", \"command\": \"bash\"}}].\n"
|
||||||
f" • Subagent assistance: {spawn}\n"
|
f" • Subagent assistance: {spawn}\n"
|
||||||
f" • Verification schedule: {follow}\n"
|
f" • Verification schedule: {follow}\n"
|
||||||
f"{advisory_section}\n"
|
f"{advisory_section}\n"
|
||||||
@@ -128,8 +128,9 @@ def wrap(job_name, job_id, agent, target, rendered, include_kpi: bool = True):
|
|||||||
has_result = "[RESULT" in rendered
|
has_result = "[RESULT" in rendered
|
||||||
bottom = (
|
bottom = (
|
||||||
"\n--- End Task ---\n\n"
|
"\n--- End Task ---\n\n"
|
||||||
f"Inspect tmux worker: box tmux capture {session_name} 30 (or attach via /tmp/tmux-muse.sock)\n"
|
f"Worker convention: name your tmux session {session_name} when you create it, then inspect via box tmux capture {session_name} 30.\n"
|
||||||
"Tools: cron.create, cron.runs, health.check, swarm.spawn, swarm.list, dm.send, box.exec, tools.list.\n"
|
"Flow in tmux: [TOOL flow.start {\"flow_id\": \"<id>\", \"command\": \"<cmd>\"}] | read delta: [TOOL flow.read {\"flow_id\": \"<id>\"}] | advance: [TOOL flow.send {\"flow_id\": \"<id>\", \"command\": \"...\"}].\n"
|
||||||
|
"Tools: flow.start, flow.read, flow.send, cron.create, health.check, swarm.spawn, dm.send, box.exec, tools.list.\n"
|
||||||
"Message a peer: [DM {\"to\": \"<agent>\", \"target\": \"<sidechat>\", \"message\": \"<text>\"}].\n"
|
"Message a peer: [DM {\"to\": \"<agent>\", \"target\": \"<sidechat>\", \"message\": \"<text>\"}].\n"
|
||||||
"Query box: [TOOL box.exec {\"action\": \"<fleet-status|dm-log|job-get|...>\"}] — [TOOL tools.list {}] lists every op.\n"
|
"Query box: [TOOL box.exec {\"action\": \"<fleet-status|dm-log|job-get|...>\"}] — [TOOL tools.list {}] lists every op.\n"
|
||||||
)
|
)
|
||||||
|
|||||||
+167
-22
@@ -109,7 +109,18 @@ def iter_result_markers(text):
|
|||||||
"""Yield (job_id, result_text) for every [RESULT <job_id>] marker in text."""
|
"""Yield (job_id, result_text) for every [RESULT <job_id>] marker in text."""
|
||||||
current_re = lookup_engine.get_result_regex() if HAS_LOOKUP_ENGINE else RESULT_RE
|
current_re = lookup_engine.get_result_regex() if HAS_LOOKUP_ENGINE else RESULT_RE
|
||||||
for m in current_re.finditer(text or ""):
|
for m in current_re.finditer(text or ""):
|
||||||
yield m.group(1).strip(), m.group(2).strip()
|
gd = m.groupdict()
|
||||||
|
if "summary" in gd:
|
||||||
|
# Engine shape: [RESULT <id>] [STATUS] <summary>. The
|
||||||
|
# status word is optional (None for bare markers); keep
|
||||||
|
# it when present so FAIL/ERROR still trips failure
|
||||||
|
# detection downstream.
|
||||||
|
status = (m.group("status") or "").strip()
|
||||||
|
summary = (m.group("summary") or "").strip()
|
||||||
|
result_text = f"{status} {summary}".strip() if status else summary
|
||||||
|
yield m.group("job_id").strip(), result_text
|
||||||
|
else:
|
||||||
|
yield m.group(1).strip(), m.group(2).strip()
|
||||||
|
|
||||||
|
|
||||||
# Verb markers for the digest response protocol:
|
# Verb markers for the digest response protocol:
|
||||||
@@ -316,9 +327,19 @@ def result_has_evidence(result_text):
|
|||||||
return bool(_PROOF_EVIDENCE_RE.search(result_text or ""))
|
return bool(_PROOF_EVIDENCE_RE.search(result_text or ""))
|
||||||
|
|
||||||
|
|
||||||
|
# Automated in-thread proof requests disabled per fleet governance decision (2026-10-09)
|
||||||
|
PROOF_REQUESTS_ENABLED = False
|
||||||
|
|
||||||
def maybe_request_proof(agent, thread_id, job_id, result_text, dry_run=False):
|
def maybe_request_proof(agent, thread_id, job_id, result_text, dry_run=False):
|
||||||
"""Ask for checkable evidence when a success RESULT has none.
|
"""Ask for checkable evidence when a success RESULT has none.
|
||||||
|
|
||||||
|
Disabled by default per fleet decision 2026-10-09: automated in-thread proof
|
||||||
|
challenges trigger adversarial rejection loops and waste agent quota.
|
||||||
|
"""
|
||||||
|
if not PROOF_REQUESTS_ENABLED:
|
||||||
|
return False
|
||||||
|
"""Ask for checkable evidence when a success RESULT has none.
|
||||||
|
|
||||||
One-shot per (thread, job) via the nudge tracker. Returns True when a
|
One-shot per (thread, job) via the nudge tracker. Returns True when a
|
||||||
proof followup was scheduled.
|
proof followup was scheduled.
|
||||||
"""
|
"""
|
||||||
@@ -559,6 +580,42 @@ def format_tool_result_for_chat(op, raw_output):
|
|||||||
out = out[:900] + "\n…(truncated, refine the call for detail)"
|
out = out[:900] + "\n…(truncated, refine the call for detail)"
|
||||||
return f"box result:\n```\n{out}\n```"
|
return f"box result:\n```\n{out}\n```"
|
||||||
|
|
||||||
|
if op == "flow.start" and isinstance(data, dict):
|
||||||
|
if not data.get("ok"):
|
||||||
|
return f"Flow start failed: {data.get('error')}"
|
||||||
|
return f"Flow `{data.get('flow_id')}` started in pane `{data.get('session')}` (status: {data.get('status')})."
|
||||||
|
|
||||||
|
if op == "flow.read" and isinstance(data, dict):
|
||||||
|
if not data.get("ok"):
|
||||||
|
return f"Flow read failed: {data.get('error')}"
|
||||||
|
st = data.get("status", "unknown")
|
||||||
|
ec = data.get("exit_code")
|
||||||
|
ec_str = f" (exit_code: {ec})" if ec is not None else ""
|
||||||
|
pm = data.get("prompt_match")
|
||||||
|
prompt_str = f"\nPrompt waiting: {pm.get('text', pm)}" if pm else ""
|
||||||
|
delta = data.get("delta", "").strip()
|
||||||
|
trunc = f" (last {data.get('lines_read')} lines)" if data.get("truncated") else ""
|
||||||
|
body = f"\n```\n{delta}\n```" if delta else " (no new output)"
|
||||||
|
return f"Flow `{data.get('flow_id')}` [{st}]{ec_str}{prompt_str}{trunc}:{body}"
|
||||||
|
|
||||||
|
if op == "flow.send" and isinstance(data, dict):
|
||||||
|
if not data.get("ok"):
|
||||||
|
return f"Flow send failed: {data.get('error')}"
|
||||||
|
kind = "command" if data.get("is_command") else "keys"
|
||||||
|
return f"Flow `{data.get('flow_id')}` sent {kind}: `{data.get('sent')}` (status: {data.get('status')})."
|
||||||
|
|
||||||
|
if op == "flow.list" and isinstance(data, dict):
|
||||||
|
flows = data.get("flows", [])
|
||||||
|
if not flows:
|
||||||
|
return "No active flows."
|
||||||
|
lines = [f"{len(flows)} flows:"]
|
||||||
|
for f in flows[:8]:
|
||||||
|
lines.append(f" • {f.get('flow_id')} [{f.get('status')}]: {f.get('session')} (cmd: {str(f.get('command', 'bash'))[:30]})")
|
||||||
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
if op == "flow.stop" and isinstance(data, dict):
|
||||||
|
return f"Flow `{data.get('flow_id')}` stopped."
|
||||||
|
|
||||||
# General fallback: compact JSON capped to 400 chars
|
# General fallback: compact JSON capped to 400 chars
|
||||||
s = json.dumps(data)
|
s = json.dumps(data)
|
||||||
return s[:400] + "..." if len(s) > 400 else s
|
return s[:400] + "..." if len(s) > 400 else s
|
||||||
@@ -623,11 +680,59 @@ def is_fail_result(result_text):
|
|||||||
return t.startswith(FAIL_PREFIXES)
|
return t.startswith(FAIL_PREFIXES)
|
||||||
|
|
||||||
|
|
||||||
|
RECENCY_WINDOW_SEC = 10800
|
||||||
|
|
||||||
|
_JOB_ID_RE = re.compile(r"^(.+)-(\d{8})-(\d{6})-([0-9a-f]{8})$")
|
||||||
|
|
||||||
|
|
||||||
|
def dispatched_families_since(job_log_path, window_sec=RECENCY_WINDOW_SEC,
|
||||||
|
now=None):
|
||||||
|
"""Job families dispatched inside the window.
|
||||||
|
|
||||||
|
Scans job-log.jsonl for job_sent/job_dispatched events newer than
|
||||||
|
``window_sec`` and returns their family names (the job id minus the
|
||||||
|
trailing -YYYYMMDD-HHMMSS-<hash> run suffix). Missing, unreadable,
|
||||||
|
or malformed input yields an empty set, never an exception.
|
||||||
|
"""
|
||||||
|
now = now or datetime.now(timezone.utc)
|
||||||
|
cutoff = now.timestamp() - window_sec
|
||||||
|
fams = set()
|
||||||
|
try:
|
||||||
|
handle = open(job_log_path, "r", encoding="utf-8")
|
||||||
|
except OSError:
|
||||||
|
return fams
|
||||||
|
with handle:
|
||||||
|
for line in handle:
|
||||||
|
line = line.strip()
|
||||||
|
if not line:
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
event = json.loads(line)
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
if event.get("type") not in ("job_sent", "job_dispatched"):
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
ts = datetime.fromisoformat(
|
||||||
|
str(event.get("ts")).replace("Z", "+00:00")).timestamp()
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
if ts < cutoff:
|
||||||
|
continue
|
||||||
|
match = _JOB_ID_RE.match(str(event.get("job_id") or ""))
|
||||||
|
if match:
|
||||||
|
fams.add(match.group(1))
|
||||||
|
return fams
|
||||||
|
|
||||||
|
|
||||||
def get_monitored_threads(target_agent=None):
|
def get_monitored_threads(target_agent=None):
|
||||||
"""
|
"""
|
||||||
Build dict of threads to monitor per agent:
|
Build dict of threads to monitor per agent:
|
||||||
{ agent: [ {"id": "<uuid>", "name": "<alias>"} ] }
|
{ agent: [ {"id": "<uuid>", "name": "<alias>"} ] }
|
||||||
Filters to permanent channels, threads with pending followups, or recent threads (< 3h).
|
Filters to permanent channels, threads with pending followups,
|
||||||
|
recently created threads (< 3h), or threads whose job family was
|
||||||
|
dispatched recently (< 3h) so old persistent sidechats that still
|
||||||
|
receive prompts stay monitored.
|
||||||
"""
|
"""
|
||||||
agents = [target_agent] if target_agent else VALID_AGENTS
|
agents = [target_agent] if target_agent else VALID_AGENTS
|
||||||
threads_by_agent = {a: [] for a in agents}
|
threads_by_agent = {a: [] for a in agents}
|
||||||
@@ -642,6 +747,7 @@ def get_monitored_threads(target_agent=None):
|
|||||||
|
|
||||||
PERM_KEYWORDS = ("coord", "tasks", "task", "brain", "heartbeat", "sync", "audit", "main-loop")
|
PERM_KEYWORDS = ("coord", "tasks", "task", "brain", "heartbeat", "sync", "audit", "main-loop")
|
||||||
now = datetime.now(timezone.utc)
|
now = datetime.now(timezone.utc)
|
||||||
|
recently_dispatched = dispatched_families_since(JOB_LOG, now=now)
|
||||||
|
|
||||||
state_files = [JOB_SIDECHATS_FILE, WAKE_SIDECHATS_FILE]
|
state_files = [JOB_SIDECHATS_FILE, WAKE_SIDECHATS_FILE]
|
||||||
for sf in state_files:
|
for sf in state_files:
|
||||||
@@ -684,7 +790,9 @@ def get_monitored_threads(target_agent=None):
|
|||||||
if isinstance(val, dict) and val.get("archived") and not is_pending:
|
if isinstance(val, dict) and val.get("archived") and not is_pending:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
if not (is_perm or is_pending or is_recent):
|
is_dispatched = key in recently_dispatched
|
||||||
|
|
||||||
|
if not (is_perm or is_pending or is_recent or is_dispatched):
|
||||||
continue
|
continue
|
||||||
|
|
||||||
existing = [t["id"] for t in threads_by_agent[agent]]
|
existing = [t["id"] for t in threads_by_agent[agent]]
|
||||||
@@ -896,8 +1004,18 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
|||||||
append_jsonl(CHAT_HISTORY_LOG, record)
|
append_jsonl(CHAT_HISTORY_LOG, record)
|
||||||
|
|
||||||
if author == "assistant":
|
if author == "assistant":
|
||||||
markers = list(iter_result_markers(text))
|
try:
|
||||||
verbs = list(iter_verb_markers(text))
|
markers = list(iter_result_markers(text))
|
||||||
|
verbs = list(iter_verb_markers(text))
|
||||||
|
except Exception as e:
|
||||||
|
# One poison message must not wedge the batch: without
|
||||||
|
# this, the same crash repeats every cycle, the
|
||||||
|
# watermark never advances past it, and the thread's
|
||||||
|
# followups nag to escalation despite answered work.
|
||||||
|
sys.stderr.write(
|
||||||
|
"warning: marker extraction failed, treating as "
|
||||||
|
f"plain reply: {e}\n")
|
||||||
|
markers, verbs = [], []
|
||||||
|
|
||||||
# Synthesize [RESULT <job-id>] DECLINE if assistant explicitly refuses the task in plain text
|
# Synthesize [RESULT <job-id>] DECLINE if assistant explicitly refuses the task in plain text
|
||||||
if not markers and not verbs and detect_explicit_refusal(text):
|
if not markers and not verbs and detect_explicit_refusal(text):
|
||||||
@@ -946,19 +1064,27 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
|||||||
try:
|
try:
|
||||||
import muse_hybrid
|
import muse_hybrid
|
||||||
thread_url = f"https://box.muse-dev.online/thread/{thread_id}"
|
thread_url = f"https://box.muse-dev.online/thread/{thread_id}"
|
||||||
tool_hint = (
|
if op.startswith("flow."):
|
||||||
f"[Runtime Context: {thread_url}]\n"
|
flow_id = t_args.get("flow_id", "<flow_id>") if isinstance(t_args, dict) else "<flow_id>"
|
||||||
f"Tools: EMIT one [TOOL <op> <args>] line per action (you do not run it;"
|
tool_hint = (
|
||||||
f" the runtime executes it and replies here). curl -sk -X POST"
|
f"[Flow Directive: advance with [TOOL flow.send {{\"flow_id\": \"{flow_id}\", \"command\": \"...\"}}]"
|
||||||
f" https://exec.muse-dev.online/exec works too.\n"
|
f" | read with [TOOL flow.read {{\"flow_id\": \"{flow_id}\"}}]"
|
||||||
f" • [TOOL tools.list {{}}] — discover every op dynamically\n"
|
f" | close with [RESULT <job_id>] OK]"
|
||||||
f" • [TOOL swarm.spawn {{\"count\": 1, \"task\": \"<task>\"}}] — spawn subagents\n"
|
)
|
||||||
f" • [DM {{\"to\": \"<agent>\", \"target\": \"<sidechat>\", \"message\": \"<text>\"}}] — send a DM\n"
|
else:
|
||||||
f" • [TOOL box.exec {{\"action\": \"fleet-status\"}}] — call box (read-only actions)\n"
|
tool_hint = (
|
||||||
f" • [TOOL followup.create {{\"in_m\": 5, \"prompt\": \"<reminder>\"}}]\n"
|
f"[Runtime Context: {thread_url}]\n"
|
||||||
f" • [TOOL health.check {{}}]\n\n"
|
f"Tools: EMIT one [TOOL <op> <args>] line per action (you do not run it;"
|
||||||
f"[Directive: Take next action or close with [RESULT <job_id>] <summary>]"
|
f" the runtime executes it and replies here). curl -sk -X POST"
|
||||||
)
|
f" https://exec.muse-dev.online/exec works too.\n"
|
||||||
|
f" • [TOOL tools.list {{}}] — discover every op dynamically\n"
|
||||||
|
f" • [TOOL swarm.spawn {{\"count\": 1, \"task\": \"<task>\"}}] — spawn subagents\n"
|
||||||
|
f" • [DM {{\"to\": \"<agent>\", \"target\": \"<sidechat>\", \"message\": \"<text>\"}}] — send a DM\n"
|
||||||
|
f" • [TOOL box.exec {{\"action\": \"fleet-status\"}}] — call box (read-only actions)\n"
|
||||||
|
f" • [TOOL followup.create {{\"in_m\": 5, \"prompt\": \"<reminder>\"}}]\n"
|
||||||
|
f" • [TOOL health.check {{}}]\n\n"
|
||||||
|
f"[Directive: Take next action or close with [RESULT <job_id>] <summary>]"
|
||||||
|
)
|
||||||
if t_ok:
|
if t_ok:
|
||||||
clean_msg = format_tool_result_for_chat(op, t_res)
|
clean_msg = format_tool_result_for_chat(op, t_res)
|
||||||
resp_text = f"Tool result (`{op}`):\n{clean_msg}\n\n{tool_hint}"
|
resp_text = f"Tool result (`{op}`):\n{clean_msg}\n\n{tool_hint}"
|
||||||
@@ -969,7 +1095,12 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
|||||||
sys.stderr.write(f"warning: failed to post tool response back to thread: {te}\n")
|
sys.stderr.write(f"warning: failed to post tool response back to thread: {te}\n")
|
||||||
|
|
||||||
if markers or verbs:
|
if markers or verbs:
|
||||||
|
seen_jobs = set()
|
||||||
for job_id, result_text in markers:
|
for job_id, result_text in markers:
|
||||||
|
if job_id in seen_jobs:
|
||||||
|
# Same verdict restated in one message: log once.
|
||||||
|
continue
|
||||||
|
seen_jobs.add(job_id)
|
||||||
is_fail = is_fail_result(result_text)
|
is_fail = is_fail_result(result_text)
|
||||||
job_results += 1
|
job_results += 1
|
||||||
|
|
||||||
@@ -983,6 +1114,11 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
|||||||
"thread_id": thread_id,
|
"thread_id": thread_id,
|
||||||
"msg_id": mid,
|
"msg_id": mid,
|
||||||
}
|
}
|
||||||
|
if result_text.startswith("DECLINE:"):
|
||||||
|
# Synthesized (or explicit) decline: still a
|
||||||
|
# non-success (no chaining), but the auditor
|
||||||
|
# buckets it as declined, not a failure.
|
||||||
|
job_record["outcome"] = "declined"
|
||||||
if not dry_run:
|
if not dry_run:
|
||||||
append_jsonl(JOB_LOG, job_record)
|
append_jsonl(JOB_LOG, job_record)
|
||||||
# Check if this is a swarm slot result: sw-YYYYMMDD-HHMMSS-xxxx/<slot>
|
# Check if this is a swarm slot result: sw-YYYYMMDD-HHMMSS-xxxx/<slot>
|
||||||
@@ -1000,10 +1136,12 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo
|
|||||||
trigger_chain_next(job_id, result_text, success=not is_fail)
|
trigger_chain_next(job_id, result_text, success=not is_fail)
|
||||||
if not is_fail:
|
if not is_fail:
|
||||||
try:
|
try:
|
||||||
maybe_request_proof(agent, thread_id, job_id, result_text)
|
maybe_request_proof(agent, thread_id, job_id, result_text,
|
||||||
|
dry_run=dry_run)
|
||||||
except Exception as pe:
|
except Exception as pe:
|
||||||
sys.stderr.write(f"warning: proof check failed: {pe}\n")
|
sys.stderr.write(f"warning: proof check failed: {pe}\n")
|
||||||
archive_ephemeral_thread(agent, thread_id, job_id=job_id)
|
archive_ephemeral_thread(agent, thread_id, job_id=job_id,
|
||||||
|
dry_run=dry_run)
|
||||||
clear_matching_followups(followups, agent, thread_id, mid, text,
|
clear_matching_followups(followups, agent, thread_id, mid, text,
|
||||||
dry_run, job_id=job_id, verb="RESULT")
|
dry_run, job_id=job_id, verb="RESULT")
|
||||||
for verb, job_id in verbs:
|
for verb, job_id in verbs:
|
||||||
@@ -1122,11 +1260,13 @@ def harvest_agent_thread(cdp, agent, thread_info, watermarks, followups, dry_run
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def archive_ephemeral_thread(agent, thread_id, job_id=None):
|
def archive_ephemeral_thread(agent, thread_id, job_id=None, dry_run=False):
|
||||||
"""
|
"""
|
||||||
If thread_id belongs to an ephemeral job or one-off check,
|
If thread_id belongs to an ephemeral job or one-off check,
|
||||||
archive it via hybrid gateway and tag it as archived in job-sidechats.json.
|
archive it via hybrid gateway and tag it as archived in job-sidechats.json.
|
||||||
"""
|
"""
|
||||||
|
if dry_run:
|
||||||
|
return
|
||||||
if not thread_id or not re.fullmatch(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", thread_id.lower()):
|
if not thread_id or not re.fullmatch(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", thread_id.lower()):
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -1699,10 +1839,15 @@ def clear_matching_followups(followups, agent, thread_id, mid, text, dry_run=Fal
|
|||||||
|
|
||||||
# Match by job_id (from [RESULT <job_id>] or [VERB <job_id>]) --
|
# Match by job_id (from [RESULT <job_id>] or [VERB <job_id>]) --
|
||||||
# works regardless of thread_uuid or target. This is an ADDITIONAL
|
# works regardless of thread_uuid or target. This is an ADDITIONAL
|
||||||
# path, not a replacement.
|
# path, not a replacement. Also matches the followup's own key
|
||||||
|
# (dm_id): agents quote the DM id from nudge text ([RESULT
|
||||||
|
# <dm_id>]), which differs from job_id on DM-ordered followups
|
||||||
|
# (observed live: [RESULT f4293153] vs job ml-muse-*).
|
||||||
match_job = False
|
match_job = False
|
||||||
if job_id and f_rec.get("job_id") and f_rec.get("job_id") == job_id:
|
if job_id and f_rec.get("job_id") and f_rec.get("job_id") == job_id:
|
||||||
match_job = True
|
match_job = True
|
||||||
|
elif job_id and job_id == f_id:
|
||||||
|
match_job = True
|
||||||
|
|
||||||
# Non-RESULT verbs are job-scoped: they must not acknowledge/resolve
|
# Non-RESULT verbs are job-scoped: they must not acknowledge/resolve
|
||||||
# unrelated pending followups that merely share the thread. RESULT
|
# unrelated pending followups that merely share the thread. RESULT
|
||||||
|
|||||||
@@ -609,6 +609,7 @@ def test_wired_dm_send_asserts_post_nav_url_before_send():
|
|||||||
assert "assert_pre_send_placement" in src, \
|
assert "assert_pre_send_placement" in src, \
|
||||||
"gate exists but dm_send never calls it"
|
"gate exists but dm_send never calls it"
|
||||||
_orig_run_full = dm.run_full
|
_orig_run_full = dm.run_full
|
||||||
|
_orig_sleep = dm.time.sleep
|
||||||
_calls = []
|
_calls = []
|
||||||
|
|
||||||
def _stub(cmd, timeout=60):
|
def _stub(cmd, timeout=60):
|
||||||
@@ -619,6 +620,9 @@ def test_wired_dm_send_asserts_post_nav_url_before_send():
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
dm.run_full = _stub
|
dm.run_full = _stub
|
||||||
|
# Settle sleeps (1s/2s per gate call) are production pacing, not
|
||||||
|
# asserted behavior: skip them like the browser subprocess above.
|
||||||
|
dm.time.sleep = lambda s: None
|
||||||
# 1. UUID-known thread, correct placement -> pass
|
# 1. UUID-known thread, correct placement -> pass
|
||||||
_stub.url = "https://muse.ai/thread/" + UUID_A
|
_stub.url = "https://muse.ai/thread/" + UUID_A
|
||||||
ok, detail = gate("opm", "pipe-x", UUID_A, direct_nav_done=True)
|
ok, detail = gate("opm", "pipe-x", UUID_A, direct_nav_done=True)
|
||||||
@@ -641,6 +645,7 @@ def test_wired_dm_send_asserts_post_nav_url_before_send():
|
|||||||
assert ok is True, f"re-nav path should pass: {detail}"
|
assert ok is True, f"re-nav path should pass: {detail}"
|
||||||
finally:
|
finally:
|
||||||
dm.run_full = _orig_run_full
|
dm.run_full = _orig_run_full
|
||||||
|
dm.time.sleep = _orig_sleep
|
||||||
return
|
return
|
||||||
nav_i = src.find("sidechat use")
|
nav_i = src.find("sidechat use")
|
||||||
send_i = src.find("Send with verification retries")
|
send_i = src.find("Send with verification retries")
|
||||||
@@ -754,9 +759,141 @@ def main():
|
|||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
|
|
||||||
|
# --------------------------------------------------------------------------
|
||||||
|
# harvester resurrection -- dm_id markers, dry-run purity, scheduling
|
||||||
|
# --------------------------------------------------------------------------
|
||||||
|
|
||||||
|
def test_wired_harvester_dmid_marker_resolves():
|
||||||
|
"""Real clear_matching_followups: [RESULT <dm_id>] resolves a
|
||||||
|
DM-ordered followup whose job_id differs (live f4293153 pattern:
|
||||||
|
marker quoted the nudge's DM id, record job was ml-muse-*).
|
||||||
|
Unrelated thread isolates the dm_id path from thread matching."""
|
||||||
|
harv = _load("harvester_under_test", "response-harvester.py")
|
||||||
|
rec = _mk_rec(thread_uuid=UUID_B, job_id="ml-muse-20261007-013210")
|
||||||
|
fups = {"f4293153": rec}
|
||||||
|
harv.clear_matching_followups(fups, "646", "unrelated-thread", "mid-9",
|
||||||
|
"[RESULT f4293153] done", dry_run=True,
|
||||||
|
job_id="f4293153", verb="RESULT")
|
||||||
|
assert rec.get("status") == "resolved", (
|
||||||
|
"DEVIATION: [RESULT <dm_id>] does not resolve its followup -- "
|
||||||
|
"clear_matching_followups() matches marker ids against job_id "
|
||||||
|
f"only, never the followup key (status={rec.get('status')!r})")
|
||||||
|
# ... and a wrong id must not resolve.
|
||||||
|
rec2 = _mk_rec(thread_uuid=UUID_B, job_id="ml-muse-20261007-013210")
|
||||||
|
fups2 = {"f4293153": rec2}
|
||||||
|
harv.clear_matching_followups(fups2, "646", "unrelated-thread", "mid-9",
|
||||||
|
"[RESULT deadbeef] done", dry_run=True,
|
||||||
|
job_id="deadbeef", verb="RESULT")
|
||||||
|
assert rec2.get("status") == "pending", (
|
||||||
|
f"wrong marker id wrongly resolved (status={rec2.get('status')!r})")
|
||||||
|
|
||||||
|
|
||||||
|
def test_wired_harvester_dry_run_has_no_side_effects():
|
||||||
|
"""process_messages(dry_run=True) with an evidence-less RESULT must
|
||||||
|
still extract the marker but must not fire proof followups,
|
||||||
|
archive threads, or persist anything."""
|
||||||
|
harv = _load("harvester_under_test", "response-harvester.py")
|
||||||
|
calls = []
|
||||||
|
saved = {n: getattr(harv, n) for n in
|
||||||
|
("execute_agent_tool", "archive_ephemeral_thread",
|
||||||
|
"append_jsonl", "save_json_file")}
|
||||||
|
harv.execute_agent_tool = lambda *a, **k: calls.append("exec") or (True, {})
|
||||||
|
harv.archive_ephemeral_thread = (
|
||||||
|
lambda *a, **k: calls.append("archive"))
|
||||||
|
harv.append_jsonl = lambda *a, **k: calls.append("append")
|
||||||
|
harv.save_json_file = lambda *a, **k: calls.append("save")
|
||||||
|
try:
|
||||||
|
msgs = [{"id": "m1", "author": "assistant",
|
||||||
|
"text": "[RESULT j1] done",
|
||||||
|
"ts": "2026-10-07T00:00:00+00:00"}]
|
||||||
|
new, wm, nres = harv.process_messages(
|
||||||
|
msgs, "646", UUID_A, "t", None, {}, dry_run=True)
|
||||||
|
finally:
|
||||||
|
for n, fn in saved.items():
|
||||||
|
setattr(harv, n, fn)
|
||||||
|
assert nres == 1, "dry-run must still extract markers"
|
||||||
|
assert calls == [], f"dry-run leaked side effects: {calls}"
|
||||||
|
|
||||||
|
|
||||||
|
def test_wired_result_markers_bare_and_status_forms():
|
||||||
|
"""iter_result_markers handles the engine's 3-group shape: a bare
|
||||||
|
[RESULT <id>] <text> (status None) must not crash, and a status
|
||||||
|
token must survive into the result text for fail detection."""
|
||||||
|
harv = _load("harvester_under_test", "response-harvester.py")
|
||||||
|
assert list(harv.iter_result_markers("[RESULT f4293153] done")) == [
|
||||||
|
("f4293153", "done")], "bare RESULT marker must extract cleanly"
|
||||||
|
jid, text = list(harv.iter_result_markers("[RESULT j9] FAIL blew up"))[0]
|
||||||
|
assert jid == "j9" and "FAIL" in text and "blew up" in text, (
|
||||||
|
f"status token must survive into result text (got {jid!r} {text!r})")
|
||||||
|
|
||||||
|
|
||||||
|
def test_wired_poison_message_does_not_wedge_batch():
|
||||||
|
"""A marker-extraction crash degrades to plain-reply handling so
|
||||||
|
sibling messages still process and the watermark keeps advancing."""
|
||||||
|
harv = _load("harvester_under_test", "response-harvester.py")
|
||||||
|
real_iter = harv.iter_result_markers
|
||||||
|
real_nudge = harv.maybe_nudge_untagged_sidechat
|
||||||
|
def boom(text):
|
||||||
|
if "POISON" in (text or ""):
|
||||||
|
raise RuntimeError("boom")
|
||||||
|
return real_iter(text)
|
||||||
|
harv.iter_result_markers = boom
|
||||||
|
harv.maybe_nudge_untagged_sidechat = lambda *a, **k: None
|
||||||
|
try:
|
||||||
|
msgs = [
|
||||||
|
{"id": "m1", "author": "assistant",
|
||||||
|
"text": "POISON [RESULT x] y",
|
||||||
|
"ts": "2026-10-07T00:00:00+00:00"},
|
||||||
|
{"id": "m2", "author": "assistant",
|
||||||
|
"text": "[RESULT j2] ok",
|
||||||
|
"ts": "2026-10-07T00:01:00+00:00"},
|
||||||
|
]
|
||||||
|
new, wm, nres = harv.process_messages(
|
||||||
|
msgs, "646", UUID_A, "t", None, {}, dry_run=True)
|
||||||
|
finally:
|
||||||
|
harv.iter_result_markers = real_iter
|
||||||
|
harv.maybe_nudge_untagged_sidechat = real_nudge
|
||||||
|
assert nres == 1, "sibling marker must still extract"
|
||||||
|
assert [m["id"] for m in new] == ["m1", "m2"], \
|
||||||
|
"both messages must process past the poison one"
|
||||||
|
|
||||||
|
|
||||||
|
def test_harvester_timer_unit_wired():
|
||||||
|
"""The harvester must be scheduler-owned: unit files exist, the
|
||||||
|
service runs --once, and the timer fires on a short cadence.
|
||||||
|
Ingestion died silently for ~22h with no unit at all."""
|
||||||
|
root = BIN_DIR.parent
|
||||||
|
svc = (root / "systemd" / "response-harvester.service").read_text()
|
||||||
|
tmr = (root / "systemd" / "response-harvester.timer").read_text()
|
||||||
|
assert "response-harvester.py" in svc and "--once" in svc, \
|
||||||
|
"service must run the harvester --once"
|
||||||
|
assert "OnUnitActiveSec=" in tmr, "timer needs a repeat cadence"
|
||||||
|
assert "WantedBy=timers.target" in tmr, "timer must target timers.target"
|
||||||
|
|
||||||
|
|
||||||
|
def test_collection_adapter_is_single_and_pytest_opted_out():
|
||||||
|
"""Collection-shape guard (no 3x duplicates): exactly one TestCase
|
||||||
|
adapter is reachable from module globals (the adapter loop must not
|
||||||
|
leak a `_fn` alias that pytest collects as a second class), and the
|
||||||
|
adapter opts out of pytest (`__test__ = False`) so the module-level
|
||||||
|
functions are pytest's single source while unittest discovery still
|
||||||
|
runs the adapter."""
|
||||||
|
cases = [v for v in list(globals().values())
|
||||||
|
if inspect.isclass(v) and issubclass(v, unittest.TestCase)]
|
||||||
|
assert len(cases) == 1, (
|
||||||
|
f"expected exactly 1 TestCase adapter, found {len(cases)} "
|
||||||
|
f"(stray aliases reintroduce duplicate collection)")
|
||||||
|
assert TestFollowupFixes.__test__ is False, (
|
||||||
|
"TestFollowupFixes must set __test__ = False so pytest collects "
|
||||||
|
"each test once via the module-level functions")
|
||||||
|
|
||||||
|
|
||||||
class TestFollowupFixes(unittest.TestCase):
|
class TestFollowupFixes(unittest.TestCase):
|
||||||
"""unittest discovery adapter for contract and wired test functions."""
|
"""unittest discovery adapter for contract and wired test functions."""
|
||||||
pass
|
# pytest collects the module-level functions; skip the adapter so each
|
||||||
|
# test runs once. (unittest discovery ignores __test__ and still runs
|
||||||
|
# the adapter, which is its only view of this file's tests.)
|
||||||
|
__test__ = False
|
||||||
|
|
||||||
|
|
||||||
for _name, _fn in list(globals().items()):
|
for _name, _fn in list(globals().items()):
|
||||||
@@ -767,6 +904,11 @@ for _name, _fn in list(globals().items()):
|
|||||||
return _runner
|
return _runner
|
||||||
setattr(TestFollowupFixes, _name, _bind(_fn))
|
setattr(TestFollowupFixes, _name, _bind(_fn))
|
||||||
|
|
||||||
|
# Drop the loop temporaries: after the final iteration `_fn` aliases
|
||||||
|
# TestFollowupFixes, and pytest collects TestCase subclasses regardless of
|
||||||
|
# name -- that stray alias was the third copy (module fn + adapter + `_fn`).
|
||||||
|
del _name, _fn
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
sys.exit(main())
|
sys.exit(main())
|
||||||
|
|||||||
+82
-15
@@ -13,6 +13,7 @@ Supports:
|
|||||||
- A/B/C choice prompts -> "A"
|
- A/B/C choice prompts -> "A"
|
||||||
- Numbered menus -> "1"
|
- Numbered menus -> "1"
|
||||||
- y/n confirmation prompts -> "y"
|
- y/n confirmation prompts -> "y"
|
||||||
|
- Interview navigate+select menus (cursor on 1 -> Enter)
|
||||||
- Press Enter prompts -> "Enter"
|
- Press Enter prompts -> "Enter"
|
||||||
- Safety guardrails (passwords, passkeys, destructive commands are never auto-approved)
|
- Safety guardrails (passwords, passkeys, destructive commands are never auto-approved)
|
||||||
4. State persistence & audit logging:
|
4. State persistence & audit logging:
|
||||||
@@ -137,6 +138,21 @@ DEFAULT_RULES: List[MatchRule] = [
|
|||||||
description="Confirms y/n at end of terminal line",
|
description="Confirms y/n at end of terminal line",
|
||||||
press_enter=True,
|
press_enter=True,
|
||||||
),
|
),
|
||||||
|
MatchRule(
|
||||||
|
id="interview_select",
|
||||||
|
name="Interview Menu (cursor on 1)",
|
||||||
|
pattern=(r"\?\s*\n"
|
||||||
|
r"(?:[^\n]*\n){0,8}"
|
||||||
|
r"[ \t]*(?:›|>)[ \t]*1\.[ \t]+\S[^\n]*\n"
|
||||||
|
r"(?:[^\n]*\n){0,10}"
|
||||||
|
r"[ \t]*2\.[ \t]+\S"),
|
||||||
|
response_key="Enter",
|
||||||
|
category="enter",
|
||||||
|
enabled=True,
|
||||||
|
description=("Selects highlighted option 1 on navigate+select "
|
||||||
|
"menus (cursor on 1. + 2. + ?-question above)"),
|
||||||
|
press_enter=False,
|
||||||
|
),
|
||||||
MatchRule(
|
MatchRule(
|
||||||
id="enter_to_continue",
|
id="enter_to_continue",
|
||||||
name="Press Enter to Continue",
|
name="Press Enter to Continue",
|
||||||
@@ -188,9 +204,22 @@ class AutoApproverState:
|
|||||||
try:
|
try:
|
||||||
with open(STATE_FILE) as f:
|
with open(STATE_FILE) as f:
|
||||||
data = json.load(f)
|
data = json.load(f)
|
||||||
return cls(**data)
|
st = cls(**data)
|
||||||
except Exception:
|
except Exception:
|
||||||
return cls()
|
return cls()
|
||||||
|
# Migrate: append built-in rules missing from stored state (a new
|
||||||
|
# default must reach the daemon without wiping operator toggles).
|
||||||
|
try:
|
||||||
|
have = {r.get("id") for r in st.rules
|
||||||
|
if isinstance(r, dict)}
|
||||||
|
missing = [asdict(r) for r in DEFAULT_RULES
|
||||||
|
if r.id not in have]
|
||||||
|
if missing:
|
||||||
|
st.rules.extend(missing)
|
||||||
|
st.save()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return st
|
||||||
|
|
||||||
|
|
||||||
# =====================================================================
|
# =====================================================================
|
||||||
@@ -300,6 +329,23 @@ def capture_pane_text(socket_path: str, pane_id: str, lines: int = 30) -> str:
|
|||||||
|
|
||||||
MUSE_COMMAND_HINTS = ("muse-bin", "muse-code")
|
MUSE_COMMAND_HINTS = ("muse-bin", "muse-code")
|
||||||
|
|
||||||
|
# (socket, pane) ever observed running a muse runtime. pane_current_command
|
||||||
|
# flickers to the child tool while the agent works, so a muse pane stays
|
||||||
|
# muse-owned when its foreground reads "python3" (observed live: the hint
|
||||||
|
# gate missed tool-running panes and both daemons stacked 'y' answers).
|
||||||
|
_MUSE_PANES_SEEN = set()
|
||||||
|
|
||||||
|
# tmux rule category -> muse watcher kind for verified sends. Text-input
|
||||||
|
# categories verify render + submit with one retry; single-key widgets
|
||||||
|
# (and unknown categories) stay blind.
|
||||||
|
_CATEGORY_KIND_MAP = {
|
||||||
|
"choice": "letter",
|
||||||
|
"menu": "numbered",
|
||||||
|
"confirm": "yn",
|
||||||
|
"muse_code": "muse-approval",
|
||||||
|
"enter": None,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def should_defer_to_muse_watcher(socket_path: str, pane_id: str,
|
def should_defer_to_muse_watcher(socket_path: str, pane_id: str,
|
||||||
current_command: str) -> bool:
|
current_command: str) -> bool:
|
||||||
@@ -309,16 +355,25 @@ def should_defer_to_muse_watcher(socket_path: str, pane_id: str,
|
|||||||
panes (stability + re-verify + once-per-prompt + decided-block
|
panes (stability + re-verify + once-per-prompt + decided-block
|
||||||
guard). When its daemon is alive for this socket:pane, tmux must
|
guard). When its daemon is alive for this socket:pane, tmux must
|
||||||
skip the pane entirely, or both daemons answer the same prompt
|
skip the pane entirely, or both daemons answer the same prompt
|
||||||
within the same second ('11' + stray keys, observed live). Never
|
within the same second ('11' + stray keys, observed live; later the
|
||||||
raises: import or liveness failures mean no owner, handle here.
|
same hole stacked 'y' answers when the foreground flickered to a
|
||||||
|
child tool mid-poll). Never raises: import or liveness failures
|
||||||
|
mean no owner, handle here.
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
cmd = current_command or ""
|
|
||||||
if not any(h in cmd for h in MUSE_COMMAND_HINTS):
|
|
||||||
return False
|
|
||||||
import muse_choice_watcher as mcw
|
import muse_choice_watcher as mcw
|
||||||
alive = getattr(mcw, "watcher_alive", mcw.is_running)
|
cmd = current_command or ""
|
||||||
return alive(socket_path, pane_id) is not None
|
key = (socket_path, pane_id)
|
||||||
|
if any(h in cmd for h in MUSE_COMMAND_HINTS):
|
||||||
|
_MUSE_PANES_SEEN.add(key)
|
||||||
|
alive = getattr(mcw, "watcher_alive", mcw.is_running)
|
||||||
|
return alive(socket_path, pane_id) is not None
|
||||||
|
if key in _MUSE_PANES_SEEN:
|
||||||
|
alive = getattr(mcw, "watcher_alive", mcw.is_running)
|
||||||
|
return alive(socket_path, pane_id) is not None
|
||||||
|
# Never observed as muse: cheap pidfile check only (covers a
|
||||||
|
# watcher racing ahead of our first observation of the pane).
|
||||||
|
return mcw.is_running(socket_path, pane_id) is not None
|
||||||
except Exception:
|
except Exception:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
@@ -568,15 +623,25 @@ class AutoApproverRunner:
|
|||||||
})
|
})
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Execute key dispatch
|
# Execute key dispatch through the verified send path:
|
||||||
|
# literal text paced apart from Enter (a single-call
|
||||||
|
# burst arrives as paste and lands a newline in
|
||||||
|
# composers instead of submitting, then re-fires past
|
||||||
|
# dedup and stacks). Text-input categories also verify
|
||||||
|
# render + submit with one retry; single-key widgets
|
||||||
|
# stay blind.
|
||||||
success = False
|
success = False
|
||||||
|
detail = {"verified": None, "retried": False}
|
||||||
if not self.dry_run:
|
if not self.dry_run:
|
||||||
args = ["send-keys", "-t", p.pane_id, verdict.key]
|
import muse_choice_watcher as mcw
|
||||||
if verdict.press_enter or verdict.key == "Enter":
|
want_enter = (verdict.key != "Enter"
|
||||||
if verdict.key != "Enter":
|
and bool(verdict.press_enter))
|
||||||
args.append("Enter")
|
ok, detail = mcw.send_answer(
|
||||||
rc, _, _ = run_tmux_cmd(p.socket, *args)
|
p.socket, p.pane_id, verdict.key,
|
||||||
success = (rc == 0)
|
enter=want_enter,
|
||||||
|
kind=_CATEGORY_KIND_MAP.get(verdict.category),
|
||||||
|
sig=sig)
|
||||||
|
success = bool(ok)
|
||||||
else:
|
else:
|
||||||
success = True # dry-run simulated
|
success = True # dry-run simulated
|
||||||
|
|
||||||
@@ -597,6 +662,8 @@ class AutoApproverRunner:
|
|||||||
"excerpt": verdict.excerpt,
|
"excerpt": verdict.excerpt,
|
||||||
"dry_run": self.dry_run,
|
"dry_run": self.dry_run,
|
||||||
"success": success,
|
"success": success,
|
||||||
|
"verified": detail["verified"],
|
||||||
|
"retried": detail["retried"],
|
||||||
}
|
}
|
||||||
self.record_audit(event)
|
self.record_audit(event)
|
||||||
actions_taken.append(event)
|
actions_taken.append(event)
|
||||||
|
|||||||
+90
-17
@@ -2,10 +2,11 @@
|
|||||||
"""tmux_server_watchdog.py — Death-capture for tmux servers.
|
"""tmux_server_watchdog.py — Death-capture for tmux servers.
|
||||||
|
|
||||||
Runs on a 1-minute systemd timer. Remembers each known socket's server
|
Runs on a 1-minute systemd timer. Remembers each known socket's server
|
||||||
pid; when a server dies or its pid changes without a witnessed death,
|
identity (pid + /proc starttime + ppid + cmdline); when a server dies,
|
||||||
appends a forensics bundle (dmesg OOM/kill lines, memory, uptime,
|
its pid changes, or its pid is recycled under us without a witnessed
|
||||||
journal tail) to logs/tmux-server-deaths.jsonl so the next "tmux
|
death, appends a forensics bundle (dmesg OOM/kill lines, memory,
|
||||||
crashed" leaves evidence instead of a mystery.
|
uptime, journal tail) to logs/tmux-server-deaths.jsonl so the next
|
||||||
|
"tmux crashed" leaves evidence instead of a mystery.
|
||||||
|
|
||||||
Read-only against tmux itself: one `display-message -p` probe per
|
Read-only against tmux itself: one `display-message -p` probe per
|
||||||
socket. Never raises; a watchdog must not need its own watchdog.
|
socket. Never raises; a watchdog must not need its own watchdog.
|
||||||
@@ -55,9 +56,46 @@ def probe(socket_path):
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def collect_forensics(socket_path, last_pid):
|
def proc_identity(pid):
|
||||||
|
"""Identity dict for a pid: starttime defeats PID-reuse confusion.
|
||||||
|
|
||||||
|
Never raises; on any failure returns {"pid": pid} so callers can
|
||||||
|
still snapshot. starttime is the raw /proc starttime tick (field
|
||||||
|
22), stable for the life of the process."""
|
||||||
|
ident = {"pid": pid}
|
||||||
|
try:
|
||||||
|
with open("/proc/%d/stat" % pid) as f:
|
||||||
|
parts = f.read().rsplit(")", 1)[1].split()
|
||||||
|
# After "(comm)": state ppid pgrp session tty_nr ... starttime
|
||||||
|
# is field 22 overall, i.e. parts[19] after the split above.
|
||||||
|
ident["ppid"] = int(parts[1])
|
||||||
|
ident["starttime"] = int(parts[19])
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
with open("/proc/%d/cmdline" % pid, "rb") as f:
|
||||||
|
raw = f.read().replace(b"\0", b" ").decode(
|
||||||
|
"utf-8", "replace").strip()
|
||||||
|
if raw:
|
||||||
|
ident["cmd"] = raw[:200]
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return ident
|
||||||
|
|
||||||
|
|
||||||
|
def probe_identity(socket_path):
|
||||||
|
"""Enriched snapshot for a socket: identity dict or None."""
|
||||||
|
pid = probe(socket_path)
|
||||||
|
if pid is None:
|
||||||
|
return None
|
||||||
|
return proc_identity(pid)
|
||||||
|
|
||||||
|
|
||||||
|
def collect_forensics(socket_path, last_pid, last_identity=None):
|
||||||
"""Best-effort death evidence. Dict of strings, never raises."""
|
"""Best-effort death evidence. Dict of strings, never raises."""
|
||||||
ev = {"ts": _now(), "socket": socket_path, "last_pid": last_pid}
|
ev = {"ts": _now(), "socket": socket_path, "last_pid": last_pid}
|
||||||
|
if last_identity:
|
||||||
|
ev["last_identity"] = last_identity
|
||||||
rc, dmesg = _run(["dmesg"], timeout=10)
|
rc, dmesg = _run(["dmesg"], timeout=10)
|
||||||
if rc != 0:
|
if rc != 0:
|
||||||
ev["dmesg"] = "unavailable: %s" % dmesg[:200]
|
ev["dmesg"] = "unavailable: %s" % dmesg[:200]
|
||||||
@@ -112,27 +150,60 @@ def append_death(ev, path=None):
|
|||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
def _as_identity(value):
|
||||||
|
"""Normalize a probed value to an identity dict (legacy int ok)."""
|
||||||
|
if value is None:
|
||||||
|
return None
|
||||||
|
if isinstance(value, dict):
|
||||||
|
return value
|
||||||
|
return {"pid": value}
|
||||||
|
|
||||||
|
|
||||||
|
def _prev_identity(prev):
|
||||||
|
ident = {"pid": prev.get("pid")}
|
||||||
|
for key in ("starttime", "ppid", "cmd"):
|
||||||
|
if prev.get(key) is not None:
|
||||||
|
ident[key] = prev[key]
|
||||||
|
return ident
|
||||||
|
|
||||||
|
|
||||||
def evaluate(previous, probed):
|
def evaluate(previous, probed):
|
||||||
"""Pure transition logic: (prev_state, {sock: pid|None}) ->
|
"""Pure transition logic: (prev_state, {sock: pid|identity|None}) ->
|
||||||
(new_state, events). Events: death | restart | started."""
|
(new_state, events). Events: death | restart | started.
|
||||||
|
|
||||||
|
Probed values may be a bare pid (legacy) or an identity dict from
|
||||||
|
probe_identity(). Same pid with a different starttime is a restart
|
||||||
|
(pid recycled under us), not steady state."""
|
||||||
new_state, events = {}, []
|
new_state, events = {}, []
|
||||||
for sock, pid in sorted(probed.items()):
|
for sock, raw in sorted(probed.items()):
|
||||||
|
ident = _as_identity(raw)
|
||||||
prev = (previous.get(sock) or {})
|
prev = (previous.get(sock) or {})
|
||||||
prev_pid = prev.get("pid")
|
prev_pid = prev.get("pid")
|
||||||
if pid is None:
|
if ident is None:
|
||||||
new_state[sock] = {"pid": None, "died": _now(),
|
new_state[sock] = {"pid": None, "died": _now(),
|
||||||
"last_pid": prev_pid}
|
"last_pid": prev_pid}
|
||||||
if prev_pid:
|
if prev_pid:
|
||||||
events.append({"type": "death", "socket": sock,
|
events.append({"type": "death", "socket": sock,
|
||||||
"last_pid": prev_pid})
|
"last_pid": prev_pid,
|
||||||
|
"last_identity": _prev_identity(prev)})
|
||||||
else:
|
else:
|
||||||
new_state[sock] = {"pid": pid, "since": _now()}
|
pid = ident.get("pid")
|
||||||
|
new_state[sock] = dict(ident, since=_now())
|
||||||
if prev_pid and prev_pid != pid:
|
if prev_pid and prev_pid != pid:
|
||||||
# Changed with no witnessed death: restart inside one
|
# Changed with no witnessed death: restart inside one
|
||||||
# tick gap (or pid recycled under us). Treat as a
|
# tick gap. Worth a forensics note.
|
||||||
# restart, still worth a forensics note.
|
|
||||||
events.append({"type": "restart", "socket": sock,
|
events.append({"type": "restart", "socket": sock,
|
||||||
"old_pid": prev_pid, "pid": pid})
|
"old_pid": prev_pid, "pid": pid,
|
||||||
|
"last_identity": _prev_identity(prev)})
|
||||||
|
elif (prev_pid and prev_pid == pid
|
||||||
|
and prev.get("starttime") is not None
|
||||||
|
and ident.get("starttime") is not None
|
||||||
|
and prev["starttime"] != ident["starttime"]):
|
||||||
|
# Same pid, different process: pid recycled under us.
|
||||||
|
events.append({"type": "restart", "socket": sock,
|
||||||
|
"old_pid": prev_pid, "pid": pid,
|
||||||
|
"pid_reused": True,
|
||||||
|
"last_identity": _prev_identity(prev)})
|
||||||
elif not prev_pid and prev.get("died"):
|
elif not prev_pid and prev.get("died"):
|
||||||
events.append({"type": "started", "socket": sock,
|
events.append({"type": "started", "socket": sock,
|
||||||
"pid": pid})
|
"pid": pid})
|
||||||
@@ -144,18 +215,20 @@ def evaluate(previous, probed):
|
|||||||
|
|
||||||
def check(sockets=None, dry_run=False):
|
def check(sockets=None, dry_run=False):
|
||||||
"""Probe, transition state, log deaths. Returns summary dict."""
|
"""Probe, transition state, log deaths. Returns summary dict."""
|
||||||
probed = {s: probe(s) for s in (sockets or KNOWN_SOCKETS)}
|
probed = {s: probe_identity(s) for s in (sockets or KNOWN_SOCKETS)}
|
||||||
previous = read_state()
|
previous = read_state()
|
||||||
new_state, events = evaluate(previous, probed)
|
new_state, events = evaluate(previous, probed)
|
||||||
for ev in events:
|
for ev in events:
|
||||||
if ev["type"] == "death":
|
if ev["type"] == "death":
|
||||||
bundle = collect_forensics(ev["socket"], ev["last_pid"])
|
bundle = collect_forensics(ev["socket"], ev["last_pid"],
|
||||||
|
ev.get("last_identity"))
|
||||||
bundle["event"] = "death"
|
bundle["event"] = "death"
|
||||||
if not dry_run:
|
if not dry_run:
|
||||||
append_death(bundle)
|
append_death(bundle)
|
||||||
ev["forensics"] = bundle
|
ev["forensics"] = bundle
|
||||||
elif ev["type"] == "restart":
|
elif ev["type"] == "restart":
|
||||||
bundle = collect_forensics(ev["socket"], ev["old_pid"])
|
bundle = collect_forensics(ev["socket"], ev["old_pid"],
|
||||||
|
ev.get("last_identity"))
|
||||||
bundle["event"] = "restart-gap-missed"
|
bundle["event"] = "restart-gap-missed"
|
||||||
if not dry_run:
|
if not dry_run:
|
||||||
append_death(bundle)
|
append_death(bundle)
|
||||||
|
|||||||
@@ -7,3 +7,5 @@ operator-pip ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIMq02n0LpsksyQzWAWQ1mS8gKOonqFA
|
|||||||
pip ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIMq02n0LpsksyQzWAWQ1mS8gKOonqFALNDqbPGqXhq4T operator-pip
|
pip ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIMq02n0LpsksyQzWAWQ1mS8gKOonqFALNDqbPGqXhq4T operator-pip
|
||||||
operator-dev ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICwHn0kmRa6SFPbr2+z75s0gRlvBCGR633Ag7gTqiYPa dev@netvm
|
operator-dev ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICwHn0kmRa6SFPbr2+z75s0gRlvBCGR633Ag7gTqiYPa dev@netvm
|
||||||
dev ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICwHn0kmRa6SFPbr2+z75s0gRlvBCGR633Ag7gTqiYPa dev@netvm
|
dev ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAICwHn0kmRa6SFPbr2+z75s0gRlvBCGR633Ag7gTqiYPa dev@netvm
|
||||||
|
def ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEn6qqPrW7Vc77pUEBnLRDBF+yX11qyWzDTjZ2+FtL7b def@netvm
|
||||||
|
operator-def ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEn6qqPrW7Vc77pUEBnLRDBF+yX11qyWzDTjZ2+FtL7b def@netvm
|
||||||
|
|||||||
@@ -177,6 +177,44 @@ Agents can emit structured tool calls in sidechats:
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
## 4.1. Agentic Flows in Tmux Panes (`box flow` & `[TOOL flow.*]`)
|
||||||
|
|
||||||
|
Chromebox browser contexts prune and store chat history aggressively, making direct in-chat execution of long-running build, test, and shell tasks token-expensive and prone to context loss.
|
||||||
|
|
||||||
|
To overcome this, Chromebox agents offload multi-turn execution to persistent tmux panes on `/tmp/tmux-muse.sock` using the **Flow Engine** (`bin/flow_engine.py`). Raw stdout/stderr streams to disk (`logs/flows/<flow_id>.log`), and agents read back only concise status and incremental output deltas.
|
||||||
|
|
||||||
|
### Lifecycle & Primitives:
|
||||||
|
1. **Start Flow**:
|
||||||
|
Spawns pane `flow-<agent>-<id>` and launches command wrapped with an exit code sentinel.
|
||||||
|
```text
|
||||||
|
[TOOL flow.start {"flow_id": "audit-tests", "command": "python3 -m unittest discover -s tests"}]
|
||||||
|
```
|
||||||
|
*CLI:* `box flow start audit-tests -c "python3 -m unittest discover -s tests"`
|
||||||
|
|
||||||
|
2. **Read Incremental Delta & State**:
|
||||||
|
Inspects the pane for execution state (`working`, `idle`, `waiting_prompt`, `finished`, `failed`), exit code, and reads newly appended log output since the last read cursor.
|
||||||
|
```text
|
||||||
|
[TOOL flow.read {"flow_id": "audit-tests"}]
|
||||||
|
```
|
||||||
|
*CLI:* `box flow read audit-tests --lines 40`
|
||||||
|
|
||||||
|
3. **Advance or Respond to Prompts**:
|
||||||
|
Sends follow-up commands or keystrokes (such as interactive menu selections) without re-running the whole prompt.
|
||||||
|
```text
|
||||||
|
[TOOL flow.send {"flow_id": "audit-tests", "command": "git diff"}]
|
||||||
|
[TOOL flow.send {"flow_id": "audit-tests", "keys": "1"}]
|
||||||
|
```
|
||||||
|
*CLI:* `box flow send audit-tests "git status" --command`
|
||||||
|
|
||||||
|
4. **List & Stop**:
|
||||||
|
```text
|
||||||
|
[TOOL flow.list {}]
|
||||||
|
[TOOL flow.stop {"flow_id": "audit-tests"}]
|
||||||
|
```
|
||||||
|
*CLI:* `box flow list` / `box flow stop audit-tests`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
## 5. Direct Operator Directives & Prompt Envelope Specification
|
## 5. Direct Operator Directives & Prompt Envelope Specification
|
||||||
|
|
||||||
When jobs are dispatched to agents via `bin/job-dispatch.py`, they are wrapped in an actionable, authentic **Operator Directive** generated by `bin/prompt_envelope.py`.
|
When jobs are dispatched to agents via `bin/job-dispatch.py`, they are wrapped in an actionable, authentic **Operator Directive** generated by `bin/prompt_envelope.py`.
|
||||||
|
|||||||
+33
-9
@@ -1,13 +1,15 @@
|
|||||||
# MUSE-AUTH-CLI Decision Record
|
# MUSE-AUTH-CLI Decision Record
|
||||||
|
|
||||||
Status: **Draft** — taken over in this checkout 2026-10-07 per user choice.
|
Status: **Final** — accepted 2026-10-07 (user chose "accept Final with
|
||||||
Only explicit user acceptance moves this document (or any decision) to Final.
|
a recorded amendment," waiving done-means item 3; see Amendment A1).
|
||||||
|
|
||||||
Handoff note: a prior grill session settled D1–D11 and U1 and reportedly
|
Handoff note: a prior grill session settled D1–D11 and U1 and reportedly
|
||||||
marked its own record Final, but that file lives in another checkout (absent
|
marked its own record Final, but that file lives in another checkout (absent
|
||||||
here; this repo has no MUSE-AUTH-CLI.md, PI-AGENT-AUTH.md, OPERATORS.md, or
|
here). D1–D11 full text never arrived; per Amendment A1 the item is WAIVED,
|
||||||
agy-auth-switch). D1–D11 details below are CARRIED, not verified — their full
|
not verified — the decisions' substance stands proven by shipped, tested
|
||||||
text needs a paste or peer handoff before this record can go Final.
|
implementations (resume pool, session bind, P1/P2) plus live verification
|
||||||
|
(2026-10-07: 27/27 unique session IDs, workspace scoping exact, refs
|
||||||
|
resolve, profiles annotate).
|
||||||
|
|
||||||
## Goal
|
## Goal
|
||||||
|
|
||||||
@@ -34,11 +36,12 @@ standard billing cycles (not a rolling 30-day window).
|
|||||||
Decisions exist; codification as an OPERATORS.md amendment delta is the U2
|
Decisions exist; codification as an OPERATORS.md amendment delta is the U2
|
||||||
follow-on and is UNRESOLVED.
|
follow-on and is UNRESOLVED.
|
||||||
|
|
||||||
### D1–D11 (remaining detail) — CARRIED, text unavailable
|
### D1–D11 (remaining detail) — WAIVED per Amendment A1
|
||||||
|
|
||||||
Full decision text was settled in the prior session but is not present in
|
Full decision text was settled in the prior session but never arrived in
|
||||||
this checkout. CARRIED as-is; paste or peer handoff required to verify.
|
this checkout. Waived: re-verification by transcript would add words, not
|
||||||
This record cannot go Final until they are quoted or re-settled here.
|
evidence. If the original text surfaces and contradicts built behavior,
|
||||||
|
built behavior wins unless a new interview reopens the item.
|
||||||
|
|
||||||
## Scope contract (ACCEPTED 2026-10-07; user chose "accept the scope as written")
|
## Scope contract (ACCEPTED 2026-10-07; user chose "accept the scope as written")
|
||||||
|
|
||||||
@@ -51,6 +54,15 @@ This record cannot go Final until they are quoted or re-settled here.
|
|||||||
interview; accepting this record never approves them.
|
interview; accepting this record never approves them.
|
||||||
- "Go"/"do it all" authorize only the boundary above.
|
- "Go"/"do it all" authorize only the boundary above.
|
||||||
|
|
||||||
|
## Amendment A1 (ACCEPTED 2026-10-07 with Final)
|
||||||
|
|
||||||
|
Done-means item (3) ("D1–D11 text verified or re-settled") is WAIVED.
|
||||||
|
Rationale: the decisions' substance is verified by shipped, tested
|
||||||
|
implementations and live checks, not by recovering the lost transcript.
|
||||||
|
Recorded per the scope contract: this amendment is the explicit owner
|
||||||
|
approval for the narrowed completion boundary. U1, P1, P2, P3 stand as
|
||||||
|
settled; implementation and U2 remain separate stages.
|
||||||
|
|
||||||
## Settled Decisions (New)
|
## Settled Decisions (New)
|
||||||
|
|
||||||
### P1. Push/pull transfer file set — SETTLED (Credentials + Metadata)
|
### P1. Push/pull transfer file set — SETTLED (Credentials + Metadata)
|
||||||
@@ -86,3 +98,15 @@ muse-bin identity and watcher coverage is untouched. Tests:
|
|||||||
tests/test_muse_session_bind.py (13). Follow-ups for the owning lanes:
|
tests/test_muse_session_bind.py (13). Follow-ups for the owning lanes:
|
||||||
wire `box runtime launch` / resume-pool `resume` through the binder,
|
wire `box runtime launch` / resume-pool `resume` through the binder,
|
||||||
and arm a reap timer once the profile store (P1) exists.
|
and arm a reap timer once the profile store (P1) exists.
|
||||||
|
|
||||||
|
Cross-agent note (2026-10-07, factual, no decision change): the peer's
|
||||||
|
wrapper is DEPLOYED as `muse-code` (symlink to
|
||||||
|
`~/Account(s)/muse_wrapper.py`); 3 live sessions observed bound under
|
||||||
|
it (profile `def`), alongside unbound direct-`muse-bin` sessions.
|
||||||
|
Peer monitor daemons were absent on inspection; a fingerprint one-shot
|
||||||
|
showed all sessions in sync (no drift, nothing written). Monitor
|
||||||
|
reliability is the peer lane; reap-by-scan stays the immune
|
||||||
|
complement. The original D1–D11 text was recovered (peer's
|
||||||
|
MUSE-AUTH-CLI.md) and reviewed: no contradiction with built behavior;
|
||||||
|
the A1 waiver stands. Convergence proposal (open): peer adopts a bind
|
||||||
|
record, NetVM reap learns the peer dirname pattern.
|
||||||
|
|||||||
@@ -150,3 +150,41 @@ cat shared/operators/SOUL.md | ssh -o StrictHostKeyChecking=no -J super@34.139.3
|
|||||||
2. **Safety Gates on Amendments**: `box md amend` automatically validates that amendments do not remove checklists or revert `SOUL.md` to passive templates.
|
2. **Safety Gates on Amendments**: `box md amend` automatically validates that amendments do not remove checklists or revert `SOUL.md` to passive templates.
|
||||||
3. **Relative Paths in Hatch RPC**: Hatch WebSocket RPC rejects absolute paths (`/SOUL.md` fails; `SOUL.md` succeeds).
|
3. **Relative Paths in Hatch RPC**: Hatch WebSocket RPC rejects absolute paths (`/SOUL.md` fails; `SOUL.md` succeeds).
|
||||||
4. **Dual Access Redundancy**: If SSH reverse tunnels drop, Hatch WebSocket RPC is independent of SSH and can be used immediately to inspect logs, repair `authorized_keys`, or restart watchdog scripts.
|
4. **Dual Access Redundancy**: If SSH reverse tunnels drop, Hatch WebSocket RPC is independent of SSH and can be used immediately to inspect logs, repair `authorized_keys`, or restart watchdog scripts.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 5. SSH Access-Management Decisions (DRAFT — grill interview in progress)
|
||||||
|
|
||||||
|
> Status: DRAFT. Each decision below is written as the interview settles it.
|
||||||
|
> Nothing here is Final until the owner explicitly accepts the full text.
|
||||||
|
> Context: 2026-10-07 key-resolution run — all 5 agents refused dial-in key
|
||||||
|
> install via chat relay (impersonation-pattern defense); keys were placed
|
||||||
|
> via the operator Hatch channel instead; file modes remain the open gap.
|
||||||
|
|
||||||
|
### Scope contract (SETTLED — Draft)
|
||||||
|
- **Artifact boundary**: Section 5 of this file (the decision record) PLUS
|
||||||
|
approval of execution stages E1–E3 below. Out of scope: code changes,
|
||||||
|
other doc rewrites, and any new PR or task program beyond E1–E3.
|
||||||
|
- **Done means**: Scope + D1–D5 + E1–E3 all written as settled text; the
|
||||||
|
owner explicitly accepts the full section; then it flips to Final.
|
||||||
|
- **Stages**: E1–E3 are approved here as plans with named owners and
|
||||||
|
verification steps. Ending the interview never authorizes
|
||||||
|
implementation — execution needs a separate explicit request afterward.
|
||||||
|
- Set by owner choice ("1" = wider-boundary alternative) on 2026-10-07.
|
||||||
|
|
||||||
|
### D1. `.ssh/authorized_keys` validator allowlist (UNRESOLVED)
|
||||||
|
- Whether the exact-match allowlist in `agent_md.py` (`MD_ALLOWED_SUBPATHS`)
|
||||||
|
stays as the permanent operator key-install mechanism.
|
||||||
|
|
||||||
|
### D2. Authority boundary: platform writes vs relayed instructions (UNRESOLVED)
|
||||||
|
- Whether operator Hatch writes are a legitimate access-grant channel when
|
||||||
|
agents refuse the same grant via chat relay, and under what conditions.
|
||||||
|
|
||||||
|
### D3. bl→VM jump-key provisioning (UNRESOLVED)
|
||||||
|
- The sanctioned process for getting bl operator SSH access to the jump host.
|
||||||
|
|
||||||
|
### D4. def/dev tunnel restoration (UNRESOLVED)
|
||||||
|
- Who provisions tunnel identities and VM-side authorization once jump works.
|
||||||
|
|
||||||
|
### D5. File-mode gap on the Hatch write path (UNRESOLVED)
|
||||||
|
- How `authorized_keys` gets to 600 given the gateway cannot set modes.
|
||||||
|
|||||||
+996
-424
File diff suppressed because it is too large
Load Diff
@@ -1,12 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "646-exec-health",
|
|
||||||
"agent": "646",
|
|
||||||
"description": "Monitor exec server health every 10 minutes",
|
|
||||||
"prompt_template": "Exec server health check.\nJob ID: {job_id}\nTime: {datetime}\n\nCheck https://34-139-37-135.sslip.io/exec/health and report status. Reply with [RESULT {job_id}] OK/FAIL.",
|
|
||||||
"schedule": "*/10 * * * *",
|
|
||||||
"timeout": 120,
|
|
||||||
"on_failure": "alert",
|
|
||||||
"dm_target": "646 tasks",
|
|
||||||
"sidechat": {"create": false},
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "646-hourly-checkin",
|
|
||||||
"description": "Hourly operational check-in for 646 during active daytime hours",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "0 8-22 * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"on_failure": "alert",
|
|
||||||
"dm_target": "646 tasks",
|
|
||||||
"sidechat": {"create": false},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "1h",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm"
|
|
||||||
},
|
|
||||||
"prompt_template": "Hourly operational check-in for 646.\nJob ID: {job_id}\nTime: {datetime}\n\nPlease report briefly in this thread:\n(1) Current tasks in flight\n(2) VM / service health status\n(3) Any blockers or peer coordination items\n\nReply with [RESULT {job_id}] and your status."
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a01",
|
|
||||||
"description": "646 auto-work: scan box job-list for claimable manual/pending jobs",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "3,33 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 work scan: run box job-list and find manual or pending jobs assigned to you or unclaimed. Claim the oldest actionable one with box job-trigger and start it. If nothing actionable, report idle and take no further action.\n[TOOL swarm.list {}]\n[TOOL dm.read {\"agent\": \"646\", \"limit\": 5}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a01",
|
|
||||||
"name_template": "auto-work-646-a01-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a02",
|
|
||||||
"description": "646 auto-work: 646 dispatch review: failed/stale dispatch triage",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "7,37 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 dispatch review: check your recent dispatches for failed or stale ones (box job-status). Retry a failed dispatch once with box job-trigger; if it fails twice, escalate to opm via box notify. Nothing broken: report all-clear.\n[TOOL dm.read {\"agent\": \"646\", \"limit\": 5}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a02",
|
|
||||||
"name_template": "auto-work-646-a02-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a03",
|
|
||||||
"description": "646 auto-work: 646 chain check: advance chained jobs waiting on you",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "11,41 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 chain check: look for chained jobs where you are the next step (box job-next on your recent job ids). If a next step is waiting on you, dispatch it. If no chains are pending, report idle.\n[TOOL swarm.list {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a03",
|
|
||||||
"name_template": "auto-work-646-a03-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a04",
|
|
||||||
"description": "646 auto-work: 646 vars check: act on scope variables signaling work",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "13,43 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 vars check: run box vars-list and read variables in your scope (for example pulse_interval_m). If a variable signals pending work or a changed threshold, act on it; otherwise report steady.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a04",
|
|
||||||
"name_template": "auto-work-646-a04-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a05",
|
|
||||||
"description": "646 auto-work: 646 node health: fleet-status check on warp-646",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "17,47 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 node health: run box fleet-status and check your node (warp-646, CDP 9430). If proc_alive or cdp_ok is false, run the watchdog repair path and verify with a second read-back. Healthy: report ok.\n[TOOL health.check {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a05",
|
|
||||||
"name_template": "auto-work-646-a05-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a06",
|
|
||||||
"description": "646 auto-work: 646 alert triage: watchdog-alerts in your scope",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "19,49 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 alert triage: run box watchdog-alerts and look for alerts in your scope. Triage the newest one: fix it, or escalate to opm with box notify. No alerts: report clear.\n[TOOL service.status {\"unit\": \"board.service\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a06",
|
|
||||||
"name_template": "auto-work-646-a06-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a07",
|
|
||||||
"description": "646 auto-work: 646 service sweep: board/harvester service health",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "23,53 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 service sweep: verify service health the box-service-health way (fleet-status plus service checks for board and harvester). Restart one failed allowlisted service, verify it recovered, otherwise escalate. All healthy: report ok.\n[TOOL service.status {\"unit\": \"response-harvester.timer\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a07",
|
|
||||||
"name_template": "auto-work-646-a07-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a08",
|
|
||||||
"description": "646 auto-work: 646 browser check: CDP latency and chrome errors",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "27,57 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 browser check: run box cdp-latency and chrome-errors for your profile. If latency is spiking or new FATAL errors appear since the watermark, restart chromium via the watchdog path and verify. Otherwise report healthy.\n[TOOL health.check {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a08",
|
|
||||||
"name_template": "auto-work-646-a08-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a09",
|
|
||||||
"description": "646 auto-work: 646 digest scan: act on newest actionable digest",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "1,31 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 digest scan: read your task sidechat for the newest actionable digest. ACK it and act, or mark it no-action with a one-line reason. Nothing actionable: take no further action.\n[TOOL dm.read {\"agent\": \"646\", \"limit\": 10}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a09",
|
|
||||||
"name_template": "auto-work-646-a09-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a10",
|
|
||||||
"description": "646 auto-work: 646 DM sweep: reply to unanswered DMs",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "9,39 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 DM sweep: run box dm-log and check for unanswered DMs addressed to you. Reply to the newest one needing a response, or escalate to opm if blocked. None waiting: report clear.\n[TOOL dm.read {\"agent\": \"646\", \"limit\": 10}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a10",
|
|
||||||
"name_template": "auto-work-646-a10-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a11",
|
|
||||||
"description": "646 auto-work: 646 loop check: resolve pending/stale digest followups",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "5,35 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 loop check: run box loop-status for agent 646 and look for pending or stale digest followups. Resolve or nudge the oldest stale one according to its declared followup. All resolved: report ok.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a11",
|
|
||||||
"name_template": "auto-work-646-a11-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a12",
|
|
||||||
"description": "646 auto-work: 646 swarm scan: claim an open swarm slot",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "15,45 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 swarm scan: run box swarm-list and find swarms with open slots in your scope. Attach to one slot and complete it, then report the result. No open slots: report idle.\n[TOOL swarm.list {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a12",
|
|
||||||
"name_template": "auto-work-646-a12-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a13",
|
|
||||||
"description": "646 auto-work: Cross-op scan: claim unclaimed work from #jobs/board",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "21,51 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "Cross-op scan: check #jobs and the board for unclaimed work other operators posted. If something fits 646 scope, claim it and start; otherwise note what you saw in one line. Nothing new: report idle.\n[TOOL swarm.list {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a13",
|
|
||||||
"name_template": "auto-work-646-a13-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a14",
|
|
||||||
"description": "646 auto-work: Fleet watch: nudge operators that look stuck",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "25,55 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "Fleet watch: run box fleet-status and compare queue depth and recent activity across muse, pip, opm, def, dev. If an operator looks stuck (growing queue, stale watermark), send them a nudge via box notify. Everyone moving: report ok.\n[TOOL health.check {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a14",
|
|
||||||
"name_template": "auto-work-646-a14-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a15",
|
|
||||||
"description": "646 auto-work: 646 harvest follow-up: pick up tagged follow-up work",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "29,59 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 harvest follow-up: check the latest opm-swarm-harvest summary for follow-up work tagged to you. Pick up the top item and complete it, or report none tagged.\n[TOOL swarm.list {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a15",
|
|
||||||
"name_template": "auto-work-646-a15-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a16",
|
|
||||||
"description": "646 auto-work: Fleet pulse check: main-loop and heartbeat status",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "2,32 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "Fleet pulse check: run box main-loop status and check heartbeat status. If the main loop or heartbeat shows errors affecting your scope, investigate and fix or escalate to opm. Green across the board: report ok.\n[TOOL service.status {\"unit\": \"board.service\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a16",
|
|
||||||
"name_template": "auto-work-646-a16-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a17",
|
|
||||||
"description": "646 auto-work: 646 timer audit: fix orphaned/disabled/erroring timers",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "6,36 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 timer audit: run box timer-list and look for your timers that are orphaned, disabled, or erroring. Re-enable or fix one, or report all healthy.\n[TOOL cron.status {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a17",
|
|
||||||
"name_template": "auto-work-646-a17-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a18",
|
|
||||||
"description": "646 auto-work: 646 quality gate: box quality check must stay green",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "12,42 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 quality gate: run box quality check. If any check fails, investigate the failing validator and fix it, or escalate with the exact failure text. All green: report the pass count.\n[TOOL health.check {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a18",
|
|
||||||
"name_template": "auto-work-646-a18-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a19",
|
|
||||||
"description": "646 auto-work: 646 failure review: fix one recurring timer failure",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "18,48 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 failure review: look at recent runs of your auto-work timers (box timer-status) and find recurring failures. Fix the root cause of one repeat failure, or escalate with evidence. No repeats: report clean.\n[TOOL cron.status {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a19",
|
|
||||||
"name_template": "auto-work-646-a19-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-646-a20",
|
|
||||||
"description": "646 auto-work: 646 open-work digest: 5-line summary to task thread",
|
|
||||||
"agent": "646",
|
|
||||||
"schedule": "24,54 * * * *",
|
|
||||||
"timeout": 300,
|
|
||||||
"prompt_template": "646 open-work digest: summarize your currently open work items (pending jobs, unacked digests, open swarm slots) into a 5-line digest in your task thread. Keep it short; informational only.\n[TOOL dm.read {\"agent\": \"646\", \"limit\": 5}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-646-a20",
|
|
||||||
"name_template": "auto-work-646-a20-{date}"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i01",
|
|
||||||
"description": "Auto-work canary for dev worker: node self-check",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "3 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Canary self-check: verify your node is reachable and healthy. Quiet run; no main-chat posts.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i01",
|
|
||||||
"name_template": "auto-work-dev-i01-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i02",
|
|
||||||
"description": "Auto-work tool exercise for dev worker",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "9 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Tool exercise: run a service.status tool call for board.service and summarize the outcome in one line. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i02",
|
|
||||||
"name_template": "auto-work-dev-i02-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i03",
|
|
||||||
"description": "Auto-work canary for dev worker: node self-check",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "15 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Canary self-check: verify your node is reachable and healthy. Quiet run; no main-chat posts.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i03",
|
|
||||||
"name_template": "auto-work-dev-i03-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i04",
|
|
||||||
"description": "Auto-work swarm slot for dev worker",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "21 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Swarm slot: spawn one ephemeral worker via box swarm on your scope and collect its result. Quiet run; no main-chat posts.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i04",
|
|
||||||
"name_template": "auto-work-dev-i04-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i05",
|
|
||||||
"description": "Auto-work tool exercise for dev worker",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "27 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Tool exercise: run a service.status tool call for board.service and summarize the outcome in one line. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i05",
|
|
||||||
"name_template": "auto-work-dev-i05-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i06",
|
|
||||||
"description": "Auto-work canary for dev worker: node self-check",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "33 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Canary self-check: verify your node is reachable and healthy. Quiet run; no main-chat posts.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i06",
|
|
||||||
"name_template": "auto-work-dev-i06-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i07",
|
|
||||||
"description": "Auto-work swarm slot for dev worker",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "39 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Swarm slot: spawn one ephemeral worker via box swarm on your scope and collect its result. Quiet run; no main-chat posts.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i07",
|
|
||||||
"name_template": "auto-work-dev-i07-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i08",
|
|
||||||
"description": "Auto-work tool exercise for dev worker",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "45 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Tool exercise: run a service.status tool call for board.service and summarize the outcome in one line. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i08",
|
|
||||||
"name_template": "auto-work-dev-i08-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i09",
|
|
||||||
"description": "Auto-work canary for dev worker: node self-check",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "51 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Canary self-check: verify your node is reachable and healthy. Quiet run; no main-chat posts.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i09",
|
|
||||||
"name_template": "auto-work-dev-i09-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i10",
|
|
||||||
"description": "Auto-work swarm job for dev worker pool",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "13 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Swarm-slot exercise: take one small self-contained task (e.g. lint one file, summarize one doc section) and do it fully, then report a one-line summary of what you did. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i10",
|
|
||||||
"name_template": "auto-work-dev-i10-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i11",
|
|
||||||
"description": "Auto-work tools job for def worker pool",
|
|
||||||
"agent": "def",
|
|
||||||
"schedule": "20 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Tool-call exercise: run a service.status tool call against board.service and report the result briefly, then report what you did in one line. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i11",
|
|
||||||
"name_template": "auto-work-dev-i11-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i12",
|
|
||||||
"description": "Auto-work result job for dev worker pool",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "27 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Result-return check: do one small real task of your choosing, verify you can read it back, then report a one-line summary confirming completion. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i12",
|
|
||||||
"name_template": "auto-work-dev-i12-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i13",
|
|
||||||
"description": "Auto-work canary job for def worker pool",
|
|
||||||
"agent": "def",
|
|
||||||
"schedule": "34 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Canary self-check: verify your node is reachable and healthy. Quiet run; no main-chat posts.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i13",
|
|
||||||
"name_template": "auto-work-dev-i13-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i14",
|
|
||||||
"description": "Auto-work swarm job for dev worker pool",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "41 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Swarm-slot exercise: take one small self-contained task (e.g. lint one file, summarize one doc section) and do it fully, then report a one-line summary of what you did. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i14",
|
|
||||||
"name_template": "auto-work-dev-i14-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i15",
|
|
||||||
"description": "Auto-work tools job for def worker pool",
|
|
||||||
"agent": "def",
|
|
||||||
"schedule": "48 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Tool-call exercise: run a service.status tool call against board.service and report the result briefly, then report what you did in one line. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i15",
|
|
||||||
"name_template": "auto-work-dev-i15-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i16",
|
|
||||||
"description": "Auto-work result job for dev worker pool",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "55 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Result-return check: do one small real task of your choosing, verify you can read it back, then report a one-line summary confirming completion. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i16",
|
|
||||||
"name_template": "auto-work-dev-i16-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i17",
|
|
||||||
"description": "Auto-work canary job for def worker pool",
|
|
||||||
"agent": "def",
|
|
||||||
"schedule": "2 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Canary self-check: verify your node is reachable and healthy. Quiet run; no main-chat posts.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i17",
|
|
||||||
"name_template": "auto-work-dev-i17-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i18",
|
|
||||||
"description": "Auto-work swarm job for dev worker pool",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "9 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Swarm-slot exercise: take one small self-contained task (e.g. lint one file, summarize one doc section) and do it fully, then report a one-line summary of what you did. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i18",
|
|
||||||
"name_template": "auto-work-dev-i18-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i19",
|
|
||||||
"description": "Auto-work tools job for def worker pool",
|
|
||||||
"agent": "def",
|
|
||||||
"schedule": "16 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Tool-call exercise: run a service.status tool call against board.service and report the result briefly, then report what you did in one line. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i19",
|
|
||||||
"name_template": "auto-work-dev-i19-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-dev-i20",
|
|
||||||
"description": "Auto-work result job for dev worker pool",
|
|
||||||
"agent": "dev",
|
|
||||||
"schedule": "23 * * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Result-return check: do one small real task of your choosing, verify you can read it back, then report a one-line summary confirming completion. Quiet run.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-dev-i20",
|
|
||||||
"name_template": "auto-work-dev-i20-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h01",
|
|
||||||
"description": "Box quality check (core) \u2014 hourly",
|
|
||||||
"schedule": "13 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Box quality check (core).\nJob ID: {job_id}\nTime: {datetime}\n\nRun:\n[TOOL health.check {}]\n\nThen run the box quality gate:\n[EXEC quality.check {}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h01",
|
|
||||||
"reuse_key": "auto-work-health-h01"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h02",
|
|
||||||
"description": "Box quality check (fleet nodes) \u2014 hourly",
|
|
||||||
"schedule": "43 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nBox quality check across fleet nodes.\nRun:\n[TOOL health.check {}]\n\nThen:\n[EXEC quality.check {\"scope\": \"fleet\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h02",
|
|
||||||
"reuse_key": "auto-work-health-h02"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h03",
|
|
||||||
"description": "Quality validate: job \u2014 hourly",
|
|
||||||
"schedule": "11 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nValidate box job definitions.\nRun:\n[EXEC quality.validate {\"target\": \"job\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h03",
|
|
||||||
"reuse_key": "auto-work-health-h03"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h04",
|
|
||||||
"description": "Quality validate: status \u2014 hourly",
|
|
||||||
"schedule": "41 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nValidate box status surface.\nRun:\n[EXEC quality.validate {\"target\": \"status\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h04",
|
|
||||||
"reuse_key": "auto-work-health-h04"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h05",
|
|
||||||
"description": "Quality validate: heartbeat \u2014 hourly",
|
|
||||||
"schedule": "26 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nValidate heartbeat pipeline.\nRun:\n[EXEC quality.validate {\"target\": \"heartbeat\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h05",
|
|
||||||
"reuse_key": "auto-work-health-h05"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h06",
|
|
||||||
"description": "Quality check deep: validators catalog \u2014 hourly",
|
|
||||||
"schedule": "56 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nDeep quality check on input validators and error-code catalog.\nRun:\n[TOOL health.check {}]\n\nThen:\n[EXEC quality.check {\"scope\": \"validators\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h06",
|
|
||||||
"reuse_key": "auto-work-health-h06"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h07",
|
|
||||||
"description": "Box service health deep: endpoints \u2014 hourly",
|
|
||||||
"schedule": "3 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nDeep service-health check: box endpoints.\nRun:\n[TOOL service.status {\"unit\": \"board.service\"}]\n[TOOL health.check {}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h07",
|
|
||||||
"reuse_key": "auto-work-health-h07"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h08",
|
|
||||||
"description": "Box service health deep: data/audit \u2014 hourly",
|
|
||||||
"schedule": "33 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nDeep service-health check: data layer and audit log.\nRun:\n[TOOL service.status {\"unit\": \"response-harvester.timer\"}]\n[TOOL health.check {}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h08",
|
|
||||||
"reuse_key": "auto-work-health-h08"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h09",
|
|
||||||
"description": "Box HTTP health deep: all routes \u2014 hourly",
|
|
||||||
"schedule": "18 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nDeep HTTP health: all box routes.\nRun:\n[TOOL health.check {}]\n\nThen:\n[TOOL cron.runs {}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h09",
|
|
||||||
"reuse_key": "auto-work-health-h09"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h10",
|
|
||||||
"description": "Box HTTP health deep: auth/pin paths \u2014 hourly",
|
|
||||||
"schedule": "48 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nDeep HTTP health: auth and PIN paths.\nRun:\n[TOOL health.check {}]\n\nThen:\n[TOOL cron.status {}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h10",
|
|
||||||
"reuse_key": "auto-work-health-h10"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h11",
|
|
||||||
"description": "Quality check: timer hygiene (orphans) \u2014 hourly",
|
|
||||||
"schedule": "8 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nTimer hygiene: find orphan or disabled timers.\nRun:\n[TOOL cron.status {}]\n\nThen:\n[EXEC quality.check {\"scope\": \"timers\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h11",
|
|
||||||
"reuse_key": "auto-work-health-h11"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h12",
|
|
||||||
"description": "Quality check: job chain integrity \u2014 hourly",
|
|
||||||
"schedule": "38 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nJob chain integrity: verify chain_next links resolve.\nRun:\n[EXEC quality.check {\"scope\": \"chains\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h12",
|
|
||||||
"reuse_key": "auto-work-health-h12"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h13",
|
|
||||||
"description": "Box service health: relay/exec bridge \u2014 hourly",
|
|
||||||
"schedule": "23 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nRelay and exec-bridge health.\nRun:\n[TOOL service.status {\"unit\": \"board.service\"}]\n[TOOL health.check {}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h13",
|
|
||||||
"reuse_key": "auto-work-health-h13"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h14",
|
|
||||||
"description": "Box HTTP health: dashboard SPA \u2014 hourly",
|
|
||||||
"schedule": "53 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nDashboard SPA health.\nRun:\n[TOOL health.check {}]\n\nThen fetch the dashboard route:\n[TOOL web.fetch {\"url\": \"https://box.muse-dev.online/\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h14",
|
|
||||||
"reuse_key": "auto-work-health-h14"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h15",
|
|
||||||
"description": "Failure rollup: collect FAILs, alert \u2014 hourly",
|
|
||||||
"schedule": "28 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nFailure rollup: scan recent job results for FAIL outcomes and summarize.\nRun:\n[TOOL cron.runs {}]\n\nThen:\n[EXEC quality.validate {\"target\": \"status\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h15",
|
|
||||||
"reuse_key": "auto-work-health-h15"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h16",
|
|
||||||
"description": "cron.runs audit: missed/failed runs \u2014 hourly",
|
|
||||||
"schedule": "58 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nAudit scheduled runs for misses and failures.\nRun:\n[TOOL cron.runs {}]\n[TOOL cron.status {}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h16",
|
|
||||||
"reuse_key": "auto-work-health-h16"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h17",
|
|
||||||
"description": "Quality validate: dm-log route \u2014 hourly",
|
|
||||||
"schedule": "16 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nValidate dm-log pipeline.\nRun:\n[EXEC quality.validate {\"target\": \"dm-log\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h17",
|
|
||||||
"reuse_key": "auto-work-health-h17"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h18",
|
|
||||||
"description": "Quality check: vars/strategy stores \u2014 hourly",
|
|
||||||
"schedule": "46 * * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nVars and strategy store health.\nRun:\n[TOOL vars.list {}]\n[TOOL health.check {}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h18",
|
|
||||||
"reuse_key": "auto-work-health-h18"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h19",
|
|
||||||
"description": "Box quality check (sidechat reuse keys) \u2014 daily",
|
|
||||||
"schedule": "6 5 * * *",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nSidechat reuse-key audit: verify persistent job threads resolve.\nRun:\n[EXEC quality.check {\"scope\": \"sidechats\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h19",
|
|
||||||
"reuse_key": "auto-work-health-h19"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-health-h20",
|
|
||||||
"description": "Weekly deep quality report \u2014 Sundays",
|
|
||||||
"schedule": "36 6 * * 0",
|
|
||||||
"agent": "646",
|
|
||||||
"timeout": 1800,
|
|
||||||
"prompt_template": "Job ID: {job_id}\nTime: {datetime}\n\nWeekly deep quality report: full gate suite plus trend summary.\nRun:\n[TOOL health.check {}]\n\nThen:\n[EXEC quality.check {\"scope\": \"all\"}]\n\nEnd with your verdict: OK if all green, otherwise FAIL plus a one-line summary of what failed.",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"name_template": "auto-work-health-h20",
|
|
||||||
"reuse_key": "auto-work-health-h20"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"escalate": "opm",
|
|
||||||
"expect_reply": true,
|
|
||||||
"nudges": 2,
|
|
||||||
"route": "auto-work-health",
|
|
||||||
"timeout": "30m"
|
|
||||||
},
|
|
||||||
"on_failure": "alert",
|
|
||||||
"chain_next": null
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c01",
|
|
||||||
"description": "Muse work sweep: claimable jobs",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "7 0 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Sweep for work muse can claim. 1) Find manual, unclaimed, or failed jobs in the job list. 2) Check the muse task thread for open items. Claim up to 2 muse-suitable jobs and start the first; leave the rest. Box: box job-list, box timer-list, box fleet-status. If nothing actionable, report NO-ACTION with a 3-line summary.\n[TOOL job-list {\"agent\": \"muse\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c01",
|
|
||||||
"name_template": "auto-work-muse-c01-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c02",
|
|
||||||
"description": "Muse work sweep: stale followups",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "19 1 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Sweep stale followups. 1) Find expired or unanswered followups routed to muse. 2) Nudge once where a nudge is due; escalate to opm where retries are exhausted; close what is done. If nothing actionable, report NO-ACTION with a 3-line summary.\n[TOOL job-list {\"agent\": \"muse\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c02",
|
|
||||||
"name_template": "auto-work-muse-c02-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c03",
|
|
||||||
"description": "Muse work sweep: task thread pickup",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "31 2 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Review the muse task thread. Pick up the oldest actionable item, do it, and report back. If nothing is actionable, verify thread health and report NO-ACTION with a 3-line summary.\n[TOOL job-list {\"agent\": \"muse\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c03",
|
|
||||||
"name_template": "auto-work-muse-c03-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c04",
|
|
||||||
"description": "Muse work sweep: board and jobs channel",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "43 3 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Scan the board and #jobs for unclaimed work orders or review requests. Claim at most 1 that fits muse scope; post a claim note so others do not duplicate. If nothing fits, report NO-ACTION with a 3-line summary.\n[TOOL job-list {\"agent\": \"muse\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c04",
|
|
||||||
"name_template": "auto-work-muse-c04-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c05",
|
|
||||||
"description": "Muse work sweep: stalled swarms",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "55 4 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Check the swarm list for stalled or partial swarms. Harvest completed slot results, kill swarms stale over 2h, and report one summary. If all healthy, report NO-ACTION with a 3-line summary.\n[TOOL swarm.list {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c05",
|
|
||||||
"name_template": "auto-work-muse-c05-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c06",
|
|
||||||
"description": "Cross-op scan: pip",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "7 6 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Check pip's task thread and recent activity for gaps or stalled items muse can take. If pip is stuck, post a scoped offer of help in the coordination thread; do not duplicate her work. Report findings either way.\n[TOOL job-list {\"agent\": \"muse\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c06",
|
|
||||||
"name_template": "auto-work-muse-c06-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c07",
|
|
||||||
"description": "Cross-op scan: 646",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "19 7 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Check 646's task thread and recent results for gaps muse can cover. Claim only what is clearly unowned; report what you found either way.\n[TOOL job-list {\"agent\": \"muse\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c07",
|
|
||||||
"name_template": "auto-work-muse-c07-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c08",
|
|
||||||
"description": "Cross-op scan: opm/dev/def",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "31 8 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Scan opm, dev, and def scopes for unowned or aging work. Surface the top 3 items to the muse task thread with a take-or-leave recommendation; claim at most 1. If nothing, report NO-ACTION.\n[TOOL job-list {\"agent\": \"muse\"}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c08",
|
|
||||||
"name_template": "auto-work-muse-c08-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c09",
|
|
||||||
"description": "Cross-op scan: alerts and lobby",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "43 9 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Scan #lobby, #fleet-status, and watchdog alerts for anything mentioning muse or unowned. Acknowledge alerts in range; escalate anything out of scope to opm. If quiet, report NO-ACTION.\n[TOOL health.check {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c09",
|
|
||||||
"name_template": "auto-work-muse-c09-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
{
|
|
||||||
"name": "auto-work-muse-c10",
|
|
||||||
"description": "Muse node health check",
|
|
||||||
"agent": "muse",
|
|
||||||
"schedule": "55 10 * * *",
|
|
||||||
"timeout": 600,
|
|
||||||
"prompt_template": "Verify the muse node: warp-muse netns up, CDP 9410 responsive, queue depth 0, egress healthy. If the browser is wedged, restart via the allowlisted service action and verify recovery. Report status.\n[TOOL health.check {}]",
|
|
||||||
"sidechat": {
|
|
||||||
"create": true,
|
|
||||||
"reuse_key": "auto-work-muse-c10",
|
|
||||||
"name_template": "auto-work-muse-c10-{date}"
|
|
||||||
},
|
|
||||||
"followup": {
|
|
||||||
"expect_reply": true,
|
|
||||||
"timeout": "30m",
|
|
||||||
"nudges": 1,
|
|
||||||
"escalate": "opm",
|
|
||||||
"route": "auto-work"
|
|
||||||
},
|
|
||||||
"chain_next": null,
|
|
||||||
"on_failure": "alert"
|
|
||||||
}
|
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user