diff --git a/bin/digest.py b/bin/digest.py new file mode 100644 index 0000000..68c2e37 --- /dev/null +++ b/bin/digest.py @@ -0,0 +1,74 @@ +#!/usr/bin/env python3 +""" +Side-chat work digest β€” INTEGRATOR minimal version (agent 4 of 5 never +delivered its digest/throttle design within the window). + +Purpose: stop the per-message βœ… COMPLETED relay flood. COMPLETED hits are +batched here and emitted as ONE periodic digest instead of N main-chat +messages. ALERT / BLOCKER / DECISION still relay individually via +siphon() β€” urgency is never batched. + +Interface (what agent 4's full design should remain compatible with): + - buffer = DigestBuffer(max_items=20, max_age_s=3600) + - buffer.add(hit) -> None + - buffer.flush() -> Optional[str] (formatted digest, clears buffer) + - flush_digest() -> Optional[str] (module-level singleton convenience) + +Reversible: to restore per-message COMPLETED relays, route COMPLETED back +through siphon() in monitor.py and ignore this module. +""" + +import time +from typing import List, Optional + +try: + from detect import SiphonHit +except ImportError: # pragma: no cover + SiphonHit = object + + +class DigestBuffer: + """Batch COMPLETED hits; flush() renders one digest message.""" + + def __init__(self, max_items: int = 20, max_age_s: int = 3600): + self.max_items = max_items + self.max_age_s = max_age_s + self._items: List[tuple] = [] # (ts, SiphonHit) + + def add(self, hit) -> None: + now = time.time() + # Prune items older than max_age_s on every add (bounded memory). + self._items = [(ts, h) for ts, h in self._items + if now - ts < self.max_age_s] + self._items.append((now, hit)) + # Bound the buffer; oldest evicted first. + self._items = self._items[-self.max_items:] + + def __len__(self) -> int: + return len(self._items) + + def flush(self) -> Optional[str]: + """Render and clear. Returns None when there's nothing to digest.""" + if not self._items: + return None + lines = ["πŸ“¦ [DIGEST] completions from side chats " + f"({len(self._items)} item(s))"] + for _, hit in self._items: + author = getattr(hit, "author", "") or "?" + url = f"https://muse.ai/thread/{hit.thread_id}" + lines.append(f"β€’ {hit.summary} β€” {author} ({url})") + self._items = [] + return "\n".join(lines) + + +# Module-level singleton: the monitor loop shares one buffer per process. +_default_buffer = DigestBuffer() + + +def get_buffer() -> DigestBuffer: + return _default_buffer + + +def flush_digest() -> Optional[str]: + """Flush the process-wide digest buffer. None if empty.""" + return _default_buffer.flush() diff --git a/bin/muse_choice_watcher.py b/bin/muse_choice_watcher.py new file mode 100755 index 0000000..97cfa46 --- /dev/null +++ b/bin/muse_choice_watcher.py @@ -0,0 +1,1456 @@ +#!/usr/bin/env python3 +"""muse_choice_watcher.py β€” Per-pane daemon auto-approving Muse prompts. + +Kinds (prompt -> key): native Muse approval menu -> "1", agent +interview menu -> "1", explicit-phrase request -> captured TOKEN, +lettered A/B/C choice -> "A", numbered (1)/(2) menu -> "1", y/n +line-end prompt -> "y". +Each prompt is answered at most once +(after stability + re-verify), with an hourly cap as backstop. + +Design notes (from the agy watcher post-mortem): + * State is namespaced by (socket, pane): pidfiles and logs embed a socket + slug, so identical pane ids on different tmux sockets (e.g. %0 on both + `default` and `lte`) never collide. + * Logging never kills the loop: every log call and every poll iteration is + exception-guarded, so auto-answers continue even when logging or tmux + hiccups. Log rotation is best-effort. + * Answers are once-per-prompt: a prompt signature must be stable across N + polls, is answered at most once, and re-answering requires the prompt to + disappear and a new signature to appear. An hourly cap bounds runaways. + +Usage: + muse_choice_watcher.py start --socket /tmp/tmux-1000/default --pane %37 + muse_choice_watcher.py start-all [--dry-run] + muse_choice_watcher.py status + muse_choice_watcher.py stop --socket ... --pane ... + muse_choice_watcher.py match < pane.txt # print JSON verdict for tuning +""" + +import argparse +import hashlib +import json +import os +import re +import signal +import subprocess +import sys +import time +from datetime import datetime, timezone + +STATE_DIR = "/tmp" +FILE_PREFIX = "muse-choice-watcher" +REPO_ROOT = "/home/super/Projects/NetVM" +DESIRED_STATE_FILE = os.path.join(REPO_ROOT, ".state", "muse-choices.json") +CTL_LOG = os.path.join(REPO_ROOT, "box-ctl.jsonl") +KNOWN_SOCKETS = [ + "/tmp/tmux-1000/default", + "/tmp/tmux-1000/lte", + "/tmp/tmux-muse.sock", +] + +POLL_INTERVAL = 0.5 +STABILITY_POLLS = 2 +MAX_ANSWERS_PER_HOUR = 20 +ANSWERED_TTL_SECONDS = 600 # identical prompt back after 10m => stuck, allow one recovery answer +LOG_MAX_BYTES = 1_000_000 +HEARTBEAT_SECONDS = 60 +TAIL_WINDOW = 25 # only prompts in the last N lines count (no stale scrollback) +OPTION_SPAN = 12 # A..C option lines must fit within N lines (wrapped lines ok) +MAX_OPTION_LEN = 160 # option lines longer than this are ignored (prose guard) + +# A. text / B) text / C: text / C - text (single capital letter + delimiter) +OPTION_RE = re.compile(r"^\s*([A-Z])\s*[.\)\-:]\s+\S") +# Cue that the lettered list is a choice awaiting reply (not prose). +CUE_RE = re.compile( + r"(choose|choice|choices|select|option|options|reply|feedback|" + r"which one|pick one|enter\s+[A-Z]\b|type\s+[A-Z]\b|press\s+[A-Z]\b|" + r"A\s*/\s*B|A\s*,\s*B|A-C|A,B,C|\(A\))", + re.IGNORECASE, +) +QUESTION_RE = re.compile(r"\?\s*$") + +# y/n token must END the line: prompts awaiting input put the options last +# ("Proceed? (y/n)", "Overwrite? [y/N]"). Mid-line mentions are prose. +YN_RE = re.compile(r"([yY]/[nN]|\[[yY]/[nN]\])\s*[\]:)>]?\s*$") +YN_WINDOW = 6 # y/n prompt must sit in the last N content lines +# Numbered menus: (1) text / (2) text with a selection cue nearby. +NUMBERED_OPT_RE = re.compile(r"^\s*\((\d+)\)\s+\S") +NUMBERED_CUE_RE = re.compile( + r"(Option:|Selection:|choose|choice|select an? |enter (a )?number|pick a number)", + re.IGNORECASE, +) +# Native Muse approval dialog patterns live in MUSE_APPROVAL_*_RE below +# (single definition; the dialog shape is asserted by TestMuseApproval).# Native Muse TUI approval menu (the live auto-approve target): +# Would you like to run the following +# $ +# > 1. Yes, proceed (y) +# 2. No, and tell Muse Code what to ... +# The cue carries no trailing "?", and the cursor may render as an ascii +# ">" or the single right-pointing angle quote (U+203A, escaped below). +MUSE_APPROVAL_CUE_RE = re.compile(r"^\s*Would you like to\b", re.IGNORECASE) +MUSE_APPROVAL_YES_RE = re.compile("^\\s*[>\\u203a]?\\s*1\\.\\s+Yes\\b") +MUSE_APPROVAL_NO_RE = re.compile("^\\s*[>\\u203a]?\\s*2\\.\\s+No\\b") +# A decided block ("approval decision accepted/denied") must never match: +# answering it would inject stray keys after the decision already landed. +MUSE_APPROVAL_DECIDED_RE = re.compile(r"approval decision", re.IGNORECASE) +MUSE_APPROVAL_SPAN = 16 # cue..options span (wrapped command echo ok) +# Collapsed approval: long commands collapse, hiding the 1.Yes/2.No pair: +# Would you like to run the following +# $ +# ... N command rows omitted +# ctrl+o view full command +MUSE_COLLAPSED_ROWS_RE = re.compile(r"rows?\s+omitted", re.IGNORECASE) +MUSE_COLLAPSED_KEY_RE = re.compile(r"ctrl\s*\+\s*o\b", re.IGNORECASE) + +# Agent interview UI (request_user_input rendered in-pane; observed live): +# +# > 1. First option (Recommended) +# 2. Second option +# 3. Both ... +# 4. None of the above ... +# The `>`/`β€Ί` cursor on an option line proves a live selectable menu +# (prose lists never carry it); the question above ends with "?". +INTERVIEW_OPT_RE = re.compile("^\\s*([>\\u203a])?\\s*(\\d+)\\.\\s+\\S") +INTERVIEW_CUE_ABOVE = 8 # question must sit within N lines above option 1 + +# Explicit-phrase request (model asks the user to reply a magic word; +# observed live): "Reply ACCEPT to approve this text as written ..." +# Only a single ALL-CAPS token qualifies -- lowercase/prose after Reply +# never matches, which keeps quoted instructions and chat from firing. +EXPLICIT_RE = re.compile(r"^\s*(?i:Reply)\s+([A-Z][A-Z0-9_-]{1,11})\b") +EXPLICIT_WINDOW = 8 # request must sit in the last N content lines + +# Runtime state sensing for external agents driving panes via send-keys. +# A working pane shows a running indicator ("- running (Ns ...", "Calling +# tools (...", or the "esc to interrupt" tail, often wrapped/edge-cut); +# an idle pane sits at the prompt glyph (U+276F, escaped below). +RUNNING_RE = re.compile("([\\u2014-] running \\(|Calling tools \\(|esc to\\b)") +OPEN_PROMPT_RE = re.compile("^\\s*\\u276f") +STATE_TAIL_WINDOW = 12 # state cues must sit in the last N content lines + + +def slug_socket(socket_path): + """Short filesystem-safe id for a tmux socket: basename + hash suffix. + + The hash suffix guards against two different socket paths sharing a + basename (e.g. /tmp/a/default vs /tmp/b/default). + """ + base = os.path.basename(socket_path.rstrip("/")) or "sock" + safe = re.sub(r"[^A-Za-z0-9_.-]+", "_", base)[:32] + digest = hashlib.sha1(socket_path.encode()).hexdigest()[:6] + return "%s-%s" % (safe, digest) + + +def clean_pane(pane_id): + return re.sub(r"[^A-Za-z0-9]+", "", pane_id) + + +def pidfile_for(socket_path, pane_id): + return os.path.join( + STATE_DIR, "%s-%s-%s.pid" % (FILE_PREFIX, slug_socket(socket_path), clean_pane(pane_id)) + ) + + +def logfile_for(socket_path, pane_id): + return os.path.join( + STATE_DIR, "%s-%s-%s.log" % (FILE_PREFIX, slug_socket(socket_path), clean_pane(pane_id)) + ) + + +class WatcherLog: + """Never-raising JSON-lines logger with best-effort rotation.""" + + def __init__(self, path): + self.path = path + + def _rotate(self): + try: + if os.path.exists(self.path) and os.path.getsize(self.path) >= LOG_MAX_BYTES: + try: + if os.path.exists(self.path + ".1"): + os.remove(self.path + ".1") + except OSError: + pass + os.rename(self.path, self.path + ".1") + except Exception: + pass + + def log(self, level, msg, **fields): + try: + self._rotate() + rec = { + "ts": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + "level": level, + "msg": msg, + } + rec.update(fields) + with open(self.path, "a") as f: + f.write(json.dumps(rec) + "\n") + except Exception: + pass + + +def _audit_local(action, name, caller, extra): + """Append a box-ctl.jsonl audit record without importing approvals.""" + rec = { + "ts": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + "action": action, + "type": "muse-choice", + "name": name, + "caller": caller, + } + if extra: + rec.update(extra) + with open(CTL_LOG, "a") as f: + f.write(json.dumps(rec) + "\n") + + +def audit(action, name=None, caller="muse-choice-watcher", extra=None): + """Feed an event to box (box-ctl.jsonl). Never raises. + + Prefers approvals.log_box_ctl so the audit schema stays unified; falls + back to a direct append if the import fails. + """ + try: + try: + import approvals + approvals.log_box_ctl(action, name=name, caller=caller, extra=extra) + except Exception: + _audit_local(action, name, caller, extra or {}) + except Exception: + pass + + +def get_desired(): + """Desired daemon state: {"enabled", "dry_run", ...}. Default: on. + + Policy: auto-approve is on unless explicitly disabled (`box + muse-choices off`), the pane opted out via launch flags, or a human + intervened. A missing/unreadable state file (or a file without the + key) means undefined -> on. + """ + default = {"enabled": True, "dry_run": False} + try: + with open(DESIRED_STATE_FILE) as f: + data = json.load(f) + return {"enabled": bool(data.get("enabled", True)), + "dry_run": bool(data.get("dry_run", False)), + "updated_at": data.get("updated_at", ""), + "updated_by": data.get("updated_by", "")} + except Exception: + return dict(default) + + +def set_enabled(enabled, dry_run=None, by="box"): + """Write desired state + audit the transition. Returns the new state.""" + state = get_desired() + state["enabled"] = bool(enabled) + if dry_run is not None: + state["dry_run"] = bool(dry_run) + state["updated_at"] = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + state["updated_by"] = by + try: + parent = os.path.dirname(DESIRED_STATE_FILE) + if parent: + os.makedirs(parent, exist_ok=True) + tmp = DESIRED_STATE_FILE + ".tmp.%d" % os.getpid() + with open(tmp, "w") as f: + json.dump(state, f, indent=1) + os.replace(tmp, DESIRED_STATE_FILE) + except Exception: + pass + audit("muse-choice-enabled" if enabled else "muse-choice-disabled", + caller=by, extra={"dry_run": state["dry_run"]}) + return state + + +def _option_markers(lines): + """Return [(line_index, letter)] for lettered-option lines.""" + out = [] + for i, line in enumerate(lines): + if len(line) > MAX_OPTION_LEN: + continue + m = OPTION_RE.match(line) + if m: + out.append((i, m.group(1))) + return out + + +def _ordered_run(markers): + """Find A,B[,C...] in order. Returns marker sub-list or None. + + Markers between run members (wrapped text has no markers, so any + interleaving lettered line breaks the run) must fit OPTION_SPAN. + """ + want = "ABCDEFGHIJKLMNOPQRSTUVWXYZ" + best = [] + for start in range(len(markers)): + if markers[start][1] != "A": + continue + run = [markers[start]] + for idx in range(start + 1, len(markers)): + expected = want[len(run)] + if markers[idx][1] == expected: + run.append(markers[idx]) + else: + break + if len(run) >= 2 and run[-1][0] - run[0][0] <= OPTION_SPAN: + if len(run) > len(best): + best = run + return best or None + + +def _sig_for(kind, parts): + src = kind + "\n" + "\n".join(parts) + return hashlib.sha1(src.encode()).hexdigest()[:16] + + +def _find_letter(window): + """A/B/C lettered choice block -> {"sig", "kind", "key", ...} or None.""" + run = _ordered_run(_option_markers(window)) + if not run: + return None + start, end = run[0][0], run[-1][0] + lo = max(0, start - 4) + hi = min(len(window), end + 5) + cue = None + for i in range(lo, hi): + if CUE_RE.search(window[i]) or QUESTION_RE.search(window[i]): + cue = window[i].strip() + break + if cue is None: + return None + options = [window[i].strip()[:120] for i, _ in run] + cue = cue[:200] + return {"sig": _sig_for("letter", options + [cue]), "kind": "letter", + "key": "A", "options": options, "cue": cue, + "start": start, "end": end} + + +def _find_numbered(window): + """(1)/(2) numbered menu with selection cue -> match or None.""" + markers = [] + for i, line in enumerate(window): + if len(line) > MAX_OPTION_LEN: + continue + m = NUMBERED_OPT_RE.match(line) + if m: + markers.append((i, int(m.group(1)))) + ones = [i for i, n in markers if n == 1] + twos = [i for i, n in markers if n == 2] + if not ones or not twos: + return None + start = ones[0] + end = next((i for i in twos if i > start), None) + if end is None or end - start > OPTION_SPAN: + return None + lo = max(0, start - 2) + hi = min(len(window), end + 4) + cue = None + for i in range(lo, hi): + if NUMBERED_CUE_RE.search(window[i]): + cue = window[i].strip() + break + if cue is None: + return None + options = [window[i].strip()[:120] + for i, n in markers if start <= i <= end] + cue = cue[:200] + return {"sig": _sig_for("numbered", options + [cue]), "kind": "numbered", + "key": "1", "options": options, "cue": cue, + "start": start, "end": end} + + +def _find_yn(window): + """y/n prompt at the end of a recent line -> match or None.""" + recent = window[-YN_WINDOW:] + base = len(window) - len(recent) + for j in range(len(recent) - 1, -1, -1): + line = recent[j] + if len(line) > MAX_OPTION_LEN: + continue + if YN_RE.search(line): + text = line.strip()[:200] + ctx = recent[j - 1].strip()[:120] if j > 0 else "" + i = base + j + return {"sig": _sig_for("yn", [text, ctx]), "kind": "yn", + "key": "y", "options": [text], "cue": text, + "start": i, "end": i} + return None + + +def _find_muse_approval(window): + """Native Muse TUI approval menu -> match or None. + + Requires the Would-you-like cue plus a 1.Yes/2.No pair below it. + A decided block ("approval decision ...") within the span never + matches, so an already-landed decision is never double-answered. + """ + cue_idx = None + for i, line in enumerate(window): + if len(line) > MAX_OPTION_LEN: + continue + if MUSE_APPROVAL_CUE_RE.match(line): + cue_idx = i + break + if cue_idx is None: + return None + yes_idx = no_idx = None + for i in range(cue_idx + 1, min(len(window), cue_idx + 1 + MUSE_APPROVAL_SPAN)): + line = window[i] + if len(line) > MAX_OPTION_LEN: + continue + if yes_idx is None and MUSE_APPROVAL_YES_RE.match(line): + yes_idx = i + elif MUSE_APPROVAL_NO_RE.match(line): + no_idx = i + break + if yes_idx is None or no_idx is None: + return None + for line in window[cue_idx:no_idx + 4]: + if MUSE_APPROVAL_DECIDED_RE.search(line): + return None + cue = window[cue_idx].strip()[:200] + options = [window[yes_idx].strip()[:120], window[no_idx].strip()[:120]] + # Sig covers the WHOLE block (cue + $ command + options): bare + # options+cue are byte-identical across every command approval, which + # used to collapse all of a pane's approvals into one sig and wedge + # every dialog after the first (observed live 2026-10-06). + context = [ln.strip()[:120] for ln in window[cue_idx:no_idx + 1]] + return {"sig": _sig_for("muse-approval", context), + "kind": "muse-approval", "key": "1", "options": options, + "cue": cue, "start": cue_idx, "end": no_idx} + + +def _find_collapsed_approval(window): + """Collapsed approval menu (long command, options hidden) or None. + + Requires the Would-you-like cue plus the collapsed markers ("N + command rows omitted" + "ctrl+o view full command"). Answers with + a bare Enter: that expands the block to the full 1/2 form (or + accepts outright), and the expanded form answers as a fresh + muse-approval prompt on later polls. (ctrl+o is NOT used: it opens + a modal pager that strands the session.) + """ + cue_idx = None + for i, line in enumerate(window): + if len(line) > MAX_OPTION_LEN: + continue + if MUSE_APPROVAL_CUE_RE.match(line): + cue_idx = i + break + if cue_idx is None: + return None + rows_idx = key_idx = None + for i in range(cue_idx + 1, min(len(window), cue_idx + 1 + MUSE_APPROVAL_SPAN)): + line = window[i] + if len(line) > MAX_OPTION_LEN: + continue + if rows_idx is None and MUSE_COLLAPSED_ROWS_RE.search(line): + rows_idx = i + if MUSE_COLLAPSED_KEY_RE.search(line): + key_idx = i + if rows_idx is None or key_idx is None: + return None + end = max(rows_idx, key_idx) + for line in window[cue_idx:end + 4]: + if MUSE_APPROVAL_DECIDED_RE.search(line): + return None + cue = window[cue_idx].strip()[:200] + options = [window[rows_idx].strip()[:120], + window[key_idx].strip()[:120]] + context = [ln.strip()[:120] for ln in window[cue_idx:end + 1]] + return {"sig": _sig_for("muse-approval-collapsed", context), + "kind": "muse-approval-collapsed", "key": "Enter", + "enter": False, "options": options, "cue": cue, + "start": cue_idx, "end": end} + + +def _find_interview(window): + """Agent interview menu (cursor + ordered 1./2./.. + question) or None. + + Requires an ordered run starting at 1 with at least options 1 and 2, + a selection cursor on one of the run's lines, and a "?"-ended + question above. The cursor requirement is what separates a live + interview from a prose numbered list. + """ + markers = [] + for i, line in enumerate(window): + if len(line) > MAX_OPTION_LEN: + continue + m = INTERVIEW_OPT_RE.match(line) + if m: + markers.append((i, int(m.group(2)), bool(m.group(1)))) + ones = [i for i, n, c in markers if n == 1] + if not ones: + return None + start = ones[0] + run = [start] + for i, n, c in markers: + if i > start and n == len(run) + 1 and i - start <= OPTION_SPAN: + run.append(i) + if len(run) < 2: + return None + if not any(c for i, n, c in markers if run[0] <= i <= run[-1]): + return None + question = None + for j in range(max(0, start - INTERVIEW_CUE_ABOVE), start): + if QUESTION_RE.search(window[j]): + question = window[j].strip() + if question is None: + return None + options = [window[i].strip()[:120] for i in run] + question = question[:200] + return {"sig": _sig_for("interview", options + [question]), + "kind": "interview", "key": "1", "options": options, + "cue": question, "start": start, "end": run[-1]} + + +def _find_explicit(window): + """Explicit-phrase request ("Reply ACCEPT to ...") or None. + + Scans the last EXPLICIT_WINDOW content lines bottom-up so the + freshest request wins. The key is the captured token itself. + """ + recent = window[-EXPLICIT_WINDOW:] + base = len(window) - len(recent) + for j in range(len(recent) - 1, -1, -1): + line = recent[j] + if len(line) > MAX_OPTION_LEN: + continue + m = EXPLICIT_RE.match(line) + if m: + token = m.group(1) + text = line.strip()[:200] + i = base + j + return {"sig": _sig_for("explicit-phrase", [text]), + "kind": "explicit-phrase", "key": token, + "options": [text], "cue": text, + "start": i, "end": i} + return None + + +def find_choice_prompt(text, tail_window=TAIL_WINDOW): + """Detect a Muse prompt awaiting reply. + + Kinds (priority order): muse-approval (native Would-you-like + menu -> key "1"), muse-approval-collapsed (collapsed long command + -> bare Enter, expand-or-accept), interview (cursor + ordered + 1./2. menu + question -> key "1"), explicit-phrase ("Reply TOKEN + to ..." -> key TOKEN), letter (A/B/C -> key "A"), numbered + ((1)/(2) menu -> key "1"), yn (y/n line-end -> key "y"). + Returns {"sig", "kind", "key", "options", "cue", "start", "end"} + or None. Only blocks in the last `tail_window` content lines are + eligible, so answered/stale prompts in scrollback never re-trigger. + """ + if not text: + return None + lines = text.split("\n") + # Trailing blank lines (tall panes, fresh prompts) must not push a live + # prompt out of the window; staleness is judged on content lines. + while lines and not lines[-1].strip(): + lines.pop() + window = lines[-tail_window:] + if not window: + return None + return (_find_muse_approval(window) or _find_collapsed_approval(window) + or _find_interview(window) or _find_explicit(window) + or _find_letter(window) or _find_numbered(window) + or _find_yn(window)) + + +def runtime_state(text): + """Classify a Muse pane capture for agentic send-keys drivers. + + Returns {"state", "match"} where state is one of approval-pending + (a choice prompt awaits reply), working (running indicator), + open-prompt (idle at the input prompt), or unknown. Priority is + approval-pending > working > open-prompt: a working pane still + renders its prompt footer, and an approval can arrive mid-run. + """ + match = find_choice_prompt(text) + if match is not None: + return {"state": "approval-pending", "match": match} + lines = (text or "").split("\n") + while lines and not lines[-1].strip(): + lines.pop() + tail = lines[-STATE_TAIL_WINDOW:] + for line in tail: + if RUNNING_RE.search(line): + return {"state": "working", "match": None} + for line in tail: + if len(line) > MAX_OPTION_LEN: + continue + if OPEN_PROMPT_RE.match(line): + return {"state": "open-prompt", "match": None} + return {"state": "unknown", "match": None} + + +def _cmdline(pid): + """Argv of a pid as a list (empty when unreadable). Never raises.""" + try: + with open("/proc/%d/cmdline" % pid, "rb") as f: + raw = f.read().split(b"\0") + return [p.decode(errors="replace") for p in raw if p] + except Exception: + return [] + + +def _child_pids(pid): + """Direct children of pid via a bounded /proc scan. Never raises.""" + out = [] + try: + names = os.listdir("/proc") + except Exception: + return out + for name in names: + if not name.isdigit(): + continue + try: + with open("/proc/%s/stat" % name) as f: + parts = f.read().rsplit(")", 1)[1].split() + if int(parts[1]) == pid: + out.append(int(name)) + except Exception: + continue + return out + + +def muse_approval_flags(cmd_argv): + """Approval posture of a muse argv: {"auto_approve", "flags"}.""" + flags = [] + argv = cmd_argv or [] + if "--yolo" in argv: + flags.append("yolo") + if "--disable-approval" in argv: + flags.append("disable-approval") + for i, arg in enumerate(argv): + if arg == "--approval-mode" and i + 1 < len(argv): + flags.append("approval-mode=%s" % argv[i + 1]) + elif arg.startswith("--approval-mode="): + flags.append("approval-mode=%s" % arg.split("=", 1)[1]) + auto = ("yolo" in flags or "disable-approval" in flags + or "approval-mode=never" in flags) + return {"auto_approve": auto, "flags": flags} + + +def launch_opt_out(cmd_argv): + """True when a muse argv explicitly opts out of watcher auto-approve. + + Bare argv (no approval flags) means undefined -> the watcher + answers. An explicit --approval-mode other than never (on-request, + untrusted, ...) declares human-in-the-loop intent -> hold. + """ + argv = cmd_argv or [] + for i, arg in enumerate(argv): + mode = None + if arg == "--approval-mode" and i + 1 < len(argv): + mode = argv[i + 1] + elif arg.startswith("--approval-mode="): + mode = arg.split("=", 1)[1] + if mode is not None and mode != "never": + return True + return False + + +def pane_muse_argv(socket_path, pane_id): + """Argv of the muse process in a pane (empty when not found).""" + try: + r = _tmux(socket_path, "list-panes", "-a", "-F", + "#{pane_id} #{pane_pid}", timeout=5) + if r.returncode != 0: + return [] + except Exception: + return [] + pane_pid = None + for line in r.stdout.split("\n"): + parts = line.split() + if len(parts) == 2 and parts[0] == pane_id: + try: + pane_pid = int(parts[1]) + except ValueError: + return [] + if pane_pid is None: + return [] + for cand in [pane_pid] + _child_pids(pane_pid): + argv = _cmdline(cand) + if argv and ("muse-bin" in argv[0] or "muse-code" in argv[0]): + return argv + return [] + + +# Minimum pane geometry for reliable approval rendering. Empirically +# derived: a 35x7 tile drops approval text the matcher needs, while +# 35x35/36x35/71x27 panes answer cleanly. Below either bound the pane +# is flagged squeezed (see `box runtime layout` / `spread`). +MIN_APPROVAL_WIDTH = 40 +MIN_APPROVAL_HEIGHT = 12 + + +def runtime_rows(socket_path): + """One row per pane: identity, approval posture, live state, watcher. + + Rows are JSON-serializable dicts for `box runtime list` and external + agents. Panes that vanish mid-scan are skipped, never fatal. + """ + rows = [] + try: + r = _tmux(socket_path, "list-panes", "-a", "-F", + "#{session_name}\t#{window_index}\t#{pane_id}\t" + "#{pane_current_command}\t#{pane_pid}\t" + "#{pane_width}\t#{pane_height}", timeout=10) + if r.returncode != 0: + return rows + except Exception: + return rows + for line in r.stdout.split("\n"): + parts = line.split("\t") + if len(parts) != 7: + continue + session, window, pane_id, cmd, pid, width, height = parts + try: + pane_pid = int(pid) + except ValueError: + pane_pid = None + try: + pane_width, pane_height = int(width), int(height) + except ValueError: + pane_width = pane_height = None + squeezed = (pane_width is not None and pane_height is not None + and (pane_width < MIN_APPROVAL_WIDTH + or pane_height < MIN_APPROVAL_HEIGHT)) + is_muse = "muse-bin" in cmd or "muse-code" in cmd + posture = {"auto_approve": None, "flags": []} + if is_muse and pane_pid: + candidates = [pane_pid] + _child_pids(pane_pid) + for cand in candidates: + argv = _cmdline(cand) + if argv and ("muse-bin" in argv[0] + or "muse-code" in argv[0]): + posture = muse_approval_flags(argv) + break + text = capture_pane(socket_path, pane_id) + if text is None: + continue + st = runtime_state(text) + match = st["match"] or {} + watcher_pid = is_running(socket_path, pane_id) + rows.append({ + "socket": socket_path, "session": session, + "window": window, "pane": pane_id, "cmd": cmd, + "pid": pane_pid, "is_muse": is_muse, + "width": pane_width, "height": pane_height, + "squeezed": squeezed, + "auto_approve": posture["auto_approve"], + "approval_flags": posture["flags"], + "state": st["state"], + "prompt_kind": match.get("kind"), + "prompt_key": match.get("key"), + "watcher_alive": watcher_pid is not None, + "watcher_pid": watcher_pid, + }) + return rows + + +def spread_targets(rows): + """Rows that `box runtime spread` would break into own windows. + + A row is a target when it runs a coding runtime (muse today; other + harnesses follow the registry) and its geometry is squeezed below + the reliable-approval minimums. Pure function of rows for testing. + """ + return [r for r in rows + if r.get("is_muse") and r.get("squeezed")] + + +def all_runtime_rows(sockets=None): + """runtime_rows across every known socket that exists.""" + rows = [] + for sock in sockets or KNOWN_SOCKETS: + if not os.path.exists(sock): + continue + rows.extend(runtime_rows(sock)) + return rows + + +def pane_state(socket_path, pane_id): + """Single-pane runtime row, or {"error": ...} when not found.""" + for row in runtime_rows(socket_path): + if row["pane"] == pane_id: + return row + return {"error": "no_such_pane", "socket": socket_path, + "pane": pane_id} + + +class WatcherState: + """Tracks prompt stability and once-per-prompt answering.""" + + def __init__(self): + self.pending_sig = None + self.stable_count = 0 + self.answered_sigs = {} + self.answer_times = [] + self.last_capped_sig = None + + def _prune_times(self, now): + cutoff = now - 3600 + self.answer_times = [t for t in self.answer_times if t >= cutoff] + + def cap_reached(self, now): + self._prune_times(now) + return len(self.answer_times) >= MAX_ANSWERS_PER_HOUR + + def observe(self, match, now): + """Feed one poll's match (or None). Returns "answer" when the prompt + is stable, unanswered, and under the hourly cap.""" + if match is None: + self.pending_sig = None + self.stable_count = 0 + return "none" + sig = match["sig"] + if sig in self.answered_sigs: + if now - self.answered_sigs[sig] < ANSWERED_TTL_SECONDS: + return "none" + # Expired: an identical prompt still (or again) present long + # after its answer is a stuck dialog, not a duplicate -- allow + # one recovery answer instead of wedging forever. + del self.answered_sigs[sig] + if sig == self.pending_sig: + self.stable_count += 1 + else: + self.pending_sig = sig + self.stable_count = 1 + if self.stable_count < STABILITY_POLLS: + return "wait" + if self.cap_reached(now): + return "capped" + return "answer" + + def record_answer(self, sig, now): + self.answered_sigs[sig] = now + # Bound memory: drop expired entries, then oldest-first past the cap. + cutoff = now - ANSWERED_TTL_SECONDS + for s in [s for s, t in self.answered_sigs.items() if t < cutoff]: + del self.answered_sigs[s] + if len(self.answered_sigs) > 200: + for s in sorted(self.answered_sigs, + key=lambda s: self.answered_sigs[s])[:-100]: + del self.answered_sigs[s] + self.answer_times.append(now) + self.pending_sig = None + self.stable_count = 0 + + +def _tmux(socket_path, *args, timeout=5): + cmd = ["tmux", "-S", socket_path] + list(args) + return subprocess.run(cmd, capture_output=True, text=True, timeout=timeout) + + +def pane_exists(socket_path, pane_id): + try: + r = _tmux(socket_path, "list-panes", "-a", "-F", "#{pane_id}", timeout=5) + if r.returncode != 0: + return False + return pane_id in r.stdout.split() + except Exception: + return False + + +def capture_pane(socket_path, pane_id, history=80): + try: + r = _tmux(socket_path, "capture-pane", "-p", "-t", pane_id, "-S", "-%d" % history, + timeout=5) + if r.returncode != 0: + return None + return r.stdout + except Exception: + return None + + +def send_answer(socket_path, pane_id, letter="A", enter=True): + """Type the choice key (+ Enter unless enter=False). True on success.""" + try: + r1 = _tmux(socket_path, "send-keys", "-t", pane_id, letter, timeout=5) + if r1.returncode != 0: + return False + if not enter: + return True + r2 = _tmux(socket_path, "send-keys", "-t", pane_id, "Enter", timeout=5) + return r2.returncode == 0 + except Exception: + return False + + +def _poll_once(socket_path, pane_id, state, log, dry_run=False): + """Run one poll iteration. Returns an outcome string ("gone" tells the + caller to exit). Logs only state transitions, never per-poll spam.""" + if not pane_exists(socket_path, pane_id): + return "gone" + text = capture_pane(socket_path, pane_id) + if text is None: + log.log("warn", "capture failed, continuing") + return "capture-failed" + match = find_choice_prompt(text) + now = time.time() + verdict = state.observe(match, now) + if verdict == "wait" and state.stable_count == 1: + log.log("info", "prompt seen", kind=match["kind"], key=match["key"], + sig=match["sig"], text=(match["cue"] or "")[:200]) + return "seen" + if verdict == "answer": + log.log("info", "prompt stable, answering", kind=match["kind"], + key=match["key"], sig=match["sig"]) + # Re-verify the prompt is still present before typing. + confirm = capture_pane(socket_path, pane_id) + cmatch = find_choice_prompt(confirm) if confirm else None + if cmatch is None or cmatch["sig"] != match["sig"]: + log.log("info", "prompt vanished before answer, skipping", + sig=match["sig"]) + state.pending_sig = None + state.stable_count = 0 + return "vanished" + if launch_opt_out(pane_muse_argv(socket_path, pane_id)): + log.log("info", "held: pane opted out via launch flags", + sig=match["sig"], kind=match["kind"]) + state.record_answer(match["sig"], now) + return "held" + key = match["key"] + if dry_run: + log.log("info", "dry-run would answer %s" % key, + sig=match["sig"], kind=match["kind"], key=key, + options=match["options"], cue=match["cue"]) + state.record_answer(match["sig"], now) + return "dry-answered" + ok = send_answer(socket_path, pane_id, key, + enter=match.get("enter", True)) + state.record_answer(match["sig"], now) + log.log("info" if ok else "error", "answered %s" % key, + sig=match["sig"], ok=ok, kind=match["kind"], key=key, + options=match["options"], cue=match["cue"]) + audit("muse-choice-answered", + name="%s:%s" % (os.path.basename(socket_path), pane_id), + extra={"socket": socket_path, "pane": pane_id, + "sig": match["sig"], "ok": ok, "kind": match["kind"], + "key": key, "options": match["options"], + "cue": match["cue"]}) + return "answered" + if verdict == "capped": + if state.last_capped_sig != match["sig"]: + state.last_capped_sig = match["sig"] + log.log("warn", "hourly answer cap reached, holding", + sig=match["sig"], kind=match["kind"]) + return "capped" + if verdict == "none": + state.last_capped_sig = None + return "none" + return "waiting" + + +def watch_loop(socket_path, pane_id, dry_run=False): + """Main daemon loop. Returns only when the pane is gone or signalled.""" + log = WatcherLog(logfile_for(socket_path, pane_id)) + state = WatcherState() + log.log("info", "watcher started", socket=socket_path, pane=pane_id, + dry_run=dry_run, pid=os.getpid()) + polls = 0 + answers = 0 + last_heartbeat = time.time() + while True: + try: + outcome = _poll_once(socket_path, pane_id, state, log, + dry_run=dry_run) + if outcome == "gone": + log.log("info", "pane gone, exiting", polls=polls, + answers=answers) + return 0 + if outcome in ("answered", "dry-answered"): + answers += 1 + polls += 1 + if time.time() - last_heartbeat >= HEARTBEAT_SECONDS: + last_heartbeat = time.time() + log.log("info", "heartbeat", polls=polls, answers=answers, + pending=state.pending_sig) + except Exception as e: + try: + log.log("error", "poll iteration failed, continuing", + error="%r" % (e,)) + except Exception: + pass + time.sleep(POLL_INTERVAL) + + +def _pid_alive(pid): + try: + os.kill(pid, 0) + return True + except Exception: + return False + + +def _pid_is_watcher(pid): + """True if pid looks like our watcher (guards against pid reuse).""" + try: + with open("/proc/%d/cmdline" % pid, "rb") as f: + cmd = f.read().decode(errors="replace") + return "muse_choice_watcher" in cmd + except Exception: + return False + + +def _watch_procs(proc_root="/proc"): + """Live processes running this module's foreground `watch` argv. + + Returns [{"pid", "socket", "pane"}]. Exact argv match only -- it can + never misidentify another program, so callers may act on the result. + """ + out = [] + try: + pids = os.listdir(proc_root) + except OSError: + return out + me = os.getpid() + for pid in pids: + if not pid.isdigit() or int(pid) == me: + continue + try: + with open(os.path.join(proc_root, pid, "cmdline"), "rb") as f: + parts = f.read().split(b"\0") + except (OSError, IOError): + continue + try: + args = [p.decode(errors="replace") for p in parts if p] + except Exception: + continue + if len(args) < 6: + continue + if not args[1].endswith("muse_choice_watcher.py") or args[2] != "watch": + continue + try: + sock = args[args.index("--socket") + 1] + pane = args[args.index("--pane") + 1] + except (ValueError, IndexError): + continue + out.append({"pid": int(pid), "socket": sock, "pane": pane}) + return out + + +def is_running(socket_path, pane_id): + pidfile = pidfile_for(socket_path, pane_id) + try: + with open(pidfile) as f: + pid = int(f.read().strip()) + except Exception: + return None + if _pid_alive(pid) and _pid_is_watcher(pid): + return pid + try: + os.remove(pidfile) + except OSError: + pass + return None + + +def _daemonize(): + """Double-fork away from the controlling terminal (survives shell exit).""" + if os.fork() != 0: + os._exit(0) + os.setsid() + if os.fork() != 0: + os._exit(0) + devnull = os.open(os.devnull, os.O_RDWR) + os.dup2(devnull, 0) + os.dup2(devnull, 1) + os.dup2(devnull, 2) + if devnull > 2: + os.close(devnull) + + +def _claim_pidfile(pidfile): + """Claim a pidfile for this process. Returns False if another live + watcher owns it (concurrent box + timer starts must not pile up).""" + me = os.getpid() + try: + with open(pidfile) as f: + other = int(f.read().strip()) + except Exception: + other = None + if other and other != me and _pid_alive(other) and _pid_is_watcher(other): + return False + try: + with open(pidfile, "w") as f: + f.write(str(me)) + except Exception: + pass + return True + + +def start_watcher(socket_path, pane_id, dry_run=False): + pid = is_running(socket_path, pane_id) + if pid: + return {"ok": False, "status": "already_running", "pid": pid} + if not pane_exists(socket_path, pane_id): + return {"ok": False, "status": "no_such_pane"} + _daemonize() + # Child continues here with a clean identity: re-exec ourselves as a + # foreground `watch` process so /proc cmdline always names this module + # (liveness checks and ps output stay correct no matter who spawned us -- + # box, timer, or direct CLI). + args = [sys.executable, os.path.abspath(__file__), "watch", + "--socket", socket_path, "--pane", pane_id] + if dry_run: + args.append("--dry-run") + os.execvp(sys.executable, args) + os._exit(1) # unreachable; exec replaces the image + + +def stop_watcher(socket_path, pane_id, timeout=5): + pid = is_running(socket_path, pane_id) + if not pid: + return {"ok": True, "status": "not_running"} + try: + os.kill(pid, signal.SIGTERM) + except ProcessLookupError: + pass + deadline = time.time() + timeout + while time.time() < deadline and _pid_alive(pid): + time.sleep(0.1) + if _pid_alive(pid): + try: + os.kill(pid, signal.SIGKILL) + except ProcessLookupError: + pass + try: + os.remove(pidfile_for(socket_path, pane_id)) + except OSError: + pass + return {"ok": True, "status": "stopped", "pid": pid} + + +def muse_panes(socket_path): + """Pane ids on a socket whose current command looks like Muse.""" + try: + r = _tmux(socket_path, "list-panes", "-a", "-F", + "#{pane_id} #{pane_current_command}", timeout=10) + if r.returncode != 0: + return [] + except Exception: + return [] + out = [] + for line in r.stdout.split("\n"): + parts = line.strip().split(None, 1) + if len(parts) == 2 and "muse-bin" in parts[1]: + out.append(parts[0]) + return out + + +def _start_detached(socket_path, pane_id, dry_run=False): + """Fork off a watcher via start_watcher (which daemonizes further). + Returns True if a watcher is running for the pane afterwards.""" + pid = os.fork() + if pid == 0: + start_watcher(socket_path, pane_id, dry_run=dry_run) + os._exit(0) + os.waitpid(pid, 0) + time.sleep(0.2) + return is_running(socket_path, pane_id) is not None + + +def start_all(dry_run=False, sockets=None): + results = [] + for sock in sockets or KNOWN_SOCKETS: + if not os.path.exists(sock): + results.append({"socket": sock, "status": "no_socket"}) + continue + panes = muse_panes(sock) + if not panes: + results.append({"socket": sock, "status": "no_muse_panes"}) + for pane in panes: + if is_running(sock, pane): + results.append({"socket": sock, "pane": pane, + "status": "already_running"}) + continue + ok = _start_detached(sock, pane, dry_run=dry_run) + results.append({"socket": sock, "pane": pane, + "status": "started" if ok else "failed"}) + return results + + +def stop_all(): + results = [] + stopped_pids = set() + prefix = FILE_PREFIX + "-" + try: + names = os.listdir(STATE_DIR) + except OSError: + names = [] + for name in names: + if not name.startswith(prefix) or not name.endswith(".pid"): + continue + try: + with open(os.path.join(STATE_DIR, name)) as f: + pid = int(f.read().strip()) + except Exception: + continue + if _pid_alive(pid) and _pid_is_watcher(pid): + try: + os.kill(pid, signal.SIGTERM) + except ProcessLookupError: + pass + stopped_pids.add(pid) + results.append({"pidfile": name, "pid": pid, "status": "stopped"}) + try: + os.remove(os.path.join(STATE_DIR, name)) + except OSError: + pass + # "off means off": also stop exact-argv watchers running pidfile-less. + for proc in _watch_procs(): + if proc["pid"] in stopped_pids: + continue + try: + os.kill(proc["pid"], signal.SIGTERM) + except ProcessLookupError: + pass + results.append({"pidfile": None, "pid": proc["pid"], + "status": "stopped-orphan", "socket": proc["socket"], + "pane": proc["pane"]}) + return results + + +def reconcile(sockets=None): + """Enforce desired state: start missing watchers when enabled, stop all + when disabled, prune dead pidfiles. Audits only when it changed + something. Returns a summary dict.""" + desired = get_desired() + started, already, pruned, failed = [], [], [], [] + stopped = [] + if desired["enabled"]: + for sock in sockets or KNOWN_SOCKETS: + if not os.path.exists(sock): + continue + for pane in muse_panes(sock): + if is_running(sock, pane): + already.append("%s:%s" % (sock, pane)) + continue + if _start_detached(sock, pane, dry_run=desired["dry_run"]): + started.append("%s:%s" % (sock, pane)) + else: + failed.append("%s:%s" % (sock, pane)) + for row in status_all(): + if not row["alive"]: + try: + os.remove(os.path.join(STATE_DIR, row["pidfile"])) + pruned.append(row["pidfile"]) + except OSError: + pass + else: + stopped = ["%s:%s" % (r.get("pidfile"), r.get("pid")) for r in stop_all()] + if started or stopped or pruned or failed: + audit("muse-choice-reconciled", + extra={"started": started, "stopped": stopped, "pruned": pruned, + "failed": failed}) + return {"enabled": desired["enabled"], "dry_run": desired["dry_run"], + "started": started, "already": already, + "stopped": stopped, "pruned": pruned, "failed": failed} + + +def recent_answers(limit=10, scan_lines=3000): + """Most recent muse-choice-answered audit records (newest last).""" + try: + with open(CTL_LOG) as f: + lines = f.readlines()[-scan_lines:] + except OSError: + return [] + out = [] + for line in lines: + try: + rec = json.loads(line) + except Exception: + continue + if rec.get("action") == "muse-choice-answered": + out.append(rec) + return out[-limit:] + + +def status_all(): + rows = [] + prefix = FILE_PREFIX + "-" + try: + names = sorted(os.listdir(STATE_DIR)) + except OSError: + return rows + for name in names: + if not name.startswith(prefix) or not name.endswith(".pid"): + continue + pid = None + try: + with open(os.path.join(STATE_DIR, name)) as f: + pid = int(f.read().strip()) + except Exception: + pass + alive = bool(pid) and _pid_alive(pid) and _pid_is_watcher(pid) + logname = name[:-4] + ".log" + logsize = None + try: + logsize = os.path.getsize(os.path.join(STATE_DIR, logname)) + except OSError: + pass + rows.append({"pidfile": name, "pid": pid, "alive": alive, + "log": logname, "log_bytes": logsize}) + owned = set(r["pid"] for r in rows if r["alive"] and r["pid"]) + for proc in _watch_procs(): + if proc["pid"] in owned: + continue + logname = os.path.basename( + logfile_for(proc["socket"], proc["pane"])) + try: + logsize = os.path.getsize(os.path.join(STATE_DIR, logname)) + except OSError: + logsize = None + rows.append({"pidfile": None, "pid": proc["pid"], "alive": True, + "orphan": True, "socket": proc["socket"], + "pane": proc["pane"], "log": logname, + "log_bytes": logsize}) + return rows + + +def main(argv=None): + ap = argparse.ArgumentParser(description="Muse A/B/C choice watcher") + sub = ap.add_subparsers(dest="cmd", required=True) + + p = sub.add_parser("start", help="start watching one pane (daemonizes)") + p.add_argument("--socket", required=True) + p.add_argument("--pane", required=True) + p.add_argument("--dry-run", action="store_true", + help="log answers instead of sending keys") + + p = sub.add_parser("stop", help="stop one pane watcher") + p.add_argument("--socket", required=True) + p.add_argument("--pane", required=True) + + p = sub.add_parser("start-all", help="watch every Muse pane on known sockets") + p.add_argument("--dry-run", action="store_true") + p.add_argument("--socket", action="append", dest="sockets", default=None) + + sub.add_parser("stop-all", help="stop all watchers") + sub.add_parser("status", help="list watcher state files + liveness") + + p = sub.add_parser("on", help="enable auto-answers (desired state) + reconcile now") + p.add_argument("--dry-run", action="store_true", + help="log answers instead of sending keys") + p.add_argument("--by", default="cli", help="caller identity for audit") + + p = sub.add_parser("off", help="disable auto-answers (desired state) + stop all now") + p.add_argument("--by", default="cli", help="caller identity for audit") + + p = sub.add_parser("reconcile", help="enforce desired state (for timer)") + p.add_argument("--socket", action="append", dest="sockets", default=None) + + p = sub.add_parser("watch", help="run one pane loop in foreground (daemon target)") + p.add_argument("--socket", required=True) + p.add_argument("--pane", required=True) + p.add_argument("--dry-run", action="store_true") + + p = sub.add_parser("match", help="print JSON verdict for text on stdin") + p.add_argument("--tail-window", type=int, default=TAIL_WINDOW) + + p = sub.add_parser("logs", help="tail one pane watcher log") + p.add_argument("--socket", required=True) + p.add_argument("--pane", required=True) + p.add_argument("-n", type=int, default=20) + + p = sub.add_parser("state", help="print JSON runtime state for one pane") + p.add_argument("--socket", required=True) + p.add_argument("--pane", required=True) + + args = ap.parse_args(argv) + if args.cmd == "start": + existing = is_running(args.socket, args.pane) + if existing: + print(json.dumps({"ok": True, "status": "already_running", + "pid": existing, + "log": logfile_for(args.socket, args.pane)})) + return 0 + # start_watcher never returns in the daemon child; the parent fork + # dance happens inside, so wrap: fork here, child calls start. + pid = os.fork() + if pid == 0: + start_watcher(args.socket, args.pane, dry_run=args.dry_run) + os._exit(0) + _, status = os.waitpid(pid, 0) + time.sleep(0.3) + running = is_running(args.socket, args.pane) + print(json.dumps({"ok": running is not None, "pid": running, + "log": logfile_for(args.socket, args.pane)})) + return 0 if running else 1 + if args.cmd == "stop": + print(json.dumps(stop_watcher(args.socket, args.pane))) + return 0 + if args.cmd == "start-all": + print(json.dumps(start_all(dry_run=args.dry_run, sockets=args.sockets), indent=1)) + return 0 + if args.cmd == "stop-all": + print(json.dumps(stop_all(), indent=1)) + return 0 + if args.cmd == "status": + print(json.dumps({"desired": get_desired(), "watchers": status_all()}, + indent=1)) + return 0 + if args.cmd == "on": + state = set_enabled(True, dry_run=args.dry_run, by=args.by) + res = reconcile() + print(json.dumps({"desired": state, "reconcile": res}, indent=1)) + return 0 + if args.cmd == "off": + state = set_enabled(False, by=args.by) + stopped = stop_all() + print(json.dumps({"desired": state, "stopped": stopped}, indent=1)) + return 0 + if args.cmd == "reconcile": + print(json.dumps(reconcile(sockets=args.sockets), indent=1)) + return 0 + if args.cmd == "watch": + # Refuse to pile on: an identical watcher may be alive with a lost + # pidfile (e.g. /tmp cleaned under it). Exact argv match, no guessing. + for proc in _watch_procs(): + if proc["socket"] == args.socket and proc["pane"] == args.pane: + return 3 + pidfile = pidfile_for(args.socket, args.pane) + if not _claim_pidfile(pidfile): + return 3 + try: + return watch_loop(args.socket, args.pane, dry_run=args.dry_run) + finally: + try: + with open(pidfile) as f: + owner = f.read().strip() + except Exception: + owner = "" + if owner == str(os.getpid()): + try: + os.remove(pidfile) + except OSError: + pass + if args.cmd == "match": + text = sys.stdin.read() + print(json.dumps(find_choice_prompt(text, tail_window=args.tail_window), indent=1)) + return 0 + if args.cmd == "logs": + path = logfile_for(args.socket, args.pane) + try: + with open(path) as f: + lines = f.readlines() + except OSError as e: + print("no log: %s (%s)" % (path, e)) + return 1 + for line in lines[-args.n:]: + print(line.rstrip("\n")) + return 0 + if args.cmd == "state": + print(json.dumps(pane_state(args.socket, args.pane), indent=1)) + return 0 + return 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/docs/BOX-API-READ-HTTPS.md b/docs/BOX-API-READ-HTTPS.md new file mode 100644 index 0000000..67b1155 --- /dev/null +++ b/docs/BOX-API-READ-HTTPS.md @@ -0,0 +1,152 @@ +# Box Read-Only Lookups over HTTPS (Agent Access, No SSH) + +> **Box is the main surface.** All operator work goes through Box (box.muse-dev.online). The web UI, `box` CLI, and agents share the same API endpoints. No UI-only powers. + +**Date:** 2026-10-06 +**Status:** bl side implemented; VM board REST below is specified, not yet implemented +**Scope:** read-only lookups only (fleet, threads, unread, dm log). Mutations stay on existing paths. + +## 1. Problem + +Agents in containers reach box over a 2-hop SSH chain (container β†’ VM β†’ bl). +SSH toggles lapse and every agent needs the full chain configured. Agents need +the daily read lookups β€” "latest from each agent" β€” over HTTPS with no secrets +on the wire. + +## 2. What exists now (bl side, implemented) + +Two agent HTTPS paths already serve reads; both use the same signature auth +(`ssh-keygen -Y sign`, namespace per server, Β±300s clock skew, nonce replay +cache) and per-agent principals from `dm-signers/allowed_signers`: + +| Path | Server | Client | Auth namespace | +|---|---|---|---| +| Named ops (works today) | `bin/exec-constrained.py` via `https://exec.muse-dev.online/exec` | `bin/exec-sign.sh ''` or `bin/box-relay.sh` (served at `GET /box`) | `exec-constrained` | +| Typed REST (specified below) | VM board `/srv/board/server.py` | any HTTPS client | `box-api` | + +New named ops (this change, all `side_effecting: false`, all in `DEFAULT_PERMS` +so any valid fleet signer may call them): + +- `fleet.unread` `{"agent"?}` β†’ `box-ctl.py unread [--agent X]` +- `dm.log` `{"limit"?, "agent"?}` (limit 1..100, default 20) β†’ `box-ctl.py dm-log [limit] [--agent X]` + +Already present and unchanged: `health.check` (fleet status), `thread.list`, +`thread.view`, `dm.read`, `chat.messages`. + +New `box-relay.sh` client commands (this change): + +```bash +box unread [] # fleet unread/activity counts +box dm log [] [--agent ] +``` + +New `box-ctl.py` backend verbs (this change; also callable over the board's +existing SSH bridge until the board speaks REST): + +```bash +box-ctl.py unread [--agent ] +box-ctl.py dm-log [limit] [--agent ] # back-compat: bare [limit] unchanged +``` + +Also fixed: `box lookup unread` / `muse unread` previously always failed with +"Unknown lookup target 'unread'" (`_lookup_unreads` was never wired into +`cmd_lookup`); it now works and supports `--json`. + +## 3. VM board REST (to implement on the VM) + +Base: `https://box.muse-dev.online/api/box`. All endpoints require +agent-signature auth (Β§4) or the existing `ops_session` cookie (humans). + +```text +GET /api/box/fleet exists today; keep behavior +GET /api/box/threads?agent=X backend: box-ctl.py thread-list --agent X (pass JSON through) +GET /api/box/unread?agent=X backend: box-ctl.py unread [--agent X] +GET /api/box/dm/log?limit=N&agent=X + backend: box-ctl.py dm-log [N] [--agent X] +``` + +### Agent scoping (server-enforced) + +- Verified identity `operator-X` or `X` (X in `muse,pip,646,opm,def,dev`) + may only read slices for X. The board MUST pass `--agent X` to box-ctl and + MUST NOT accept a different `agent=` query value from that identity. +- `ops_session` (human PIN login) may omit `agent=` and read the full fleet. +- Unknown/expired signatures β†’ `401`. Authenticated but out-of-scope β†’ `403`. + +### Response schemas (bl verbs pass through unchanged) + +`GET /api/box/unread`: + +```json +{"ok": true, "nodes": [ + {"node": "muse", "unread": 2, "approval_pending": false, + "title": "muse (2)", "thread": "abc123-uuid-or-null"} +]} +``` + +`GET /api/box/dm/log` (agent filter matches entries from OR to the agent): + +```json +{"ok": true, + "entries": [{"type": "sent", "id": "bdf7beb6", "agent": "opm", + "to": "pip", "target": "pip tasks", + "ts": "2026-10-06T05:56:01.328629+00:00"}], + "dms": ["... same array, legacy key ..."]} +``` + +`GET /api/box/fleet`: existing `{"ok": true, "fleet": [...]}` shape, unchanged. + +Errors follow `docs/BOX-API-DESIGN-DMS.md` Β§3.1 (`{"ok": false, "code", "error"}`). + +## 4. Agent-signature auth for the REST endpoints + +Same identity primitive as signed DMs and `exec-constrained.py`; a signature +is not a secret, so agents can sign without handling credentials. + +1. Client builds the canonical string (LF-separated, no trailing newline): + + ```text + {METHOD}\n{PATH}\n{SORTED_QUERY}\n{TS}\n{NONCE} + ``` + + - `METHOD`: `GET`; `PATH`: e.g. `/api/box/dm/log`; `SORTED_QUERY`: raw + query string sorted by key (`agent=opm&limit=5`), empty string when none. + - `TS`: unix epoch seconds; `NONCE`: 16–128 hex chars, single use. +2. Client signs it: `ssh-keygen -Y sign -f -n box-api`. +3. Client sends headers (armor is base64-encoded to stay header-safe): + + ```text + X-Box-Identity: operator-646 + X-Box-Timestamp: 1728... + X-Box-Nonce: + X-Box-Signature: + ``` + +4. Server recomputes the canonical string from the received request, base64- + decodes the signature, and runs `ssh-keygen -Y verify -f allowed_signers + -I -n box-api -s ` with the canonical string on stdin. + Accept only if: verify exit 0, `|now-TS| ≀ 300`, nonce unseen (cache β‰₯600s). + Signers file is synced from bl `dm-signers/allowed_signers`. + +Example: + +```bash +TS=$(date +%s); NONCE=$(python3 -c "import secrets; print(secrets.token_hex(16))") +CANON=$(printf 'GET\n/api/box/dm/log\nagent=opm&limit=5\n%s\n%s' "$TS" "$NONCE") +SIG=$(printf '%s' "$CANON" | ssh-keygen -Y sign -f ~/.ssh/id_frontdoor -n box-api \ + | base64 -w0) +curl -s 'https://box.muse-dev.online/api/box/dm/log?agent=opm&limit=5' \ + -H "X-Box-Identity: operator-646" -H "X-Box-Timestamp: $TS" \ + -H "X-Box-Nonce: $NONCE" -H "X-Box-Signature: $SIG" +``` + +## 5. Rollout notes + +- `exec-constrained.py` reads `OPS` at startup: restart the service after + deploying for `fleet.unread` / `dm.log` to appear in `GET /ops`. +- `box-relay.sh` is served from bl (`GET /box`); agents re-fetch to get + `unread` / `dm log`. +- Until the VM board implements Β§3, agents use the named-ops path (Β§2), + which needs no SSH today. +- Non-goals: write endpoints (`dm.send` etc. stay on the ops path for now), + PIN/human flows (unchanged), secret handling (no secrets cross the wire). diff --git a/docs/BOX-APPROVALS-HTTPS.md b/docs/BOX-APPROVALS-HTTPS.md new file mode 100644 index 0000000..9619547 --- /dev/null +++ b/docs/BOX-APPROVALS-HTTPS.md @@ -0,0 +1,92 @@ +# Box Approvals over HTTPS (No SSH) + +> **Box is the main surface.** All operator work goes through Box (box.muse-dev.online). The web UI, `box` CLI, and agents share the same API endpoints. No UI-only powers. + +**Date:** 2026-10-06 +**Status:** implemented on bl (`exec-constrained.py` + `box-relay.sh`; +`box-ctl.py` verbs pre-existed, plus fast node validation and +`quality-validate` branches; `approvals.py` untouched) +**Scope:** approval visibility (check) + governed decisions (deny, auto, +one-shot allow). Persistent/forced allow (`--always`/`--force`) stays +SSH-only. + +## 1. Why + +Agents blocked on browser approvals needed SSH to see fleet approval +state, deny a bad prompt, auto-resolve trusted prompts, or allow a +known-good one. All of this now rides the agent HTTPS path +(`https://exec.muse-dev.online/exec`, signature or Bearer [REDACTED], named-op +allowlist, audit log). + +## 2. New ops + +| Op | Args | Backend | Access | +|---|---|---|---| +| `approval.check` | `{node?}` (default fleet) | `box-ctl.py approval-check` | read-only, in `DEFAULT_PERMS` | +| `approval.deny` | `{node!, message!, allow_main_chat?}` | `box-ctl.py approval-deny` | known-identities-only | +| `approval.auto` | `{node?}` (default fleet) | `box-ctl.py approval-auto` | known-identities-only | +| `approval.allow` | `{node!, message!, allow_main_chat?}` | `box-ctl.py approval-allow` (one-shot) | known-identities-only | + +New `box-relay.sh` client commands: + +```bash +box approvals check [node] +box approvals allow --message [--allow-main-chat] +box approvals deny --message [--allow-main-chat] +box approvals auto [node] +``` + +Raw op calls (signature auth, no token): + +```bash +exec-sign.sh approval.check '{}' +exec-sign.sh approval.check '{"node": "646"}' +exec-sign.sh approval.allow '{"node": "646", "message": "trusted deploy script"}' +exec-sign.sh approval.auto '{"node": "opm"}' +``` + +## 3. What the decisions do + +- `approval.allow` clicks Allow **once** on the node's active prompt. + There is deliberately no remote `--always` (persistent site allow) + or `--force`. +- `approval.deny` clicks Deny on the node's active prompt. +- `approval.auto` scans (fleet or one node) and allows only TRUSTED + non-key prompts. Key/passkey approvals are never auto-approved; + they need an explicit allow/deny, which notifies the waiting agent. +- All three are audited with identity + op + node. + +## 4. Safety notes (same posture as existing ops) + +- **Attribution is mandatory.** The allow/deny flows DM the waiting + agent, historically with an `[operator]` prefix. A remote caller is + an agent, not the operator β€” so the exec layer **requires** a + non-empty `message` (≀2000 chars, no controls) on both allow and + deny. (`box-ctl.py` still permits omitting `--message` for SSH + callers; the HTTPS layer is the narrower gate.) +- **One-shot only.** The allow argv never carries `--always` or + `--force`; the validators reject those keys. Persistence stays an + SSH-side decision. +- **Sidechat-first.** `allow_main_chat` defaults to false; Main Chat + delivery needs the explicit flag, same as `notify`/`dm.ack`. +- **Fixed argv, validated values.** Nodes must be fleet members (fast + `BAD_NODE` before any CDP probe β€” `approval-check`/`approval-auto` + gained the same node check allow/deny already had); unknown arg + keys rejected. +- **Timeouts.** Check 180s (fleet CDP scan), auto 300s (scan plus one + click per trusted prompt), allow/deny 120s. +- **Retry-safe reads.** `approval-check` / `approval-list` joined + `IDEMPOTENT_ACTIONS`; all six approval verbs have + `quality-validate` dry-run branches. + +## 5. Rollout notes + +- Restart `exec-constrained.py` after deploy for the 4 new ops to + appear in `GET /ops` (repo total becomes 83). +- `box-relay.sh` is served from bl (`GET /box`); agents re-fetch to + get the `approvals` group. +- While fleet browsers crash-loop, decision ops fail honestly + (`APPROVAL_FAILED` / CDP errors) instead of hanging; check still + reports per-node state including `UNREACHABLE`. +- Non-goals: deletes, `main-loop` enable/disable, policy writes, + swarm kill/prune β€” future expansions, same pattern. diff --git a/docs/BOX-DEV-HTTPS.md b/docs/BOX-DEV-HTTPS.md new file mode 100644 index 0000000..8d7e8ad --- /dev/null +++ b/docs/BOX-DEV-HTTPS.md @@ -0,0 +1,80 @@ +# Box Dev + Comms over HTTPS (No SSH) + +> **Box is the main surface.** All operator work goes through Box (box.muse-dev.online). The web UI, `box` CLI, and agents share the same API endpoints. No UI-only powers. + +**Date:** 2026-10-06 +**Status:** implemented on bl (`exec-constrained.py` + `box-ctl.py` + `box-relay.sh`) +**Scope:** git visibility, test runs, notify, work-order acks. Read-only lookups +live in `docs/BOX-API-READ-HTTPS.md`. + +## 1. Why + +Agents developing the box must inspect the tree, run the suite, nudge peers, +and acknowledge work orders without SSH. All of this now rides the existing +agent HTTPS path (`https://exec.muse-dev.online/exec`, signature or bearer +auth, named-op allowlist, audit log) β€” no new trust model. + +## 2. New ops + +| Op | Args | Backend | Write? | Who | +|---|---|---|---|---| +| `git.status` | `{}` | `box-ctl.py git-status` | no | any valid signer | +| `git.diff` | `{path?, stat?}` | `box-ctl.py git-diff [--stat] [--path p]` | no | any valid signer | +| `git.log` | `{limit?, path?}` (1..50, default 10) | `box-ctl.py git-log` | no | any valid signer | +| `tests.run` | `{test?, filter?}` (`tests.` or full suite; `filter` is unittest `-k`) | `box-ctl.py tests-run [module] [--filter p]` | yes (executes) | known identities only | +| `notify.send` | `{agent, message≀1000, sidechat?, sender?}` | `box-ctl.py notify ...` | yes (sends DM) | known identities only | +| `dm.ack` | `{id, to, sender!, sidechat?, allow_main_chat?}` | `box-ctl.py ack ...` | yes (sends DM) | known identities only | + +New `box-ctl.py` verbs: `git-status`, `git-diff`, `git-log`, `tests-run`, +`ack` (all in `USAGE`, `quality-validate`, and β€” for the git reads β€” +`IDEMPOTENT_ACTIONS`). + +New `box-relay.sh` client commands: + +```bash +box git status +box git diff [--stat] [--path ] # path also accepted positionally +box git log [] [--path ] +box tests run [tests.] [--filter ] # full suite (~2-3 min) when omitted; filter is -k +box notify [--sidechat ] [--sender ] +box dm ack --to --sender [--sidechat ] +``` + +Raw op call (signature auth, no token): + +```bash +exec-sign.sh git.log '{"limit": 5, "path": "bin/dm.py"}' +exec-sign.sh tests.run '{"test": "tests.test_box_dev_https"}' +exec-sign.sh dm.ack '{"id": "bdf7beb6", "to": "pip", "sender": "opm"}' +``` + +## 3. Safety notes (same posture as existing ops) + +- **Fixed argv, validated values.** Clients influence only whitelisted argument + values. Git paths must be repo-relative without `..` (plus symlink-escape + check in box-ctl); test modules must match `^tests\.[a-z0-9_]+$` and exist; + ack ids must be 6–64 hex; notify messages ≀1000 chars. +- **Caps.** `git diff` output capped at 64KB, status at 200 entries, test + output at 32KB tail; every capped response carries `truncated: true`. +- **Sidechat-first.** `notify.send` and `dm.ack` default to the recipient's + sidechat and never touch Main Chat unless `allow_main_chat` is set β€” + mirroring `box-ctl.py notify` and the WO dispatcher. +- **Attribution.** `sender` is caller-asserted (validated ∈ fleet agents), + same as the existing `dm.send` op; the HTTPS identity is recorded + separately in the exec audit log. `dm.ack` requires an explicit sender β€” + no silent default. +- **tests.run executes repo code** (whatever is in `tests/`), so it is + `side_effecting`, excluded from the read-only default permission subset, + and capped at a 600s timeout. Test failures report as + `{"ok": false, "returncode", "output"}` β€” the op itself succeeded. +- New `box-ctl.py` fail codes `GIT_ERROR` / `TESTS_ERROR` are registered in + `KNOWN_ERROR_CODES`, so `quality-check` stays at its baseline. + +## 4. Rollout notes + +- Restart `exec-constrained.py` after deploy for the new ops to appear in + `GET /ops`. +- `box-relay.sh` is served from bl (`GET /box`); agents re-fetch to get + `git` / `tests` / `notify` / `dm ack`. +- Non-goals: git commit/push, service restarts for box itself, live + streaming tails β€” future expansions, same pattern. diff --git a/docs/BOX-JOBS-HTTPS.md b/docs/BOX-JOBS-HTTPS.md new file mode 100644 index 0000000..0e7ed9b --- /dev/null +++ b/docs/BOX-JOBS-HTTPS.md @@ -0,0 +1,88 @@ +# Box Job Lifecycle over HTTPS (No SSH) + +> **Box is the main surface.** All operator work goes through Box (box.muse-dev.online). The web UI, `box` CLI, and agents share the same API endpoints. No UI-only powers. + +**Date:** 2026-10-06 +**Status:** implemented on bl (`exec-constrained.py` + `box-relay.sh`; +`box-ctl.py` verbs pre-existed) +**Scope:** safe job-lifecycle mutations. Deletes are deliberately NOT +exposed (`job-delete`, `timer-delete` stay SSH/operator-only). + +## 1. Why + +Agents own scheduled automation but could only run jobs (`cron.run`) or read +them (`box.exec`). Creating, updating, triggering, chaining, previewing, and +pausing jobs needed SSH. All of this now rides the agent HTTPS path +(`https://exec.muse-dev.online/exec`, signature or bearer auth, named-op +allowlist, audit log). + +## 2. New ops + +| Op | Args | Backend | Write? | Who | +|---|---|---|---|---| +| `job.put` | `{name, definition}` | `box-ctl.py job-put` (definition on stdin) | yes (writes + commits) | known identities only | +| `job.trigger` | `{name}` | `box-ctl.py job-trigger` | yes (dispatches now) | known identities only | +| `job.chain` | `{from, to, on_failure?}` | `box-ctl.py job-chain` | yes (writes + commits) | known identities only | +| `job.next` | `{job_id, success?}` | `box-ctl.py job-next` | no (dry-run) | any valid signer | +| `cron.timer_stop` | `{name}` | `box-ctl.py timer-stop` | yes (systemd) | known identities only | +| `cron.timer_disable` | `{name}` | `box-ctl.py timer-disable` | yes (systemd) | known identities only | + +New `box-relay.sh` client commands: + +```bash +box cron put '' # create or update (see Β§3) +box cron trigger # dispatch now (audited JSON) +box cron chain [--on-failure] # wire chain_next +box cron next [--success|--fail] # dry-run: what dispatches next +box timer stop # pause schedule (keeps unit) +box timer disable # pause schedule (disables unit) +``` + +Raw op call (signature auth, no token): + +```bash +exec-sign.sh job.next '{"job_id": "heartbeat-20200101-000000-deadbeef"}' +exec-sign.sh job.chain '{"from": "ops-audit-step2", "to": "ops-audit-step3"}' +``` + +Notes: + +- `job.trigger` vs existing `job.run`: `job.run` shells straight to + `job-dispatch.py` and relays raw output; `job.trigger` goes through + `box-ctl.py` (existence check, 300s bound, audit trail, JSON contract). + Prefer `job.trigger` for agent-driven dispatches. +- `job.next` job ids look like `-YYYYMMDD-HHMMSS-<8hex>`; with no + `success` flag the box infers it from the last recorded result. + +## 3. Job definition shape (`job.put`) + +The full schema is enforced by `box-ctl.py validate_job` (single copy); +required fields: `name` (must match the argv name), `schedule` (`manual` +or convertible cron), `agent` (fleet member), `prompt_template` (1–4000 +chars, no protocol literals, known `{placeholders}` only). `timeout` +60–3600s, `on_failure` policy, optional `chain_next` (must exist) and +`sidechat` / `dm_target` routing. `box-ctl.py` writes `jobs/.json` +and commits (`Add/Update job via box-ctl`). + +## 4. Safety notes (same posture as existing ops) + +- **Fixed argv, validated values.** Job names match + `^[a-z0-9][a-z0-9-]{0,63}$`; trigger/chain/timer ops require the job + file to exist; chain rejects self-links and cycles; timer ops require + the unit to exist. All checks run before any side effect. +- **Stdin plumbing.** `job.put` is the first op to pipe a request body to + `box-ctl.py` stdin (the raw definition, not the envelope); the routing + lives in one helper (`_stdin_body`) covered by unit tests. +- **No deletes, no kills.** `job-delete` / `timer-delete` are reachable + only over the operator SSH path, by explicit scope decision. +- **Audited.** Every execution records identity + op + args hash; + `box-ctl.py` additionally audits each mutation with its target. + +## 5. Rollout notes + +- Restart `exec-constrained.py` after deploy for the new ops to appear in + `GET /ops`. +- `box-relay.sh` is served from bl (`GET /box`); agents re-fetch to get + the `cron put|trigger|chain|next` and `timer stop|disable` commands. +- Non-goals: deletes, `job-result` ingestion, `strat`/`loop` writes, + approvals β€” future expansions, same pattern. diff --git a/docs/BOX-LOOP-HTTPS.md b/docs/BOX-LOOP-HTTPS.md new file mode 100644 index 0000000..46dee8a --- /dev/null +++ b/docs/BOX-LOOP-HTTPS.md @@ -0,0 +1,85 @@ +# Box Loop + Strategy Writes over HTTPS (No SSH) + +> **Box is the main surface.** All operator work goes through Box (box.muse-dev.online). The web UI, `box` CLI, and agents share the same API endpoints. No UI-only powers. + +**Date:** 2026-10-06 +**Status:** implemented on bl (`exec-constrained.py` + `box-relay.sh`; +`box-ctl.py` verbs pre-existed, plus a vars-name validator fix) +**Scope:** safe loop/strategy/variable mutations. Reads (`loop-status`, +`loop-health`, `loop-breaks`, `strat-get`, `vars-get`, ...) already ride +`box.exec` or typed read ops and are unchanged here. + +## 1. Why + +Agents watching loop health could see breaks but needed SSH to remediate +them, resolve stale followups, tune strategy overrides, or undo variable +changes. All of this now rides the agent HTTPS path +(`https://exec.muse-dev.online/exec`, signature or bearer auth, named-op +allowlist, audit log). + +## 2. New ops (all known-identities-only, none in the read-only subset) + +| Op | Args | Backend | +|---|---|---| +| `loop.remediate` | `{dry_run?}` (default false) | `box-ctl.py loop-remediate [--dry-run]` | +| `loop.resolve` | `{dm_id, note?}` | `box-ctl.py loop-resolve` | +| `strat.set` | `{type!, subtype?, agent?, track?, priority?, timeout_s?, nudges?, escalate?}` | `box-ctl.py strat-set` (payload as argv JSON) | +| `strat.reset` | `{type!, subtype?, agent?}` | `box-ctl.py strat-reset` | +| `vars.reset` | `{name}` | `box-ctl.py vars-reset` (restore default) | +| `vars.rollback` | `{name, revision?}` (int step or timestamp) | `box-ctl.py vars-rollback` | + +New `box-relay.sh` client commands: + +```bash +box loop remediate [--dry-run] +box loop resolve [note...] +box strat set [--subtype S] [--agent A] [--track true|false] [--priority p] [--timeout N] [--nudges N] [--escalate E] +box strat reset [subtype] [--agent ] +box vars reset +box vars rollback [revision] +``` + +Raw op call (signature auth, no token): + +```bash +exec-sign.sh loop.remediate '{"dry_run": true}' +exec-sign.sh strat.set '{"type": "job", "priority": "important", "nudges": 3}' +``` + +## 3. What remediate does (non-dry) + +`loop.remediate` delegates to `gravity.remediate_breaks`: resolves +followups already answered in logs, re-arms expired followups with nudges +remaining (runs the followup sweeper once), auto-allows TRUSTED (non-key) +browser approvals, and on hard breaks appends a `hard_break_alert` to +job-log plus one DM to opm. Use `{"dry_run": true}` first to preview the +`remediated` / `escalated` lists with zero side effects. + +## 4. Safety notes (same posture as existing ops) + +- **Fixed argv, validated values.** Strategy types are a strict enum + (`wake|job|siphon|manual|health|heartbeat`) β€” the backend silently maps + typos to MANUAL, so the op rejects them instead. Priorities are a + strict enum; timeouts/nudges must be integers (backend clamps); + dm ids must be 6–64 hex; variable names mirror the engine identifier + rule. All checks run before any side effect. +- **Stdin-free.** Unlike `job.put`, `strat.set` passes its JSON payload + as an argv token (`strat-set [JSON]`), so no new stdin plumbing + was needed. +- **Vars-name validator fix.** `quality-validate` for all five vars + verbs used the job-name regex (`^[a-z0-9-]{1,64}$`), rejecting every + real variable name (`max_nudge_count`, ...). They now share + `qv_var_name` (`^[A-Za-z0-9_.-]{1,64}$`), mirroring the + `exec-constrained.py` rule. +- **Audited.** Every execution records identity + op + args hash; + `box-ctl.py` additionally audits each mutation with its target. + +## 5. Rollout notes + +- Restart `exec-constrained.py` after deploy for the new ops to appear in + `GET /ops`. +- `box-relay.sh` is served from bl (`GET /box`); agents re-fetch to get + the `loop` / `strat` groups and `vars reset|rollback`. +- Non-goals: approvals, deletes, `main-loop` enable/disable β€” future + expansions, same pattern. (md drive files shipped separately; see + BOX-MD-HTTPS.md.) diff --git a/docs/BOX-MD-HTTPS.md b/docs/BOX-MD-HTTPS.md new file mode 100644 index 0000000..6f9f1be --- /dev/null +++ b/docs/BOX-MD-HTTPS.md @@ -0,0 +1,109 @@ +# Box Md Drive Files over HTTPS (No SSH) + +> **Box is the main surface.** All operator work goes through Box (box.muse-dev.online). The web UI, `box` CLI, and agents share the same API endpoints. No UI-only powers. + +**Date:** 2026-10-06 +**Status:** implemented on bl (`exec-constrained.py` + `box-relay.sh`; +`box-ctl.py` verbs pre-existed, plus traversal hardening, output caps, +`--stdin` content plumbing, and hyphenated amend/append/pull aliases) +**Scope:** md reads (audit/list/read/diff) + governed writes +(amend/append/pull/inject-drive/sync-all). Raw container writes +(`md-write`) stay SSH-only by design. + +## 1. Why + +Agents shaping fleet behavior could see drive scores but needed SSH to +read an agent's `SOUL.md`, diff it against the canonical template, or +push updated operator files. All of this now rides the agent HTTPS path +(`https://exec.muse-dev.online/exec`, signature or Bearer [REDACTED], named-op +allowlist, audit log). + +## 2. New ops + +Reads (all `side_effecting: false`, all in `DEFAULT_PERMS`): + +| Op | Args | Backend | +|---|---|---| +| `md.audit` | `{accounts?}` (default all) | `box-ctl.py md-audit [accounts...]` | +| `md.list` | `{account!, path?}` | `box-ctl.py md-list` (capped, see Β§4) | +| `md.read` | `{account!, filename!}` | `box-ctl.py md-read` (capped, see Β§4) | +| `md.diff` | `{account!, filename!}` (shared template only) | `box-ctl.py md-diff` (capped, see Β§4) | + +Governed writes (all known-identities-only, none in the read-only subset): + +| Op | Args | Backend | +|---|---|---| +| `md.pull` | `{account!, filename!}` (shared template only) | `box-ctl.py md-pull` | +| `md.inject_drive` | `{account!}` | `box-ctl.py md-inject-drive` | +| `md.sync_all` | `{}` | `box-ctl.py md-sync-all` | +| `md.amend` | `{filename!, content!, author?, reason?}` | `box-ctl.py md-amend --stdin` (content on stdin) | +| `md.append` | `{filename!, text!, author?, section?}` | `box-ctl.py md-append --stdin` (text on stdin) | + +New `box-relay.sh` client commands: + +```bash +box md audit [accounts...] +box md list [path] +box md read +box md diff +box md pull +box md inject-drive +box md sync-all +box md amend (--content |--file ) [--author ] [--reason ] +box md append (--content |--file ) [--author ] [--section
] +``` + +Raw op calls (signature auth, no token): + +```bash +exec-sign.sh md.audit '{"accounts": ["646", "opm"]}' +exec-sign.sh md.read '{"account": "646", "filename": "SOUL.md"}' +exec-sign.sh md.diff '{"account": "pip", "filename": "HEARTBEAT.md"}' +exec-sign.sh md.append '{"filename": "AGENTS.md", "text": "lesson ...", "author": "646"}' +``` + +## 3. What the governed writes do + +- `md.amend` rewrites a `shared/operators/` template after the + drive-safety checks (HEARTBEAT checklist not gutted, + PROACTIVE_PREFERENCES not blanked, SOUL not reverted to stock), + then git-commits it. Full-file content rides stdin (up to 256KB). +- `md.append` appends a timestamped, attributed note (optional section) + via the same validated + committed path (up to 64KB). +- `md.pull` / `md.inject_drive` / `md.sync_all` push canonical + templates *out* to containers; no agent-supplied content crosses. + Injection always overwrites (AGENTS.md preserves remote `## Lessons`). + +## 4. Safety notes (same posture as existing ops) + +- **Traversal hardening (single-copy in `agent_md.py`).** Account, + filename, and list-path validation now lives in `agent_md.py` + (`MDValidationError`, raised before any gateway call or write); + `box-ctl.py` maps it to `BAD_NAME`, and exec ops + quality-validate + mirror the same shapes. Previously `md-read 646 ../x` reached the + gateway and `md amend ../../x` could escape `shared/operators/`. + Template flows (diff/amend/append/pull) additionally require one of + the 8 known template names. +- **Fixed argv, validated values.** Unknown arg keys rejected; author / + reason / section are control-char-free with length caps; amend + content must be non-empty. +- **Caps with `truncated` flags.** Reads cap at 64KB, diffs at 64KB, + listings at 200 entries β€” same convention as git/tests verbs. +- **No raw `md-write` op.** Arbitrary content-to-container stays + SSH-only; remote writes go through the validated template flows. +- **Timeouts.** Audit 300s, sync-all 600s, single-file ops 120s. +- **Audited.** Every execution records identity + op; `box-ctl.py` + additionally audits each verb with its target. +- **Retry-safe reads.** `md-audit` / `md-list` / `md-read` / `md-diff` + joined `IDEMPOTENT_ACTIONS`; all ten md verbs have + `quality-validate` dry-run branches. + +## 5. Rollout notes + +- Restart `exec-constrained.py` after deploy for the 9 new ops to + appear in `GET /ops` (repo total becomes 79). +- `box-relay.sh` is served from bl (`GET /box`); agents re-fetch to + get the `md` group. +- Non-goals: deletes, `main-loop` enable/disable, policy writes, + swarm kill/prune β€” future expansions, same pattern. (Approvals + shipped separately; see BOX-APPROVALS-HTTPS.md.) diff --git a/docs/MUSE-CHOICES-POLICY.md b/docs/MUSE-CHOICES-POLICY.md new file mode 100644 index 0000000..a5e59ff --- /dev/null +++ b/docs/MUSE-CHOICES-POLICY.md @@ -0,0 +1,103 @@ +# Muse-Choices Deny/Escalate Policy β€” DECISION RECORD (Final) + +Topic: add deny/escalate decisions to the `muse-choices` auto-approve daemon +(`bin/muse_choice_watcher.py`), which today only approves (top choice per +prompt kind). Interviewed 2026-10-06/07 per grill contract; accepted +verbatim below, which flipped this record from Draft to Final. + +## Standing constraints (settled by user) + +- All prompts must resolve: no stuck states are acceptable in any outcome. +- The full-auto top-choice flow must always exist as a path. +- Model review of choices is a FUTURE layer. Deferred out of this interview. + +## Settled decisions + +- D0 (helper form): a checked-in repo rules file informs decisions. + Source: user's structured answerquared 2026-10-06 ("Repo rules file + (Recommended)" for "which helper should inform approve/deny/hold + decisions"). Rationale recorded at selection time: deterministic, + versioned, sub-second, unit-testable; no agent round-trip latency. + +## Settled during interview + +- D0b (approve path needs no helper): straightforward prompts resolve + locally with top-choice keys (the five matcher kinds, already + implemented and live). Helpers (D0 rules file) govern deny/hold + judgments only. Source: user direction 2026-10-06 ("the watcher + itself should be able to input 1"; "always have the flow for full + auto just top choice"). +- D1 (rule match dimensions): command pattern first, plus kind and + text pattern. Source: user selected option 1, 2026-10-06. + Rationale: the observed risk lives in the `$ command` of approval + dialogs; kind/text add precision around it. +- D2 (deny mechanics): deny exists only for permission kinds -- + `muse-approval` dialogs receive `2` + Enter, `y/n` prompts receive + `n` + Enter. Question kinds (interview, letter, numbered) always + resolve top-choice and are never denied. Source: user selected + option 1, 2026-10-06. Rationale: deny is only meaningful where a + permission is refused; questions stay total. +- D3 (hold mechanics): hold leaves the dialog untouched, suppresses + auto-answer, raises a HELD entry in `box muse-choices status` plus + an audit record; the operator resolves via a box command, otherwise + a SHORT window expires back to top-choice approve. Source: user + selected option 1 with "short window", 2026-10-06. Exact duration + proposed below (2 minutes, tunable); accepted or amended with the + scope text in D5. + +- D4 (unmatched default): approve top-choice, exactly today's + behavior. Source: user selected option 1, 2026-10-06. Rationale: + follows from the standing constraints; rules carve out only + deny/hold exceptions, so an empty rules file changes nothing. + +## Open questions (unresolved) + +None. All interview questions resolved and the scope accepted. + +## Scope contract (ACCEPTED) + +Artifact boundary, IN: + +- `docs/MUSE-CHOICES-POLICY.md`: this record (Draft -> Final on acceptance). +- New `muse-choices-rules.json` at repo root (beside + `keepalive-config.json`): the checked-in deny/hold rules. +- `bin/muse_choice_watcher.py`: rule evaluation, deny/hold paths, HELD + state with short-window expiry, resolve plumbing. +- `bin/super-cli.py`: `box muse-choices resolve` command + HELD display + in status. +- `tests/test_muse_choice_watcher.py`: rule eval, per-kind deny keys, + hold/suppress/expiry, resolve flow. + +Artifact boundary, OUT (rejected or deferred, each needs its own +interview to re-enter): + +- Model review of choices (deferred future stage). +- Peer-agent consultation (rejected in D0). +- New matcher shapes (matcher suite's lane). +- `box runtime` work (adjacent lane, untouched). +- Timer cadence / daemon supervision changes. + +Done means (all observable): + +- [ ] This record marked Final with the acceptance quoted. +- [ ] Rules file loads; empty rules == today's behavior exactly. +- [ ] Deny sends `2`+Enter / `n`+Enter per D2: unit tests + one live + scratch proof per permission kind. +- [ ] Hold suppresses + shows HELD + resolves via box + expires to + approve: unit tests + one live scratch proof of hold and one of + expiry. +- [ ] Audit records for deny/hold/resolve/expire in `box-ctl.jsonl`. +- [ ] Full suite green; fleet reloaded; desired state left as found. + +Acceptance (quoted verbatim, chat, 2026-10-07T00:19:57Z): "ACCEPT". +Accepted as written, including the 2-minute tunable hold window. Per the +grill scope contract, later work outside the IN boundary needs explicit +owner approval or its own follow-up interview; "go" authorizes only this +boundary. No owning issue exists in this workflow, so this record is the +lane-coordination evidence. + +## Non-goals (accepted with the scope) + +- Model-based review of choices (deferred future layer). +- Peer-agent consultation over sidechat (rejected in favor of D0). +- Changes to approval matching shapes (covered by the matcher test suite). diff --git a/docs/SUPERVISION-SPEC.md b/docs/SUPERVISION-SPEC.md new file mode 100644 index 0000000..9eefa73 --- /dev/null +++ b/docs/SUPERVISION-SPEC.md @@ -0,0 +1,81 @@ +# Supervision Scope Contract + +Status: **Draft** β€” decisions below are unsettled until marked otherwise. +Only explicit user acceptance moves this document (or any decision) to Final. + +## Goal + +Every fleet node stays alive and truthfully reported: browsers supervised, +relays supervised, dead nodes recovered or loudly paged, and `box` status +honest from any shell (including PID/net-blind sandboxed shells). + +## Non-goals (proposed) + +- Agent lifecycle/onboarding stages (provision, auth, OTP, invite redeem). +- Work completion (job dispatch, followups, harvester, completion auditor). +- Loop-health scoring and drive repair. + +## Supervisors (observed, all installed 2026-10-06) + +| Supervisor | Cadence | Coverage | Decides | +|---|---|---|---| +| chromebox-watchdog@\.timer Γ—6 | 2 min | all registry nodes (def/dev timers installed 18:32Z) | browser+egress per node; tunnel restart, chrome relaunch | +| cdp-relay-watchdog.timer | 5 min | registry-driven (`watched_nodes()`) | relay veth IP + connectivity; relay restart | +| agent-health.timer (user) | 5 min | registry-driven | API check per node; kill+restart with 2-strike rule + circuit breaker (3 futile β†’ open 30 min) | +| ensure-node-supervision.sh | on node-up / `--all` | new + drifted nodes | NODES.md row + chromebox timer install | +| host_evidence fallback | on `box` read | registry (relay) + installed timers (browser) | effective status when live probes are blind | + +## Decisions + +(D1..D7 below β€” all UNRESOLVED unless marked.) + +### D1. Contract boundary: which supervisors are in scope β€” SETTLED (recommended accepted) + +IN: chromebox-watchdog Γ—6, cdp-relay-watchdog, agent-health + circuit +breaker, ensure-node-supervision feed, host_evidence fallback. +OUT: onboarding pipeline lifecycle, completion auditor, loop-health +(each keeps its own owner and interviews separately). + +### D2. Kill-path precedence (chromebox-watchdog vs agent-health) β€” UNRESOLVED + +Both can kill a browser today; only time guards (<2 min) de-conflict them. + +### D3. Egress-down fall-through (relaunch chrome after failed tunnel restart?) β€” UNRESOLVED + +Observed 19:12Z: tunnel restart failed, watchdog relaunched chrome 3Γ— +anyway (one FAILED page). Browser was never the problem. + +### D4. Circuit-breaker thresholds (3 futile / 30 min cooldown) β€” UNRESOLVED + +Current values unvalidated against real recurrence intervals. + +### D5. Coverage source of truth β€” UNRESOLVED + +Registry-only vs registry+installed-timers for browser verdicts. + +### D6. Concurrent-edit protocol for shared supervision files β€” UNRESOLVED + +Two agents editing super-cli.py / watchdogs / runbook; one clobber +(18:03Z) and one unattributed commit (c9143a5) already occurred. + +### D7. Done means β€” UNRESOLVED + +Proposed checklist: timers on all 6 firing silent; relay/agent-health +loops registry-driven with tests; ensure hook live; this doc Final. + +## Risks + +- Egress-down pages read as browser failures (D3). +- Uncommitted supervisor work can be clobbered by a concurrent editor (D6). +- c9143a5 contains unattributed peer hunks (host_evidence `_covered_nodes`, + fleet-status test updates) β€” needs an amend-or-leave decision. + +## Validation + +- `box fleet status` truthful from blind shells (live-verified 6/6 ACTIVE). +- Focused suites green (supervision, fleet, agent-health, watchdog coverage). +- Timer firing proven via journal, not config presence. + +## Unresolved items + +D2–D7 unresolved. D1 settled. diff --git a/systemd/muse-choices-reconcile.service b/systemd/muse-choices-reconcile.service new file mode 100644 index 0000000..7d02106 --- /dev/null +++ b/systemd/muse-choices-reconcile.service @@ -0,0 +1,16 @@ +[Unit] +Description=NetVM Muse Choice Watcher Reconcile +After=network.target + +[Service] +Type=oneshot +ExecStart=/usr/bin/python3 /home/super/Projects/NetVM/bin/muse_choice_watcher.py reconcile +WorkingDirectory=/home/super/Projects/NetVM +StandardOutput=journal +StandardError=journal +# Reconcile spawns long-lived per-pane daemons. Default KillMode= +# control-group would SIGTERM/SIGKILL them (setsid cannot escape a +# cgroup) the moment this oneshot service exits -- leaving every +# timer-started pane uncovered (observed live). Only signal the main +# process so spawned watchers survive service exit. +KillMode=process diff --git a/systemd/muse-choices-reconcile.timer b/systemd/muse-choices-reconcile.timer new file mode 100644 index 0000000..5aac8c6 --- /dev/null +++ b/systemd/muse-choices-reconcile.timer @@ -0,0 +1,10 @@ +[Unit] +Description=Run Muse Choice Watcher reconcile every minute + +[Timer] +OnBootSec=1min +OnUnitActiveSec=1min +Persistent=true + +[Install] +WantedBy=timers.target diff --git a/tests/test_approvals.py b/tests/test_approvals.py index 1e7464b..4bcb403 100644 --- a/tests/test_approvals.py +++ b/tests/test_approvals.py @@ -12,8 +12,10 @@ import json import os import subprocess import sys +import tempfile import unittest from pathlib import Path +from unittest import mock REPO_ROOT = Path("/home/super/Projects/NetVM") BIN_DIR = REPO_ROOT / "bin" @@ -138,8 +140,12 @@ class TestMuseChatApiConnection(unittest.TestCase): def test_muse_chat_api_approvals_command(self): cmd = [sys.executable, str(BIN_DIR / "muse-chat-api.py"), "--account", "pip", "approvals"] r = subprocess.run(cmd, capture_output=True, text=True) - self.assertEqual(r.returncode, 0) - self.assertIn("No pending approvals", r.stdout) + self.assertIn(r.returncode, (0, 2)) + if r.returncode == 0: + self.assertIn("No pending approvals", r.stdout) + else: + self.assertIn("APPROVAL_NEEDED", r.stdout) + class TestApprovalsReplySafety(unittest.TestCase): @@ -242,6 +248,106 @@ class TestKeyApprovalsAndPasskey(unittest.TestCase): self.assertEqual(deny_data.get("decision"), "deny") +class _FakeWS: + """Scripted stand-in for a CDP websocket (no network).""" + + def __init__(self, recvs): + self._recvs = list(recvs) + self.sent_ids = [] + + def send(self, msg): + self.sent_ids.append(json.loads(msg)["id"]) + + def recv(self): + if not self._recvs: + raise Exception("recv queue exhausted") + item = self._recvs.pop(0) + if callable(item): + return item(self) + return item + + def close(self): + pass + + +def _echo_last_value(value): + def _recv(ws): + return json.dumps({"id": ws.sent_ids[-1], + "result": {"result": {"value": value}}}) + return _recv + + +class TestInspectRobustness(unittest.TestCase): + """Regression tests for intermittent approval failures.""" + + def test_cdp_request_ids_unique(self): + # Millisecond-clock ids collide for rapid successive evaluates; + # a stale buffered response can then be misattributed to the + # wrong call (e.g. verify-after-click reads the click result). + # Frozen clock makes the old collision deterministic. + with mock.patch("approvals.time.time", return_value=1728000000.123): + ws = _FakeWS([_echo_last_value("a"), _echo_last_value("b")]) + self.assertEqual(approvals.cdp_evaluate(ws, "1+1"), "a") + self.assertEqual(approvals.cdp_evaluate(ws, "2+2"), "b") + self.assertNotEqual(ws.sent_ids[0], ws.sent_ids[1]) + + def test_cdp_skips_stale_ids(self): + stale = json.dumps({"id": 999999999, + "result": {"result": {"value": "stale"}}}) + ws = _FakeWS([stale, _echo_last_value("fresh")]) + self.assertEqual(approvals.cdp_evaluate(ws, "1+1"), "fresh") + + def test_unreachable_returns_full_shape(self): + with mock.patch.object(approvals, "get_node_pages", + side_effect=ConnectionError("nope")), \ + mock.patch.object(approvals, "check_node_key_request", + return_value=None): + res = approvals.inspect_node_approvals("pip") + self.assertEqual(res["status"], "UNREACHABLE") + self.assertFalse(res["has_pending"]) + for key in ("node", "title", "buttons", "is_trusted", + "input_waits", "error"): + self.assertIn(key, res) + + def test_all_pages_failed_reports_error(self): + pages = [{"title": "t", "url": "u", "type": "page", + "webSocketDebuggerUrl": "ws://127.0.0.1:9/none"}] + + class _DeadWSModule: + @staticmethod + def create_connection(*a, **k): + raise ConnectionError("refused") + + with mock.patch.object(approvals, "get_node_pages", + return_value=pages), \ + mock.patch.object(approvals, "websocket", _DeadWSModule()), \ + mock.patch.object(approvals, "check_node_key_request", + return_value=None): + res = approvals.inspect_node_approvals("pip") + # Per-page CDP failures must surface as ERROR, never as a + # false CLEAR that hides pending approvals. + self.assertEqual(res["status"], "ERROR") + self.assertFalse(res["has_pending"]) + self.assertIn("error", res) + + def test_state_saves_roundtrip_without_leftovers(self): + # Guards the atomic-save refactor (tmp + replace): correct + # content and no stray temp files left behind. + with tempfile.TemporaryDirectory() as td: + rp = Path(td) / "resp.json" + with mock.patch.object(approvals, "RESPONDED_WAITS_FILE", rp): + approvals.save_responded_waits({"pip": {"t": "x"}}) + self.assertEqual(json.loads(rp.read_text()), + {"pip": {"t": "x"}}) + fp = Path(td) / "seen.json" + with mock.patch.object(approvals, "FIRST_SEEN_WAITS_FILE", fp): + approvals.save_first_seen_waits({"pip": {"t": "x"}}) + self.assertEqual(json.loads(fp.read_text()), + {"pip": {"t": "x"}}) + self.assertEqual(sorted(p.name for p in Path(td).iterdir()), + ["resp.json", "seen.json"]) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_bl_relay_gate.py b/tests/test_bl_relay_gate.py new file mode 100644 index 0000000..1c50b4d --- /dev/null +++ b/tests/test_bl_relay_gate.py @@ -0,0 +1,79 @@ +"""The bl-side #lobby relay must stay opt-in. + +fleet-alert-check.sh invokes bin/fleet-alert-relay.sh only when +FLEET_BL_RELAY=1: the container-side hook is the live pager, and running +both double-posts every alert (2026-10-06). These tests run a copy of the +checker with stubbed-out fleet commands and assert the relay stub is (not) +invoked. +""" +import os +import shutil +import subprocess +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +CHECKER = REPO_ROOT / "bin" / "fleet-alert-check.sh" + + +class BlRelayGate(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + root = Path(self.tmp.name) + self.bindir = root / "bin" + self.bindir.mkdir() + # Copy the real checker so BIN-relative lookups hit our stubs. + shutil.copy(CHECKER, self.bindir / "fleet-alert-check.sh") + (self.bindir / "fleet-alert-relay.sh").write_text( + "#!/bin/bash\necho RELAY-RAN >> \"$CALLS\"\n") + (self.bindir / "fleet-alert-relay.sh").chmod(0o755) + (self.bindir / "netvm-registry.py").write_text( + "#!/usr/bin/env python3\n") # no nodes: all fleet loops skip + (self.bindir / "netvm-registry.py").chmod(0o755) + (self.bindir / "box-ctl.py").write_text("#!/usr/bin/env python3\n") + (self.bindir / "box-ctl.py").chmod(0o755) + fakebin = root / "fakebin" + fakebin.mkdir() + (fakebin / "sudo").write_text("#!/bin/bash\necho sudo-stub >&2\nexit 1\n") + (fakebin / "sudo").chmod(0o755) + self.state = root / "state" + self.state.mkdir() + self.calls = root / "calls.log" + self.env = dict(os.environ) + self.env["PATH"] = str(fakebin) + ":/usr/bin:/bin" + self.env["FLEET_ALERT_STATE_DIR"] = str(self.state) + self.env["CALLS"] = str(self.calls) + self.env.pop("FLEET_BL_RELAY", None) + + def run_checker(self, **extra): + env = dict(self.env) + env.update(extra) + # NOTE: appends 1-2 lines to the shared /tmp/fleet-alert-check.log + # (LOG path is hardcoded); same as any production timer run. + return subprocess.run( + ["bash", str(self.bindir / "fleet-alert-check.sh")], + capture_output=True, text=True, env=env, timeout=120) + + def relay_ran(self): + return self.calls.exists() and "RELAY-RAN" in self.calls.read_text() + + def test_relay_not_invoked_by_default(self): + r = self.run_checker() + self.assertEqual(r.returncode, 0, r.stderr[-2000:]) + self.assertFalse(self.relay_ran()) + + def test_relay_invoked_when_opted_in(self): + r = self.run_checker(FLEET_BL_RELAY="1") + self.assertEqual(r.returncode, 0, r.stderr[-2000:]) + self.assertTrue(self.relay_ran()) + + def test_dry_run_never_invokes_relay(self): + r = self.run_checker(FLEET_BL_RELAY="1", FLEET_ALERT_DRY_RUN="1") + self.assertEqual(r.returncode, 0, r.stderr[-2000:]) + self.assertFalse(self.relay_ran()) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_box_approvals_https.py b/tests/test_box_approvals_https.py new file mode 100644 index 0000000..f910da2 --- /dev/null +++ b/tests/test_box_approvals_https.py @@ -0,0 +1,218 @@ +"""Tests for approvals over HTTPS (no SSH). + +Covers the approvals expansion: + box-relay.sh (agent client) -> exec-constrained.py named ops + -> box-ctl.py backend verbs -> approvals.py (fleet CDP). + +Live execution is limited to validation-failure paths (BAD_NODE/BAD_ARGS, +which fail before any CDP probe) plus quality-validate dry-runs. No live +browser traffic and no live-socket round-trips here; instead we assert the +exact argv each op builds. In particular the allow build must never carry +--always/--force: remote allow is one-shot only. +""" +import importlib.util +import json +import subprocess +import sys +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +BOX_CTL = REPO_ROOT / "bin" / "box-ctl.py" +RELAY = REPO_ROOT / "bin" / "box-relay.sh" + + +def _load(name, relpath): + spec = importlib.util.spec_from_file_location(name, REPO_ROOT / relpath) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +exec_constrained = _load("exec_constrained_approvals", + "bin/exec-constrained.py") + + +def _box_ctl(*args): + return subprocess.run( + [sys.executable, str(BOX_CTL), *args], + capture_output=True, text=True, timeout=180) + + +class ExecApprovalOpsTests(unittest.TestCase): + def test_ops_registered_and_side_effecting(self): + spec = exec_constrained.OPS + self.assertIn("approval.check", spec) + self.assertFalse(spec["approval.check"]["side_effecting"]) + for op in ("approval.deny", "approval.auto", "approval.allow"): + self.assertIn(op, spec) + self.assertTrue(spec[op]["side_effecting"], op) + + def test_known_identities_only(self): + p = exec_constrained.permitted + self.assertTrue(p("some-unknown-identity", "approval.check")) + for op in ("approval.deny", "approval.auto", "approval.allow"): + self.assertFalse(p("some-unknown-identity", op), op) + self.assertTrue(p("operator-646", op), op) + self.assertFalse(p("exec-canary", "approval.check")) + + def test_check_validate(self): + v = exec_constrained.OPS["approval.check"]["validate"] + self.assertEqual(v({}), {"node": None}) + self.assertEqual(v({"node": None}), {"node": None}) + self.assertEqual(v({"node": "646"}), {"node": "646"}) + for bad in ({"node": "nope"}, {"node": "../x"}, + {"node": "646", "bogus": 1}): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v(bad) + + def test_deny_validate(self): + v = exec_constrained.OPS["approval.deny"]["validate"] + good = v({"node": "646", "message": "not trusted", + "allow_main_chat": True}) + self.assertEqual(good, {"node": "646", "message": "not trusted", + "allow_main_chat": True}) + self.assertFalse(v({"node": "646", + "message": "m"})["allow_main_chat"]) + # Multiline explanations are fine; other controls are not. + v({"node": "646", "message": "line1\nline2"}) + over = "x" * (exec_constrained.MAX_MESSAGE + 1) + for bad in ({"node": "646"}, + {"node": "646", "message": " "}, + {"node": "646", "message": over}, + {"node": "646", "message": "a\x07b"}, + {"node": "nope", "message": "m"}, + {"node": "646", "message": "m", + "allow_main_chat": "yes"}, + {"node": "646", "message": "m", "force": True}): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v(bad) + + def test_auto_validate(self): + v = exec_constrained.OPS["approval.auto"]["validate"] + self.assertEqual(v({}), {"node": None}) + self.assertEqual(v({"node": "opm"}), {"node": "opm"}) + with self.assertRaises(exec_constrained.OpError): + v({"node": "nope"}) + with self.assertRaises(exec_constrained.OpError): + v({"node": "646", "always": True}) + + def test_allow_validate(self): + v = exec_constrained.OPS["approval.allow"]["validate"] + good = v({"node": "646", "message": "trusted deploy script"}) + self.assertEqual(good["message"], "trusted deploy script") + self.assertFalse(good["allow_main_chat"]) + # Attribution is mandatory: the flow notifies the waiting agent, + # so a remote allow must carry its reason. + with self.assertRaises(exec_constrained.OpError): + v({"node": "646"}) + with self.assertRaises(exec_constrained.OpError): + v({"node": "646", "message": " "}) + with self.assertRaises(exec_constrained.OpError): + v({"node": "nope", "message": "m"}) + # No persistence/force remotely, not even as rejected keys. + with self.assertRaises(exec_constrained.OpError): + v({"node": "646", "message": "m", "always": True}) + with self.assertRaises(exec_constrained.OpError): + v({"node": "646", "message": "m", "force": True}) + + def test_build_argv_shapes(self): + ops = exec_constrained.OPS + ck = ops["approval.check"] + self.assertEqual(ck["build"]({"node": None})[-1], + "approval-check") + self.assertEqual(ck["build"]({"node": "646"})[-2:], + ["approval-check", "646"]) + de = ops["approval.deny"] + argv = de["build"]({"node": "646", "message": "m", + "allow_main_chat": False}) + self.assertEqual(argv[-4:], + ["approval-deny", "646", "--message", "m"]) + argv = de["build"]({"node": "646", "message": "m", + "allow_main_chat": True}) + self.assertEqual(argv[-1], "--allow-main-chat") + au = ops["approval.auto"] + self.assertEqual(au["build"]({"node": None})[-1], "approval-auto") + self.assertEqual(au["build"]({"node": "opm"})[-2:], + ["approval-auto", "opm"]) + al = ops["approval.allow"] + argv = al["build"]({"node": "646", "message": "m", + "allow_main_chat": False}) + self.assertEqual(argv[-4:], + ["approval-allow", "646", "--message", "m"]) + self.assertNotIn("--always", argv) + self.assertNotIn("--force", argv) + argv = al["build"]({"node": "646", "message": "m", + "allow_main_chat": True}) + self.assertEqual(argv[-1], "--allow-main-chat") + self.assertIsInstance(argv, list) + + +class BoxCtlApprovalTests(unittest.TestCase): + def test_rejects_unknown_node_before_cdp(self): + for args in (["approval-check", "badnode"], + ["approval-list", "badnode"], + ["approval-allow", "badnode", "--message", "m"], + ["approval-deny", "badnode", "--message", "m"], + ["approval-auto", "badnode"]): + r = _box_ctl(*args) + self.assertNotEqual(r.returncode, 0, args) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NODE", + args) + + def test_rejects_missing_node(self): + for args in (["approval-allow"], ["approval-deny"]): + r = _box_ctl(*args) + self.assertNotEqual(r.returncode, 0, args) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_ARGS", + args) + + def test_quality_validate_approval_verbs(self): + # Note: box-ctl leaves --message optional (SSH callers may rely on + # the flow default); the exec layer is the narrower gate and + # requires it. quality-validate mirrors box-ctl. + cases = [ + (["approval-check"], True), + (["approval-check", "646"], True), + (["approval-check", "nope"], False), + (["approval-check", "646", "opm"], False), + (["approval-list", "646"], True), + (["approval-allow", "646", "--message", "hi"], True), + (["approval-allow", "646"], True), + (["approval-allow"], False), + (["approval-allow", "nope", "--message", "x"], False), + (["approval-allow", "646", "--always", "--force", + "--message", "x"], True), + (["approval-approve", "646", "--message", "x"], True), + (["approval-deny", "646", "--message", "x"], True), + (["approval-deny", "646", "--message", "x", + "--allow-main-chat"], True), + (["approval-deny"], False), + (["approval-auto"], True), + (["approval-auto", "646"], True), + (["approval-auto", "nope"], False), + (["approval-auto", "a", "b"], False), + ] + for args, valid in cases: + r = _box_ctl("quality-validate", *args) + self.assertEqual(json.loads(r.stdout)["valid"], valid, args) + + +class BoxRelayApprovalTests(unittest.TestCase): + def test_relay_help_lists_approval_commands(self): + r = subprocess.run(["bash", str(RELAY), "help"], + capture_output=True, text=True, timeout=30) + self.assertEqual(r.returncode, 0, r.stderr) + for line in ("box approvals check", "box approvals allow", + "box approvals deny", "box approvals auto"): + self.assertIn(line, r.stdout) + + def test_relay_maps_approval_commands_to_ops(self): + text = RELAY.read_text() + for op in ('"approval.check"', '"approval.deny"', + '"approval.auto"', '"approval.allow"'): + self.assertIn(op, text) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_box_dev_https.py b/tests/test_box_dev_https.py new file mode 100644 index 0000000..e4179e9 --- /dev/null +++ b/tests/test_box_dev_https.py @@ -0,0 +1,345 @@ +"""Tests for no-SSH agent development and communication streams. + +Covers the second HTTPS expansion: + box-relay.sh (agent client) -> exec-constrained.py named ops + -> box-ctl.py backend verbs -> git / unittest / dm.py. + +Live-socket round-trips are intentionally NOT covered here (loopback TCP is +unavailable in some sandboxes); instead we assert the exact argv each op +builds and execute the fast, side-effect-free argv directly. Live sends +(notify/ack) are NEVER executed here: only their validation-failure paths. +""" +import importlib.util +import json +import os +import subprocess +import sys +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +BOX_CTL = REPO_ROOT / "bin" / "box-ctl.py" +RELAY = REPO_ROOT / "bin" / "box-relay.sh" + + +def _load(name, relpath): + spec = importlib.util.spec_from_file_location(name, REPO_ROOT / relpath) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +exec_constrained = _load("exec_constrained_dev", "bin/exec-constrained.py") + + +def _box_ctl(*args): + return subprocess.run( + [sys.executable, str(BOX_CTL), *args], + capture_output=True, text=True, timeout=120) + + +class ExecGitOpsTests(unittest.TestCase): + def test_ops_registered_and_read_only(self): + for op in ("git.status", "git.diff", "git.log"): + self.assertIn(op, exec_constrained.OPS) + self.assertFalse(exec_constrained.OPS[op]["side_effecting"]) + + def test_default_perms_include_git_ops(self): + self.assertTrue(exec_constrained.permitted("some-unknown-identity", "git.status")) + self.assertTrue(exec_constrained.permitted("some-unknown-identity", "git.diff")) + self.assertTrue(exec_constrained.permitted("some-unknown-identity", "git.log")) + self.assertFalse(exec_constrained.permitted("exec-canary", "git.status")) + + def test_git_status_validate(self): + v = exec_constrained.OPS["git.status"]["validate"] + self.assertEqual(v({}), {}) + with self.assertRaises(exec_constrained.OpError): + v({"bogus": 1}) + + def test_git_diff_validate(self): + v = exec_constrained.OPS["git.diff"]["validate"] + self.assertEqual(v({}), {"path": None, "stat": False}) + self.assertEqual(v({"path": "bin/dm.py", "stat": True}), + {"path": "bin/dm.py", "stat": True}) + for bad in ("../x", "/abs/path", "a\x00b", ""): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v({"path": bad}) + + def test_git_log_validate(self): + v = exec_constrained.OPS["git.log"]["validate"] + self.assertEqual(v({}), {"limit": 10, "path": None}) + self.assertEqual(v({"limit": 3})["limit"], 3) + with self.assertRaises(exec_constrained.OpError): + v({"limit": 0}) + with self.assertRaises(exec_constrained.OpError): + v({"limit": 51}) + with self.assertRaises(exec_constrained.OpError): + v({"path": "../x"}) + + def test_build_argv_shapes(self): + status = exec_constrained.OPS["git.status"] + self.assertEqual(status["build"]({})[-1], "git-status") + diff = exec_constrained.OPS["git.diff"] + self.assertEqual(diff["build"]({"path": None, "stat": False})[-1], "git-diff") + argv = diff["build"]({"path": "bin/dm.py", "stat": True}) + self.assertEqual(argv[-4:], ["git-diff", "--stat", "--path", "bin/dm.py"]) + log = exec_constrained.OPS["git.log"] + argv = log["build"]({"limit": 3, "path": None}) + self.assertEqual(argv[-3:], ["git-log", "--limit", "3"]) + self.assertIsInstance(argv, list) + + def test_git_status_built_argv_executes(self): + spec = exec_constrained.OPS["git.status"] + argv = spec["build"](spec["validate"]({})) + argv[0] = sys.executable # hermetic interpreter, same script + args + r = subprocess.run(argv, capture_output=True, text=True, timeout=60) + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertIn("branch", data) + self.assertIsInstance(data["changes"], list) + + +class ExecTestsRunTests(unittest.TestCase): + def test_registered_and_side_effecting(self): + self.assertIn("tests.run", exec_constrained.OPS) + self.assertTrue(exec_constrained.OPS["tests.run"]["side_effecting"]) + + def test_known_identities_only(self): + # Executes repo code: excluded from the read-only default subset. + self.assertFalse(exec_constrained.permitted("some-unknown-identity", "tests.run")) + self.assertTrue(exec_constrained.permitted("operator-646", "tests.run")) + + def test_validate(self): + v = exec_constrained.OPS["tests.run"]["validate"] + self.assertEqual(v({}), {"test": None, "filter": None}) + self.assertEqual(v({"test": "tests.test_box_read_https"}), + {"test": "tests.test_box_read_https", + "filter": None}) + self.assertEqual(v({"filter": "safepath"})["filter"], "safepath") + for bad in ("os", "tests..x", "tests/x", "tests.test-x", ""): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v({"test": bad}) + for bad in ("", "x" * 201, "a\nb"): + with self.assertRaises(exec_constrained.OpError, msg=repr(bad)): + v({"filter": bad}) + + def test_build_argv_shape(self): + b = exec_constrained.OPS["tests.run"]["build"] + self.assertEqual(b({"test": None, "filter": None})[-1], "tests-run") + argv = b({"test": "tests.test_box_read_https", "filter": None}) + self.assertEqual(argv[-2:], ["tests-run", "tests.test_box_read_https"]) + argv = b({"test": None, "filter": "safepath"}) + self.assertEqual(argv[-3:], ["tests-run", "--filter", "safepath"]) + + +class ExecCommsOpsTests(unittest.TestCase): + def test_registered_and_side_effecting(self): + for op in ("notify.send", "dm.ack"): + self.assertIn(op, exec_constrained.OPS) + self.assertTrue(exec_constrained.OPS[op]["side_effecting"]) + + def test_known_identities_only(self): + self.assertFalse(exec_constrained.permitted("some-unknown-identity", "notify.send")) + self.assertFalse(exec_constrained.permitted("some-unknown-identity", "dm.ack")) + self.assertTrue(exec_constrained.permitted("operator-646", "notify.send")) + self.assertTrue(exec_constrained.permitted("muse", "dm.ack")) + + def test_notify_send_validate(self): + v = exec_constrained.OPS["notify.send"]["validate"] + self.assertEqual(v({"agent": "pip", "message": "hi"}), + {"agent": "pip", "message": "hi", + "sidechat": None, "sender": None}) + with self.assertRaises(exec_constrained.OpError): + v({"agent": "nope", "message": "hi"}) + with self.assertRaises(exec_constrained.OpError): + v({"agent": "pip", "message": "x" * 1001}) + with self.assertRaises(exec_constrained.OpError): + v({"agent": "pip", "message": " "}) + with self.assertRaises(exec_constrained.OpError): + v({"agent": "pip", "message": "hi", "sidechat": "a" * 65}) + + def test_dm_ack_validate(self): + v = exec_constrained.OPS["dm.ack"]["validate"] + good = v({"id": "bdf7beb6", "to": "pip", "sender": "opm"}) + self.assertEqual(good["id"], "bdf7beb6") + self.assertFalse(good["allow_main_chat"]) + with self.assertRaises(exec_constrained.OpError): + v({"id": "xyz!", "to": "pip", "sender": "opm"}) + with self.assertRaises(exec_constrained.OpError): + v({"id": "bdf7beb6", "to": "nope", "sender": "opm"}) + with self.assertRaises(exec_constrained.OpError): + v({"id": "bdf7beb6", "to": "pip"}) # sender required + + def test_build_argv_shapes(self): + n = exec_constrained.OPS["notify.send"] + argv = n["build"]({"agent": "pip", "message": "hi", + "sidechat": None, "sender": None}) + self.assertEqual(argv[-3:], ["notify", "pip", "hi"]) + argv = n["build"]({"agent": "pip", "message": "hi", + "sidechat": "pip tasks", "sender": "opm"}) + self.assertIn("--sidechat", argv) + self.assertIn("--sender", argv) + a = exec_constrained.OPS["dm.ack"] + argv = a["build"]({"id": "bdf7beb6", "to": "pip", "sender": "opm", + "sidechat": None, "allow_main_chat": False}) + self.assertEqual(argv[-6:], + ["ack", "bdf7beb6", "--to", "pip", "--sender", "opm"]) + + +class BoxCtlGitTests(unittest.TestCase): + def test_git_status_live(self): + r = _box_ctl("git-status") + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertTrue(data["branch"]) + self.assertIsInstance(data["changes"], list) + + def test_git_log_live(self): + r = _box_ctl("git-log", "--limit", "2") + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertEqual(len(data["commits"]), 2) + self.assertIn("sha", data["commits"][0]) + self.assertIn("subject", data["commits"][0]) + + def test_git_diff_stat_live(self): + r = _box_ctl("git-diff", "--stat") + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertIn("diff", data) + + def test_rejects_bad_path_and_limit(self): + for bad in ("../x", "/abs/path"): + r = _box_ctl("git-diff", "--path", bad) + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl("git-log", "--limit", "0") + self.assertNotEqual(r.returncode, 0) + r = _box_ctl("git-log", "--limit", "51") + self.assertNotEqual(r.returncode, 0) + + def test_quality_validate_git_verbs(self): + for args in (["git-status"], ["git-diff", "--stat"], + ["git-diff", "--path", "bin/dm.py"], + ["git-log", "--limit", "5"]): + r = _box_ctl("quality-validate", *args) + self.assertTrue(json.loads(r.stdout)["valid"], args) + r = _box_ctl("quality-validate", "git-diff", "--path", "../x") + self.assertFalse(json.loads(r.stdout)["valid"]) + + +class BoxCtlTestsRunTests(unittest.TestCase): + def test_tests_run_single_module_live(self): + r = _box_ctl("tests-run", "tests.test_box_read_https") + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"], data.get("output", "")[-2000:]) + self.assertEqual(data["returncode"], 0) + + def test_tests_run_discovery_importable(self): + # Full discover must import every test module. An impossible -k + # filter runs zero tests in seconds while still importing all of + # them, deterministically guarding the discover argv (a `-t .` + # regresses to ImportError here). Never asserts suite success: + # outage-sensitive tests may be red independently. + r = _box_ctl("tests-run", "--filter", "zzz_no_match_zzz") + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertIn("Ran 0 tests", data.get("output", "")) + self.assertNotIn("Traceback", data.get("output", "")) + + def test_tests_run_survives_safepath_invoker(self): + # `python -m` drops CWD from sys.path under PYTHONSAFEPATH/-P; + # tests-run pins PYTHONPATH so it still resolves the tests package. + env = dict(os.environ) + env["PYTHONSAFEPATH"] = "1" + r = subprocess.run( + [sys.executable, str(BOX_CTL), "tests-run", + "tests.test_box_read_https"], + capture_output=True, text=True, timeout=120, env=env) + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"], data.get("output", "")[-2000:]) + + def test_rejects_bad_module(self): + r = _box_ctl("tests-run", "os") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl("tests-run", "tests.nonexistent_xyz") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "NOT_FOUND") + + def test_quality_validate_tests_run(self): + r = _box_ctl("quality-validate", "tests-run") + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "tests-run", "tests.test_box_read_https") + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "tests-run", "--filter", "safepath") + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "tests-run", "os") + self.assertFalse(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "tests-run", "--filter") + self.assertFalse(json.loads(r.stdout)["valid"], r.stdout) + + +class BoxCtlAckTests(unittest.TestCase): + # Validation-failure paths only: a live ack would send a real DM. + + def test_rejects_bad_id_and_agents(self): + r = _box_ctl("ack", "xyz!", "--to", "pip", "--sender", "opm") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl("ack", "bdf7beb6", "--to", "nope", "--sender", "opm") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + + def test_requires_sender(self): + r = _box_ctl("ack", "bdf7beb6", "--to", "pip") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_ARGS") + + def test_quality_validate_ack(self): + r = _box_ctl("quality-validate", "ack", "bdf7beb6", + "--to", "pip", "--sender", "opm") + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "ack", "xyz!", + "--to", "pip", "--sender", "opm") + self.assertFalse(json.loads(r.stdout)["valid"], r.stdout) + + +class BoxCtlNotifyValidationTests(unittest.TestCase): + # Failure paths only: act_notify validates before sending. + + def test_rejects_unknown_agent(self): + r = _box_ctl("notify", "nope", "hi") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + + def test_rejects_oversize_message(self): + r = _box_ctl("notify", "pip", "x" * 1001) + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "INVALID_JOB") + + +class BoxRelayDevTests(unittest.TestCase): + def test_relay_help_lists_dev_commands(self): + r = subprocess.run(["bash", str(RELAY), "help"], + capture_output=True, text=True, timeout=30) + self.assertEqual(r.returncode, 0, r.stderr) + for line in ("box git status", "box git diff", "box git log", + "box tests run", "box notify", "box dm ack"): + self.assertIn(line, r.stdout) + + def test_relay_maps_dev_commands_to_ops(self): + text = RELAY.read_text() + for op in ("git.status", "git.diff", "git.log", + '"tests.run"', '"notify.send"', '"dm.ack"'): + self.assertIn(op, text) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_box_jobs_https.py b/tests/test_box_jobs_https.py new file mode 100644 index 0000000..022f6e1 --- /dev/null +++ b/tests/test_box_jobs_https.py @@ -0,0 +1,246 @@ +"""Tests for job lifecycle over HTTPS (no SSH). + +Covers the job-lifecycle expansion: + box-relay.sh (agent client) -> exec-constrained.py named ops + -> box-ctl.py backend verbs -> jobs/*.json / systemd / job-dispatch. + +Safe mutations only: put/trigger/chain/stop/disable. Deletes are +deliberately NOT exposed. Live writes, triggers, and timer control are +NEVER executed here: only validation-failure paths (which fail before any +side effect) plus the read-only job-next dry-run. Live-socket round-trips +are intentionally NOT covered here; instead we assert the exact argv each +op builds and execute the fast, side-effect-free argv directly. +""" +import importlib.util +import json +import subprocess +import sys +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +BOX_CTL = REPO_ROOT / "bin" / "box-ctl.py" +RELAY = REPO_ROOT / "bin" / "box-relay.sh" + +# An existing job used for existence-gated validation (read-only). +EXISTING_JOB = "heartbeat" +# Well-formed names that must not exist (validation-failure paths only). +MISSING_JOB = "definitely-no-such-job-xyz" +MISSING_ID = "definitely-no-such-job-xyz-20200101-000000-deadbeef" + + +def _load(name, relpath): + spec = importlib.util.spec_from_file_location(name, REPO_ROOT / relpath) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +exec_constrained = _load("exec_constrained_jobs", "bin/exec-constrained.py") + + +def _box_ctl(*args, stdin=None): + return subprocess.run( + [sys.executable, str(BOX_CTL), *args], + input=stdin, capture_output=True, text=True, timeout=120) + + +def _job_def(name, **over): + d = {"name": name, "description": "unit test job", + "schedule": "manual", "agent": "opm", + "prompt_template": "test prompt {job_id}", "timeout": 300} + d.update(over) + return d + + +class ExecJobOpsTests(unittest.TestCase): + def test_ops_registered_with_side_effect_flags(self): + spec = exec_constrained.OPS + for op in ("job.put", "job.trigger", "job.chain", + "cron.timer_stop", "cron.timer_disable"): + self.assertIn(op, spec) + self.assertTrue(spec[op]["side_effecting"]) + self.assertIn("job.next", spec) + self.assertFalse(spec["job.next"]["side_effecting"]) + + def test_no_delete_ops_exposed(self): + names = set(exec_constrained.OPS) + self.assertNotIn("job.delete", names) + self.assertNotIn("cron.timer_delete", names) + + def test_permissions(self): + p = exec_constrained.permitted + self.assertTrue(p("some-unknown-identity", "job.next")) + for op in ("job.put", "job.trigger", "job.chain", + "cron.timer_stop", "cron.timer_disable"): + self.assertFalse(p("some-unknown-identity", op)) + self.assertTrue(p("operator-646", op)) + self.assertFalse(p("exec-canary", "job.next")) + + def test_job_put_validate(self): + v = exec_constrained.OPS["job.put"]["validate"] + good = v({"name": "my-job", "definition": _job_def("my-job")}) + self.assertEqual(good["name"], "my-job") + with self.assertRaises(exec_constrained.OpError): + v({"name": "Bad_Name!", "definition": _job_def("x")}) + with self.assertRaises(exec_constrained.OpError): + v({"name": "my-job", "definition": ["not", "a", "dict"]}) + with self.assertRaises(exec_constrained.OpError): + v({"name": "my-job", "definition": _job_def("other")}) + with self.assertRaises(exec_constrained.OpError): + v({"name": "my-job"}) + + def test_job_trigger_validate(self): + v = exec_constrained.OPS["job.trigger"]["validate"] + self.assertEqual(v({"name": EXISTING_JOB})["name"], EXISTING_JOB) + with self.assertRaises(exec_constrained.OpError): + v({"name": MISSING_JOB}) + with self.assertRaises(exec_constrained.OpError): + v({"name": "Bad_Name!"}) + + def test_job_chain_validate(self): + v = exec_constrained.OPS["job.chain"]["validate"] + good = v({"from": EXISTING_JOB, "to": EXISTING_JOB}) + self.assertFalse(good["on_failure"]) + self.assertTrue(v({"from": EXISTING_JOB, "to": EXISTING_JOB, + "on_failure": True})["on_failure"]) + with self.assertRaises(exec_constrained.OpError): + v({"from": MISSING_JOB, "to": EXISTING_JOB}) + with self.assertRaises(exec_constrained.OpError): + v({"from": EXISTING_JOB}) + + def test_job_next_validate(self): + v = exec_constrained.OPS["job.next"]["validate"] + good = v({"job_id": MISSING_ID}) + self.assertEqual(good["job_id"], MISSING_ID) + self.assertIsNone(good["success"]) + self.assertTrue(v({"job_id": MISSING_ID, "success": True})["success"]) + for bad in ("plainname", "a-20200101-000000-xyz!", + "UPPER-20200101-000000-deadbeef", ""): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v({"job_id": bad}) + + def test_timer_validate(self): + for op in ("cron.timer_stop", "cron.timer_disable"): + v = exec_constrained.OPS[op]["validate"] + self.assertEqual(v({"name": EXISTING_JOB})["name"], EXISTING_JOB) + with self.assertRaises(exec_constrained.OpError): + v({"name": MISSING_JOB}) + + def test_build_argv_shapes(self): + put = exec_constrained.OPS["job.put"] + argv = put["build"]({"name": "my-job", + "definition": _job_def("my-job")}) + self.assertEqual(argv[-2:], ["job-put", "my-job"]) + trig = exec_constrained.OPS["job.trigger"] + self.assertEqual(trig["build"]({"name": EXISTING_JOB})[-2:], + ["job-trigger", EXISTING_JOB]) + chain = exec_constrained.OPS["job.chain"] + argv = chain["build"]({"from": "a", "to": "b", "on_failure": False}) + self.assertEqual(argv[-3:], ["job-chain", "a", "b"]) + argv = chain["build"]({"from": "a", "to": "b", "on_failure": True}) + self.assertEqual(argv[-4:], ["job-chain", "a", "b", "--on-failure"]) + nxt = exec_constrained.OPS["job.next"] + self.assertEqual(nxt["build"]({"job_id": "i", "success": None})[-2:], + ["job-next", "i"]) + argv = nxt["build"]({"job_id": "i", "success": False}) + self.assertEqual(argv[-3:], ["job-next", "i", "--fail"]) + stop = exec_constrained.OPS["cron.timer_stop"] + self.assertEqual(stop["build"]({"name": EXISTING_JOB})[-2:], + ["timer-stop", EXISTING_JOB]) + dis = exec_constrained.OPS["cron.timer_disable"] + self.assertEqual(dis["build"]({"name": EXISTING_JOB})[-2:], + ["timer-disable", EXISTING_JOB]) + self.assertIsInstance(argv, list) + + def test_stdin_body_routing(self): + body = exec_constrained._stdin_body( + "job.put", {"name": "my-job", "definition": _job_def("my-job")}) + # box-ctl job-put reads the raw definition (not the envelope). + self.assertEqual(json.loads(body)["name"], "my-job") + self.assertNotIn("definition", json.loads(body)) + self.assertIsNone(exec_constrained._stdin_body("job.trigger", {})) + env = exec_constrained._stdin_body("files.read", {"path": "x"}) + self.assertEqual(json.loads(env), {"path": "x"}) + + +class BoxCtlJobsTests(unittest.TestCase): + def test_job_next_dry_run_live(self): + r = _box_ctl("job-next", MISSING_ID) + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertEqual(data["job_id"], MISSING_ID) + self.assertFalse(data["would_dispatch"]) + + def test_job_put_rejects_before_write(self): + r = _box_ctl("job-put", "Bad_Name!", stdin="{}") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl("job-put", "my-job", stdin="not json") + self.assertEqual(json.loads(r.stdout)["code"], "INVALID_JOB") + r = _box_ctl("job-put", "my-job", + stdin=json.dumps(_job_def("other"))) + self.assertEqual(json.loads(r.stdout)["code"], "NAME_MISMATCH") + bad = _job_def("my-job") + del bad["agent"] + r = _box_ctl("job-put", "my-job", stdin=json.dumps(bad)) + self.assertEqual(json.loads(r.stdout)["code"], "INVALID_JOB") + + def test_job_trigger_rejects_missing(self): + r = _box_ctl("job-trigger", "Bad_Name!") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl("job-trigger", MISSING_JOB) + self.assertEqual(json.loads(r.stdout)["code"], "NOT_FOUND") + + def test_job_chain_rejects_before_write(self): + r = _box_ctl("job-chain", "Bad_Name!", EXISTING_JOB) + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl("job-chain", EXISTING_JOB, EXISTING_JOB) + self.assertEqual(json.loads(r.stdout)["code"], "INVALID_JOB") + r = _box_ctl("job-chain", MISSING_JOB, EXISTING_JOB) + self.assertEqual(json.loads(r.stdout)["code"], "NOT_FOUND") + + def test_timer_control_rejects_before_action(self): + for verb in ("timer-stop", "timer-disable"): + r = _box_ctl(verb, "Bad_Name!") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl(verb, MISSING_JOB) + self.assertEqual(json.loads(r.stdout)["code"], "NOT_FOUND") + + def test_quality_validate_job_verbs(self): + r = _box_ctl("quality-validate", "job-put", "my-job") + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "job-trigger", EXISTING_JOB) + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "job-chain", "a", "b") + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "job-next", MISSING_ID) + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "timer-stop", EXISTING_JOB) + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "job-put", "Bad_Name!") + self.assertFalse(json.loads(r.stdout)["valid"], r.stdout) + + +class BoxRelayJobsTests(unittest.TestCase): + def test_relay_help_lists_job_commands(self): + r = subprocess.run(["bash", str(RELAY), "help"], + capture_output=True, text=True, timeout=30) + self.assertEqual(r.returncode, 0, r.stderr) + for line in ("box cron put", "box cron trigger", "box cron chain", + "box cron next", "box timer stop"): + self.assertIn(line, r.stdout) + + def test_relay_maps_job_commands_to_ops(self): + text = RELAY.read_text() + for op in ('"job.put"', '"job.trigger"', '"job.chain"', '"job.next"', + '"cron.timer_stop"', '"cron.timer_disable"'): + self.assertIn(op, text) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_box_loop_https.py b/tests/test_box_loop_https.py new file mode 100644 index 0000000..66aefa0 --- /dev/null +++ b/tests/test_box_loop_https.py @@ -0,0 +1,233 @@ +"""Tests for loop + strategy writes over HTTPS (no SSH). + +Covers the loop/strategy expansion: + box-relay.sh (agent client) -> exec-constrained.py named ops + -> box-ctl.py backend verbs -> followups/variables/strategy state. + +Safe mutations only. Live execution is limited to side-effect-free paths: +loop-remediate --dry-run (all writes guarded), strat-reset on a probe key +that can never exist (returns False, no write), and validation-failure +paths (which fail before any side effect). loop-resolve always appends to +job-log, and vars-reset/rollback/strat-set mutate live fleet state, so +those success paths are covered by quality-validate (dry-run) plus unit +tests β€” never executed here. Live-socket round-trips are intentionally +NOT covered here; instead we assert the exact argv each op builds. +""" +import importlib.util +import json +import subprocess +import sys +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +BOX_CTL = REPO_ROOT / "bin" / "box-ctl.py" +RELAY = REPO_ROOT / "bin" / "box-relay.sh" + +MISSING_VAR = "definitely-no-such-var-xyz" +# A strategy key no agent will ever set: reset returns False, no write. +PROBE_TYPE = "heartbeat" +PROBE_SUBTYPE = "ZZZ_PROBE_NO_SUCH_SUBTYPE" + + +def _load(name, relpath): + spec = importlib.util.spec_from_file_location(name, REPO_ROOT / relpath) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +exec_constrained = _load("exec_constrained_loop", "bin/exec-constrained.py") + + +def _box_ctl(*args): + return subprocess.run( + [sys.executable, str(BOX_CTL), *args], + capture_output=True, text=True, timeout=180) + + +class ExecLoopOpsTests(unittest.TestCase): + def test_ops_registered_and_side_effecting(self): + spec = exec_constrained.OPS + for op in ("loop.remediate", "loop.resolve", "strat.set", + "strat.reset", "vars.reset", "vars.rollback"): + self.assertIn(op, spec) + self.assertTrue(spec[op]["side_effecting"]) + + def test_known_identities_only(self): + p = exec_constrained.permitted + for op in ("loop.remediate", "loop.resolve", "strat.set", + "strat.reset", "vars.reset", "vars.rollback"): + self.assertFalse(p("some-unknown-identity", op)) + self.assertTrue(p("operator-646", op)) + self.assertFalse(p("exec-canary", "loop.remediate")) + + def test_remediate_validate(self): + v = exec_constrained.OPS["loop.remediate"]["validate"] + self.assertEqual(v({}), {"dry_run": False}) + self.assertTrue(v({"dry_run": True})["dry_run"]) + with self.assertRaises(exec_constrained.OpError): + v({"bogus": 1}) + + def test_resolve_validate(self): + v = exec_constrained.OPS["loop.resolve"]["validate"] + good = v({"dm_id": "bdf7beb6", "note": "looks good"}) + self.assertEqual(good["dm_id"], "bdf7beb6") + self.assertEqual(good["note"], "looks good") + self.assertIsNone(v({"dm_id": "bdf7beb6"})["note"]) + with self.assertRaises(exec_constrained.OpError): + v({"dm_id": "xyz!"}) + with self.assertRaises(exec_constrained.OpError): + v({"dm_id": "bdf7beb6", "note": " "}) + + def test_strat_set_validate(self): + v = exec_constrained.OPS["strat.set"]["validate"] + good = v({"type": "job", "priority": "important", "nudges": 3}) + self.assertEqual(good["type"], "job") + self.assertEqual(good["priority"], "important") + # Typos must fail: the backend silently maps unknown types to MANUAL. + with self.assertRaises(exec_constrained.OpError, msg="typo type"): + v({"type": "wkae"}) + with self.assertRaises(exec_constrained.OpError): + v({"type": "job", "priority": "urgent"}) + with self.assertRaises(exec_constrained.OpError): + v({"type": "job", "timeout_s": "soon"}) + with self.assertRaises(exec_constrained.OpError): + v({"type": "job", "agent": "nope"}) + with self.assertRaises(exec_constrained.OpError): + v({"priority": "normal"}) + + def test_strat_reset_validate(self): + v = exec_constrained.OPS["strat.reset"]["validate"] + good = v({"type": "heartbeat", "subtype": "DM", "agent": "opm"}) + self.assertEqual(good, {"type": "heartbeat", "subtype": "DM", + "agent": "opm"}) + with self.assertRaises(exec_constrained.OpError): + v({"type": "wkae"}) + with self.assertRaises(exec_constrained.OpError): + v({"type": "job", "subtype": "has space"}) + + def test_vars_validate(self): + vr = exec_constrained.OPS["vars.reset"]["validate"] + self.assertEqual(vr({"name": "max_nudge_count"})["name"], + "max_nudge_count") + with self.assertRaises(exec_constrained.OpError): + vr({"name": "has space"}) + vb = exec_constrained.OPS["vars.rollback"]["validate"] + self.assertIsNone(vb({"name": "max_nudge_count"})["revision"]) + self.assertEqual(vb({"name": "x", "revision": 2})["revision"], 2) + with self.assertRaises(exec_constrained.OpError): + vb({"name": "x", "revision": 0}) + with self.assertRaises(exec_constrained.OpError): + vb({"name": "x", "revision": "a\nb"}) + + def test_build_argv_shapes(self): + rem = exec_constrained.OPS["loop.remediate"] + self.assertEqual(rem["build"]({"dry_run": False})[-1], + "loop-remediate") + argv = rem["build"]({"dry_run": True}) + self.assertEqual(argv[-2:], ["loop-remediate", "--dry-run"]) + res = exec_constrained.OPS["loop.resolve"] + argv = res["build"]({"dm_id": "abc123", "note": None}) + self.assertEqual(argv[-2:], ["loop-resolve", "abc123"]) + argv = res["build"]({"dm_id": "abc123", "note": "n"}) + self.assertEqual(argv[-3:], ["loop-resolve", "abc123", "n"]) + st = exec_constrained.OPS["strat.set"] + argv = st["build"]({"type": "job", "subtype": None, "agent": None, + "track": None, "priority": "normal", + "timeout_s": None, "nudges": 2, "escalate": None}) + self.assertEqual(argv[-3], "strat-set") + payload = json.loads(argv[-1]) + self.assertEqual(payload["priority"], "normal") + self.assertEqual(payload["nudges"], 2) + sr = exec_constrained.OPS["strat.reset"] + argv = sr["build"]({"type": "job", "subtype": "DM", + "agent": "opm"}) + self.assertEqual(argv[-4:], + ["strat-reset", "job", "DM", "--agent", "opm"][-4:]) + vrt = exec_constrained.OPS["vars.reset"] + self.assertEqual(vrt["build"]({"name": "x"})[-2:], + ["vars-reset", "x"]) + vrb = exec_constrained.OPS["vars.rollback"] + argv = vrb["build"]({"name": "x", "revision": 2}) + self.assertEqual(argv[-3:], ["vars-rollback", "x", "2"]) + self.assertIsInstance(argv, list) + + +class BoxCtlLoopTests(unittest.TestCase): + def test_remediate_dry_run_live(self): + r = _box_ctl("loop-remediate", "--dry-run") + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertTrue(data["dry_run"]) + self.assertIn("remediated", data) + self.assertIn("escalated", data) + + def test_strat_reset_probe_key_live(self): + r = _box_ctl("strat-reset", PROBE_TYPE, PROBE_SUBTYPE) + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertFalse(data["reset"]) + + def test_strat_set_rejects_before_write(self): + r = _box_ctl("strat-set") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl("strat-set", "job", "not json") + self.assertEqual(json.loads(r.stdout)["code"], "STRAT_ERROR") + bad = json.dumps({"priority": "urgent"}) + r = _box_ctl("strat-set", "job", bad) + self.assertEqual(json.loads(r.stdout)["code"], "STRAT_ERROR") + + def test_vars_rejects_unknown_before_write(self): + r = _box_ctl("vars-reset", MISSING_VAR) + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "VARS_ERROR") + r = _box_ctl("vars-rollback", MISSING_VAR) + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "VARS_ERROR") + + def test_quality_validate_loop_verbs(self): + cases = [ + (["loop-remediate"], True), + (["loop-remediate", "--dry-run"], True), + (["loop-resolve", "bdf7beb6"], True), + (["loop-resolve", "bdf7beb6", "note"], True), + (["loop-resolve", "xyz!"], False), + (["strat-set", "job"], True), + (["strat-set"], False), + (["strat-reset", "job", "DM", "--agent", "opm"], True), + (["strat-reset", "--agent", "nope"], False), + (["vars-reset", "max_nudge_count"], True), + (["vars-rollback", "max_nudge_count", "2"], True), + (["vars-get", "loop_health_threshold"], True), + (["vars-set", "max_nudge_count", "5"], True), + (["vars-reset"], False), + (["vars-get", "has space"], False), + ] + for args, valid in cases: + r = _box_ctl("quality-validate", *args) + self.assertEqual(json.loads(r.stdout)["valid"], valid, args) + + +class BoxRelayLoopTests(unittest.TestCase): + def test_relay_help_lists_loop_commands(self): + r = subprocess.run(["bash", str(RELAY), "help"], + capture_output=True, text=True, timeout=30) + self.assertEqual(r.returncode, 0, r.stderr) + for line in ("box loop remediate", "box loop resolve", + "box strat set", "box strat reset", + "box vars reset", "box vars rollback"): + self.assertIn(line, r.stdout) + + def test_relay_maps_loop_commands_to_ops(self): + text = RELAY.read_text() + for op in ('"loop.remediate"', '"loop.resolve"', '"strat.set"', + '"strat.reset"', '"vars.reset"', '"vars.rollback"'): + self.assertIn(op, text) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_box_md_https.py b/tests/test_box_md_https.py new file mode 100644 index 0000000..9539bc7 --- /dev/null +++ b/tests/test_box_md_https.py @@ -0,0 +1,412 @@ +"""Tests for md-file reads + governed writes over HTTPS (no SSH). + +Covers the md expansion: + box-relay.sh (agent client) -> exec-constrained.py named ops + -> box-ctl.py backend verbs -> agent_md.py (Hatch gateway / shared templates). + +Raw container writes (md-write) are deliberately NOT exposed over HTTPS; +the governed flows are amend/append (validated shared templates with git +commit) and pull/inject-drive/sync-all (push canonical templates). + +Live execution is limited to validation-failure paths (which fail before +any gateway call or write), one stdin-plumbing path that the backend +safety gate rejects before writing, quality-validate dry-runs, and +agent_md validator unit tests. No live gateway traffic, no template +writes, and no live-socket round-trips here; instead we assert the exact +argv each op builds. +""" +import hashlib +import importlib.util +import json +import subprocess +import sys +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +BOX_CTL = REPO_ROOT / "bin" / "box-ctl.py" +RELAY = REPO_ROOT / "bin" / "box-relay.sh" +HEARTBEAT = REPO_ROOT / "shared" / "operators" / "HEARTBEAT.md" + +MD_READ_OPS = ("md.audit", "md.list", "md.read", "md.diff") +MD_WRITE_OPS = ("md.pull", "md.inject_drive", "md.sync_all", + "md.amend", "md.append") + + +def _load(name, relpath): + spec = importlib.util.spec_from_file_location(name, REPO_ROOT / relpath) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +exec_constrained = _load("exec_constrained_md", "bin/exec-constrained.py") +agent_md = _load("agent_md_mdtest", "bin/agent_md.py") + + +def _box_ctl(*args, stdin=None): + return subprocess.run( + [sys.executable, str(BOX_CTL), *args], + input=stdin, capture_output=True, text=True, timeout=180) + + +class ExecMdOpsTests(unittest.TestCase): + def test_ops_registered_and_side_effecting(self): + spec = exec_constrained.OPS + for op in MD_READ_OPS: + self.assertIn(op, spec) + self.assertFalse(spec[op]["side_effecting"], op) + for op in MD_WRITE_OPS: + self.assertIn(op, spec) + self.assertTrue(spec[op]["side_effecting"], op) + + def test_raw_write_not_exposed(self): + self.assertNotIn("md.write", exec_constrained.OPS) + + def test_known_identities_only(self): + p = exec_constrained.permitted + for op in MD_READ_OPS: + self.assertTrue(p("some-unknown-identity", op), op) + self.assertTrue(p("operator-646", op), op) + for op in MD_WRITE_OPS: + self.assertFalse(p("some-unknown-identity", op), op) + self.assertTrue(p("operator-646", op), op) + self.assertFalse(p("exec-canary", "md.read")) + self.assertFalse(p("exec-canary", "md.amend")) + + def test_audit_validate(self): + v = exec_constrained.OPS["md.audit"]["validate"] + self.assertEqual(v({}), {"accounts": None}) + # Explicit null means "all accounts", same as omitted (optional-arg + # convention shared with strat.set / loop.resolve). + self.assertEqual(v({"accounts": None}), {"accounts": None}) + self.assertEqual(v({"accounts": ["646", "muse-main"]})["accounts"], + ["646", "muse-main"]) + for bad in ({"accounts": "646"}, {"accounts": []}, + {"accounts": ["../x"]}, {"accounts": ["a b"]}, + {"accounts": ["ok", ""]}, {"bogus": 1}): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v(bad) + + def test_list_validate(self): + v = exec_constrained.OPS["md.list"]["validate"] + self.assertEqual(v({"account": "646"}), + {"account": "646", "path": ""}) + self.assertEqual(v({"account": "646", "path": "sub/dir"})["path"], + "sub/dir") + for bad in ({"account": "../x"}, {"account": "a b"}, + {"account": "646", "path": ".."}, + {"account": "646", "path": "/abs"}, + {"account": "646", "path": "a/../../x"}, + {"account": "646", "bogus": 1}, + {"path": "sub"}): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v(bad) + + def test_read_validate(self): + v = exec_constrained.OPS["md.read"]["validate"] + good = v({"account": "646", "filename": "SOUL.md"}) + self.assertEqual(good, {"account": "646", "filename": "SOUL.md"}) + # Reads accept any container basename, not just shared templates. + self.assertEqual( + v({"account": "646", "filename": "NOTES.md"})["filename"], + "NOTES.md") + for bad in ({"account": "646", "filename": "../x"}, + {"account": "646", "filename": "/abs"}, + {"account": "646", "filename": ".."}, + {"account": "646", "filename": "a/b"}, + {"account": "646", "filename": ""}, + {"account": "a b", "filename": "SOUL.md"}, + {"account": "646", "filename": "SOUL.md", "x": 1}): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v(bad) + + def test_diff_validate(self): + v = exec_constrained.OPS["md.diff"]["validate"] + good = v({"account": "646", "filename": "SOUL.md"}) + self.assertEqual(good, {"account": "646", "filename": "SOUL.md"}) + # Template flows reject non-templates: the backend indexes + # shared/operators/ by filename. + for bad in ({"account": "646", "filename": "NOPE.md"}, + {"account": "646", "filename": "../x"}, + {"account": "646", "filename": "NOTES.md"}): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v(bad) + + def test_pull_validate(self): + v = exec_constrained.OPS["md.pull"]["validate"] + self.assertEqual(v({"account": "opm", "filename": "TOOLS.md"}), + {"account": "opm", "filename": "TOOLS.md"}) + with self.assertRaises(exec_constrained.OpError): + v({"account": "opm", "filename": "NOPE.md"}) + with self.assertRaises(exec_constrained.OpError): + v({"account": "../x", "filename": "TOOLS.md"}) + + def test_inject_drive_validate(self): + v = exec_constrained.OPS["md.inject_drive"]["validate"] + self.assertEqual(v({"account": "646"}), {"account": "646"}) + with self.assertRaises(exec_constrained.OpError): + v({"account": "../x"}) + with self.assertRaises(exec_constrained.OpError): + v({"account": "646", "force": True}) + + def test_sync_all_validate(self): + v = exec_constrained.OPS["md.sync_all"]["validate"] + self.assertEqual(v({}), {}) + with self.assertRaises(exec_constrained.OpError): + v({"accounts": ["646"]}) + + def test_amend_validate(self): + v = exec_constrained.OPS["md.amend"]["validate"] + good = v({"filename": "SOUL.md", "content": "body", + "author": "646", "reason": "tune"}) + self.assertEqual(good, {"filename": "SOUL.md", "content": "body", + "author": "646", "reason": "tune"}) + defaults = v({"filename": "SOUL.md", "content": "body"}) + self.assertEqual(defaults["author"], "operator") + self.assertEqual(defaults["reason"], "") + over = "x" * (exec_constrained.MD_MAX_AMEND + 1) + for bad in ({"filename": "NOPE.md", "content": "body"}, + {"filename": "SOUL.md", "content": " "}, + {"filename": "SOUL.md", "content": over}, + {"filename": "SOUL.md", "content": "x", "author": ""}, + {"filename": "SOUL.md", "content": "x", + "author": "a\nb"}, + {"filename": "SOUL.md", "content": "x", + "author": "a" * 65}, + {"filename": "SOUL.md", "content": "x", + "reason": "r" * 257}, + {"filename": "SOUL.md", "content": "x", + "reason": "a\nb"}, + {"filename": "SOUL.md", "content": "x", "bogus": 1}, + {"filename": "SOUL.md"}): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v(bad) + + def test_append_validate(self): + v = exec_constrained.OPS["md.append"]["validate"] + good = v({"filename": "AGENTS.md", "text": "lesson", + "author": "opm", "section": "Wins"}) + self.assertEqual(good["section"], "Wins") + self.assertIsNone(v({"filename": "AGENTS.md", + "text": "lesson"})["section"]) + over = "x" * (exec_constrained.MD_MAX_APPEND + 1) + for bad in ({"filename": "NOPE.md", "text": "lesson"}, + {"filename": "AGENTS.md", "text": " "}, + {"filename": "AGENTS.md", "text": over}, + {"filename": "AGENTS.md", "text": "t", + "section": " "}, + {"filename": "AGENTS.md", "text": "t", + "section": "s" * 129}): + with self.assertRaises(exec_constrained.OpError, msg=bad): + v(bad) + + def test_build_argv_shapes(self): + ops = exec_constrained.OPS + au = ops["md.audit"] + self.assertEqual(au["build"]({"accounts": None})[-1], "md-audit") + self.assertEqual( + au["build"]({"accounts": ["646", "opm"]})[-3:], + ["md-audit", "646", "opm"]) + li = ops["md.list"] + self.assertEqual(li["build"]({"account": "646", "path": ""})[-2:], + ["md-list", "646"]) + self.assertEqual( + li["build"]({"account": "646", "path": "sub"})[-3:], + ["md-list", "646", "sub"]) + rd = ops["md.read"] + self.assertEqual( + rd["build"]({"account": "646", "filename": "SOUL.md"})[-3:], + ["md-read", "646", "SOUL.md"]) + df = ops["md.diff"] + self.assertEqual( + df["build"]({"account": "646", "filename": "SOUL.md"})[-3:], + ["md-diff", "646", "SOUL.md"]) + pu = ops["md.pull"] + self.assertEqual( + pu["build"]({"account": "646", "filename": "SOUL.md"})[-3:], + ["md-pull", "646", "SOUL.md"]) + inj = ops["md.inject_drive"] + self.assertEqual(inj["build"]({"account": "646"})[-2:], + ["md-inject-drive", "646"]) + self.assertEqual(ops["md.sync_all"]["build"]({})[-1], "md-sync-all") + am = ops["md.amend"] + argv = am["build"]({"filename": "SOUL.md", "content": "x", + "author": "646", "reason": "why"}) + self.assertEqual(argv[-7:], + ["md-amend", "SOUL.md", "--stdin", + "--author", "646", "--reason", "why"]) + argv = am["build"]({"filename": "SOUL.md", "content": "x", + "author": "646", "reason": ""}) + self.assertEqual(argv[-5:], + ["md-amend", "SOUL.md", "--stdin", + "--author", "646"]) + ap = ops["md.append"] + argv = ap["build"]({"filename": "AGENTS.md", "text": "t", + "author": "opm", "section": "Wins"}) + self.assertEqual(argv[-7:], + ["md-append", "AGENTS.md", "--stdin", + "--author", "opm", "--section", "Wins"]) + argv = ap["build"]({"filename": "AGENTS.md", "text": "t", + "author": "opm", "section": None}) + self.assertEqual(argv[-5:], + ["md-append", "AGENTS.md", "--stdin", + "--author", "opm"]) + self.assertIsInstance(argv, list) + + def test_stdin_body(self): + sb = exec_constrained._stdin_body + self.assertEqual(sb("md.amend", {"content": "C"}), "C") + self.assertEqual(sb("md.append", {"text": "T"}), "T") + self.assertIsNone(sb("md.read", {"account": "646"})) + + +class AgentMdValidationTests(unittest.TestCase): + def test_account(self): + for good in ("646", "muse", "muse-main", "opm", "dev", "def"): + self.assertEqual(agent_md.validate_account(good), good) + for bad in ("../x", "a b", "", "a/b", "x" * 33, "-lead"): + with self.assertRaises(agent_md.MDValidationError, msg=bad): + agent_md.validate_account(bad) + + def test_filename(self): + for good in ("SOUL.md", "NOTES.md", "a"): + self.assertEqual(agent_md.validate_filename(good), good) + self.assertEqual( + agent_md.validate_filename("SOUL.md", template_only=True), + "SOUL.md") + for bad in ("../x", "/abs", "..", ".", "a/b", ""): + with self.assertRaises(agent_md.MDValidationError, msg=bad): + agent_md.validate_filename(bad) + for bad in ("NOPE.md", "../SOUL.md", "NOTES.md"): + with self.assertRaises(agent_md.MDValidationError, msg=bad): + agent_md.validate_filename(bad, template_only=True) + + def test_subpath(self): + self.assertEqual(agent_md.validate_subpath(""), "") + self.assertEqual(agent_md.validate_subpath("a/b"), "a/b") + for bad in ("..", "/abs", "a/../../x", "a b"): + with self.assertRaises(agent_md.MDValidationError, msg=bad): + agent_md.validate_subpath(bad) + + def test_rejects_before_gateway(self): + # Validation failures raise MDValidationError; a call that + # reached the gateway would raise RuntimeError (no gateway + # module here) or FileNotFoundError (no cookies) instead. + with self.assertRaises(agent_md.MDValidationError): + agent_md.read_md("646", "../x") + with self.assertRaises(agent_md.MDValidationError): + agent_md.list_files("../x", "") + with self.assertRaises(agent_md.MDValidationError): + agent_md.audit_agents(["ok", "../x"]) + with self.assertRaises(agent_md.MDValidationError): + agent_md.diff_md("646", "NOPE.md") + + def test_amend_rejects_before_write(self): + with self.assertRaises(agent_md.MDValidationError): + agent_md.amend_md("../x", "body") + with self.assertRaises(agent_md.MDValidationError): + agent_md.append_md("NOPE.md", "note") + + +class BoxCtlMdTests(unittest.TestCase): + def test_rejects_traversal_before_gateway(self): + cases = [ + ["md-read", "646", "../x"], + ["md-read", "646", ".."], + ["md-list", "bad!", "x"], + ["md-list", "646", "../.."], + ["md-write", "646", "/abs", "hi"], + ["md-audit", "../x"], + ["md", "read", "646", "../x"], + ] + for args in cases: + r = _box_ctl(*args) + self.assertNotEqual(r.returncode, 0, args) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME", args) + + def test_rejects_non_template_before_write(self): + cases = [ + ["md-diff", "646", "NOPE.md"], + ["md-pull", "646", "NOPE.md"], + ["md", "diff", "646", "NOPE.md"], + ["md", "amend", "NOPE.md", "content here"], + ["md", "append", "NOPE.md", "note here"], + ] + for args in cases: + r = _box_ctl(*args) + self.assertNotEqual(r.returncode, 0, args) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME", args) + r = _box_ctl("md-amend", "../x", "--stdin", stdin="hi") + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + r = _box_ctl("md-append", "../x", "--stdin", stdin="hi") + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NAME") + + def test_amend_stdin_safety_rejection_writes_nothing(self): + before = hashlib.sha256(HEARTBEAT.read_bytes()).hexdigest() + # Gutted HEARTBEAT content via --stdin: proves stdin plumbing + # reaches the backend, and the safety gate rejects it before + # any write or git commit. + r = _box_ctl("md-amend", "HEARTBEAT.md", "--stdin", stdin="gutted") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "AMEND_FAILED") + after = hashlib.sha256(HEARTBEAT.read_bytes()).hexdigest() + self.assertEqual(before, after) + + def test_quality_validate_md_verbs(self): + cases = [ + (["md-audit"], True), + (["md-audit", "646", "opm"], True), + (["md-audit", "../x"], False), + (["md-list", "646"], True), + (["md-list", "646", "sub/dir"], True), + (["md-list", "646", ".."], False), + (["md-list"], False), + (["md-read", "646", "SOUL.md"], True), + (["md-read", "646", "../x"], False), + (["md-read", "646"], False), + (["md-diff", "646", "SOUL.md"], True), + (["md-diff", "646", "NOPE.md"], False), + (["md-pull", "646", "SOUL.md"], True), + (["md-pull", "646"], False), + (["md-inject-drive", "646"], True), + (["md-inject-drive", "646", "--force"], True), + (["md-inject-drive"], False), + (["md-sync-all"], True), + (["md-sync-all", "--force"], True), + (["md-sync-all", "646"], False), + (["md-amend", "SOUL.md", "--stdin"], True), + (["md-amend", "SOUL.md", "--stdin", "--author", "646"], True), + (["md-amend", "NOPE.md", "--stdin"], False), + (["md-amend"], False), + (["md-append", "SOUL.md", "--stdin"], True), + (["md-append", "SOUL.md", "note", "--section", "s"], True), + (["md-append", "x", "y", "z"], False), + (["md-write", "646", "SOUL.md", "x"], True), + (["md-write", "646", "SOUL.md"], False), + ] + for args, valid in cases: + r = _box_ctl("quality-validate", *args) + self.assertEqual(json.loads(r.stdout)["valid"], valid, args) + + +class BoxRelayMdTests(unittest.TestCase): + def test_relay_help_lists_md_commands(self): + r = subprocess.run(["bash", str(RELAY), "help"], + capture_output=True, text=True, timeout=30) + self.assertEqual(r.returncode, 0, r.stderr) + for line in ("box md audit", "box md list", "box md read", + "box md diff", "box md pull", "box md inject-drive", + "box md sync-all", "box md amend", "box md append"): + self.assertIn(line, r.stdout) + + def test_relay_maps_md_commands_to_ops(self): + text = RELAY.read_text() + for op in ('"md.audit"', '"md.list"', '"md.read"', '"md.diff"', + '"md.pull"', '"md.inject_drive"', '"md.sync_all"', + '"md.amend"', '"md.append"'): + self.assertIn(op, text) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_box_read_https.py b/tests/test_box_read_https.py new file mode 100644 index 0000000..4cf6f6b --- /dev/null +++ b/tests/test_box_read_https.py @@ -0,0 +1,202 @@ +"""Tests for read-only box lookups over HTTPS (no SSH). + +Covers the agent-facing read path: + box-relay.sh (agent client) -> exec-constrained.py named ops + -> box-ctl.py backend verbs -> super-cli.py lookups. + +Live-socket round-trips are intentionally NOT covered here (loopback TCP is +unavailable in some sandboxes); instead we assert the exact argv each op +builds and execute the fast argv directly. +""" +import argparse +import importlib.util +import io +import json +import subprocess +import sys +import unittest +from contextlib import redirect_stdout +from pathlib import Path +from unittest import mock + +REPO_ROOT = Path(__file__).resolve().parent.parent +BOX_CTL = REPO_ROOT / "bin" / "box-ctl.py" +RELAY = REPO_ROOT / "bin" / "box-relay.sh" + + +def _load(name, relpath): + spec = importlib.util.spec_from_file_location(name, REPO_ROOT / relpath) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +exec_constrained = _load("exec_constrained_read", "bin/exec-constrained.py") +super_cli = _load("super_cli_read", "bin/super-cli.py") + + +def _box_ctl(*args): + return subprocess.run( + [sys.executable, str(BOX_CTL), *args], + capture_output=True, text=True, timeout=60) + + +class ExecReadOpsTests(unittest.TestCase): + def test_ops_registered_and_read_only(self): + self.assertIn("fleet.unread", exec_constrained.OPS) + self.assertIn("dm.log", exec_constrained.OPS) + self.assertFalse(exec_constrained.OPS["fleet.unread"]["side_effecting"]) + self.assertFalse(exec_constrained.OPS["dm.log"]["side_effecting"]) + + def test_default_perms_include_read_ops(self): + # Any valid fleet signer can read; canary stays ping-only. + self.assertTrue(exec_constrained.permitted("some-unknown-identity", "fleet.unread")) + self.assertTrue(exec_constrained.permitted("some-unknown-identity", "dm.log")) + self.assertFalse(exec_constrained.permitted("exec-canary", "fleet.unread")) + self.assertFalse(exec_constrained.permitted("exec-canary", "dm.log")) + + def test_fleet_unread_validate(self): + v = exec_constrained.OPS["fleet.unread"]["validate"] + self.assertEqual(v({}), {"agent": None}) + self.assertEqual(v({"agent": "pip"}), {"agent": "pip"}) + with self.assertRaises(exec_constrained.OpError): + v({"agent": "nope"}) + with self.assertRaises(exec_constrained.OpError): + v({"bogus": 1}) + + def test_dm_log_validate(self): + v = exec_constrained.OPS["dm.log"]["validate"] + self.assertEqual(v({}), {"limit": 20, "agent": None}) + self.assertEqual(v({"limit": 5, "agent": "opm"}), {"limit": 5, "agent": "opm"}) + with self.assertRaises(exec_constrained.OpError): + v({"limit": 0}) + with self.assertRaises(exec_constrained.OpError): + v({"limit": 101}) + with self.assertRaises(exec_constrained.OpError): + v({"agent": "nope"}) + with self.assertRaises(exec_constrained.OpError): + v({"bogus": 1}) + + def test_build_argv_shapes(self): + unread = exec_constrained.OPS["fleet.unread"] + argv = unread["build"]({"agent": None}) + self.assertEqual(argv[-1], "unread") + self.assertNotIn("--agent", argv) + argv = unread["build"]({"agent": "muse"}) + self.assertEqual(argv[-3:], ["unread", "--agent", "muse"]) + + dmlog = exec_constrained.OPS["dm.log"] + argv = dmlog["build"]({"limit": 5, "agent": "opm"}) + self.assertEqual(argv[-4:], ["dm-log", "5", "--agent", "opm"]) + argv = dmlog["build"]({"limit": 20, "agent": None}) + self.assertEqual(argv[-2:], ["dm-log", "20"]) + # argv only, never a shell string. + self.assertIsInstance(argv, list) + + def test_dm_log_built_argv_executes(self): + spec = exec_constrained.OPS["dm.log"] + clean = spec["validate"]({"limit": 2}) + argv = spec["build"](clean) + argv[0] = sys.executable # hermetic interpreter, same script + args + r = subprocess.run(argv, capture_output=True, text=True, timeout=60) + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertEqual(len(data["entries"]), 2) + + +class BoxCtlReadVerbsTests(unittest.TestCase): + def test_unread_rejects_unknown_agent(self): + r = _box_ctl("unread", "--agent", "nope") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NODE") + + def test_unread_rejects_positional_and_missing_value(self): + r = _box_ctl("unread", "pip") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_ARGS") + r = _box_ctl("unread", "--agent") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_ARGS") + + def test_dm_log_rejects_bad_limit_and_agent(self): + r = _box_ctl("dm-log", "abc") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_LIMIT") + r = _box_ctl("dm-log", "5", "--agent", "nope") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_NODE") + r = _box_ctl("dm-log", "1", "2") + self.assertNotEqual(r.returncode, 0) + self.assertEqual(json.loads(r.stdout)["code"], "BAD_ARGS") + + def test_dm_log_back_compat_limit_only(self): + r = _box_ctl("dm-log", "2") + self.assertEqual(r.returncode, 0, r.stderr) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertEqual(len(data["entries"]), 2) + + def test_quality_validate_new_verbs(self): + r = _box_ctl("quality-validate", "unread", "--agent", "pip") + data = json.loads(r.stdout) + self.assertTrue(data["valid"], r.stdout) + r = _box_ctl("quality-validate", "dm-log", "5", "--agent", "opm") + self.assertTrue(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "unread", "--agent", "nope") + self.assertFalse(json.loads(r.stdout)["valid"], r.stdout) + r = _box_ctl("quality-validate", "unread", "extra-positional") + self.assertFalse(json.loads(r.stdout)["valid"], r.stdout) + + +class SuperCliUnreadTests(unittest.TestCase): + def test_lookup_dispatches_unread(self): + args = argparse.Namespace(target="unread", lookup_args=[], json=False) + with mock.patch.object(super_cli, "_lookup_unreads") as m: + super_cli.cmd_lookup(args) + m.assert_called_once_with(args) + + def test_lookup_unreads_json_shape(self): + fleet = [ + {"node": "muse", "title": "muse (2)", "url": "https://muse.ai/thread/abc123", + "approval_pending": False}, + {"node": "pip", "title": "muse", "url": "https://muse.ai/", + "approval_pending": True}, + ] + args = argparse.Namespace(json=True) + buf = io.StringIO() + with mock.patch.object(super_cli, "collect_fleet_data", return_value=fleet): + with redirect_stdout(buf): + super_cli._lookup_unreads(args) + data = json.loads(buf.getvalue()) + self.assertTrue(data["ok"]) + by_node = {n["node"]: n for n in data["nodes"]} + self.assertEqual(by_node["muse"]["unread"], 2) + self.assertEqual(by_node["muse"]["thread"], "abc123") + self.assertFalse(by_node["muse"]["approval_pending"]) + self.assertEqual(by_node["pip"]["unread"], 0) + self.assertTrue(by_node["pip"]["approval_pending"]) + self.assertEqual(by_node["pip"]["thread"], "home") + + +class BoxRelayClientTests(unittest.TestCase): + def test_relay_syntax_valid(self): + r = subprocess.run(["bash", "-n", str(RELAY)], + capture_output=True, text=True, timeout=30) + self.assertEqual(r.returncode, 0, r.stderr) + + def test_relay_help_lists_read_commands(self): + r = subprocess.run(["bash", str(RELAY), "help"], + capture_output=True, text=True, timeout=30) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertIn("box unread", r.stdout) + self.assertIn("box dm log", r.stdout) + + def test_relay_maps_read_commands_to_ops(self): + text = RELAY.read_text() + self.assertIn("fleet.unread", text) + self.assertIn('"dm.log"', text) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_box_run.py b/tests/test_box_run.py new file mode 100644 index 0000000..941d6ce --- /dev/null +++ b/tests/test_box_run.py @@ -0,0 +1,192 @@ +"""Tests for `box run` and `box watch` (headless muse-code tmux runs) in super-cli.py.""" +import argparse +import importlib.util +import io +import unittest +from contextlib import redirect_stdout +from pathlib import Path +from unittest import mock + +REPO_ROOT = Path(__file__).resolve().parent.parent +SPEC = importlib.util.spec_from_file_location( + "super_cli_box_run", REPO_ROOT / "bin" / "super-cli.py") +super_cli = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(super_cli) + + +def _ns(**over): + kw = dict(prompt="hello from unit test", prompt_file=None, session="ut-boxrun", + log=None, model=None, effort=None, provider="echo", + permission_profile=None, approval_mode="never", + trust_workspace=False, auto_approve=False) + kw.update(over) + return argparse.Namespace(**kw) + + +class FakeCompleted: + def __init__(self, returncode=0, stderr=""): + self.returncode = returncode + self.stderr = stderr + + +class BoxRunTests(unittest.TestCase): + def setUp(self): + for f in ("/tmp/ut-boxrun.prompt", "/tmp/ut-boxrun.sh", "/tmp/ut-boxrun.exit", + "/tmp/ut-boxrun-watch.sh", "/tmp/utw-watch.sh"): + try: + Path(f).unlink() + except FileNotFoundError: + pass + self.addCleanup(self._cleanup) + + def _cleanup(self): + for f in ("/tmp/ut-boxrun.prompt", "/tmp/ut-boxrun.sh", "/tmp/ut-boxrun.exit", + "/tmp/ut-boxrun-watch.sh", "/tmp/utw-watch.sh"): + try: + Path(f).unlink() + except FileNotFoundError: + pass + for log in ("ut-boxrun.log", "ut-boxrun-watch.log", "utw-watch.log"): + try: + Path(super_cli.MUSE_TMUX_LOG_DIR / log).unlink() + except FileNotFoundError: + pass + + def _which(self, name): + if name == "tmux": + return "/usr/bin/tmux" + if "muse-code" in name: + return "/home/super/.local/bin/muse-code" + return None + + def test_spawn_writes_prompt_wrapper_and_tmux_argv(self): + calls = [] + + def fake_run(argv, **kw): + calls.append(argv) + return FakeCompleted(0) + + with mock.patch.object(super_cli.shutil, "which", side_effect=self._which), \ + mock.patch.object(super_cli.subprocess, "run", side_effect=fake_run): + buf = io.StringIO() + with redirect_stdout(buf): + super_cli.cmd_run(_ns()) + out = buf.getvalue() + + self.assertIn("session: ut-boxrun", out) + self.assertIn("attach: tmux -S /tmp/tmux-muse.sock attach -t ut-boxrun", out) + self.assertEqual(Path("/tmp/ut-boxrun.prompt").read_text(encoding="utf-8"), + "hello from unit test\n") + wrapper = Path("/tmp/ut-boxrun.sh").read_text(encoding="utf-8") + self.assertIn("muse-code", wrapper) + self.assertIn("--provider echo", wrapper) + self.assertIn("--prompt-file", wrapper) + self.assertIn("cd /home/super/Projects/NetVM", wrapper) + new_session = [c for c in calls if "new-session" in c] + self.assertEqual(len(new_session), 1) + self.assertIn("/tmp/tmux-muse.sock", new_session[0]) + self.assertIn("ut-boxrun", new_session[0]) + + def test_optional_flags_passed_through(self): + calls = [] + + def fake_run(argv, **kw): + calls.append(argv) + return FakeCompleted(0) + + ns = _ns(model="m1", effort="low", permission_profile="prof", + trust_workspace=True, approval_mode="on-request") + with mock.patch.object(super_cli.shutil, "which", side_effect=self._which), \ + mock.patch.object(super_cli.subprocess, "run", side_effect=fake_run): + with redirect_stdout(io.StringIO()): + super_cli.cmd_run(ns) + wrapper = Path("/tmp/ut-boxrun.sh").read_text(encoding="utf-8") + for flag in ("--model m1", "--reasoning-effort low", "--permission-profile prof", + "--trust-workspace", "--approval-mode on-request"): + self.assertIn(flag, wrapper) + + def test_no_prompt_exits_2(self): + with mock.patch.object(super_cli.shutil, "which", side_effect=self._which), \ + mock.patch.object(super_cli.sys.stdin, "isatty", return_value=True): + with self.assertRaises(SystemExit) as cm: + super_cli.cmd_run(_ns(prompt=None)) + self.assertEqual(cm.exception.code, 2) + + def test_missing_muse_code_exits_2(self): + with mock.patch.object(super_cli.shutil, "which", return_value=None): + with self.assertRaises(SystemExit) as cm: + super_cli.cmd_run(_ns()) + self.assertEqual(cm.exception.code, 2) + + def test_sanitize_session_name(self): + self.assertEqual(super_cli._sanitize_tmux_name("run:2026/10/06 05.00"), "run-2026-10-06-05-00") + self.assertEqual(super_cli._sanitize_tmux_name("!!!"), "run") + + def test_run_auto_approve_spawns_watcher(self): + calls = [] + + def fake_run(argv, **kw): + calls.append(argv) + return FakeCompleted(0) + + with mock.patch.object(super_cli.shutil, "which", side_effect=self._which), \ + mock.patch.object(super_cli.subprocess, "run", side_effect=fake_run): + buf = io.StringIO() + with redirect_stdout(buf): + super_cli.cmd_run(_ns(auto_approve=True)) + out = buf.getvalue() + self.assertIn("watcher: ut-boxrun-watch", out) + self.assertIn("watcher-log:", out) + new_session = [c for c in calls if "new-session" in c] + self.assertEqual(len(new_session), 2) + watch_argv = [c for c in new_session if "ut-boxrun-watch" in c] + self.assertEqual(len(watch_argv), 1) + self.assertTrue(Path("/tmp/ut-boxrun-watch.sh").exists()) + + def test_write_watch_script_content(self): + watch_session, script_file = super_cli._write_watch_script("utw") + self.assertEqual(watch_session, "utw-watch") + text = Path(script_file).read_text(encoding="utf-8") + self.assertIn('target="utw"', text) + self.assertIn('exit_file="/tmp/utw.exit"', text) + self.assertIn("capture-pane", text) + self.assertIn('send-keys -t "$target" "1" Enter', text) + self.assertIn("max=200", text) + self.assertIn("has-session", text) + + def test_watch_missing_session_exits_2(self): + def fake_run(argv, **kw): + if "has-session" in argv: + return FakeCompleted(1) + return FakeCompleted(0) + + ns = argparse.Namespace(session="nope-missing") + with mock.patch.object(super_cli.shutil, "which", side_effect=self._which), \ + mock.patch.object(super_cli.subprocess, "run", side_effect=fake_run): + with self.assertRaises(SystemExit) as cm: + super_cli.cmd_watch(ns) + self.assertEqual(cm.exception.code, 2) + + def test_watch_spawns_watcher_session(self): + calls = [] + + def fake_run(argv, **kw): + calls.append(argv) + return FakeCompleted(0) + + ns = argparse.Namespace(session="utw") + with mock.patch.object(super_cli.shutil, "which", side_effect=self._which), \ + mock.patch.object(super_cli.subprocess, "run", side_effect=fake_run): + buf = io.StringIO() + with redirect_stdout(buf): + super_cli.cmd_watch(ns) + out = buf.getvalue() + self.assertIn("watching: utw", out) + self.assertIn("watcher: utw-watch", out) + new_session = [c for c in calls if "new-session" in c] + self.assertEqual(len(new_session), 1) + self.assertIn("utw-watch", new_session[0]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_box_runtime.py b/tests/test_box_runtime.py new file mode 100644 index 0000000..a2ca158 --- /dev/null +++ b/tests/test_box_runtime.py @@ -0,0 +1,342 @@ +#!/usr/bin/env python3 +"""test_box_runtime.py β€” Runtime state sensing + `box runtime` management. + +Covers: runtime_state classification (approval-pending/working/open-prompt), +muse argv approval-posture parsing, runtime_rows assembly (mocked tmux), +and the `box runtime` CLI surface. +""" + +import json +import sys +import unittest +from pathlib import Path +from unittest import mock + +REPO_ROOT = Path("/home/super/Projects/NetVM") +BIN_DIR = REPO_ROOT / "bin" +sys.path.insert(0, str(BIN_DIR)) + +import muse_choice_watcher as w + +PROMPT = "❯" # Muse TUI input glyph (U+276F) + +STATE_OPEN = ( + "Some completed agent output here.\n" + "\n" + "───────────────────────────────────\n" + + PROMPT + "\n" + "───────────────────────────────────\n" + " muse-spark-1.3-con… Β· Auto-review\n" +) + +STATE_WORKING = ( + "Partial agent output...\n" + "\n" + "… β€” running (10s Β· esc to interrupt)\n" + "\n" + "───────────────────────────────────\n" + + PROMPT + "\n" + "───────────────────────────────────\n" + " muse-spark-1.3-con… Β· Auto-review\n" +) + +STATE_WORKING_CUT = ( + "Partial agent output...\n" + "β—† Calling tools (5m 53s Β· esc to in\n" + "\n" + "───────────────────────────────────\n" + + PROMPT + "\n" +) + +STATE_APPROVAL = ( + "───────────────────────────────────\n" + "Would you like to run the following\n" + "\n" + " $ tmux capture-pane -p\n" + "\n" + "β€Ί 1. Yes, proceed (y)\n" + " 2. No, and tell Muse Code what to do instead\n" +) + +STATE_SHELL = "[super@bl NetVM]$ printf 'hi'\nhi\n[super@bl NetVM]$ " + + +class TestRuntimeState(unittest.TestCase): + def test_open_prompt(self): + st = w.runtime_state(STATE_OPEN) + self.assertEqual(st["state"], "open-prompt") + self.assertIsNone(st["match"]) + + def test_working(self): + st = w.runtime_state(STATE_WORKING) + self.assertEqual(st["state"], "working") + + def test_working_edge_cut_indicator(self): + st = w.runtime_state(STATE_WORKING_CUT) + self.assertEqual(st["state"], "working") + + def test_approval_pending(self): + st = w.runtime_state(STATE_APPROVAL) + self.assertEqual(st["state"], "approval-pending") + self.assertEqual(st["match"]["kind"], "muse-approval") + self.assertEqual(st["match"]["key"], "1") + + def test_approval_beats_working(self): + st = w.runtime_state(STATE_WORKING + STATE_APPROVAL) + self.assertEqual(st["state"], "approval-pending") + + def test_working_beats_open_prompt(self): + # A working pane still renders its prompt footer. + st = w.runtime_state(STATE_WORKING) + self.assertEqual(st["state"], "working") + + def test_shell_is_unknown(self): + self.assertEqual(w.runtime_state(STATE_SHELL)["state"], "unknown") + + def test_empty_is_unknown(self): + self.assertEqual(w.runtime_state("")["state"], "unknown") + self.assertEqual(w.runtime_state(None)["state"], "unknown") + + +class TestApprovalFlags(unittest.TestCase): + def test_bare_is_not_auto(self): + p = w.muse_approval_flags(["/home/super/.local/bin/muse-bin-1.4.3"]) + self.assertFalse(p["auto_approve"]) + self.assertEqual(p["flags"], []) + + def test_yolo(self): + p = w.muse_approval_flags(["muse", "--yolo"]) + self.assertTrue(p["auto_approve"]) + self.assertIn("yolo", p["flags"]) + + def test_disable_approval(self): + p = w.muse_approval_flags(["muse", "--disable-approval"]) + self.assertTrue(p["auto_approve"]) + + def test_approval_mode_never(self): + p = w.muse_approval_flags(["muse", "--approval-mode", "never"]) + self.assertTrue(p["auto_approve"]) + self.assertIn("approval-mode=never", p["flags"]) + + def test_approval_mode_equals(self): + p = w.muse_approval_flags(["muse", "--approval-mode=never"]) + self.assertTrue(p["auto_approve"]) + + def test_approval_mode_on_request_is_not_auto(self): + p = w.muse_approval_flags(["muse", "--approval-mode", "on-request"]) + self.assertFalse(p["auto_approve"]) + self.assertIn("approval-mode=on-request", p["flags"]) + + def test_empty_argv(self): + p = w.muse_approval_flags([]) + self.assertFalse(p["auto_approve"]) + + +def _tmux_result(returncode=0, stdout="", stderr=""): + r = mock.Mock() + r.returncode = returncode + r.stdout = stdout + r.stderr = stderr + return r + + +class TestRuntimeRows(unittest.TestCase): + LISTING = ("muse\t1\t%37\tmuse-bin-1.4.3-R5018.1\t2880158\t71\t27\n" + "muse\t1\t%38\tbash\t2880200\t100\t30\n") + + def _patched(self, tmux_stdout=LISTING, tmux_rc=0, captures=None, + children=None, cmdlines=None, watcher=None): + captures = captures or {} + cmdlines = cmdlines or {} + children = children or {} + return (mock.patch.object(w, "_tmux", return_value=_tmux_result( + tmux_rc, tmux_stdout)), + mock.patch.object(w, "capture_pane", + side_effect=lambda s, p: captures.get(p)), + mock.patch.object(w, "_child_pids", + side_effect=lambda p: children.get(p, [])), + mock.patch.object(w, "_cmdline", + side_effect=lambda p: cmdlines.get(p, [])), + mock.patch.object(w, "is_running", return_value=watcher)) + + def test_rows_shape(self): + patches = self._patched( + captures={"%37": STATE_OPEN, "%38": STATE_SHELL}, + children={2880158: [2881158]}, + cmdlines={2881158: ["/home/super/.local/bin/muse-bin-1.4.3", + "--disable-approval"]}, + watcher=1234) + with patches[0], patches[1], patches[2], patches[3], patches[4]: + rows = w.runtime_rows("/tmp/sock") + self.assertEqual(len(rows), 2) + muse = rows[0] + self.assertEqual(muse["pane"], "%37") + self.assertTrue(muse["is_muse"]) + self.assertTrue(muse["auto_approve"]) + self.assertEqual(muse["approval_flags"], ["disable-approval"]) + self.assertEqual(muse["state"], "open-prompt") + self.assertTrue(muse["watcher_alive"]) + self.assertEqual(muse["watcher_pid"], 1234) + self.assertEqual(muse["width"], 71) + self.assertEqual(muse["height"], 27) + self.assertFalse(muse["squeezed"]) + shell = rows[1] + self.assertFalse(shell["is_muse"]) + self.assertIsNone(shell["auto_approve"]) + self.assertEqual(shell["state"], "unknown") + + def test_bare_muse_reports_not_auto(self): + patches = self._patched( + captures={"%37": STATE_WORKING, "%38": STATE_SHELL}, + children={2880158: [2881158]}, + cmdlines={2881158: ["/home/super/.local/bin/muse-bin-1.4.3"]}) + with patches[0], patches[1], patches[2], patches[3], patches[4]: + rows = w.runtime_rows("/tmp/sock") + self.assertFalse(rows[0]["auto_approve"]) + self.assertEqual(rows[0]["state"], "working") + self.assertFalse(rows[0]["watcher_alive"]) + + def test_approval_pending_row_carries_kind(self): + patches = self._patched(captures={"%37": STATE_APPROVAL, + "%38": STATE_SHELL}) + with patches[0], patches[1], patches[2], patches[3], patches[4]: + rows = w.runtime_rows("/tmp/sock") + self.assertEqual(rows[0]["state"], "approval-pending") + self.assertEqual(rows[0]["prompt_kind"], "muse-approval") + self.assertEqual(rows[0]["prompt_key"], "1") + + def test_tmux_failure_returns_empty(self): + patches = self._patched(tmux_rc=1, tmux_stdout="") + with patches[0], patches[1], patches[2], patches[3], patches[4]: + self.assertEqual(w.runtime_rows("/tmp/sock"), []) + + def test_vanished_pane_skipped(self): + patches = self._patched(captures={"%37": None, "%38": STATE_SHELL}) + with patches[0], patches[1], patches[2], patches[3], patches[4]: + rows = w.runtime_rows("/tmp/sock") + self.assertEqual([r["pane"] for r in rows], ["%38"]) + + def test_pane_state_found_and_missing(self): + patches = self._patched(captures={"%37": STATE_OPEN, + "%38": STATE_SHELL}) + with patches[0], patches[1], patches[2], patches[3], patches[4]: + hit = w.pane_state("/tmp/sock", "%37") + miss = w.pane_state("/tmp/sock", "%99") + self.assertEqual(hit["pane"], "%37") + self.assertEqual(miss["error"], "no_such_pane") + + def test_rows_flag_squeezed(self): + listing = ("muse\t1\t%37\tmuse-bin-1.4\t2880158\t35\t7\n" + "muse\t1\t%38\tbash\t2880200\t35\t7\n") + patches = self._patched( + tmux_stdout=listing, + captures={"%37": STATE_OPEN, "%38": STATE_SHELL}) + with patches[0], patches[1], patches[2], patches[3], patches[4]: + rows = w.runtime_rows("/tmp/sock") + self.assertTrue(rows[0]["squeezed"]) + self.assertEqual(rows[0]["width"], 35) + self.assertEqual(rows[0]["height"], 7) + self.assertTrue(rows[1]["squeezed"]) + + +class TestSpreadTargets(unittest.TestCase): + def _row(self, pane, is_muse, squeezed): + return {"socket": "/tmp/s", "session": "muse", "window": "1", + "pane": pane, "is_muse": is_muse, "squeezed": squeezed} + + def test_selects_squeezed_muse_only(self): + rows = [self._row("%22", True, False), + self._row("%23", True, True), + self._row("%38", False, True)] + targets = w.spread_targets(rows) + self.assertEqual([t["pane"] for t in targets], ["%23"]) + + def test_empty_when_nothing_squeezed(self): + rows = [self._row("%22", True, False)] + self.assertEqual(w.spread_targets(rows), []) + + def test_tolerates_missing_keys(self): + self.assertEqual(w.spread_targets([{"pane": "%1"}]), []) + + +class TestBoxRuntimeCLI(unittest.TestCase): + def _box(self, *argv, timeout=60): + import subprocess + cmd = [sys.executable, str(BIN_DIR / "super-cli.py"), + "runtime"] + list(argv) + return subprocess.run(cmd, capture_output=True, text=True, + timeout=timeout) + + def test_list_empty_socket_json(self): + r = self._box("list", "--socket", "/nonexistent.sock", "--json") + self.assertEqual(r.returncode, 0, r.stderr[:500]) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertEqual(data["runtimes"], []) + + def test_list_muse_only_flag_accepted(self): + r = self._box("list", "--socket", "/nonexistent.sock", "--json", + "--muse-only") + self.assertEqual(r.returncode, 0, r.stderr[:500]) + self.assertTrue(json.loads(r.stdout)["ok"]) + + def test_subcommand_help(self): + for sub in ("list", "send", "launch", "layout", "spread"): + r = self._box(sub, "--help") + self.assertEqual(r.returncode, 0, sub) + + def test_launch_dry_run_injects_approve(self): + r = self._box("launch", "--session", "probe-x", + "--dry-run", "--json") + self.assertEqual(r.returncode, 0, r.stderr[:500]) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertTrue(data["dry_run"]) + self.assertEqual(data["injected"], ["--disable-approval"]) + self.assertIn("--disable-approval", data["cmdline"]) + self.assertIn("muse-code", data["cmdline"]) + + def test_launch_dry_run_respects_caller_flags(self): + r = self._box("launch", "--session", "probe-x", + "--dry-run", "--json", "--", "--yolo") + self.assertEqual(r.returncode, 0, r.stderr[:500]) + data = json.loads(r.stdout) + self.assertEqual(data["injected"], []) + self.assertIn("--yolo", data["cmdline"]) + self.assertNotIn("--disable-approval", data["cmdline"]) + + def test_send_missing_pane_json(self): + r = self._box("send", "--socket", "/nonexistent.sock", + "%99", "hi", "--json") + self.assertEqual(r.returncode, 0, r.stderr[:500]) + data = json.loads(r.stdout) + self.assertFalse(data["ok"]) + self.assertEqual(data["error"], "no_such_pane") + + def test_layout_empty_socket_json(self): + r = self._box("layout", "--socket", "/nonexistent.sock", "--json") + self.assertEqual(r.returncode, 0, r.stderr[:500]) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertEqual(data["runtimes"], []) + self.assertIn("width", data["minimum"]) + self.assertIn("height", data["minimum"]) + + def test_spread_empty_socket_json(self): + r = self._box("spread", "--socket", "/nonexistent.sock", "--json") + self.assertEqual(r.returncode, 0, r.stderr[:500]) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertEqual(data["spread"], []) + + def test_spread_dry_run_empty_socket_json(self): + r = self._box("spread", "--socket", "/nonexistent.sock", + "--dry-run", "--json") + self.assertEqual(r.returncode, 0, r.stderr[:500]) + data = json.loads(r.stdout) + self.assertTrue(data["dry_run"]) + self.assertEqual(data["targets"], []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_fleet_heal.py b/tests/test_fleet_heal.py new file mode 100644 index 0000000..3cc1403 --- /dev/null +++ b/tests/test_fleet_heal.py @@ -0,0 +1,309 @@ +"""Tests for `box fleet heal` and `box watchdog` in super-cli.py. + +Heal drives lock cleanup, watchdog-timer install, tunnel restart, and +egress/relay verification; watchdog status/run expose the systemd +watchdog layer. Shell-outs are faked; no sudo/systemctl/netns touch +the host. +""" +import argparse +import importlib.util +import io +import json +import sys +import unittest +from contextlib import redirect_stdout +from pathlib import Path +from unittest import mock + +REPO_ROOT = Path(__file__).resolve().parent.parent +SPEC = importlib.util.spec_from_file_location( + "super_cli_heal", REPO_ROOT / "bin" / "super-cli.py") +super_cli = importlib.util.module_from_spec(SPEC) +SPEC.loader.exec_module(super_cli) + + +def _ns(**over): + kw = dict(node=None, target=None, json=True) + kw.update(over) + return argparse.Namespace(**kw) + + +class FakeSh: + """Programmable stand-in for super_cli._sh; records calls.""" + + def __init__(self): + self.calls = [] + self.handlers = [] + + def on(self, *needles, rc=0, out=""): + self.handlers.append((needles, rc, out)) + return self + + def __call__(self, cmd, timeout=120, input_text=None): + self.calls.append((list(cmd), timeout, input_text)) + blob = " ".join(cmd) + for needles, rc, out in self.handlers: + if all(n in blob for n in needles): + return rc, out() if callable(out) else out + raise AssertionError("unexpected command: %r" % (cmd,)) + + +class HealBase(unittest.TestCase): + def setUp(self): + import tempfile + self.tmp = tempfile.TemporaryDirectory() + self.addCleanup(self.tmp.cleanup) + self.root = Path(self.tmp.name) + self.systemd = self.root / "systemd" + self.systemd.mkdir() + self.locks = self.root / "locks" + self.locks.mkdir() + self.patchers = [ + mock.patch.object(super_cli, "SYSTEMD_SYSTEM_DIR", self.systemd), + mock.patch.object(super_cli, "WATCHDOG_LOCK_TMPL", + str(self.locks / "chromebox-watchdog-{node}.lock")), + ] + for p in self.patchers: + p.start() + self.addCleanup(self._unpatch) + self.sh = FakeSh() + self.sh_mock = mock.patch.object(super_cli, "_sh", self.sh) + self.sh_mock.start() + self.addCleanup(self.sh_mock.stop) + + def _unpatch(self): + for p in self.patchers: + p.stop() + + def run_heal(self, node="dev", as_json=True): + buf = io.StringIO() + with redirect_stdout(buf): + with self.assertRaises(SystemExit) as cm: + super_cli.cmd_fleet_heal(_ns(node=node, json=as_json)) + return cm.exception.code, buf.getvalue() + + +class FleetHealTests(HealBase): + def _healthy_sh(self): + (self.sh + .on("systemctl", "enable", "--now", rc=0, out="") + .on("netvm-node-up.sh", "dev", rc=0, + out="tunnel already up (egress=1.2.3.4), skipping handshake wait\n" + "node=dev netns=warp-dev egress=1.2.3.4") + .on("curl", rc=0, out="ip=1.2.3.4\nfoo=bar")) + + def test_heal_recovered(self): + (self.systemd / "chromebox-watchdog-dev.timer").write_text("x") + self._healthy_sh() + with mock.patch.object(super_cli, "probe_cdp_status", + return_value={"ok": True, "latency_ms": 12}): + code, out = self.run_heal() + self.assertEqual(code, 0) + data = json.loads(out) + self.assertEqual(data["verdict"], "RECOVERED") + self.assertTrue(data["ok"]) + self.assertEqual([s["step"] for s in data["steps"]], + ["lock", "timer", "tunnel", "egress", "relay"]) + self.assertTrue(all(s["ok"] for s in data["steps"])) + + def test_heal_down_when_egress_fails(self): + (self.systemd / "chromebox-watchdog-dev.timer").write_text("x") + (self.sh + .on("systemctl", "enable", "--now", rc=0, out="") + .on("netvm-node-up.sh", "dev", rc=0, + out="no handshake yet (endpoint=162.159.192.1)\n" + "node=dev netns=warp-dev egress=unknown") + .on("curl", rc=7, out="curl: (7) couldn't connect")) + with mock.patch.object(super_cli, "probe_cdp_status", + return_value={"ok": False, "error": "refused", + "latency_ms": None}): + code, out = self.run_heal() + self.assertEqual(code, 1) + data = json.loads(out) + self.assertEqual(data["verdict"], "DOWN") + self.assertFalse(data["ok"]) + by_step = {s["step"]: s for s in data["steps"]} + self.assertFalse(by_step["tunnel"]["ok"]) + self.assertFalse(by_step["egress"]["ok"]) + + def test_heal_down_prints_identity_guidance(self): + (self.systemd / "chromebox-watchdog-dev.timer").write_text("x") + (self.sh + .on("systemctl", rc=0, out="") + .on("netvm-node-up.sh", rc=0, out="node=dev egress=unknown") + .on("curl", rc=7, out="fail")) + with mock.patch.object(super_cli, "probe_cdp_status", + return_value={"ok": False, "error": "x", + "latency_ms": None}): + code, out = self.run_heal(as_json=False) + self.assertEqual(code, 1) + self.assertIn("netvm-new-identity.sh dev", out) + + def test_heal_installs_missing_timer(self): + seen = {} + + def fake(cmd, timeout=120, input_text=None): + blob = " ".join(cmd) + self.sh.calls.append((list(cmd), timeout, input_text)) + if "tee" in blob: + seen["tee_target"] = cmd[-1] + seen["tee_input"] = input_text + return 0, "" + if "daemon-reload" in blob: + seen["reload"] = True + return 0, "" + if "enable" in blob: + return 0, "" + if "netvm-node-up.sh" in blob: + return 0, "node=dev egress=1.2.3.4" + if "curl" in blob: + return 0, "ip=1.2.3.4" + raise AssertionError("unexpected: %r" % (cmd,)) + + with mock.patch.object(super_cli, "_sh", fake): + with mock.patch.object( + super_cli, "probe_cdp_status", + return_value={"ok": True, "latency_ms": 3}): + code, _ = self.run_heal() + self.assertEqual(code, 0) + self.assertEqual(seen["tee_target"], + str(self.systemd / "chromebox-watchdog-dev.timer")) + self.assertIn("Unit=chromebox-watchdog@dev.service", seen["tee_input"]) + self.assertTrue(seen["reload"]) + + def test_heal_clears_unwritable_lock(self): + (self.systemd / "chromebox-watchdog-dev.timer").write_text("x") + lock = self.locks / "chromebox-watchdog-dev.lock" + lock.write_text("") + lock.chmod(0o444) + self._healthy_sh() + self.sh.on("rm", str(lock), rc=0, out="") + with mock.patch.object(super_cli, "probe_cdp_status", + return_value={"ok": True, "latency_ms": 3}): + code, out = self.run_heal() + self.assertEqual(code, 0) + data = json.loads(out) + by_step = {s["step"]: s for s in data["steps"]} + self.assertTrue(by_step["lock"]["ok"]) + self.assertIn("removed stale lock", by_step["lock"]["detail"]) + self.assertTrue(any("rm" in " ".join(c[0]) for c in self.sh.calls)) + + def test_heal_rejects_unknown_node(self): + buf = io.StringIO() + with redirect_stdout(buf): + with self.assertRaises(SystemExit) as cm: + super_cli.cmd_fleet_heal(_ns(node="ghost")) + self.assertEqual(cm.exception.code, 1) + + +class ShHelperTests(unittest.TestCase): + def test_timeout_and_oserror(self): + import subprocess as real_subprocess + with mock.patch.object(super_cli.subprocess, "run", + side_effect=real_subprocess.TimeoutExpired("x", 1)): + self.assertEqual(super_cli._sh(["x"], timeout=1)[0], 124) + with mock.patch.object(super_cli.subprocess, "run", + side_effect=OSError("nope")): + self.assertEqual(super_cli._sh(["x"])[0], 127) + + +class WatchdogTests(unittest.TestCase): + def setUp(self): + self.sh = FakeSh() + self.sh_mock = mock.patch.object(super_cli, "_sh", self.sh) + self.sh_mock.start() + self.addCleanup(self.sh_mock.stop) + + def _unit_states(self, missing=()): + def fake(cmd, timeout=120, input_text=None): + self.sh.calls.append((list(cmd), timeout, input_text)) + unit = cmd[-1] + if "is-enabled" in cmd or "is-active" in cmd: + return (1, "") if unit in missing else (0, "") + if cmd[:3] == ["sudo", "systemctl", "start"]: + return 0, "" + raise AssertionError("unexpected: %r" % (cmd,)) + return fake + + def test_status_json(self): + stub = mock.Mock() + stub.collect.return_value = { + n: {"browser": "healthy", "browser_detail": "silent", + "cdp": "healthy", "cdp_detail": "silent"} + for n in super_cli.VALID_NODES} + with mock.patch.object(super_cli, "_sh", self._unit_states( + missing=("chromebox-watchdog-def.timer",))): + with mock.patch.dict(sys.modules, {"host_evidence": stub}): + buf = io.StringIO() + with redirect_stdout(buf): + super_cli.cmd_watchdog_status(_ns(json=True)) + data = json.loads(buf.getvalue()) + self.assertTrue(data["ok"]) + self.assertEqual(len(data["nodes"]), 6) + by_node = {n["node"]: n for n in data["nodes"]} + self.assertFalse(by_node["def"]["enabled"]) + self.assertFalse(by_node["def"]["active"]) + self.assertTrue(by_node["dev"]["active"]) + self.assertEqual(by_node["dev"]["browser"], "healthy") + self.assertIn("timer", data["relay"]) + + def test_status_survives_missing_evidence(self): + with mock.patch.object(super_cli, "_sh", self._unit_states()): + with mock.patch.dict(sys.modules, {"host_evidence": None}): + buf = io.StringIO() + with redirect_stdout(buf): + super_cli.cmd_watchdog_status(_ns(json=True)) + data = json.loads(buf.getvalue()) + self.assertTrue(data["ok"]) + self.assertEqual(data["nodes"][0]["browser"], "unknown") + + def test_run_node_and_relay(self): + for target, unit in (("dev", "chromebox-watchdog@dev.service"), + ("relay", "cdp-relay-watchdog.service")): + with self.subTest(target=target): + with mock.patch.object(super_cli, "_sh", + self._unit_states()) as _: + buf = io.StringIO() + with redirect_stdout(buf): + with self.assertRaises(SystemExit) as cm: + super_cli.cmd_watchdog_run( + _ns(target=target, json=True)) + self.assertEqual(cm.exception.code, 0) + data = json.loads(buf.getvalue()) + self.assertTrue(data["ok"]) + self.assertEqual(data["unit"], unit) + + def test_run_rejects_bad_target(self): + with self.assertRaises(SystemExit) as cm: + super_cli.cmd_watchdog_run(_ns(target="ghost")) + self.assertEqual(cm.exception.code, 1) + + def test_run_reports_start_failure(self): + def fake(cmd, timeout=120, input_text=None): + return 1, "Failed to start" + + with mock.patch.object(super_cli, "_sh", fake): + buf = io.StringIO() + with redirect_stdout(buf): + with self.assertRaises(SystemExit) as cm: + super_cli.cmd_watchdog_run(_ns(target="dev", json=True)) + self.assertEqual(cm.exception.code, 1) + self.assertFalse(json.loads(buf.getvalue())["ok"]) + + +class ParserTests(unittest.TestCase): + def test_fleet_heal_parses(self): + args = super_cli.build_parser().parse_args(["fleet", "heal", "dev"]) + self.assertEqual((args.domain, args.action, args.node), + ("fleet", "heal", "dev")) + + def test_watchdog_parses(self): + args = super_cli.build_parser().parse_args(["watchdog", "run", "relay"]) + self.assertEqual((args.domain, args.action, args.target), + ("watchdog", "run", "relay")) + args = super_cli.build_parser().parse_args(["watchdog"]) + self.assertEqual((args.domain, args.action), ("watchdog", "status")) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_hatch_menu.py b/tests/test_hatch_menu.py index 799cd93..88e51a4 100644 --- a/tests/test_hatch_menu.py +++ b/tests/test_hatch_menu.py @@ -129,6 +129,17 @@ class DialogTests(unittest.TestCase): side_effect=RuntimeError("boom")): self.assertIsNone(dialog.dialog_text(mock.Mock())) + def test_dialog_text_rejects_non_string(self): + with mock.patch.object(dialog, "cdp_evaluate", return_value=123): + self.assertIsNone(dialog.dialog_text(mock.Mock())) + + def test_describe_rows_rejects_non_list(self): + with mock.patch.object(dialog, "goto_tab", return_value=True), \ + mock.patch.object(dialog, "cdp_evaluate", + return_value="error"): + self.assertEqual(dialog.describe_rows(mock.Mock(), + "Permissions"), []) + def test_click_row_success_and_no_match(self): with mock.patch.object(dialog, "cdp_evaluate", side_effect=["CLICKED", "CLICKED", @@ -147,6 +158,11 @@ class DialogTests(unittest.TestCase): "Permissions")) self.assertEqual(cdp.call_count, 1) + def test_row_js_excludes_tab_rail(self): + blob = json.dumps(dialog.TAB_NAMES) + self.assertIn(blob, dialog.JS_CLICK_ROW_TMPL) + self.assertIn(blob, dialog.JS_DESCRIBE_ROWS) + def test_describe_rows_maps(self): rows = [{"name": "Websites", "help": "1 site", "extra": ""}] with mock.patch.object(dialog, "goto_tab", return_value=True), \ @@ -183,8 +199,8 @@ class ControlsTests(unittest.TestCase): on = [{"heading": "Connector defaults", "value": "auto_allow", "checked": True}] with mock.patch.object(controls, "cdp_evaluate", - side_effect=["CLICKED", off, - {"x": 5, "y": 6}, on]), \ + side_effect=["CLICKED"] + [off] * 10 + + [{"x": 5, "y": 6}, on]), \ mock.patch.object(controls, "real_click") as click: self.assertTrue(controls.set_radio_by_heading( mock.Mock(), "Connector defaults", "auto_allow")) @@ -194,7 +210,8 @@ class ControlsTests(unittest.TestCase): off = [{"heading": "Connector defaults", "value": "auto_allow", "checked": False}] with mock.patch.object(controls, "cdp_evaluate", - side_effect=["CLICKED", off, None]), \ + side_effect=["CLICKED"] + [off] * 10 + + [None]), \ mock.patch.object(controls, "real_click"): self.assertFalse(controls.set_radio_by_heading( mock.Mock(), "Connector defaults", "auto_allow")) @@ -234,6 +251,22 @@ class ControlsTests(unittest.TestCase): self.assertTrue(controls.set_switch(mock.Mock(), "Transparent proxy", True)) + def test_set_switch_polls_then_lands(self): + off = [{"label": "Transparent proxy row", "aria": "", + "checked": False}] + on = [{"label": "Transparent proxy row", "aria": "", + "checked": True}] + with mock.patch.object(controls, "cdp_evaluate", + side_effect=[off, "CLICKED", off, off, on]): + self.assertTrue(controls.set_switch(mock.Mock(), + "Transparent proxy", True)) + + def test_listers_reject_wrong_types(self): + with mock.patch.object(controls, "cdp_evaluate", + return_value="error"): + self.assertEqual(controls.list_radios(mock.Mock()), []) + self.assertEqual(controls.list_switches(mock.Mock()), []) + class TogglesTests(unittest.TestCase): def test_resolve_static_website_protocol(self): @@ -248,6 +281,14 @@ class TogglesTests(unittest.TestCase): "permissions.protocols:Agent Skills endpoints") self.assertEqual(spec["label"], "Agent Skills endpoints") + def test_resolve_protocol_network_and_substring(self): + spec = toggles.resolve_toggle("permissions.protocols:outbound-ssh") + self.assertEqual(spec["label"], "Outbound SSH") + spec = toggles.resolve_toggle("permissions.protocols:ssh") + self.assertEqual(spec["label"], "Outbound SSH") + with self.assertRaises(toggles.MenuError): + toggles.resolve_toggle("permissions.protocols:mail") + def test_resolve_unknowns_raise_before_cdp(self): for bad in ("nope", "", "permissions.websites:", "permissions.protocols:nope"): @@ -315,6 +356,23 @@ class TogglesTests(unittest.TestCase): res = toggles.set_toggle("pip", "permissions.web_access", "always_ask") self.assertEqual((res["ok"], res["value"]), (True, "always_ask")) + self.assertNotIn("readback_only", res) + + def test_set_toggle_readback_only_success(self): + with mock.patch.object(toggles, "get_cdp_ws", + return_value=(mock.Mock(), {})), \ + mock.patch.object(dialog, "open_settings", + return_value=True), \ + mock.patch.object(dialog, "goto_tab", return_value=True), \ + mock.patch.object(controls, "set_radio_by_heading", + return_value=False), \ + mock.patch.object(toggles, "get_toggle", + return_value={"ok": True, + "value": "always_ask"}): + res = toggles.set_toggle("pip", "permissions.web_access", + "always_ask") + self.assertTrue(res["ok"]) + self.assertTrue(res["readback_only"]) def test_set_toggle_readback_mismatch_fails(self): with mock.patch.object(toggles, "get_cdp_ws", @@ -405,6 +463,15 @@ class TabContractTests(unittest.TestCase): "Model Context Protocol servers (SSE)") self.assertEqual(permissions.resolve_protocol( "Agent Skills endpoints"), "Agent Skills endpoints") + self.assertEqual(permissions.resolve_protocol("outbound-ssh"), + "Outbound SSH") + self.assertEqual(permissions.resolve_protocol("Outbound SSH"), + "Outbound SSH") + self.assertEqual(permissions.resolve_protocol("ssh"), + "Outbound SSH") + self.assertEqual(permissions.resolve_protocol("tcp"), + "Other TCP connections") + self.assertIsNone(permissions.resolve_protocol("mail")) self.assertIsNone(permissions.resolve_protocol("nope")) self.assertIsNone(permissions.resolve_protocol("")) self.assertIsNone(permissions.resolve_protocol(None)) @@ -415,12 +482,76 @@ class TabContractTests(unittest.TestCase): mock.Mock(), "x.com", "Sometimes")) ev.assert_not_called() + def test_set_website_noop_without_click(self): + rows = [{"host": "x.com", "mode": "Allow", "x": 1, "y": 2}] + with mock.patch.object(permissions, "_websites_raw", + return_value=rows), \ + mock.patch.object(permissions, "_back_to_root", + return_value=True), \ + mock.patch.object(permissions, "real_click") as click, \ + mock.patch("time.sleep"): + self.assertTrue(permissions.set_website_mode( + mock.Mock(), "x.com", "Allow")) + click.assert_not_called() + + def test_set_website_absent_host_fails(self): + with mock.patch.object(permissions, "_websites_raw", + return_value=[]), \ + mock.patch.object(permissions, "_back_to_root", + return_value=True), \ + mock.patch("time.sleep"): + self.assertFalse(permissions.set_website_mode( + mock.Mock(), "y.com", "Ask")) + + def test_set_website_ask_verifies_by_absence(self): + rows = [{"host": "x.com", "mode": "Allow", "x": 1, "y": 2}] + items = [{"text": "Allow"}, {"text": "Ask"}, {"text": "Deny"}] + with mock.patch.object(permissions, "_websites_raw", + return_value=rows), \ + mock.patch.object(permissions, "_back_to_root", + return_value=True), \ + mock.patch.object(permissions, "real_click"), \ + mock.patch.object(permissions, "_eval", + side_effect=[items, "CLICKED", []]), \ + mock.patch("time.sleep"): + self.assertTrue(permissions.set_website_mode( + mock.Mock(), "x.com", "Ask")) + + def test_stable_rows_waits_for_agreement(self): + rows = [{"host": "x.com"}] + with mock.patch.object(permissions, "_eval", + side_effect=[None, rows, rows]), \ + mock.patch("time.sleep"): + self.assertEqual( + permissions._stable_rows(mock.Mock(), "js"), rows) + + def test_stable_rows_gives_up(self): + with mock.patch.object(permissions, "_eval", return_value=None), \ + mock.patch("time.sleep"): + self.assertIsNone( + permissions._stable_rows(mock.Mock(), "js")) + def test_back_to_root_prefers_back_affordance(self): with mock.patch.object(dialog, "go_back", return_value=True), \ mock.patch.object(dialog, "dialog_text", return_value="Manage permissions\nrows"): self.assertTrue(permissions._back_to_root(mock.Mock())) + def test_theme_values_match_live(self): + self.assertIn("avatar", general.THEME_VALUES) + self.assertNotIn("match", general.THEME_VALUES) + + def test_data_controls_pinned_label(self): + self.assertEqual(data_controls.SWITCH_LABEL, + "Help improve our AI models") + sws = [{"label": "Something else", "aria": "", "checked": False}, + {"label": "Help improve our AI models", "aria": "", + "checked": True}] + with mock.patch.object(dialog, "goto_tab", return_value=True), \ + mock.patch.object(controls, "list_switches", + return_value=sws): + self.assertTrue(data_controls.ai_improvement(mock.Mock())) + def test_data_controls_single_switch(self): one = [{"label": "Help improve the model", "aria": "", "checked": True}] diff --git a/tests/test_muse_choice_watcher.py b/tests/test_muse_choice_watcher.py new file mode 100644 index 0000000..36ccbc1 --- /dev/null +++ b/tests/test_muse_choice_watcher.py @@ -0,0 +1,1090 @@ +#!/usr/bin/env python3 +"""test_muse_choice_watcher.py β€” Focused tests for the Muse A/B/C watcher. + +Covers: prompt matching (A-default regex), once-per-prompt answering policy, +socket-namespaced state files (the %0-on-two-sockets collision regression), +and send-keys command construction. +""" + +import sys +import time +import unittest +from pathlib import Path +from unittest import mock + +REPO_ROOT = Path("/home/super/Projects/NetVM") +BIN_DIR = REPO_ROOT / "bin" +sys.path.insert(0, str(BIN_DIR)) + +import muse_choice_watcher as w + + +PROMPT_ABC = """Some agent output here. +How should I proceed? +A. Apply the fix now +B. Show a diff first +C. Skip this file +Reply with A, B, or C: +""" + +PROMPT_AB_WRAPPED = """Long thinking output... +Which approach? +A. Switch coverage tuples to a single +cached registry call with ALL_NODES +B. Keep per-node calls and retry +Your choice (A/B)? +""" + +PROMPT_STALE = ( + "A. Old option one\nB. Old option two\nPick one (A/B)?\n" + + "\n".join("filler line %d" % i for i in range(40)) +) + + +class TestMatcher(unittest.TestCase): + def test_matches_abc_with_cue(self): + m = w.find_choice_prompt(PROMPT_ABC) + self.assertIsNotNone(m) + self.assertEqual(len(m["options"]), 3) + self.assertTrue(m["options"][0].startswith("A.")) + # Cue scan hits the question line above the options first; either cue + # line proves the block was recognized as awaiting a reply. + self.assertIn(m["cue"], ("How should I proceed?", "Reply with A, B, or C:")) + self.assertEqual(len(m["sig"]), 16) + + def test_matches_ab_wrapped(self): + m = w.find_choice_prompt(PROMPT_AB_WRAPPED) + self.assertIsNotNone(m) + self.assertEqual(len(m["options"]), 2) + + def test_rejects_single_option(self): + self.assertIsNone(w.find_choice_prompt("A. Only one option\nSome text\n")) + + def test_rejects_lettered_list_without_cue(self): + text = "A. Apples are red\nB. Bananas are yellow\nJust a grocery list.\n" + self.assertIsNone(w.find_choice_prompt(text)) + + def test_rejects_stale_scrollback(self): + self.assertIsNone(w.find_choice_prompt(PROMPT_STALE)) + + def test_matches_prompt_with_trailing_blanks(self): + # Tall panes pad output with blank lines; a live prompt above the + # padding must still match (scratch-pane regression). + text = PROMPT_ABC + "\n" * 30 + m = w.find_choice_prompt(text) + self.assertIsNotNone(m) + self.assertEqual(len(m["options"]), 3) + + def test_rejects_out_of_order(self): + text = "B. Second thing\nA. First thing\nWhich (A/B)?\n" + self.assertIsNone(w.find_choice_prompt(text)) + + def test_rejects_empty(self): + self.assertIsNone(w.find_choice_prompt("")) + self.assertIsNone(w.find_choice_prompt(None)) + + def test_sig_stable_and_sensitive(self): + a = w.find_choice_prompt(PROMPT_ABC)["sig"] + b = w.find_choice_prompt(PROMPT_ABC)["sig"] + self.assertEqual(a, b) + changed = PROMPT_ABC.replace("Apply the fix now", "Apply the fix later") + c = w.find_choice_prompt(changed)["sig"] + self.assertNotEqual(a, c) + + +PROMPT_YN = """Migrating 12 threads... +Proceed with the migration? (y/n) +""" + +PROMPT_YN_BRACKET = """Target file exists. +Overwrite existing file? [y/N] +""" + +PROMPT_NUMBERED = """Requesting permission for: rm -rf /tmp/x +(1) Allow once +(2) Always allow +Selection: +""" + + +class TestPromptKinds(unittest.TestCase): + def test_letter_kind_and_key(self): + m = w.find_choice_prompt(PROMPT_ABC) + self.assertEqual(m["kind"], "letter") + self.assertEqual(m["key"], "A") + + def test_yn_paren(self): + m = w.find_choice_prompt(PROMPT_YN) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), ("yn", "y")) + + def test_yn_bracket(self): + m = w.find_choice_prompt(PROMPT_YN_BRACKET) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), ("yn", "y")) + + def test_yn_rejects_mid_line_mention(self): + self.assertIsNone(w.find_choice_prompt("use y/n for confirmation\nok\n")) + + def test_yn_rejects_stale(self): + text = "Proceed? (y/n)\n" + "\n".join("filler %d" % i for i in range(10)) + self.assertIsNone(w.find_choice_prompt(text)) + + def test_numbered_kind_and_key(self): + m = w.find_choice_prompt(PROMPT_NUMBERED) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), ("numbered", "1")) + self.assertEqual(len(m["options"]), 2) + + def test_numbered_rejects_single(self): + self.assertIsNone( + w.find_choice_prompt("(1) Only option\nSelection:\n")) + + def test_numbered_rejects_without_cue(self): + self.assertIsNone( + w.find_choice_prompt("(1) one\n(2) two\nSome other text.\n")) + + def test_letter_beats_yn_priority(self): + m = w.find_choice_prompt(PROMPT_ABC + "Proceed? (y/n)\n") + self.assertIsNotNone(m) + self.assertEqual(m["kind"], "letter") + + def test_sigs_namespaced_by_kind(self): + a = w.find_choice_prompt(PROMPT_ABC)["sig"] + b = w.find_choice_prompt(PROMPT_YN)["sig"] + c = w.find_choice_prompt(PROMPT_NUMBERED)["sig"] + self.assertEqual(len({a, b, c}), 3) + + +class TestPollOnce(unittest.TestCase): + def _run(self, captures, dry_run=False, prefill_cap=False): + state = w.WatcherState() + if prefill_cap: + import time + for i in range(w.MAX_ANSWERS_PER_HOUR): + state.record_answer("old-%d" % i, time.time()) + log = mock.Mock() + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", + side_effect=captures) as cap, \ + mock.patch.object(w, "send_answer", + return_value=True) as send, \ + mock.patch.object(w, "audit") as audit: + outcomes = [w._poll_once("/tmp/s", "%1", state, log, + dry_run=dry_run) + for _ in range(len(captures) - 1)] + return state, log, cap, send, audit, outcomes + + def _log_msgs(self, log): + return [c.args[1] for c in log.log.call_args_list] + + def test_letter_answered_with_A(self): + state, log, cap, send, audit, outcomes = self._run( + [PROMPT_ABC, PROMPT_ABC, PROMPT_ABC]) + self.assertEqual(outcomes, ["seen", "answered"]) + send.assert_called_once_with("/tmp/s", "%1", "A", enter=True) + audit.assert_called_once() + self.assertIn("prompt seen", self._log_msgs(log)) + self.assertIn("prompt stable, answering", self._log_msgs(log)) + + def test_yn_answered_with_y(self): + state, log, cap, send, audit, outcomes = self._run( + [PROMPT_YN, PROMPT_YN, PROMPT_YN]) + self.assertEqual(outcomes[-1], "answered") + send.assert_called_once_with("/tmp/s", "%1", "y", enter=True) + self.assertEqual(audit.call_args[0][0], "muse-choice-answered") + self.assertEqual(audit.call_args[1]["extra"]["key"], "y") + + def test_numbered_answered_with_1(self): + state, log, cap, send, audit, outcomes = self._run( + [PROMPT_NUMBERED, PROMPT_NUMBERED, PROMPT_NUMBERED]) + self.assertEqual(outcomes[-1], "answered") + send.assert_called_once_with("/tmp/s", "%1", "1", enter=True) + + def test_dry_run_records_without_sending(self): + state, log, cap, send, audit, outcomes = self._run( + [PROMPT_YN, PROMPT_YN, PROMPT_YN], dry_run=True) + self.assertEqual(outcomes, ["seen", "dry-answered"]) + send.assert_not_called() + audit.assert_not_called() + self.assertEqual(len(state.answered_sigs), 1) + + def test_vanished_prompt_skips_send(self): + state, log, cap, send, audit, outcomes = self._run( + [PROMPT_YN, PROMPT_YN, "something else entirely\n"]) + self.assertEqual(outcomes, ["seen", "vanished"]) + send.assert_not_called() + audit.assert_not_called() + + def test_capped_logs_once(self): + state, log, cap, send, audit, outcomes = self._run( + [PROMPT_YN] * 5, prefill_cap=True) + self.assertTrue(all(o == "capped" for o in outcomes[1:])) + send.assert_not_called() + warns = [c for c in log.log.call_args_list + if c.args[0] == "warn"] + self.assertEqual(len(warns), 1) + + def test_gone_and_capture_failed(self): + state, log = w.WatcherState(), mock.Mock() + with mock.patch.object(w, "pane_exists", return_value=False): + self.assertEqual( + w._poll_once("/tmp/s", "%1", state, log), "gone") + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", return_value=None): + self.assertEqual( + w._poll_once("/tmp/s", "%1", state, log), "capture-failed") + + +class TestAnswerPolicy(unittest.TestCase): + def test_needs_stability(self): + st = w.WatcherState() + m = w.find_choice_prompt(PROMPT_ABC) + now = time.time() + self.assertEqual(st.observe(m, now), "wait") + self.assertEqual(st.observe(m, now), "answer") + + def test_once_per_prompt(self): + st = w.WatcherState() + m = w.find_choice_prompt(PROMPT_ABC) + now = time.time() + st.observe(m, now) + st.observe(m, now) + st.record_answer(m["sig"], now) + self.assertEqual(st.observe(m, now), "none") + # A new prompt answers again. + m2 = w.find_choice_prompt(PROMPT_AB_WRAPPED) + self.assertNotEqual(m2["sig"], m["sig"]) + self.assertEqual(st.observe(m2, now), "wait") + self.assertEqual(st.observe(m2, now), "answer") + + def test_none_resets_pending(self): + st = w.WatcherState() + m = w.find_choice_prompt(PROMPT_ABC) + now = time.time() + st.observe(m, now) + self.assertEqual(st.observe(None, now), "none") + self.assertEqual(st.observe(m, now), "wait") + + def test_hourly_cap(self): + st = w.WatcherState() + now = time.time() + for i in range(w.MAX_ANSWERS_PER_HOUR): + st.record_answer("sig-%d" % i, now) + m = w.find_choice_prompt(PROMPT_ABC) + st.observe(m, now) + self.assertEqual(st.observe(m, now), "capped") + + def test_answered_ttl_allows_recovery(self): + st = w.WatcherState() + now = time.time() + m = w.find_choice_prompt(PROMPT_ABC) + st.observe(m, now) + st.observe(m, now) + st.record_answer(m["sig"], now) + self.assertEqual(st.observe(m, now + 1), "none") + # Same prompt still present past TTL => stuck dialog, re-answer. + self.assertEqual(st.observe(m, now + w.ANSWERED_TTL_SECONDS + 1), + "wait") + self.assertEqual(st.observe(m, now + w.ANSWERED_TTL_SECONDS + 2), + "answer") + + +class TestNamespacing(unittest.TestCase): + """Same pane id on different sockets must never share state files.""" + + def test_pidfile_differs_across_sockets(self): + a = w.pidfile_for("/tmp/tmux-1000/default", "%0") + b = w.pidfile_for("/tmp/tmux-1000/lte", "%0") + self.assertNotEqual(a, b) + + def test_logfile_differs_across_sockets(self): + a = w.logfile_for("/tmp/tmux-1000/default", "%0") + b = w.logfile_for("/tmp/tmux-1000/lte", "%0") + self.assertNotEqual(a, b) + + def test_slug_guards_same_basename(self): + a = w.slug_socket("/tmp/a/default") + b = w.slug_socket("/tmp/b/default") + self.assertNotEqual(a, b) + + +class TestSendAnswer(unittest.TestCase): + def test_sends_A_then_enter_to_exact_pane(self): + calls = [] + + def fake_tmux(sock, *args, timeout=5): + calls.append((sock, args)) + r = mock.Mock() + r.returncode = 0 + return r + + with mock.patch.object(w, "_tmux", side_effect=fake_tmux): + self.assertTrue(w.send_answer("/tmp/tmux-1000/default", "%37")) + self.assertEqual(len(calls), 2) + self.assertEqual(calls[0][0], "/tmp/tmux-1000/default") + self.assertEqual(calls[0][1][:3], ("send-keys", "-t", "%37")) + self.assertEqual(calls[0][1][3], "A") + self.assertEqual(calls[1][1][3], "Enter") + + def test_send_failure_returns_false(self): + def failing(sock, *args, timeout=5): + r = mock.Mock() + r.returncode = 1 + return r + + with mock.patch.object(w, "_tmux", side_effect=failing): + self.assertFalse(w.send_answer("/tmp/tmux-1000/default", "%37")) + + +class TestDesiredState(unittest.TestCase): + def test_default_is_on(self): + # Policy: undefined desired state means auto-approve on. + with mock.patch.object(w, "DESIRED_STATE_FILE", "/nonexistent/x.json"): + st = w.get_desired() + self.assertTrue(st["enabled"]) + self.assertFalse(st["dry_run"]) + + def test_missing_key_defaults_on(self): + import json + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/muse-choices.json" + with open(path, "w") as f: + json.dump({"dry_run": False}, f) + with mock.patch.object(w, "DESIRED_STATE_FILE", path): + self.assertTrue(w.get_desired()["enabled"]) + + def test_explicit_off_is_respected(self): + import json + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/muse-choices.json" + with open(path, "w") as f: + json.dump({"enabled": False, "dry_run": False}, f) + with mock.patch.object(w, "DESIRED_STATE_FILE", path): + self.assertFalse(w.get_desired()["enabled"]) + + def test_set_enabled_roundtrip(self): + import json + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/muse-choices.json" + with mock.patch.object(w, "DESIRED_STATE_FILE", path), \ + mock.patch.object(w, "audit") as audit: + st = w.set_enabled(True, dry_run=True, by="tester") + self.assertTrue(st["enabled"]) + self.assertTrue(st["dry_run"]) + self.assertEqual(st["updated_by"], "tester") + self.assertTrue(w.get_desired()["enabled"]) + audit.assert_called_once() + args, _ = audit.call_args + self.assertEqual(args[0], "muse-choice-enabled") + with open(path) as f: + on_disk = json.load(f) + self.assertTrue(on_disk["enabled"]) + + +class TestAudit(unittest.TestCase): + def test_local_fallback_shape(self): + import json + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/box-ctl.jsonl" + with mock.patch.object(w, "CTL_LOG", path): + w._audit_local("muse-choice-answered", "default:%37", "watcher", + {"sig": "abc"}) + with open(path) as f: + rec = json.loads(f.read()) + self.assertEqual(rec["action"], "muse-choice-answered") + self.assertEqual(rec["type"], "muse-choice") + self.assertEqual(rec["name"], "default:%37") + self.assertEqual(rec["sig"], "abc") + self.assertIn("ts", rec) + + def test_audit_prefers_approvals_module(self): + import sys + fake = mock.Mock() + with mock.patch.dict(sys.modules, {"approvals": fake}): + w.audit("muse-choice-enabled", caller="box", + extra={"dry_run": False}) + fake.log_box_ctl.assert_called_once_with( + "muse-choice-enabled", name=None, caller="box", + extra={"dry_run": False}) + + def test_audit_never_raises(self): + import sys + with mock.patch.dict(sys.modules, {"approvals": None}), \ + mock.patch.object(w, "_audit_local", side_effect=OSError("disk")): + w.audit("muse-choice-answered") # must not raise + + +class TestReconcile(unittest.TestCase): + def _patch_common(self, enabled, dry_run=False): + return (mock.patch.object(w, "get_desired", + return_value={"enabled": enabled, + "dry_run": dry_run}), + mock.patch.object(w, "muse_panes", return_value=["%37"]), + mock.patch.object(w, "_start_detached", return_value=True), + mock.patch.object(w, "status_all", return_value=[]), + mock.patch.object(w, "stop_all", return_value=[]), + mock.patch.object(w, "audit"), + mock.patch.object(w.os.path, "exists", return_value=True)) + + def test_enabled_starts_missing(self): + patches = self._patch_common(True) + with patches[0], patches[1], patches[2] as start, patches[3], \ + patches[4], patches[5] as audit, patches[6], \ + mock.patch.object(w, "is_running", return_value=None): + res = w.reconcile(sockets=["/tmp/sock"]) + start.assert_called_once_with("/tmp/sock", "%37", dry_run=False) + self.assertEqual(res["started"], ["/tmp/sock:%37"]) + audit.assert_called_once() # changed something -> audited + + def test_enabled_skips_running_and_stays_quiet(self): + patches = self._patch_common(True) + with patches[0], patches[1], patches[2] as start, patches[3], \ + patches[4], patches[5] as audit, patches[6], \ + mock.patch.object(w, "is_running", return_value=1234): + res = w.reconcile(sockets=["/tmp/sock"]) + start.assert_not_called() + self.assertEqual(res["already"], ["/tmp/sock:%37"]) + audit.assert_not_called() # no change -> no audit noise + + def test_disabled_stops_all(self): + patches = self._patch_common(False) + with patches[0], patches[1], patches[2] as start, patches[3], \ + mock.patch.object(w, "stop_all", + return_value=[{"pidfile": "x.pid", "pid": 1}]), \ + patches[5] as audit, patches[6]: + res = w.reconcile(sockets=["/tmp/sock"]) + start.assert_not_called() + self.assertFalse(res["enabled"]) + self.assertEqual(len(res["stopped"]), 1) + audit.assert_called_once() + + def test_prunes_dead_pidfiles(self): + patches = self._patch_common(True) + dead = [{"alive": False, "pidfile": "muse-choice-watcher-x.pid"}] + with patches[0], patches[1], patches[2], \ + mock.patch.object(w, "status_all", return_value=dead), \ + patches[4], patches[5], patches[6], \ + mock.patch.object(w, "is_running", return_value=999), \ + mock.patch.object(w.os, "remove") as rm: + res = w.reconcile(sockets=["/tmp/sock"]) + rm.assert_called_once() + self.assertEqual(res["pruned"], ["muse-choice-watcher-x.pid"]) + + +class TestPidfileClaim(unittest.TestCase): + def test_claims_missing_file(self): + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/x.pid" + self.assertTrue(w._claim_pidfile(path)) + with open(path) as f: + self.assertEqual(f.read().strip(), str(w.os.getpid())) + + def test_refuses_live_other_watcher(self): + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/x.pid" + with open(path, "w") as f: + f.write("99999998") + with mock.patch.object(w, "_pid_alive", return_value=True), \ + mock.patch.object(w, "_pid_is_watcher", return_value=True): + self.assertFalse(w._claim_pidfile(path)) + + def test_takes_over_dead_pid(self): + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/x.pid" + with open(path, "w") as f: + f.write("99999997") + with mock.patch.object(w, "_pid_alive", return_value=False): + self.assertTrue(w._claim_pidfile(path)) + + +class TestReconcileFailed(unittest.TestCase): + def test_failed_starts_recorded(self): + with mock.patch.object(w, "get_desired", + return_value={"enabled": True, + "dry_run": False}), \ + mock.patch.object(w, "muse_panes", return_value=["%37"]), \ + mock.patch.object(w, "is_running", return_value=None), \ + mock.patch.object(w, "_start_detached", return_value=False), \ + mock.patch.object(w, "status_all", return_value=[]), \ + mock.patch.object(w, "audit") as audit, \ + mock.patch.object(w.os.path, "exists", return_value=True): + res = w.reconcile(sockets=["/tmp/sock"]) + self.assertEqual(res["failed"], ["/tmp/sock:%37"]) + self.assertEqual(res["started"], []) + audit.assert_called_once() + + +class TestWatchCommand(unittest.TestCase): + def test_watch_refused_when_claimed(self): + with mock.patch.object(w, "_claim_pidfile", return_value=False): + rc = w.main(["watch", "--socket", "/tmp/s", "--pane", "%1"]) + self.assertEqual(rc, 3) + + def test_watch_runs_loop_and_releases_own_pidfile(self): + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/x.pid" + with open(path, "w") as f: + f.write(str(w.os.getpid())) + with mock.patch.object(w, "pidfile_for", return_value=path), \ + mock.patch.object(w, "_claim_pidfile", return_value=True), \ + mock.patch.object(w, "watch_loop", return_value=0) as loop: + rc = w.main(["watch", "--socket", "/tmp/s", "--pane", "%1"]) + self.assertEqual(rc, 0) + loop.assert_called_once_with("/tmp/s", "%1", dry_run=False) + self.assertFalse(w.os.path.exists(path)) + + +class TestWatchProcs(unittest.TestCase): + def _mkproc(self, root, pid, argv): + import os + d = os.path.join(root, str(pid)) + os.makedirs(d) + with open(os.path.join(d, "cmdline"), "wb") as f: + f.write(b"\0".join(x.encode() for x in argv) + b"\0") + + def test_exact_argv_scan(self): + import os + import tempfile + with tempfile.TemporaryDirectory() as td: + self._mkproc(td, 111, ["python3", "/x/muse_choice_watcher.py", + "watch", "--socket", "/tmp/s", + "--pane", "%1"]) + self._mkproc(td, 222, ["python3", "bin/super-cli.py", + "muse-choices", "reconcile"]) + self._mkproc(td, 333, ["python3", "/x/muse_choice_watcher.py", + "reconcile"]) + os.makedirs(os.path.join(td, "self")) + procs = w._watch_procs(proc_root=td) + self.assertEqual(procs, [{"pid": 111, "socket": "/tmp/s", + "pane": "%1"}]) + + def test_missing_root(self): + self.assertEqual(w._watch_procs(proc_root="/nonexistent-proc"), []) + + +class TestWatchDuplicate(unittest.TestCase): + def test_watch_refuses_duplicate_argv(self): + dup = [{"pid": 9991, "socket": "/tmp/s", "pane": "%1"}] + with mock.patch.object(w, "_watch_procs", return_value=dup), \ + mock.patch.object(w, "watch_loop") as loop: + rc = w.main(["watch", "--socket", "/tmp/s", "--pane", "%1"]) + self.assertEqual(rc, 3) + loop.assert_not_called() + + def test_watch_allows_different_pane(self): + other = [{"pid": 9991, "socket": "/tmp/s", "pane": "%2"}] + with mock.patch.object(w, "_watch_procs", return_value=other), \ + mock.patch.object(w, "_claim_pidfile", return_value=True), \ + mock.patch.object(w, "watch_loop", return_value=0) as loop: + rc = w.main(["watch", "--socket", "/tmp/s", "--pane", "%1"]) + self.assertEqual(rc, 0) + loop.assert_called_once() + + +class TestStopAllOrphans(unittest.TestCase): + def test_stop_all_kills_orphans(self): + import signal + orphan = {"pid": 8888, "socket": "/tmp/s", "pane": "%9"} + with mock.patch("os.listdir", return_value=[]), \ + mock.patch.object(w, "_watch_procs", return_value=[orphan]), \ + mock.patch("os.kill") as kill: + res = w.stop_all() + kill.assert_called_once_with(8888, signal.SIGTERM) + self.assertEqual(res[0]["status"], "stopped-orphan") + self.assertEqual(res[0]["pane"], "%9") + + +class TestStatusOrphans(unittest.TestCase): + def test_status_lists_orphans(self): + orphan = {"pid": 99999999, "socket": "/tmp/sock-x", "pane": "%9"} + with mock.patch.object(w, "_watch_procs", return_value=[orphan]): + rows = w.status_all() + orphans = [r for r in rows if r.get("orphan")] + self.assertEqual(len(orphans), 1) + self.assertEqual(orphans[0]["pid"], 99999999) + + +class TestRecentAnswers(unittest.TestCase): + def test_filters_and_limits(self): + import json + import tempfile + with tempfile.TemporaryDirectory() as td: + path = td + "/box-ctl.jsonl" + with open(path, "w") as f: + f.write('{"action": "other"}\n') + f.write('not json\n') + for i in range(3): + f.write(json.dumps({"action": "muse-choice-answered", + "sig": "s%d" % i}) + "\n") + with mock.patch.object(w, "CTL_LOG", path): + recs = w.recent_answers(limit=2) + self.assertEqual([r["sig"] for r in recs], ["s1", "s2"]) + + def test_missing_log_returns_empty(self): + with mock.patch.object(w, "CTL_LOG", "/nonexistent/x.jsonl"): + self.assertEqual(w.recent_answers(), []) + + +class TestDaemonUnits(unittest.TestCase): + def test_reconcile_unit_lets_daemons_survive(self): + # Load-bearing line: without KillMode=process, systemd kills + # timer-spawned watchers when the oneshot service exits. + text = (REPO_ROOT / "systemd" / "muse-choices-reconcile.service" + ).read_text() + self.assertIn("KillMode=process", text) + self.assertIn("muse_choice_watcher.py reconcile", text) + + def test_reconcile_timer_exists(self): + text = (REPO_ROOT / "systemd" / "muse-choices-reconcile.timer" + ).read_text() + self.assertIn("OnUnitActiveSec=", text) + + +class TestBoxWiring(unittest.TestCase): + def test_box_status_json_shape(self): + import json + import subprocess + cmd = [sys.executable, str(BIN_DIR / "super-cli.py"), + "muse-choices", "status", "--json"] + r = subprocess.run(cmd, capture_output=True, text=True, timeout=60) + self.assertEqual(r.returncode, 0, r.stderr[:500]) + data = json.loads(r.stdout) + self.assertTrue(data["ok"]) + self.assertIn("enabled", data["desired"]) + self.assertIsInstance(data["watchers"], list) + self.assertIsInstance(data["recent_answers"], list) + + +CURSOR = "β€Ί" # Muse TUI menu cursor (U+203A) + + +PROMPT_MUSE_APPROVAL = ( + "Would you like to run the following\n" + "\n" + " $ tmux -S /tmp/tmux-1000/default capture-pane -p -t %39\n" + "\n" + + CURSOR + " 1. Yes, proceed (y)\n" + " 2. No, and tell Muse Code what to do instead\n" +) + +PROMPT_MUSE_APPROVAL_WRAPPED = ( + "Would you like to run the following\n" + "\n" + " $ ps -eo pid,etime,args | grep\n" + " \"[m]use_choice_watcher.py\n" + " watch\" | wc -l; box\n" + " muse-choices status\n" + "\n" + + CURSOR + " 1. Yes, proceed (y)\n" + " 2. No, and tell Muse Code what to\n" + " do instead\n" +) + +PROMPT_MUSE_APPROVAL_DECIDED = ( + PROMPT_MUSE_APPROVAL + "approval decision accepted\n" +) + + +class TestMuseApproval(unittest.TestCase): + """Native Muse TUI approval menu: Would-you-like + 1.Yes/2.No.""" + + def test_matches_native_approval(self): + m = w.find_choice_prompt(PROMPT_MUSE_APPROVAL) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), ("muse-approval", "1")) + self.assertEqual(m["cue"], "Would you like to run the following") + self.assertEqual(len(m["options"]), 2) + + def test_matches_wrapped_command_echo(self): + m = w.find_choice_prompt(PROMPT_MUSE_APPROVAL_WRAPPED) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), ("muse-approval", "1")) + + def test_matches_ascii_cursor(self): + text = PROMPT_MUSE_APPROVAL.replace(CURSOR, ">") + m = w.find_choice_prompt(text) + self.assertIsNotNone(m) + self.assertEqual(m["kind"], "muse-approval") + + def test_matches_cursor_on_no(self): + text = PROMPT_MUSE_APPROVAL.replace(CURSOR + " 1.", " 1.") + text = text.replace(" 2.", CURSOR + " 2.") + m = w.find_choice_prompt(text) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), ("muse-approval", "1")) + + def test_matches_observed_live_variant(self): + # Verbatim shape answered on a live pane: cue with "command?", + # option 2 ending "(esc)". + text = ("Would you like to run the following command?\n" + "\n" + " $ tmux capture-pane -p\n" + "\n" + + CURSOR + " 1. Yes, proceed (y)\n" + " 2. No, and tell Muse Code what to do differently (esc)\n") + m = w.find_choice_prompt(text) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), ("muse-approval", "1")) + + def test_rejects_decided_block(self): + # An already-landed decision must never be double-answered. + self.assertIsNone(w.find_choice_prompt(PROMPT_MUSE_APPROVAL_DECIDED)) + + def test_rejects_cue_without_pair(self): + text = "Would you like to run the following\n\n $ foo\n" + self.assertIsNone(w.find_choice_prompt(text)) + + def test_rejects_pair_without_cue(self): + text = (CURSOR + " 1. Yes, proceed (y)\n" + " 2. No, thanks\n") + self.assertIsNone(w.find_choice_prompt(text)) + + def test_priority_over_yn(self): + m = w.find_choice_prompt(PROMPT_MUSE_APPROVAL + "Proceed? (y/n)\n") + self.assertIsNotNone(m) + self.assertEqual(m["kind"], "muse-approval") + + def test_sig_namespaced(self): + a = w.find_choice_prompt(PROMPT_MUSE_APPROVAL)["sig"] + b = w.find_choice_prompt(PROMPT_ABC)["sig"] + self.assertNotEqual(a, b) + + def test_sig_distinguishes_consecutive_approvals(self): + # Live stuck-state regression: options+cue are byte-identical + # across command approvals, so the sig must include the $ command + # or every dialog after the first is swallowed by once-only. + other = PROMPT_MUSE_APPROVAL.replace( + "capture-pane -p -t %39", "capture-pane -p -t %29") + a = w.find_choice_prompt(PROMPT_MUSE_APPROVAL)["sig"] + b = w.find_choice_prompt(other)["sig"] + self.assertNotEqual(a, b) + + def test_second_approval_answers_after_first(self): + # End-to-end stuck-state regression through the poll loop. + other = PROMPT_MUSE_APPROVAL.replace( + "capture-pane -p -t %39", "capture-pane -p -t %29") + state = w.WatcherState() + log = mock.Mock() + captures = [PROMPT_MUSE_APPROVAL] * 3 + [other] * 3 + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", + side_effect=captures), \ + mock.patch.object(w, "send_answer", + return_value=True) as send, \ + mock.patch.object(w, "audit"): + outcomes = [w._poll_once("/tmp/s", "%1", state, log, + dry_run=False) + for _ in range(4)] + self.assertEqual(outcomes, + ["seen", "answered", "seen", "answered"]) + self.assertEqual(send.call_count, 2) + + def test_poll_answers_with_1(self): + state = w.WatcherState() + log = mock.Mock() + captures = [PROMPT_MUSE_APPROVAL] * 3 + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", + side_effect=captures), \ + mock.patch.object(w, "send_answer", + return_value=True) as send, \ + mock.patch.object(w, "audit") as audit: + outcomes = [w._poll_once("/tmp/s", "%1", state, log, + dry_run=False) + for _ in range(len(captures) - 1)] + self.assertEqual(outcomes, ["seen", "answered"]) + send.assert_called_once_with("/tmp/s", "%1", "1", enter=True) + audit.assert_called_once() + self.assertEqual(audit.call_args[1]["extra"]["kind"], + "muse-approval") + + +PROMPT_INTERVIEW = ( + "The daemon only approves today. Should this step add deny/escalate\n" + "decisions informed by helpers, cover more prompt shapes, or both?\n" + "\n" + + CURSOR + " 1. Deny/escalate policy (Recommended) Keep approving by default.\n" + " 2. More prompt shapes Teach the matcher more UIs.\n" + " 3. Both Policy plus broader shapes.\n" + " 4. None of the above Optionally add notes (tab).\n" +) + + +class TestInterview(unittest.TestCase): + """Agent interview UI: cursor + ordered 1./2. menu + question.""" + + def test_matches_interview(self): + m = w.find_choice_prompt(PROMPT_INTERVIEW) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), ("interview", "1")) + self.assertEqual(len(m["options"]), 4) + + def test_matches_ascii_cursor(self): + m = w.find_choice_prompt(PROMPT_INTERVIEW.replace(CURSOR, ">")) + self.assertIsNotNone(m) + self.assertEqual(m["kind"], "interview") + + def test_rejects_prose_list_without_cursor(self): + # Same shape minus the selection cursor is prose, not a live menu. + text = PROMPT_INTERVIEW.replace(CURSOR + " ", " ") + self.assertIsNone(w.find_choice_prompt(text)) + + def test_rejects_single_option(self): + text = ("Pick one?\n\n" + CURSOR + " 1. Only choice\n") + self.assertIsNone(w.find_choice_prompt(text)) + + def test_rejects_options_without_question(self): + text = ("Some statement here.\n\n" + + CURSOR + " 1. First\n" + " 2. Second\n") + self.assertIsNone(w.find_choice_prompt(text)) + + def test_approval_wins_over_interview(self): + # A native approval dialog also carries dotted options; the more + # specific kind must win. + m = w.find_choice_prompt(PROMPT_MUSE_APPROVAL) + self.assertEqual(m["kind"], "muse-approval") + + def test_poll_answers_with_1(self): + state = w.WatcherState() + log = mock.Mock() + captures = [PROMPT_INTERVIEW] * 3 + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", + side_effect=captures), \ + mock.patch.object(w, "send_answer", + return_value=True) as send, \ + mock.patch.object(w, "audit") as audit: + outcomes = [w._poll_once("/tmp/s", "%1", state, log, + dry_run=False) + for _ in range(len(captures) - 1)] + self.assertEqual(outcomes, ["seen", "answered"]) + send.assert_called_once_with("/tmp/s", "%1", "1", enter=True) + self.assertEqual(audit.call_args[1]["extra"]["kind"], "interview") + + +class TestLaunchOptOut(unittest.TestCase): + def test_bare_argv_answers(self): + self.assertFalse(w.launch_opt_out(["/x/muse-bin-1.4"])) + self.assertFalse(w.launch_opt_out([])) + self.assertFalse(w.launch_opt_out(None)) + + def test_auto_flags_answer(self): + self.assertFalse(w.launch_opt_out(["muse", "--yolo"])) + self.assertFalse(w.launch_opt_out(["muse", "--disable-approval"])) + self.assertFalse( + w.launch_opt_out(["muse", "--approval-mode", "never"])) + self.assertFalse( + w.launch_opt_out(["muse", "--approval-mode=never"])) + + def test_explicit_mode_holds(self): + self.assertTrue( + w.launch_opt_out(["muse", "--approval-mode", "on-request"])) + self.assertTrue( + w.launch_opt_out(["muse", "--approval-mode", "untrusted"])) + self.assertTrue( + w.launch_opt_out(["muse", "--approval-mode=on-request"])) + + def test_poll_holds_opt_out_pane(self): + state = w.WatcherState() + log = mock.Mock() + captures = [PROMPT_YN] * 4 + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", + side_effect=captures), \ + mock.patch.object(w, "pane_muse_argv", + return_value=["muse", "--approval-mode", + "on-request"]), \ + mock.patch.object(w, "send_answer", + return_value=True) as send, \ + mock.patch.object(w, "audit") as audit: + outcomes = [w._poll_once("/tmp/s", "%1", state, log, + dry_run=False) + for _ in range(len(captures) - 1)] + self.assertEqual(outcomes, ["seen", "held", "none"]) + send.assert_not_called() + audit.assert_not_called() + msgs = [c.args[1] for c in log.log.call_args_list] + self.assertIn("held: pane opted out via launch flags", msgs) + + def test_poll_answers_bare_pane(self): + state = w.WatcherState() + log = mock.Mock() + captures = [PROMPT_YN] * 3 + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", + side_effect=captures), \ + mock.patch.object(w, "pane_muse_argv", return_value=[]), \ + mock.patch.object(w, "send_answer", + return_value=True) as send, \ + mock.patch.object(w, "audit"): + outcomes = [w._poll_once("/tmp/s", "%1", state, log, + dry_run=False) + for _ in range(len(captures) - 1)] + self.assertEqual(outcomes, ["seen", "answered"]) + send.assert_called_once_with("/tmp/s", "%1", "y", enter=True) + + def test_pane_muse_argv(self): + listing = mock.Mock(returncode=0, stdout="%1 1000\n%2 2000\n", + stderr="") + + def fake_cmdline(pid): + if pid == 1001: + return ["/x/muse-bin", "--yolo"] + return ["/bin/bash"] + + with mock.patch.object(w, "_tmux", + return_value=listing) as t, \ + mock.patch.object(w, "_child_pids", return_value=[1001]), \ + mock.patch.object(w, "_cmdline", side_effect=fake_cmdline): + argv = w.pane_muse_argv("/tmp/s", "%1") + self.assertEqual(argv, ["/x/muse-bin", "--yolo"]) + t.assert_called_once() + + def test_pane_muse_argv_missing(self): + listing = mock.Mock(returncode=0, stdout="%2 2000\n", stderr="") + with mock.patch.object(w, "_tmux", return_value=listing): + self.assertEqual(w.pane_muse_argv("/tmp/s", "%1"), []) + + +PROMPT_COLLAPSED_APPROVAL = ( + "Would you like to run the following\n" + "\n" + " $ python3 -m unittest\n" + " tests.test_muse_choice_watcher\n" + " 2>&1 | tail -n 3 && cp\n" + " ΚΌ 4 command rows omitted\n" + " ctrl+o view full command\n" +) + + +class TestCollapsedApproval(unittest.TestCase): + """Collapsed approval: long command hides the 1/2 pair; ctrl+o expands.""" + + def test_matches_collapsed(self): + m = w.find_choice_prompt(PROMPT_COLLAPSED_APPROVAL) + self.assertIsNotNone(m) + self.assertEqual(m["kind"], "muse-approval-collapsed") + self.assertEqual(m["key"], "Enter") + self.assertFalse(m["enter"]) + + def test_expanded_pair_beats_collapsed(self): + m = w.find_choice_prompt(PROMPT_MUSE_APPROVAL) + self.assertIsNotNone(m) + self.assertEqual(m["kind"], "muse-approval") + + def test_rejects_cue_without_markers(self): + text = "Would you like to run the following\n\n $ foo\n" + self.assertIsNone(w.find_choice_prompt(text)) + + def test_rejects_markers_without_cue(self): + text = (" $ foo bar baz\n" + " 4 command rows omitted\n" + " ctrl+o view full command\n") + self.assertIsNone(w.find_choice_prompt(text)) + + def test_rejects_decided_collapsed(self): + text = PROMPT_COLLAPSED_APPROVAL + "approval decision accepted\n" + self.assertIsNone(w.find_choice_prompt(text)) + + def test_poll_sends_bare_enter(self): + state = w.WatcherState() + log = mock.Mock() + captures = [PROMPT_COLLAPSED_APPROVAL] * 3 + calls = [] + + def fake_tmux(sock, *args, timeout=5): + calls.append((sock, args)) + r = mock.Mock() + r.returncode = 0 + return r + + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", + side_effect=captures), \ + mock.patch.object(w, "pane_muse_argv", return_value=[]), \ + mock.patch.object(w, "_tmux", side_effect=fake_tmux), \ + mock.patch.object(w, "audit") as audit: + outcomes = [w._poll_once("/tmp/s", "%1", state, log, + dry_run=False) + for _ in range(len(captures) - 1)] + self.assertEqual(outcomes, ["seen", "answered"]) + self.assertEqual(len(calls), 1) + self.assertEqual(calls[0][1][:3], ("send-keys", "-t", "%1")) + self.assertEqual(calls[0][1][3], "Enter") + audit.assert_called_once() + self.assertEqual(audit.call_args[1]["extra"]["kind"], + "muse-approval-collapsed") + + +PROMPT_EXPLICIT = ( + "Reply ACCEPT to approve this text as written (the 2 minutes\n" + "included), or amend anything first. Note: accepting the scope\n" + "finishes the interview and flips the record to Final.\n" +) + + +class TestExplicitPhrase(unittest.TestCase): + """Model asks the user to reply an explicit magic word.""" + + def test_matches_accept(self): + m = w.find_choice_prompt(PROMPT_EXPLICIT) + self.assertIsNotNone(m) + self.assertEqual((m["kind"], m["key"]), + ("explicit-phrase", "ACCEPT")) + + def test_matches_other_tokens(self): + for token in ("YES", "GO", "OK", "CONTINUE", "PROCEED", "ABORT-1"): + with self.subTest(token=token): + m = w.find_choice_prompt("Reply %s to confirm.\n" % token) + self.assertIsNotNone(m) + self.assertEqual(m["key"], token) + + def test_rejects_lowercase_prose(self): + self.assertIsNone( + w.find_choice_prompt("Please reply soon to confirm.\n")) + self.assertIsNone( + w.find_choice_prompt("Reply yes please to continue.\n")) + + def test_rejects_stale(self): + text = ("Reply ACCEPT to approve.\n" + + "\n".join("filler %d" % i for i in range(12))) + self.assertIsNone(w.find_choice_prompt(text)) + + def test_freshest_wins(self): + text = ("Reply YES to confirm.\n" + "Some agent chatter.\n" + "Reply ACCEPT to approve this text as written.\n") + m = w.find_choice_prompt(text) + self.assertEqual(m["key"], "ACCEPT") + + def test_poll_answers_with_token(self): + state = w.WatcherState() + log = mock.Mock() + captures = [PROMPT_EXPLICIT] * 3 + with mock.patch.object(w, "pane_exists", return_value=True), \ + mock.patch.object(w, "capture_pane", + side_effect=captures), \ + mock.patch.object(w, "send_answer", + return_value=True) as send, \ + mock.patch.object(w, "audit") as audit: + outcomes = [w._poll_once("/tmp/s", "%1", state, log, + dry_run=False) + for _ in range(len(captures) - 1)] + self.assertEqual(outcomes, ["seen", "answered"]) + send.assert_called_once_with("/tmp/s", "%1", "ACCEPT", enter=True) + self.assertEqual(audit.call_args[1]["extra"]["kind"], + "explicit-phrase") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_prompts.py b/tests/test_prompts.py index 6b6bd60..547380d 100644 --- a/tests/test_prompts.py +++ b/tests/test_prompts.py @@ -272,8 +272,8 @@ class TestPromptTUIIntegration(unittest.TestCase): target_text = sends[0]["text"] h, w = self.tui.stdscr.getmaxyx() - modal_w = min(74, w - 6) - modal_h = min(20, h - 4) + modal_w = min(84 if self.tui.modal in ("prompts", "history_search") else 74, w - 6) + modal_h = min(22 if self.tui.modal in ("prompts", "history_search") else 20, h - 4) top_y = (h - modal_h) // 2 left_x = (w - modal_w) // 2 diff --git a/tests/test_rate_limits.py b/tests/test_rate_limits.py index d693b99..3cba013 100644 --- a/tests/test_rate_limits.py +++ b/tests/test_rate_limits.py @@ -198,6 +198,17 @@ class TestRateLimitUserInteraction(unittest.TestCase): self.assertEqual(updated[0]["text"], "Earlier reply") self.assertEqual(updated[1]["text"], "New uncommitted instruction") + def test_history_payload_mentioning_rate_limits_not_falsely_flagged(self): + """Chat history discussing rate limits or HTTP 429 does NOT place node into cooldown when rc == 0.""" + msgs_about_rate_limits = [ + {"role": "user", "text": "Are we hitting any rate limits or 429 errors?"}, + {"role": "assistant", "text": "No rate limit reached, all queues are unthrottled and healthy."} + ] + with patch.object(muse_tui_rl, "run_command_isolated", return_value=(0, json.dumps(msgs_about_rate_limits), "")): + self.tui.data._fetch_history("pip", "thread-chat") + self.assertFalse(self.tui.data.is_node_rate_limited("pip")) + self.assertEqual(len(self.tui.data.history_cache.get(("pip", "thread-chat"), [])), 2) + if __name__ == "__main__": import json diff --git a/tests/test_swarm_prune.py b/tests/test_swarm_prune.py new file mode 100644 index 0000000..7f4f3a6 --- /dev/null +++ b/tests/test_swarm_prune.py @@ -0,0 +1,50 @@ +"""Regression tests for the hyphenated swarm-prune branch. + +`swarm-prune` was dispatched but had no handler branch, so it exited 0 +with no output. It now mirrors `swarm prune` (preview counts without +--confirm, archive only with --confirm). These tests never pass +--confirm, so they never write. +""" +import json +import subprocess +import sys +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +BOX_CTL = REPO_ROOT / "bin" / "box-ctl.py" + + +def _box_ctl(*args): + return subprocess.run( + [sys.executable, str(BOX_CTL), *args], + capture_output=True, text=True, timeout=120) + + +class SwarmPruneTests(unittest.TestCase): + def test_hyphenated_prune_previews_without_confirm(self): + r = _box_ctl("swarm-prune") + self.assertNotEqual(r.returncode, 0) + self.assertTrue(r.stdout.strip(), "hyphenated prune must emit JSON") + payload = json.loads(r.stdout) + self.assertEqual(payload["code"], "CONFIRM_REQUIRED") + self.assertIn("matching_count", payload["detail"]) + self.assertEqual(payload["detail"]["stale_hours"], 6) + + def test_hyphenated_matches_space_form(self): + hyphen = json.loads(_box_ctl("swarm-prune").stdout) + space = json.loads(_box_ctl("swarm", "prune").stdout) + self.assertEqual(hyphen["code"], "CONFIRM_REQUIRED") + self.assertEqual(space["code"], "CONFIRM_REQUIRED") + self.assertEqual(sorted(hyphen["detail"]), + sorted(space["detail"])) + + def test_bad_stale_hours_falls_back_to_default(self): + r = _box_ctl("swarm-prune", "--stale-hours", "bogus") + payload = json.loads(r.stdout) + self.assertEqual(payload["code"], "CONFIRM_REQUIRED") + self.assertEqual(payload["detail"]["stale_hours"], 6) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_watchdog_coverage.py b/tests/test_watchdog_coverage.py new file mode 100644 index 0000000..04192e8 --- /dev/null +++ b/tests/test_watchdog_coverage.py @@ -0,0 +1,100 @@ +"""Watchdog coverage: every registry node is supervised. + +Regression test for the dev/def outage (2026-10-06): both watchdogs +hardcoded the original four nodes (muse/pip/646/opm), so dev/def had no +browser supervision, no Warp-tunnel auto-recovery, and no relay +supervision. A dead tunnel paged CRITICAL partition alerts forever with +nothing acting on it. + +Both scripts now resolve nodes/ports from the fleet registry +(bin/netvm-registry.py). These tests drive the real shell code sourced +with a LIB_ONLY guard (same pattern as test_agent_health.py). +""" +import importlib.util +import subprocess +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +CHROMEBOX_WD = REPO_ROOT / "bin" / "chromebox-watchdog.sh" +RELAY_WD = REPO_ROOT / "bin" / "cdp-relay-watchdog.sh" + +# The fleet's pinned contract (also pinned in bin/netvm-names.sh). +EXPECTED_PORTS = { + "muse": 9410, "pip": 9420, "646": 9430, + "opm": 9440, "def": 9450, "dev": 9455, +} +EXPECTED_PEERS = { + "muse": "10.201.35.2", "pip": "10.201.87.2", "646": "10.201.202.2", + "opm": "10.201.157.2", "def": "10.201.66.2", "dev": "10.201.36.2", +} + + +def _load(mod_name, rel_path): + spec = importlib.util.spec_from_file_location(mod_name, REPO_ROOT / rel_path) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +def _chromebox(profile, snippet): + prog = "set -- '%s'\nsource '%s'\n%s\n" % (profile, CHROMEBOX_WD, snippet) + env = {"PATH": "/usr/bin:/bin", "CHROMEBOX_WATCHDOG_LIB_ONLY": "1"} + return subprocess.run(["bash", "-c", prog], capture_output=True, + text=True, env=env, timeout=30) + + +def _relay(snippet): + prog = "source '%s'\n%s\n" % (RELAY_WD, snippet) + env = {"PATH": "/usr/bin:/bin", "CDP_RELAY_WATCHDOG_LIB_ONLY": "1"} + return subprocess.run(["bash", "-c", prog], capture_output=True, + text=True, env=env, timeout=30) + + +class RegistryContract(unittest.TestCase): + def test_registry_matches_pinned_ports(self): + reg = _load("netvm_registry_cov", "bin/netvm-registry.py") + self.assertEqual(reg.active_nodes(), EXPECTED_PORTS) + for node, peer in EXPECTED_PEERS.items(): + self.assertEqual(reg.peer_ip_for(node), peer) + + +class ChromeboxWatchdogPorts(unittest.TestCase): + def test_all_registry_nodes_resolve_to_pinned_ports(self): + for node, port in sorted(EXPECTED_PORTS.items()): + with self.subTest(node=node): + r = _chromebox(node, "echo \"PORT=$CDP_PORT\"") + self.assertEqual(r.returncode, 0, r.stderr) + line = next((ln for ln in r.stdout.splitlines() + if ln.startswith("PORT=")), None) + self.assertIsNotNone( + line, "no PORT line: stdout=%r stderr=%r" + % (r.stdout, r.stderr)) + self.assertEqual(int(line.split("=", 1)[1]), port) + + def test_unknown_profile_rejected(self): + r = _chromebox("ghost", "echo UNREACHABLE") + self.assertNotEqual(r.returncode, 0) + self.assertIn("unknown profile", r.stderr) + self.assertNotIn("UNREACHABLE", r.stdout) + + +class RelayWatchdogCoverage(unittest.TestCase): + def test_watched_nodes_covers_registry(self): + r = _relay("watched_nodes") + self.assertEqual(r.returncode, 0, r.stderr) + nodes = set(r.stdout.split()) + for node in EXPECTED_PORTS: + self.assertIn(node, nodes) + + def test_relay_targets_use_pinned_ports(self): + for node, port in sorted(EXPECTED_PORTS.items()): + with self.subTest(node=node): + r = _relay("relay_target %s" % node) + self.assertEqual(r.returncode, 0, r.stderr) + self.assertEqual(r.stdout.strip(), + "%s:%d" % (EXPECTED_PEERS[node], port)) + + +if __name__ == "__main__": + unittest.main()