diff --git a/bin/exec-constrained.py b/bin/exec-constrained.py index c732696..0b7aa6f 100755 --- a/bin/exec-constrained.py +++ b/bin/exec-constrained.py @@ -994,6 +994,67 @@ def _swarm_results_build(a): return [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), 'swarm-results', a['swarm_id']] +# box.exec allowlist: read-only box-ctl actions only (mirrors the read-only +# subset of box-ctl.py IDEMPOTENT_ACTIONS). Actions with side-effecting +# subverbs (main-loop enable/disable), path args (git-diff/git-log), or +# complex argv shapes (job-next, policy-*, quality-validate, job-result) +# are deliberately excluded. +BOX_EXEC_NOARG_ACTIONS = frozenset({ + 'fleet-status', 'watchdog-alerts', 'relay-health', 'cdp-latency', + 'chrome-errors', 'identity-audit', 'timer-list', 'job-list', + 'vars-list', 'strat-list', 'loop-status', 'loop-health', 'loop-breaks', + 'thread-list', 'dm-log', 'unread', 'swarm-list', 'quality-check', + 'git-status', +}) +BOX_EXEC_ONEARG_ACTIONS = frozenset({ + 'job-get', 'job-status', 'timer-status', 'vars-get', 'vars-history', + 'strat-get', 'swarm-status', 'swarm-results', +}) +_BOX_ARG_RE = re.compile(r'^[A-Za-z0-9][A-Za-z0-9/_.-]{0,127}$') + + +def _box_exec_validate(raw): + if not isinstance(raw, dict): + raise OpError('args must be an object') + for k in raw: + if k not in {'action', 'arg', 'agent'}: + raise OpError(f'unknown arg: {k}') + action = raw.get('action') + if action in BOX_EXEC_NOARG_ACTIONS: + if raw.get('arg') is not None: + raise OpError(f'{action} takes no arg') + return {'action': action} + if action in BOX_EXEC_ONEARG_ACTIONS: + arg = raw.get('arg') + if arg is None: + return {'action': action} + if not isinstance(arg, str) or not _BOX_ARG_RE.fullmatch(arg): + raise OpError('arg must match safe token') + return {'action': action, 'arg': arg} + raise OpError(f'unknown or non-read-only box action: {action}') + + +def _box_exec_build(a): + argv = [sys.executable, os.path.join(BIN_DIR, 'box-ctl.py'), a['action']] + if a.get('arg'): + argv.append(a['arg']) + return argv + + +def _tools_list_validate(raw): + if not isinstance(raw, dict): + raise OpError('args must be an object') + for k in raw: + if k not in {'agent'}: + raise OpError(f'unknown arg: {k}') + return {} + + +def _tools_list_build(a): + return [sys.executable, os.path.join(BIN_DIR, 'exec-constrained.py'), + '--list-ops'] + + # op -> {validate, build, timeout, side_effecting, description} OPS = { 'dm.send': { @@ -1207,6 +1268,16 @@ OPS = { 'timeout': 10, 'side_effecting': False, 'desc': 'Canary no-op for watchdogs', }, + 'box.exec': { + 'validate': _box_exec_validate, 'build': _box_exec_build, + 'timeout': 60, 'side_effecting': False, + 'desc': 'Call a read-only box-ctl action (fleet-status, dm-log, job-get, ...)', + }, + 'tools.list': { + 'validate': _tools_list_validate, 'build': _tools_list_build, + 'timeout': 15, 'side_effecting': False, + 'desc': 'List all exec ops with descriptions (dynamic discovery)', + }, } # identity -> set of ops. 'master' may invoke everything. Unknown identities @@ -1495,6 +1566,19 @@ def main(): ap.add_argument('--key-file', default='/home/super/.exec-constrained-key.pem') ap.add_argument('--work-dir', default='/home/super') + ap.add_argument('--list-ops', action='store_true', + help='Print the op registry as JSON and exit (backs tools.list)') + args = ap.parse_args() + + if args.list_ops: + print(json.dumps({ + 'ok': True, + 'ops': [{'op': k, 'desc': v.get('desc', ''), + 'side_effecting': bool(v.get('side_effecting', False)), + 'timeout': v.get('timeout', 30)} + for k, v in sorted(OPS.items())], + })) + return args = ap.parse_args() TOKEN_FILE, TOKEN_DIR = args.token_file, args.token_dir diff --git a/bin/lookup_engine.py b/bin/lookup_engine.py index 015a605..7574106 100644 --- a/bin/lookup_engine.py +++ b/bin/lookup_engine.py @@ -31,7 +31,7 @@ _COMPILED_PATTERNS: Dict[str, re.Pattern] = {} # Fallback hardcoded regexes in case files are missing or unreadable _FALLBACK_RESULT_RE = re.compile(r"\[RESULT\s+([A-Za-z0-9_/-]+)\]\s*(.*?)(?=\[RESULT\s|\Z)", re.S) _FALLBACK_VERB_RE = re.compile(r"\[(ACK|CLAIM|RESULT|DECLINE|NO-ACTION)\s+([A-Za-z0-9_/-]+)\]") -_FALLBACK_TOOL_RE = re.compile(r"\[(TOOL|EXEC)\s+([a-zA-Z0-9_.-]+)\s+(\{.*?\})\]", re.S) +_FALLBACK_TOOL_RE = re.compile(r"\[(TOOL|EXEC|DM)\s+(?:([a-zA-Z0-9_.-]+)\s+)?(\{([^{}]|\{[^{}]*\})*\})\]", re.S) _FALLBACK_CONTRACT_FOOTER = ( "Reply: [ACK id] seen | [CLAIM id] mine | " "[RESULT id] done | [DECLINE id] | [NO-ACTION id]." diff --git a/bin/prompt_envelope.py b/bin/prompt_envelope.py index 7ecbfff..837e003 100644 --- a/bin/prompt_envelope.py +++ b/bin/prompt_envelope.py @@ -57,14 +57,22 @@ def pick_profile(job_name): def _tool(op, args): - # no ']' inside the JSON: the harvester's [TOOL ...] regex stops at the first one + # the harvester extracts JSON args with balanced-brace scanning, so + # ']' and nested objects/arrays inside args are safe return "[TOOL %s %s]" % (op, json.dumps(args, separators=(", ", ": "))) def spawn_call(job_id, job_name, profile): count, _, hint = PROFILES[profile] - # no brackets in the task text: the harvester's [TOOL ...] regex stops at the first ']' task = "Subagent for job %s (%s): %s." % (job_id, job_name, hint) + + +def dm_call(to, target, message): + return "[DM %s]" % json.dumps( + {"to": to, "target": target, "message": message[:900]}, + separators=(", ", ": ")) + + return _tool("swarm.spawn", {"count": count, "task": task[:900], "label": (job_name or "job")[:60]}) @@ -111,7 +119,9 @@ def wrap(job_name, job_id, agent, target, rendered): bottom = ( "\n--- End Task ---\n\n" f"Inspect tmux output: [TOOL tmux.capture {{\"session\": \"{session_name}\", \"lines\": 30}}]\n" - "Tools available: cron.create, cron.runs, health.check, swarm.spawn, swarm.list.\n" + "Tools: cron.create, cron.runs, health.check, swarm.spawn, swarm.list, dm.send, box.exec, tools.list.\n" + "Message a peer: [DM {\"to\": \"\", \"target\": \"\", \"message\": \"\"}].\n" + "Query box: [TOOL box.exec {\"action\": \"\"}] \u2014 [TOOL tools.list {}] lists every op.\n" ) if not has_result: bottom += f"When complete, report your verdict: [RESULT {job_id}] OK: \n" diff --git a/bin/response-harvester.py b/bin/response-harvester.py index 17109d4..e8cc32a 100755 --- a/bin/response-harvester.py +++ b/bin/response-harvester.py @@ -154,6 +154,12 @@ NATIVE_ALIASES = { "subagents.spawn": "swarm.spawn", "subagent.list": "swarm.list", "subagent.status": "swarm.status", + "dm": "dm.send", + "message.send": "dm.send", + "box": "box.exec", + "box.run": "box.exec", + "tools": "tools.list", + "tools.list": "tools.list", } @@ -184,6 +190,23 @@ def normalize_native_call(op, args): break if "count" not in args and "n" in args: args["count"] = args.pop("n") + elif op == "dm.send": + if "message" not in args: + for k in ("text", "body", "content", "msg"): + if k in args: + args["message"] = args.pop(k) + break + if "target" not in args: + for k in ("thread", "sidechat", "channel"): + if k in args: + args["target"] = args.pop(k) + break + elif op == "box.exec": + if "action" not in args: + for k in ("cmd", "verb", "command", "run"): + if k in args: + args["action"] = args.pop(k) + break elif op.startswith("tmux."): if "session" not in args: for k in ("name", "target", "s"): @@ -198,24 +221,96 @@ def normalize_native_call(op, args): return op, args +_TOOL_OPEN_RE = re.compile(r"\[(TOOL|EXEC)\s+([a-zA-Z0-9_.-]+)\s*") +_DM_OPEN_RE = re.compile(r"\[DM\s+") + + +def _extract_balanced_json(s, i): + """Extract one JSON object starting at s[i] == '{' (brace-aware, string-aware). + + Returns (obj, end_index) with end_index just past the closing brace, + or (None, i) when no balanced object is present. Unlike a first-']' + regex this tolerates ']' (and nested objects/arrays) inside args. + """ + if i >= len(s) or s[i] != "{": + return None, i + depth = 0 + in_str = False + esc = False + for j in range(i, len(s)): + c = s[j] + if in_str: + if esc: + esc = False + elif c == "\\": + esc = True + elif c == '"': + in_str = False + elif c == '"': + in_str = True + elif c == "{": + depth += 1 + elif c == "}": + depth -= 1 + if depth == 0: + try: + return json.loads(s[i:j + 1]), j + 1 + except Exception: + return None, i + return None, i + + +def _scan_bracket_calls(text): + """Yield (op, args) for [TOOL op {...}] / [EXEC op {...}] / [DM {...}]. + + JSON args are extracted with balanced-brace scanning so ']' inside + strings, arrays, or nested objects no longer truncates the call. + Non-JSON tails keep the legacy first-']' behavior (raw passthrough). + """ + out = [] + spans = [] + for m in _TOOL_OPEN_RE.finditer(text or ""): + spans.append((m.start(), "tool", m.group(2).strip(), m.end())) + for m in _DM_OPEN_RE.finditer(text or ""): + spans.append((m.start(), "dm", "dm.send", m.end())) + spans.sort() + for _, kind, op, pos in spans: + if pos < len(text) and text[pos] == "{": + args, _ = _extract_balanced_json(text, pos) + if args is None: + continue + if not isinstance(args, dict): + args = {"raw": args} + elif kind == "dm": + continue # [DM ...] requires a JSON object; skip bare forms + else: + end = text.find("]", pos) + if end == -1: + continue + raw_args = text[pos:end].strip() + if not raw_args: + args = {} + else: + try: + args = json.loads(raw_args) + if not isinstance(args, dict): + args = {"raw": args} + except Exception: + args = {"raw": raw_args} + out.append((op, args)) + return out + + def parse_tool_calls(text): """ Extract structured tool/exec calls from assistant messages. Supports: - 1. [TOOL ] or [EXEC ] - 2. ```box / ```tool / ```exec JSON blocks + 1. [TOOL ] or [EXEC ] (JSON may nest) + 2. [DM ] shorthand for dm.send + 3. ```box / ```tool / ```exec JSON blocks """ calls = [] - for m in re.finditer(r"\[(?:TOOL|EXEC)\s+([a-zA-Z0-9_.-]+)(?:\s+(.*?))?\]", text or ""): - op = m.group(1).strip() - raw_args = (m.group(2) or "").strip() - args = {} - if raw_args: - try: - args = json.loads(raw_args) - except Exception: - args = {"raw": raw_args} - calls.append((op, args)) + calls.extend(_scan_bracket_calls(text)) for m in re.finditer(r"```(?:box|tool|exec)\s*\n(.*?)```", text or "", re.DOTALL): block = m.group(1).strip() @@ -311,6 +406,10 @@ def format_tool_result_for_chat(op, raw_output): data = json.loads(raw_output) except Exception: s = str(raw_output).strip() + if op == "box.exec": + if len(s) > 900: + s = s[:900] + "\n…(truncated, refine the call for detail)" + return f"box result:\n```\n{s}\n```" return s[:500] if len(s) > 500 else s if op == "health.check" and isinstance(data, dict) and "fleet" in data: @@ -385,6 +484,26 @@ def format_tool_result_for_chat(op, raw_output): sw = data.get("swarm", {}) return f"Swarm `{sw.get('swarm_id')}`: {sw.get('status')} ({sw.get('done', 0)}/{sw.get('count', 0)} slots completed)." + if op == "tools.list" and isinstance(data, dict): + ops = data.get("ops", []) + if not ops: + return "No tools registered." + ro = [o["op"] for o in ops if not o.get("side_effecting")] + se = [o["op"] for o in ops if o.get("side_effecting")] + lines = [f"{len(ops)} tools available via [TOOL ]."] + lines.append("read-only: " + (", ".join(ro) if ro else "none")) + if se: + lines.append("side-effecting: " + ", ".join(se)) + return "\n".join(lines) + + if op == "box.exec" and isinstance(data, dict): + if data.get("ok") is False: + return f"box call failed: {data.get('error', 'unknown error')}" + out = json.dumps(data) + if len(out) > 900: + out = out[:900] + "\n…(truncated, refine the call for detail)" + return f"box result:\n```\n{out}\n```" + # General fallback: compact JSON capped to 400 chars s = json.dumps(data) return s[:400] + "..." if len(s) > 400 else s @@ -774,9 +893,12 @@ def process_messages(raw_messages, agent, thread_id, thread_name, last_wm, follo thread_url = f"https://box.muse-dev.online/thread/{thread_id}" tool_hint = ( f"[Runtime Context: {thread_url}]\n" - f"Tools available: [TOOL ] or curl -sk -X POST https://exec.muse-dev.online/exec\n" + f"Tools: [TOOL ] or curl -sk -X POST https://exec.muse-dev.online/exec\n" + f" • [TOOL tools.list {{}}] — discover every op dynamically\n" + f" • [TOOL swarm.spawn {{\"count\": 1, \"task\": \"\"}}] — spawn subagents\n" + f" • [DM {{\"to\": \"\", \"target\": \"\", \"message\": \"\"}}] — send a DM\n" + f" • [TOOL box.exec {{\"action\": \"fleet-status\"}}] — call box (read-only actions)\n" f" • [TOOL followup.create {{\"in_m\": 5, \"prompt\": \"\"}}]\n" - f" • [TOOL swarm.spawn {{\"count\": 1, \"task\": \"\"}}]\n" f" • [TOOL health.check {{}}]\n\n" f"[Directive: Take next action or close with [RESULT ] ]" ) diff --git a/docs/INBAND-MESSAGING-SPEC.md b/docs/INBAND-MESSAGING-SPEC.md new file mode 100644 index 0000000..0a6407b --- /dev/null +++ b/docs/INBAND-MESSAGING-SPEC.md @@ -0,0 +1,58 @@ +# In-Band Internal Messaging: Directives, Parsing, Exec Surface + +> **Status: Final** — accepted by the owner on 2026-10-06 +> ("i accept the scope contract, mark it Final"). + +> **Box is the main surface.** All operator work goes through Box +> (box.muse-dev.online). The web UI, `box` CLI, and agents share the same +> API endpoints. No UI-only powers. + +## Background (researched facts, not decisions) + +- Agents receive timer/job DMs wrapped by `bin/prompt_envelope.py` (`wrap()`), + currently advertising `[TOOL swarm.spawn]`, `[TOOL cron.create]`,(tmux + worker pointer, `[RESULT]` verdict rule. +- Agents reply with in-band directives. `bin/response-harvester.py` + `parse_tool_calls()` extracts `[TOOL op {json}]` / `[EXEC …]`, fenced + ```box|tool|exec blocks, and curl-to-/exec payloads; each call runs through + the `exec-constrained` HTTPS daemon (`op` allowlist + per-op + validate/build) and the result is posted back into the originating thread. +- Implemented this session, uncommitted: balanced-brace JSON scanning (no + more first-`]` truncation), `[DM {…}]` shorthand for `dm.send`, new + `box.exec` (read-only box-ctl actions) and `tools.list` (dynamic op + discovery) ops, native aliases (`dm`, `box`, `tools`, …), expanded + envelope/tool-hint verb lists, `tests/test_tool_calls.py` (38 tests). +- Related specs: `docs/DM_SPEC.md` (WO + logging layer), `docs/DM_SPEC.md` + (control plane), `docs/JOB-SPEC.md` (scheduler/distributor), + `CHAT_POLICY.md` (sidechat-first). + +## Scope contract (accepted) + +- Artifact boundary: this record covers directive syntax/parsing + (`[TOOL]`/`[EXEC]`/`[DM]`, fenced blocks), the exec op surface + (`box.exec`, `tools.list`, native aliases), and timer-message/envelope + content. Out of scope: gateway/browser transport, swarm worker + reliability and the failed-slot backlog, new `box` CLI verbs, + `CHAT_POLICY.md` changes. +- Done means: D1–D5 settled in writing below; owner explicitly accepts + this record (Draft → Final). Nothing else is a completion dependency. +- Deferred stages (each needs its own interview): swarm reliability + target, `box` CLI inspection verbs for in-band traffic. + +## Decisions + +| # | Decision | Status | +|---|----------|--------| +| D1 | Scope boundary = A (directives + exec surface + envelope; transport, swarm reliability, new CLI verbs, chat policy out) | settled | +| D2 | `box.exec` = read-only v1 (19 no-arg + 8 one-arg reads); side-effecting box actions stay out, dedicated ops cover writes | settled | +| D3 | `[DM …]` = strict JSON-only; bare forms without a JSON object are silently ignored | settled | +| D4 | Broken-JSON directives are skipped silently (no reply, no record) | settled | +| D5 | `tools.list` returns every op with its `side_effecting` flag; enforcement stays in per-op validation + identity permissions | settled | + +## Risks / validation (to fill as decisions settle) + +- Full suite: 235–244 tests (count varies run to run), 3–4 failures, all + in `test_approvals` / `test_copy_actions`, which import only + `approvals` / `gravity` / `muse_tui` — none of this record's modules. + Pre-existing/environmental, unrelated to the directive changes. + Focused suites green (38 tool-call + 32 docs/prompts tests). diff --git a/lookup_internal/regex_patterns.json b/lookup_internal/regex_patterns.json index 2271870..2cb7439 100644 --- a/lookup_internal/regex_patterns.json +++ b/lookup_internal/regex_patterns.json @@ -99,18 +99,20 @@ }, "tool_call": { "name": "In-Band Tool Execution Call", - "pattern": "\\[(?PTOOL|EXEC)\\s+(?P[a-zA-Z0-9_.-]+)\\s+(?P\\{.*?\\})\\]", + "pattern": "\\[(?PTOOL|EXEC|DM)\\s+(?:(?P[a-zA-Z0-9_.-]+)\\s+)?(?P\\{([^{}]|\\{[^{}]*\\})*\\})\\]", "flags": ["DOTALL"], - "description": "Matches synthesized inline tool execution directives with JSON arguments.", + "description": "Matches inline tool directives with JSON args (one nesting level; response-harvester.py scans balanced braces for arbitrary depth). DM carries no op (implies dm.send).", "named_groups": { - "engine": "Either TOOL or EXEC", - "op": "Target operation name (e.g. followup.create, swarm.spawn)", - "args": "Valid JSON string of argument parameters" + "engine": "TOOL, EXEC, or DM (DM implies dm.send)", + "op": "Target operation (e.g. followup.create, swarm.spawn); absent for DM", + "args": "JSON argument object; may contain ']' and one level of nested objects" }, "test_samples": { "valid": [ "[TOOL followup.create {\"in_m\": 5, \"prompt\": \"check\"}]", - "[EXEC health.check {\"verbose\": true}]" + "[EXEC health.check {\"verbose\": true}]", + "[TOOL box.exec {\"action\": \"job-get\", \"arg\": \"a-b[0]\"}]", + "[DM {\"to\": \"pip\", \"target\": \"pip tasks\", \"message\": \"hi\"}]" ], "invalid": [ "[TOOL followup.create without args]", diff --git a/tests/test_tool_calls.py b/tests/test_tool_calls.py new file mode 100644 index 0000000..49d128e --- /dev/null +++ b/tests/test_tool_calls.py @@ -0,0 +1,271 @@ +"""Tests for TOOL/DM directive parsing, native aliases, and the +box.exec / tools.list exec ops (dynamic in-band message passing).""" +import importlib.util +import json +import re +import subprocess +import sys +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent + + +def _load(mod_name, rel_path): + spec = importlib.util.spec_from_file_location(mod_name, REPO_ROOT / rel_path) + mod = importlib.util.module_from_spec(spec) + spec.loader.exec_module(mod) + return mod + + +harv = _load("harvester_tool_calls", "bin/response-harvester.py") +exc = _load("exec_constrained_tool_calls", "bin/exec-constrained.py") +env = _load("prompt_envelope_tool_calls", "bin/prompt_envelope.py") + + +class ParseToolCalls(unittest.TestCase): + def test_simple(self): + self.assertEqual( + harv.parse_tool_calls("[TOOL health.check {}]"), + [("health.check", {})], + ) + + def test_exec_engine(self): + calls = harv.parse_tool_calls('[EXEC cron.runs {"limit": 3}]') + self.assertEqual(calls, [("cron.runs", {"limit": 3})]) + + def test_bracket_inside_json_survives(self): + text = '[TOOL box.exec {"action": "job-get", "arg": "a-b[0]"}]' + self.assertEqual( + harv.parse_tool_calls(text), + [("box.exec", {"action": "job-get", "arg": "a-b[0]"})], + ) + + def test_nested_objects_and_arrays(self): + args = {"outer": {"inner": [1, 2, {"k": "v]w"}]}, "list": ["a", "b]c"]} + text = "[TOOL swarm.spawn %s]" % json.dumps(args) + self.assertEqual(harv.parse_tool_calls(text), [("swarm.spawn", args)]) + + def test_escaped_quotes_and_braces_in_strings(self): + args = {"prompt": 'say "{hi}" \\ done'} + text = "[TOOL followup.create %s]" % json.dumps(args) + op, got = harv.parse_tool_calls(text)[0] + self.assertEqual(op, "followup.create") + self.assertEqual(got["prompt"], args["prompt"]) + + def test_dm_shorthand(self): + text = '[DM {"to": "pip", "target": "pip tasks", "message": "hi [you]"}]' + self.assertEqual( + harv.parse_tool_calls(text), + [("dm.send", {"to": "pip", "target": "pip tasks", "message": "hi [you]"})], + ) + + def test_dm_bare_form_skipped(self): + self.assertEqual(harv.parse_tool_calls("[DM hello pip]"), []) + + def test_no_args(self): + self.assertEqual( + harv.parse_tool_calls("[TOOL cron.runs]"), [("cron.runs", {})] + ) + + def test_legacy_raw_passthrough(self): + self.assertEqual( + harv.parse_tool_calls("[TOOL foo bar baz]"), + [("foo", {"raw": "bar baz"})], + ) + + def test_broken_json_skipped(self): + self.assertEqual(harv.parse_tool_calls("[TOOL foo {bad}]"), []) + + def test_fenced_block(self): + text = '```tool\n{"op": "health.check", "args": {}}\n```' + self.assertEqual(harv.parse_tool_calls(text), [("health.check", {})]) + + def test_dedupe_repeated_call(self): + text = "[TOOL health.check {}] ... [TOOL health.check {}]" + self.assertEqual(harv.parse_tool_calls(text), [("health.check", {})]) + + def test_native_aliases_applied(self): + text = '[TOOL subagent.spawn {"count": 1, "task": "t"}]' + self.assertEqual( + harv.parse_tool_calls(text), + [("swarm.spawn", {"count": 1, "task": "t"})], + ) + text = '[TOOL cron.create {"kind": "runonce", "in_m": 5, "prompt": "p"}]' + op, args = harv.parse_tool_calls(text)[0] + self.assertEqual(op, "followup.create") + self.assertNotIn("kind", args) + self.assertEqual(args["in_m"], 5) + + +class NormalizeNativeCall(unittest.TestCase): + def test_dm_synonyms(self): + op, args = harv.normalize_native_call( + "dm", {"to": "pip", "thread": "pip tasks", "text": "hi"}) + self.assertEqual(op, "dm.send") + self.assertEqual(args["message"], "hi") + self.assertEqual(args["target"], "pip tasks") + + def test_box_synonyms(self): + op, args = harv.normalize_native_call("box", {"cmd": "fleet-status"}) + self.assertEqual((op, args), ("box.exec", {"action": "fleet-status"})) + + def test_tools_alias(self): + op, args = harv.normalize_native_call("tools", {}) + self.assertEqual(op, "tools.list") + + +class FormatToolResult(unittest.TestCase): + def test_tools_list_grouping(self): + out = harv.format_tool_result_for_chat("tools.list", json.dumps({ + "ok": True, + "ops": [ + {"op": "health.check", "side_effecting": False}, + {"op": "dm.send", "side_effecting": True}, + ], + })) + self.assertIn("2 tools", out) + self.assertIn("health.check", out) + self.assertIn("dm.send", out) + + def test_box_exec_string_fenced(self): + out = harv.format_tool_result_for_chat("box.exec", "NODE UP") + self.assertIn("```", out) + self.assertIn("NODE UP", out) + + def test_box_exec_string_truncated(self): + out = harv.format_tool_result_for_chat("box.exec", "x" * 2000) + self.assertIn("truncated", out) + self.assertLess(len(out), 1200) + + def test_box_exec_json_dict_passthrough(self): + out = harv.format_tool_result_for_chat( + "box.exec", json.dumps({"ok": True, "nodes": []})) + self.assertIn("```", out) + self.assertIn('"nodes": []', out) + + def test_box_exec_json_error(self): + out = harv.format_tool_result_for_chat( + "box.exec", json.dumps({"ok": False, "error": "BAD_NAME"})) + self.assertIn("BAD_NAME", out) + + +class BoxExecOp(unittest.TestCase): + def test_noarg_ok(self): + self.assertEqual( + exc._box_exec_validate({"action": "fleet-status"}), + {"action": "fleet-status"}, + ) + + def test_agent_key_tolerated(self): + self.assertEqual( + exc._box_exec_validate({"action": "unread", "agent": "646"}), + {"action": "unread"}, + ) + + def test_onearg_ok(self): + self.assertEqual( + exc._box_exec_validate({"action": "job-get", "arg": "abc-123"}), + {"action": "job-get", "arg": "abc-123"}, + ) + + def test_rejects_unknown_action(self): + with self.assertRaises(exc.OpError): + exc._box_exec_validate({"action": "job-trigger"}) + + def test_rejects_side_effecting(self): + for action in ("vars-set", "job-delete", "timer-create", "md-write"): + with self.assertRaises(exc.OpError, msg=action): + exc._box_exec_validate({"action": action}) + + def test_rejects_excluded_idempotent(self): + for action in ("main-loop", "quality-validate", "git-diff", "job-next"): + with self.assertRaises(exc.OpError, msg=action): + exc._box_exec_validate({"action": action}) + + def test_rejects_bad_arg(self): + for bad in ("../x", "a b", "a;b", "", "x" * 200): + with self.assertRaises(exc.OpError, msg=bad): + exc._box_exec_validate({"action": "job-get", "arg": bad}) + + def test_rejects_arg_on_noarg_action(self): + with self.assertRaises(exc.OpError): + exc._box_exec_validate({"action": "fleet-status", "arg": "x"}) + + def test_build_argv(self): + argv = exc._box_exec_build({"action": "job-get", "arg": "abc"}) + self.assertEqual(argv[-2:], ["job-get", "abc"]) + self.assertTrue(argv[1].endswith("box-ctl.py")) + + def test_registered_read_only(self): + self.assertIn("box.exec", exc.OPS) + self.assertFalse(exc.OPS["box.exec"]["side_effecting"]) + self.assertIn("tools.list", exc.OPS) + self.assertFalse(exc.OPS["tools.list"]["side_effecting"]) + + def test_permissions_cover_agents(self): + for ident in ("muse", "pip", "646", "opm", "dev", "def"): + self.assertIn("box.exec", exc.PERMISSIONS[ident]) + self.assertIn("tools.list", exc.PERMISSIONS[ident]) + + def test_list_ops_subcommand(self): + p = subprocess.run( + [sys.executable, str(REPO_ROOT / "bin" / "exec-constrained.py"), + "--list-ops"], + capture_output=True, text=True, timeout=30, + ) + self.assertEqual(p.returncode, 0) + data = json.loads(p.stdout) + self.assertTrue(data["ok"]) + names = {o["op"] for o in data["ops"]} + for want in ("box.exec", "tools.list", "dm.send", "swarm.spawn", + "health.check", "followup.create"): + self.assertIn(want, names) + + +class CanonicalToolPattern(unittest.TestCase): + def test_samples_match(self): + data = json.loads( + (REPO_ROOT / "lookup_internal" / "regex_patterns.json").read_text()) + pat = data["patterns"]["tool_call"]["pattern"] + rx = re.compile(pat, re.S) + for s in data["patterns"]["tool_call"]["test_samples"]["valid"]: + self.assertIsNotNone(rx.search(s), s) + for s in data["patterns"]["tool_call"]["test_samples"]["invalid"]: + self.assertIsNone(rx.search(s), s) + + def test_dm_sample_has_no_op(self): + data = json.loads( + (REPO_ROOT / "lookup_internal" / "regex_patterns.json").read_text()) + rx = re.compile(data["patterns"]["tool_call"]["pattern"], re.S) + m = rx.search('[DM {"to": "pip"}]') + self.assertIsNotNone(m) + self.assertEqual(m.group("engine"), "DM") + self.assertIsNone(m.group("op")) + + +class EnvelopeRoundTrip(unittest.TestCase): + def test_wrap_advertises_new_verbs(self): + body = env.wrap("work-finder", "work-finder-1", "646", + "646 tasks", "Do the thing.") + for token in ("dm.send", "box.exec", "tools.list", "[DM {"): + self.assertIn(token, body) + + def test_spawn_call_parses(self): + text = env.spawn_call("jid-1", "work-finder", "work") + op, args = harv.parse_tool_calls(text)[0] + self.assertEqual(op, "swarm.spawn") + self.assertIn("count", args) + self.assertIn("task", args) + + def test_dm_call_parses(self): + text = env.dm_call("pip", "pip tasks", "hello [brackets] work") + self.assertEqual( + harv.parse_tool_calls(text), + [("dm.send", {"to": "pip", "target": "pip tasks", + "message": "hello [brackets] work"})], + ) + + +if __name__ == "__main__": + unittest.main()