#!/usr/bin/env python3 """ Side-chat to main-chat work siphon — detection rules (REPAIRED, agent 2 of 5). Fixes the false-positive ✅ COMPLETED relay at the source: "Sending is disabled until this conversation can be verified." → COMPLETED was caused by a SINGLE keyword ("verified") matching one regex. Repairs (see OUTPUT.md for rationale): 1. COMPLETED requires >= 2 DISTINCT pattern hits (weighted: the structured `[RESULT ...] OK` marker counts 2 — it is the fleet's own machine-emitted completion signal, far less ambiguous than a bare "done"). 2. Negation guards: negation/failure-state words veto COMPLETED outright (fail-closed: a negated completion claim is never relayed as complete). 3. Honest labeling: the fake "confidence 60%" (which literally meant "one regex hit") is replaced by a keyword-hit count. SiphonHit.hits is the authoritative field; `confidence` is kept for backward compatibility but must NOT be rendered as a percentage anywhere user-facing. 4. Stale suppression: a message older than 15 minutes never relays as COMPLETED. Pass message_ts (epoch seconds). monitor.py currently does NOT pass a timestamp — agent 3 / the integrator must thread message["ts"] through (see OUTPUT.md). DO NOT overwrite the original detect.py with this file until the integrator reconciles all 5 agents' outputs. """ import re import time from dataclasses import dataclass, field from typing import Optional @dataclass class SiphonHit: category: str # COMPLETED, BLOCKER, DECISION, ALERT, MILESTONE confidence: float # LEGACY — kept for API compatibility only. # Do NOT render as "confidence NN%"; it is not a # reliability measure. See `hits`. hits: int = 0 # AUTHORITATIVE — distinct keyword-pattern hits # (weighted; see COMPLETED_PATTERN_WEIGHTS). summary: str = "" # one-line summary for main chat thread_id: str = "" # source side chat message_id: str = "" # source message author: str = "" # INTEGRATOR (agent 3 absent) — the message's real # author, plumbed from message["author"] by # monitor.py. Empty = unknown; NEVER substitute the # thread's registered agent silently (see # format_siphon). # Full message text is NOT stored here — main chat gets a summary # plus a link back, never the full content (safety: no sensitive # data siphoned verbatim). # --- Keyword sets (configurable) --- COMPLETED_PATTERNS = [ re.compile(r'\b(done|completed|finished|deployed|shipped|live|verified)\b', re.I), re.compile(r'\b(all|tests?)\s+(pass|green|passing)\b', re.I), re.compile(r'\[RESULT[^\]]*\]\s*OK', re.I), re.compile(r'\b(merged|committed|pushed|published)\b', re.I), ] # Weighted hits: the structured [RESULT] OK marker is the fleet's own # machine-emitted completion signal — unambiguous enough to stand alone. COMPLETED_PATTERN_WEIGHTS = {0: 1, 1: 1, 2: 2, 3: 1} COMPLETED_MIN_WEIGHT = 2 # >= 2 distinct pattern hits (or one [RESULT] OK) BLOCKER_PATTERNS = [ re.compile(r'\b(blocked|stuck|failing|broken|down|error|failed)\b', re.I), re.compile(r'\b(need|needs|waiting)\s+(your|approval|input|decision)\b', re.I), re.compile(r'\b(can\'t|cannot|unable to)\b', re.I), re.compile(r'\[RESULT[^\]]*\]\s*(FAIL|ERROR)', re.I), ] DECISION_PATTERNS = [ re.compile(r'\b(should (i|we)|shall i|want me to)\b', re.I), re.compile(r'\b(your call|needs? your|awaiting your)\b', re.I), re.compile(r'\b(approve|approval)\b.*\?', re.I), re.compile(r'^(yes|no)\s*\?\s*$', re.I), ] ALERT_PATTERNS = [ re.compile(r'\b(security|vulnerability|breach|compromised|exploit)\b', re.I), re.compile(r'\b(urgent|critical|emergency|asap)\b', re.I), re.compile(r'\b(502|503|500)\b.*\b(error|down)\b', re.I), re.compile(r'\b(ssh|tunnel).*\b(down|broken|failed)\b', re.I), ] MILESTONE_PATTERNS = [ re.compile(r'\b(milestone|phase \d+ (complete|done)|shipped v)\b', re.I), re.compile(r'\b(all \d+ (items? )?done)\b', re.I), ] # Patterns that suppress siphoning (safety) SUPPRESS_PATTERNS = [ re.compile(r'\b(password|secret|token|key|pin)\s*[:=]', re.I), re.compile(r'-----BEGIN', re.I), # never siphon key material re.compile(r'\[do not siphon\]', re.I), # explicit opt-out marker ] # --- Negation guards: any match vetoes COMPLETED (fail-closed) --- # A completion claim in the presence of negation / failure-state language # is never relayed as ✅ COMPLETED, no matter how many keywords hit. NEGATION_GUARDS = [ # explicit negation re.compile(r'\b(not|never|no|nothing|none|neither|nor)\b', re.I), re.compile(r"\b(do not|don't|didn't|doesn't|won't|can't|cannot|isn't|aren't|" r"wasn't|weren't|haven't|hasn't|hadn't|couldn't|shouldn't)\b", re.I), # incompleteness hedges re.compile(r'\b(still|yet|pending|unfinished|incomplete)\b', re.I), # failure-state words (a "completed" message containing these is suspect) re.compile(r'\b(broken|failed|failing|failure|down|stuck|blocked|disabled|' r'error|errors|crash|crashed)\b', re.I), # hedging conjunctions ("deployed, but tests are red") re.compile(r'\b(but|however|although|though)\b', re.I), ] # Messages older than this never relay as COMPLETED (seconds). COMPLETED_MAX_AGE_S = 15 * 60 def _match_score(text: str, patterns) -> float: """Legacy confidence for non-COMPLETED categories (unchanged).""" hits = sum(1 for p in patterns if p.search(text)) if hits == 0: return 0.0 # Diminishing returns: 1 hit = 0.6, 2 = 0.8, 3+ = 0.95 return min(0.95, 0.6 + (hits - 1) * 0.2) def _completed_weight(text: str): """ Return (weighted_hits, distinct_hits, matched_pattern_indexes) for COMPLETED_PATTERNS. Weighted: [RESULT] OK counts 2. """ matched = [i for i, p in enumerate(COMPLETED_PATTERNS) if p.search(text)] weight = sum(COMPLETED_PATTERN_WEIGHTS.get(i, 1) for i in matched) return weight, len(matched), matched def _is_negated(text: str) -> bool: """True if any negation guard fires anywhere in the text.""" return any(p.search(text) for p in NEGATION_GUARDS) def _extract_summary(text: str, max_len: int = 120) -> str: """Extract a safe one-line summary. Strips to first meaningful line.""" # Take first non-empty line, truncate for line in text.strip().split('\n'): line = line.strip() if line and len(line) > 10: if len(line) > max_len: return line[:max_len - 3] + '...' return line return text[:max_len] def detect(text: str, thread_id: str, message_id: str, min_confidence: float = 0.6, message_ts: Optional[float] = None) -> Optional[SiphonHit]: """ Check a side chat message for siphon-worthy content. Returns SiphonHit or None. message_ts: epoch seconds of the original message (optional). Messages older than COMPLETED_MAX_AGE_S (15 min) never relay as COMPLETED. NOTE: monitor.py does not currently pass a timestamp — agent 3 / the integrator must thread message["ts"] through the detect() call. """ # Safety: suppress sensitive content for p in SUPPRESS_PATTERNS: if p.search(text): return None # Stale suppression applies to COMPLETED only. completed_allowed = True if message_ts is not None: try: age = time.time() - float(message_ts) if age > COMPLETED_MAX_AGE_S: completed_allowed = False except (TypeError, ValueError): pass # unparseable ts: proceed, do not fail closed on metadata # COMPLETED: >=2 distinct weighted pattern hits, no negation, not stale. completed_hits = 0 completed_conf = 0.0 if completed_allowed and not _is_negated(text): weight, distinct, _ = _completed_weight(text) if weight >= COMPLETED_MIN_WEIGHT: completed_hits = weight # legacy confidence kept for API compat; NOT a reliability measure completed_conf = min(0.95, 0.6 + (distinct - 1) * 0.2) candidates = [ ("COMPLETED", completed_conf, completed_hits), ("BLOCKER", _match_score(text, BLOCKER_PATTERNS), sum(1 for p in BLOCKER_PATTERNS if p.search(text))), ("DECISION", _match_score(text, DECISION_PATTERNS), sum(1 for p in DECISION_PATTERNS if p.search(text))), ("ALERT", _match_score(text, ALERT_PATTERNS), sum(1 for p in ALERT_PATTERNS if p.search(text))), ("MILESTONE", _match_score(text, MILESTONE_PATTERNS), sum(1 for p in MILESTONE_PATTERNS if p.search(text))), ] # Sort by confidence descending; ALERT wins ties (safety: urgency first) # Use negative confidence for descending, and ALERT as tiebreaker candidates.sort(key=lambda x: (-x[1], 0 if x[0] == "ALERT" else 1)) best_cat, best_conf, best_hits = candidates[0] if best_conf < min_confidence: return None return SiphonHit( category=best_cat, confidence=best_conf, hits=best_hits, summary=_extract_summary(text), thread_id=thread_id, message_id=message_id, ) def format_siphon(hit: SiphonHit, agent_name: str = "sidechat") -> str: """ Format a siphon message for main chat. HONEST LABELING: reports keyword hit count, never a fake "confidence %". HONEST AUTHORSHIP (integrator, agent 3 absent): attributes the message's real author when known. Falls back to the thread's registered agent only when the author is unknown — and says so explicitly, so a relay can never again launder thread ownership as authorship. """ emoji = {"COMPLETED": "✅", "BLOCKER": "🚧", "DECISION": "❓", "ALERT": "🚨", "MILESTONE": "🎯"}.get(hit.category, "📋") thread_url = f"https://muse.ai/thread/{hit.thread_id}" if hit.author: attribution = f"from {hit.author}" else: attribution = f"from {agent_name} side chat (author unverified)" return ( f"{emoji} [{hit.category}] {attribution}\n" f"{hit.summary}\n" f"→ {thread_url}\n" f"(keyword hits: {hit.hits})" ) # --- Opt-out registry --- _opt_out_threads: set = set() def opt_out(thread_id: str): """Agent opts a side chat out of siphoning.""" _opt_out_threads.add(thread_id) def opt_in(thread_id: str): """Re-enable siphoning for a side chat.""" _opt_out_threads.discard(thread_id) def is_opted_out(thread_id: str) -> bool: return thread_id in _opt_out_threads