Files

267 lines
10 KiB
Python
Raw Permalink Normal View History

#!/usr/bin/env python3
"""
Side-chat to main-chat work siphon — detection rules (REPAIRED, agent 2 of 5).
Fixes the false-positive ✅ COMPLETED relay at the source:
"Sending is disabled until this conversation can be verified." → COMPLETED
was caused by a SINGLE keyword ("verified") matching one regex.
Repairs (see OUTPUT.md for rationale):
1. COMPLETED requires >= 2 DISTINCT pattern hits (weighted: the structured
`[RESULT ...] OK` marker counts 2 — it is the fleet's own machine-emitted
completion signal, far less ambiguous than a bare "done").
2. Negation guards: negation/failure-state words veto COMPLETED outright
(fail-closed: a negated completion claim is never relayed as complete).
3. Honest labeling: the fake "confidence 60%" (which literally meant "one
regex hit") is replaced by a keyword-hit count. SiphonHit.hits is the
authoritative field; `confidence` is kept for backward compatibility
but must NOT be rendered as a percentage anywhere user-facing.
4. Stale suppression: a message older than 15 minutes never relays as
COMPLETED. Pass message_ts (epoch seconds). monitor.py currently does
NOT pass a timestamp — agent 3 / the integrator must thread
message["ts"] through (see OUTPUT.md).
DO NOT overwrite the original detect.py with this file until the integrator
reconciles all 5 agents' outputs.
"""
import re
import time
from dataclasses import dataclass, field
from typing import Optional
@dataclass
class SiphonHit:
category: str # COMPLETED, BLOCKER, DECISION, ALERT, MILESTONE
confidence: float # LEGACY — kept for API compatibility only.
# Do NOT render as "confidence NN%"; it is not a
# reliability measure. See `hits`.
hits: int = 0 # AUTHORITATIVE — distinct keyword-pattern hits
# (weighted; see COMPLETED_PATTERN_WEIGHTS).
summary: str = "" # one-line summary for main chat
thread_id: str = "" # source side chat
message_id: str = "" # source message
author: str = "" # INTEGRATOR (agent 3 absent) — the message's real
# author, plumbed from message["author"] by
# monitor.py. Empty = unknown; NEVER substitute the
# thread's registered agent silently (see
# format_siphon).
# Full message text is NOT stored here — main chat gets a summary
# plus a link back, never the full content (safety: no sensitive
# data siphoned verbatim).
# --- Keyword sets (configurable) ---
COMPLETED_PATTERNS = [
re.compile(r'\b(done|completed|finished|deployed|shipped|live|verified)\b', re.I),
re.compile(r'\b(all|tests?)\s+(pass|green|passing)\b', re.I),
re.compile(r'\[RESULT[^\]]*\]\s*OK', re.I),
re.compile(r'\b(merged|committed|pushed|published)\b', re.I),
]
# Weighted hits: the structured [RESULT] OK marker is the fleet's own
# machine-emitted completion signal — unambiguous enough to stand alone.
COMPLETED_PATTERN_WEIGHTS = {0: 1, 1: 1, 2: 2, 3: 1}
COMPLETED_MIN_WEIGHT = 2 # >= 2 distinct pattern hits (or one [RESULT] OK)
BLOCKER_PATTERNS = [
re.compile(r'\b(blocked|stuck|failing|broken|down|error|failed)\b', re.I),
re.compile(r'\b(need|needs|waiting)\s+(your|approval|input|decision)\b', re.I),
re.compile(r'\b(can\'t|cannot|unable to)\b', re.I),
re.compile(r'\[RESULT[^\]]*\]\s*(FAIL|ERROR)', re.I),
]
DECISION_PATTERNS = [
re.compile(r'\b(should (i|we)|shall i|want me to)\b', re.I),
re.compile(r'\b(your call|needs? your|awaiting your)\b', re.I),
re.compile(r'\b(approve|approval)\b.*\?', re.I),
re.compile(r'^(yes|no)\s*\?\s*$', re.I),
]
ALERT_PATTERNS = [
re.compile(r'\b(security|vulnerability|breach|compromised|exploit)\b', re.I),
re.compile(r'\b(urgent|critical|emergency|asap)\b', re.I),
re.compile(r'\b(502|503|500)\b.*\b(error|down)\b', re.I),
re.compile(r'\b(ssh|tunnel).*\b(down|broken|failed)\b', re.I),
]
MILESTONE_PATTERNS = [
re.compile(r'\b(milestone|phase \d+ (complete|done)|shipped v)\b', re.I),
re.compile(r'\b(all \d+ (items? )?done)\b', re.I),
]
# Patterns that suppress siphoning (safety)
SUPPRESS_PATTERNS = [
re.compile(r'\b(password|secret|token|key|pin)\s*[:=]', re.I),
re.compile(r'-----BEGIN', re.I), # never siphon key material
re.compile(r'\[do not siphon\]', re.I), # explicit opt-out marker
]
# --- Negation guards: any match vetoes COMPLETED (fail-closed) ---
# A completion claim in the presence of negation / failure-state language
# is never relayed as ✅ COMPLETED, no matter how many keywords hit.
NEGATION_GUARDS = [
# explicit negation
re.compile(r'\b(not|never|no|nothing|none|neither|nor)\b', re.I),
re.compile(r"\b(do not|don't|didn't|doesn't|won't|can't|cannot|isn't|aren't|"
r"wasn't|weren't|haven't|hasn't|hadn't|couldn't|shouldn't)\b", re.I),
# incompleteness hedges
re.compile(r'\b(still|yet|pending|unfinished|incomplete)\b', re.I),
# failure-state words (a "completed" message containing these is suspect)
re.compile(r'\b(broken|failed|failing|failure|down|stuck|blocked|disabled|'
r'error|errors|crash|crashed)\b', re.I),
# hedging conjunctions ("deployed, but tests are red")
re.compile(r'\b(but|however|although|though)\b', re.I),
]
# Messages older than this never relay as COMPLETED (seconds).
COMPLETED_MAX_AGE_S = 15 * 60
def _match_score(text: str, patterns) -> float:
"""Legacy confidence for non-COMPLETED categories (unchanged)."""
hits = sum(1 for p in patterns if p.search(text))
if hits == 0:
return 0.0
# Diminishing returns: 1 hit = 0.6, 2 = 0.8, 3+ = 0.95
return min(0.95, 0.6 + (hits - 1) * 0.2)
def _completed_weight(text: str):
"""
Return (weighted_hits, distinct_hits, matched_pattern_indexes) for
COMPLETED_PATTERNS. Weighted: [RESULT] OK counts 2.
"""
matched = [i for i, p in enumerate(COMPLETED_PATTERNS) if p.search(text)]
weight = sum(COMPLETED_PATTERN_WEIGHTS.get(i, 1) for i in matched)
return weight, len(matched), matched
def _is_negated(text: str) -> bool:
"""True if any negation guard fires anywhere in the text."""
return any(p.search(text) for p in NEGATION_GUARDS)
def _extract_summary(text: str, max_len: int = 120) -> str:
"""Extract a safe one-line summary. Strips to first meaningful line."""
# Take first non-empty line, truncate
for line in text.strip().split('\n'):
line = line.strip()
if line and len(line) > 10:
if len(line) > max_len:
return line[:max_len - 3] + '...'
return line
return text[:max_len]
def detect(text: str, thread_id: str, message_id: str,
min_confidence: float = 0.6,
message_ts: Optional[float] = None) -> Optional[SiphonHit]:
"""
Check a side chat message for siphon-worthy content.
Returns SiphonHit or None.
message_ts: epoch seconds of the original message (optional). Messages
older than COMPLETED_MAX_AGE_S (15 min) never relay as COMPLETED.
NOTE: monitor.py does not currently pass a timestamp — agent 3 / the
integrator must thread message["ts"] through the detect() call.
"""
# Safety: suppress sensitive content
for p in SUPPRESS_PATTERNS:
if p.search(text):
return None
# Stale suppression applies to COMPLETED only.
completed_allowed = True
if message_ts is not None:
try:
age = time.time() - float(message_ts)
if age > COMPLETED_MAX_AGE_S:
completed_allowed = False
except (TypeError, ValueError):
pass # unparseable ts: proceed, do not fail closed on metadata
# COMPLETED: >=2 distinct weighted pattern hits, no negation, not stale.
completed_hits = 0
completed_conf = 0.0
if completed_allowed and not _is_negated(text):
weight, distinct, _ = _completed_weight(text)
if weight >= COMPLETED_MIN_WEIGHT:
completed_hits = weight
# legacy confidence kept for API compat; NOT a reliability measure
completed_conf = min(0.95, 0.6 + (distinct - 1) * 0.2)
candidates = [
("COMPLETED", completed_conf, completed_hits),
("BLOCKER", _match_score(text, BLOCKER_PATTERNS),
sum(1 for p in BLOCKER_PATTERNS if p.search(text))),
("DECISION", _match_score(text, DECISION_PATTERNS),
sum(1 for p in DECISION_PATTERNS if p.search(text))),
("ALERT", _match_score(text, ALERT_PATTERNS),
sum(1 for p in ALERT_PATTERNS if p.search(text))),
("MILESTONE", _match_score(text, MILESTONE_PATTERNS),
sum(1 for p in MILESTONE_PATTERNS if p.search(text))),
]
# Sort by confidence descending; ALERT wins ties (safety: urgency first)
# Use negative confidence for descending, and ALERT as tiebreaker
candidates.sort(key=lambda x: (-x[1], 0 if x[0] == "ALERT" else 1))
best_cat, best_conf, best_hits = candidates[0]
if best_conf < min_confidence:
return None
return SiphonHit(
category=best_cat,
confidence=best_conf,
hits=best_hits,
summary=_extract_summary(text),
thread_id=thread_id,
message_id=message_id,
)
def format_siphon(hit: SiphonHit, agent_name: str = "sidechat") -> str:
"""
Format a siphon message for main chat.
HONEST LABELING: reports keyword hit count, never a fake "confidence %".
HONEST AUTHORSHIP (integrator, agent 3 absent): attributes the message's
real author when known. Falls back to the thread's registered agent only
when the author is unknown — and says so explicitly, so a relay can
never again launder thread ownership as authorship.
"""
emoji = {"COMPLETED": "✅", "BLOCKER": "🚧", "DECISION": "❓",
"ALERT": "🚨", "MILESTONE": "🎯"}.get(hit.category, "📋")
thread_url = f"https://muse.ai/thread/{hit.thread_id}"
if hit.author:
attribution = f"from {hit.author}"
else:
attribution = f"from {agent_name} side chat (author unverified)"
return (
f"{emoji} [{hit.category}] {attribution}\n"
f"{hit.summary}\n"
f"→ {thread_url}\n"
f"(keyword hits: {hit.hits})"
)
# --- Opt-out registry ---
_opt_out_threads: set = set()
def opt_out(thread_id: str):
"""Agent opts a side chat out of siphoning."""
_opt_out_threads.add(thread_id)
def opt_in(thread_id: str):
"""Re-enable siphoning for a side chat."""
_opt_out_threads.discard(thread_id)
def is_opted_out(thread_id: str) -> bool:
return thread_id in _opt_out_threads