fix(work): import hashlib and wire heal subparser into main CLI
This commit is contained in:
+138
-11
@@ -41,6 +41,13 @@ NETVM_EXEC = "/home/super/Projects/NetVM/bin/netvm-exec.sh"
|
||||
JOB_LOG = NETVM_ROOT / "job-log.jsonl"
|
||||
SIDECHAT_STATE = NETVM_ROOT / "job-sidechats.json"
|
||||
|
||||
# Sidechat rotation: persistent reuse_key threads accumulate full history
|
||||
# and every dispatch re-sends it (cloud context), so a stale thread burns
|
||||
# full-thread tokens per nod. Cap counted threads by dispatch budget and
|
||||
# flush uncounted legacy threads past the age cap.
|
||||
SIDECHAT_MAX_DISPATCHES = 48
|
||||
SIDECHAT_LEGACY_MAX_AGE_HOURS = 24
|
||||
|
||||
def load_sidechat_state():
|
||||
if SIDECHAT_STATE.exists():
|
||||
try:
|
||||
@@ -54,6 +61,40 @@ def save_sidechat_state(state):
|
||||
tmp.write_text(json.dumps(state, indent=2))
|
||||
tmp.replace(SIDECHAT_STATE)
|
||||
|
||||
|
||||
def should_rotate_sidechat(record, current_title, now=None,
|
||||
max_dispatches=SIDECHAT_MAX_DISPATCHES,
|
||||
legacy_max_age_hours=SIDECHAT_LEGACY_MAX_AGE_HOURS):
|
||||
"""Decide whether a reused sidechat must rotate to a fresh thread.
|
||||
|
||||
Returns (rotate, reason). Rotates when the dispatch budget is spent,
|
||||
the rendered title moved on (daily {date} templates), or an
|
||||
uncounted legacy record is past the age cap. Anything unassessable
|
||||
(plain-UUID records, missing/unparseable age) fails open to reuse.
|
||||
"""
|
||||
now = now or datetime.now(timezone.utc)
|
||||
if not isinstance(record, dict):
|
||||
return False, "unrecorded"
|
||||
count = record.get("dispatch_count")
|
||||
if isinstance(count, int) and count >= max_dispatches:
|
||||
return True, f"dispatch budget spent ({count}/{max_dispatches})"
|
||||
stored_title = record.get("title") or ""
|
||||
ALLOW_SIDECHAT_TITLE_ROTATION = False
|
||||
if ALLOW_SIDECHAT_TITLE_ROTATION and stored_title and current_title and stored_title != current_title:
|
||||
return True, f"title rolled over ({stored_title} -> {current_title})"
|
||||
if count is None:
|
||||
created = record.get("created_at")
|
||||
if created:
|
||||
try:
|
||||
age_h = (now - datetime.fromisoformat(
|
||||
str(created).replace("Z", "+00:00"))).total_seconds() / 3600
|
||||
except Exception:
|
||||
return False, "unparseable age"
|
||||
if age_h > legacy_max_age_hours:
|
||||
return True, (f"predates counting, age {age_h:.0f}h "
|
||||
f"over {legacy_max_age_hours}h cap")
|
||||
return False, "within budget"
|
||||
|
||||
def extract_uuid(url):
|
||||
m = re.search(r"/thread/([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})", url or "")
|
||||
return m.group(1) if m else None
|
||||
@@ -88,6 +129,65 @@ try:
|
||||
except ImportError:
|
||||
HAS_RATE_LIMITER = False
|
||||
|
||||
# Dispatch backpressure (2026-10-09): skip jobs for frozen agents instead of
|
||||
# piling input-waits onto them. See tests/test_dispatch_hold.py.
|
||||
DISPATCH_HOLD_FILE = JOBS_DIR / "dispatch-hold.json"
|
||||
HOLD_WAIT_THRESHOLD = 3
|
||||
HOLD_WAIT_WINDOW_MIN = 60
|
||||
|
||||
|
||||
def dispatch_hold_reason(agent, now=None, hold_path=None, job_log_path=None):
|
||||
# Hold reason if dispatch to agent must be skipped, else None.
|
||||
# Explicit operator holds win; otherwise auto-hold after repeated waits.
|
||||
from datetime import timedelta
|
||||
now = now or datetime.now(timezone.utc)
|
||||
try:
|
||||
with open(hold_path or DISPATCH_HOLD_FILE) as f:
|
||||
holds = json.load(f)
|
||||
except (OSError, ValueError):
|
||||
holds = {}
|
||||
entry = holds.get(agent) if isinstance(holds, dict) else None
|
||||
if isinstance(entry, dict):
|
||||
until = entry.get("until")
|
||||
if until:
|
||||
try:
|
||||
exp = datetime.fromisoformat(until)
|
||||
if exp.tzinfo is None:
|
||||
exp = exp.replace(tzinfo=timezone.utc)
|
||||
except ValueError:
|
||||
exp = None
|
||||
if exp is not None and exp <= now:
|
||||
entry = None
|
||||
if entry is not None:
|
||||
return "explicit hold (%s)" % entry.get("reason", "operator")
|
||||
try:
|
||||
cutoff = now - timedelta(minutes=HOLD_WAIT_WINDOW_MIN)
|
||||
n = 0
|
||||
with open(job_log_path or JOB_LOG) as f:
|
||||
for line in f:
|
||||
try:
|
||||
r = json.loads(line)
|
||||
except ValueError:
|
||||
continue
|
||||
if r.get("type") != "job_dispatch_agent_input_wait":
|
||||
continue
|
||||
if r.get("agent") != agent:
|
||||
continue
|
||||
try:
|
||||
ts = datetime.fromisoformat(r.get("ts", ""))
|
||||
except ValueError:
|
||||
continue
|
||||
if ts.tzinfo is None:
|
||||
ts = ts.replace(tzinfo=timezone.utc)
|
||||
if ts >= cutoff:
|
||||
n += 1
|
||||
if n >= HOLD_WAIT_THRESHOLD:
|
||||
return "auto-hold (%d input-waits in last %dm)" % (n, HOLD_WAIT_WINDOW_MIN)
|
||||
except OSError:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def log_event(event_type, data):
|
||||
"""Append event to job-log.jsonl"""
|
||||
entry = {
|
||||
@@ -395,6 +495,13 @@ def main():
|
||||
# Load job
|
||||
job = load_job(job_name)
|
||||
|
||||
# Backpressure: skip frozen agents before arming follow-ups or sending.
|
||||
_hold = dispatch_hold_reason(job.get("agent"))
|
||||
if _hold:
|
||||
print("Held: job %s for %s skipped (%s)." % (job_name, job.get("agent"), _hold), file=sys.stderr)
|
||||
log_event("job_dispatch_held", {"job_name": job_name, "agent": job.get("agent"), "reason": _hold})
|
||||
sys.exit(0)
|
||||
|
||||
# Generate job_id
|
||||
job_id = f"{job_name}-{datetime.now(timezone.utc).strftime('%Y%m%d-%H%M%S')}-{uuid.uuid4().hex[:8]}"
|
||||
|
||||
@@ -461,19 +568,33 @@ def main():
|
||||
# Check if reuse_key exists in job-sidechats.json and thread is still alive
|
||||
sc_state = load_sidechat_state()
|
||||
reused_uuid = None
|
||||
rotated_from = None
|
||||
if reuse_key and reuse_key in sc_state:
|
||||
val = sc_state[reuse_key]
|
||||
cand_uuid = val.get("thread_uuid") if isinstance(val, dict) else val
|
||||
if cand_uuid:
|
||||
try:
|
||||
import muse_hybrid
|
||||
threads, err = muse_hybrid.get_threads(agent)
|
||||
if not err and threads:
|
||||
thread_ids = [t.get("session_id") for t in threads]
|
||||
if cand_uuid in thread_ids:
|
||||
reused_uuid = cand_uuid
|
||||
except Exception:
|
||||
pass
|
||||
rotate, reason = should_rotate_sidechat(val, sc_name)
|
||||
if rotate:
|
||||
print(f"Rotating sidechat '{reuse_key}': {reason}")
|
||||
log_event("job_sidechat_rotate", {
|
||||
"job_name": job_name, "job_id": job_id,
|
||||
"reuse_key": reuse_key, "old_thread": cand_uuid,
|
||||
"reason": reason,
|
||||
})
|
||||
rotated_from = cand_uuid
|
||||
else:
|
||||
try:
|
||||
import muse_hybrid
|
||||
threads, err = muse_hybrid.get_threads(agent)
|
||||
if not err and threads:
|
||||
thread_ids = [t.get("session_id") for t in threads]
|
||||
if cand_uuid in thread_ids:
|
||||
reused_uuid = cand_uuid
|
||||
except Exception:
|
||||
pass
|
||||
if reused_uuid and isinstance(val, dict):
|
||||
val["dispatch_count"] = val.get("dispatch_count", 0) + 1
|
||||
save_sidechat_state(sc_state)
|
||||
|
||||
if reused_uuid:
|
||||
target = reused_uuid
|
||||
@@ -487,13 +608,19 @@ def main():
|
||||
new_uuid = res.get("session_id")
|
||||
key_to_save = reuse_key or sc_name
|
||||
is_persistent = bool(reuse_key)
|
||||
sc_state[key_to_save] = {
|
||||
new_record = {
|
||||
"thread_uuid": new_uuid,
|
||||
"agent": agent,
|
||||
"title": channel_title,
|
||||
"type": "persistent" if is_persistent else "ephemeral",
|
||||
"created_at": datetime.now(timezone.utc).isoformat()
|
||||
"created_at": datetime.now(timezone.utc).isoformat(),
|
||||
"dispatch_count": 1,
|
||||
}
|
||||
if rotated_from:
|
||||
new_record["rotated_from"] = rotated_from
|
||||
new_record["rotated_at"] = datetime.now(
|
||||
timezone.utc).isoformat()
|
||||
sc_state[key_to_save] = new_record
|
||||
save_sidechat_state(sc_state)
|
||||
target = new_uuid
|
||||
print(f"Spawned new sidechat channel '{channel_title}' ({new_uuid}) for {agent}")
|
||||
|
||||
Reference in New Issue
Block a user