feat(hybrid-gateway): integrate muse-cli with Cloudflare netns isolation, symmetric sidechat routing, and 646-pip sync unblock

This commit is contained in:
operator
2026-10-04 22:54:01 +00:00
parent 3d5fbe5aeb
commit 1a271b1bbd
60 changed files with 5068 additions and 55 deletions
+58 -1
View File
@@ -68,9 +68,45 @@ healthy() {
|| { HEALTH_FAIL_REASON="CDP up but no page target in list"; return 1; }
echo "$list" | grep -E -q '"url": "https://muse\.ai' \
|| { HEALTH_FAIL_REASON="CDP up but not on muse.ai"; return 1; }
warp_egress_healthy || return 1
return 0
}
# Stage 5: Warp egress health (2026-10-04). A partitioned browser (tunnel down,
# CDP green) passes stages 1-4 while being unable to reach muse.ai. Check the
# WireGuard handshake age and probe egress from inside the node's netns.
# Sets HEALTH_FAIL_REASON with a distinct "warp egress down" prefix so
# root-cause analysis can distinguish partitions from Chromium crashes.
warp_egress_healthy() {
local wg_out hs_line age_s code
wg_out="$(sudo -n ip netns exec "warp-${PROFILE}" wg show 2>/dev/null)" \
|| { HEALTH_FAIL_REASON="warp egress down (wg show failed in warp-${PROFILE})"; return 1; }
hs_line="$(printf '%s\n' "$wg_out" | grep -i "latest handshake" | head -1)"
[ -n "$hs_line" ] \
|| { HEALTH_FAIL_REASON="warp egress down (no WireGuard handshake in warp-${PROFILE})"; return 1; }
# "latest handshake: 1 minute, 41 seconds ago" -> total seconds
age_s="$(printf '%s\n' "$hs_line" | python3 -c '
import sys, re
s = sys.stdin.read()
m = re.search(r"(\d+)\s*hour", s); h = int(m.group(1)) if m else 0
m = re.search(r"(\d+)\s*minute", s); mi = int(m.group(1)) if m else 0
m = re.search(r"(\d+)\s*second", s); se = int(m.group(1)) if m else 0
print(h*3600 + mi*60 + se)
' 2>/dev/null)"
{ [ -n "$age_s" ] && [ "$age_s" -ge 0 ]; } 2>/dev/null \
|| { HEALTH_FAIL_REASON="warp egress down (unparseable handshake: $hs_line)"; return 1; }
[ "$age_s" -le 180 ] \
|| { HEALTH_FAIL_REASON="warp egress down (handshake ${age_s}s old in warp-${PROFILE})"; return 1; }
code="$(sudo -n ip netns exec "warp-${PROFILE}" curl -m 5 -s -o /dev/null -w "%{http_code}" "https://muse.ai/" 2>/dev/null)" \
|| { HEALTH_FAIL_REASON="warp egress down (egress probe curl failed in warp-${PROFILE})"; return 1; }
# Any 2xx/3xx means we reached muse.ai infra ("/" 307-redirects to the
# auth flow). The probe tests egress connectivity, not page content.
case "$code" in
2*|3*) return 0 ;;
*) HEALTH_FAIL_REASON="warp egress down (egress probe HTTP $code in warp-${PROFILE})"; return 1 ;;
esac
}
if healthy; then
exit 0
fi
@@ -101,6 +137,23 @@ if [ -n "$recent_pid" ]; then
log "browser launched recently (pid $recent_pid), skipping relaunch (probably still starting)"
exit 0
fi
# Warp-partition recovery (2026-10-04): relaunching Chrome cannot fix a dead
# Warp tunnel — the new browser would fail the same egress check and the
# watchdog would loop. If the failure is warp-egress, restart the tunnel first
# (netvm-node-up.sh is idempotent); only fall through to the Chrome relaunch
# if the tunnel does not recover.
case "$HEALTH_FAIL_REASON" in
"warp egress down"*)
log "warp partition detected ($HEALTH_FAIL_REASON), restarting tunnel via netvm-node-up.sh"
sudo -n "$NETVM_BIN/netvm-node-up.sh" "$PROFILE" >>"$LOG" 2>&1 || true
sleep 5
if healthy; then
log "tunnel restart recovered warp egress, chrome relaunch not needed"
exit 0
fi
log "tunnel restart did not recover egress ($HEALTH_FAIL_REASON), proceeding with chrome relaunch"
;;
esac
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
# bracket trick so pkill never matches its own command line
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
@@ -125,9 +178,13 @@ rm -f "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chro
# muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the
# launcher but is unreachable to the API (observed 2026-10-03: pip relaunched
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
# 9>&-: do NOT let the backgrounded launcher inherit the watchdog lock fd.
# Inherited flock fds wedge the lock forever (observed 2026-10-04: opm/muse
# launchers held their profile lock for 100+ min, every later watchdog run
# skipped as "another run in progress" — the watchdog was silently dead).
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
>>"$CHROME_LOG" 2>&1 < /dev/null &
>>"$CHROME_LOG" 2>&1 < /dev/null 9>&- &
# Retry with backoff (2026-10-04): cold starts (fresh egress IP, Cloudflare
# handshake) can take >25s for the page title to appear. A single check after
# 25s kills working-but-slow browsers. Try up to 4 times, 15s apart (~60s