feat(hybrid-gateway): integrate muse-cli with Cloudflare netns isolation, symmetric sidechat routing, and 646-pip sync unblock
This commit is contained in:
@@ -68,9 +68,45 @@ healthy() {
|
||||
|| { HEALTH_FAIL_REASON="CDP up but no page target in list"; return 1; }
|
||||
echo "$list" | grep -E -q '"url": "https://muse\.ai' \
|
||||
|| { HEALTH_FAIL_REASON="CDP up but not on muse.ai"; return 1; }
|
||||
warp_egress_healthy || return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
# Stage 5: Warp egress health (2026-10-04). A partitioned browser (tunnel down,
|
||||
# CDP green) passes stages 1-4 while being unable to reach muse.ai. Check the
|
||||
# WireGuard handshake age and probe egress from inside the node's netns.
|
||||
# Sets HEALTH_FAIL_REASON with a distinct "warp egress down" prefix so
|
||||
# root-cause analysis can distinguish partitions from Chromium crashes.
|
||||
warp_egress_healthy() {
|
||||
local wg_out hs_line age_s code
|
||||
wg_out="$(sudo -n ip netns exec "warp-${PROFILE}" wg show 2>/dev/null)" \
|
||||
|| { HEALTH_FAIL_REASON="warp egress down (wg show failed in warp-${PROFILE})"; return 1; }
|
||||
hs_line="$(printf '%s\n' "$wg_out" | grep -i "latest handshake" | head -1)"
|
||||
[ -n "$hs_line" ] \
|
||||
|| { HEALTH_FAIL_REASON="warp egress down (no WireGuard handshake in warp-${PROFILE})"; return 1; }
|
||||
# "latest handshake: 1 minute, 41 seconds ago" -> total seconds
|
||||
age_s="$(printf '%s\n' "$hs_line" | python3 -c '
|
||||
import sys, re
|
||||
s = sys.stdin.read()
|
||||
m = re.search(r"(\d+)\s*hour", s); h = int(m.group(1)) if m else 0
|
||||
m = re.search(r"(\d+)\s*minute", s); mi = int(m.group(1)) if m else 0
|
||||
m = re.search(r"(\d+)\s*second", s); se = int(m.group(1)) if m else 0
|
||||
print(h*3600 + mi*60 + se)
|
||||
' 2>/dev/null)"
|
||||
{ [ -n "$age_s" ] && [ "$age_s" -ge 0 ]; } 2>/dev/null \
|
||||
|| { HEALTH_FAIL_REASON="warp egress down (unparseable handshake: $hs_line)"; return 1; }
|
||||
[ "$age_s" -le 180 ] \
|
||||
|| { HEALTH_FAIL_REASON="warp egress down (handshake ${age_s}s old in warp-${PROFILE})"; return 1; }
|
||||
code="$(sudo -n ip netns exec "warp-${PROFILE}" curl -m 5 -s -o /dev/null -w "%{http_code}" "https://muse.ai/" 2>/dev/null)" \
|
||||
|| { HEALTH_FAIL_REASON="warp egress down (egress probe curl failed in warp-${PROFILE})"; return 1; }
|
||||
# Any 2xx/3xx means we reached muse.ai infra ("/" 307-redirects to the
|
||||
# auth flow). The probe tests egress connectivity, not page content.
|
||||
case "$code" in
|
||||
2*|3*) return 0 ;;
|
||||
*) HEALTH_FAIL_REASON="warp egress down (egress probe HTTP $code in warp-${PROFILE})"; return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
if healthy; then
|
||||
exit 0
|
||||
fi
|
||||
@@ -101,6 +137,23 @@ if [ -n "$recent_pid" ]; then
|
||||
log "browser launched recently (pid $recent_pid), skipping relaunch (probably still starting)"
|
||||
exit 0
|
||||
fi
|
||||
# Warp-partition recovery (2026-10-04): relaunching Chrome cannot fix a dead
|
||||
# Warp tunnel — the new browser would fail the same egress check and the
|
||||
# watchdog would loop. If the failure is warp-egress, restart the tunnel first
|
||||
# (netvm-node-up.sh is idempotent); only fall through to the Chrome relaunch
|
||||
# if the tunnel does not recover.
|
||||
case "$HEALTH_FAIL_REASON" in
|
||||
"warp egress down"*)
|
||||
log "warp partition detected ($HEALTH_FAIL_REASON), restarting tunnel via netvm-node-up.sh"
|
||||
sudo -n "$NETVM_BIN/netvm-node-up.sh" "$PROFILE" >>"$LOG" 2>&1 || true
|
||||
sleep 5
|
||||
if healthy; then
|
||||
log "tunnel restart recovered warp egress, chrome relaunch not needed"
|
||||
exit 0
|
||||
fi
|
||||
log "tunnel restart did not recover egress ($HEALTH_FAIL_REASON), proceeding with chrome relaunch"
|
||||
;;
|
||||
esac
|
||||
log "unhealthy ($HEALTH_FAIL_REASON), relaunching chromebox"
|
||||
# bracket trick so pkill never matches its own command line
|
||||
pat="profiles/${PROFILE:0:${#PROFILE}-1}[${PROFILE: -1}]/"
|
||||
@@ -125,9 +178,13 @@ rm -f "/home/super/.local/share/chrome-box/profiles/${PROFILE}/home/.config/chro
|
||||
# muse-chat-api.py uses — a relaunch on the wrong port looks healthy to the
|
||||
# launcher but is unreachable to the API (observed 2026-10-03: pip relaunched
|
||||
# on 9278 instead of 9420, watchdog looped on "relaunch FAILED").
|
||||
# 9>&-: do NOT let the backgrounded launcher inherit the watchdog lock fd.
|
||||
# Inherited flock fds wedge the lock forever (observed 2026-10-04: opm/muse
|
||||
# launchers held their profile lock for 100+ min, every later watchdog run
|
||||
# skipped as "another run in progress" — the watchdog was silently dead).
|
||||
systemd-run --user --scope --unit="netvm-chrome-${PROFILE}-$(date +%s)" \
|
||||
"$NETVM_BIN/netvm-chrome.sh" --headless --cdp-port "$CDP_PORT" "$PROFILE" "https://muse.ai" \
|
||||
>>"$CHROME_LOG" 2>&1 < /dev/null &
|
||||
>>"$CHROME_LOG" 2>&1 < /dev/null 9>&- &
|
||||
# Retry with backoff (2026-10-04): cold starts (fresh egress IP, Cloudflare
|
||||
# handshake) can take >25s for the page title to appear. A single check after
|
||||
# 25s kills working-but-slow browsers. Try up to 4 times, 15s apart (~60s
|
||||
|
||||
Reference in New Issue
Block a user