From 87be6ae4fc84d4abdb705b2db185c04fa5785cd3 Mon Sep 17 00:00:00 2001 From: operator Date: Fri, 9 Oct 2026 12:57:28 +0000 Subject: [PATCH] feat(supervisor): implement and verify automated tunnel recovery supervisor with test coverage --- cloud-uptime/uptime-watcher.sh | 130 +++++++++++++++++++++++++++++++++ tests/test_uptime_watcher.py | 37 ++++++++++ 2 files changed, 167 insertions(+) create mode 100755 cloud-uptime/uptime-watcher.sh create mode 100644 tests/test_uptime_watcher.py diff --git a/cloud-uptime/uptime-watcher.sh b/cloud-uptime/uptime-watcher.sh new file mode 100755 index 0000000..18921b7 --- /dev/null +++ b/cloud-uptime/uptime-watcher.sh @@ -0,0 +1,130 @@ +#!/usr/bin/env bash +# uptime-watcher.sh — simple hatch-hook watcher: spawn/rebuild from spec. +# +# Register as a hatch hook (id `uptime-watcher`, poll 120s, timeout 300s) +# alongside tunnel-keeper. Each poll it guarantees the three things a +# container rebuild destroys: +# 1. provisioning — runs recover-after-rebuild.sh on a fresh root fs +# 2. supervisor — respawns gcp-tunnel-up.sh if it died +# 3. cron jobs — reinstalls crontab from ~/workspace/cron/*.persist +# +# It also verifies the VM-side SSH forward answers a banner, and wakes the +# operator (rate-limited, 30 min) only when something stays broken across +# polls. Silent on success. Safe to run by hand or from cron too. +set -u + +# --- runtime (hatch hook functions, or local fallbacks) --- +if [ -n "${HATCH_HOOK_RUNTIME:-}" ] && [ -f "$HATCH_HOOK_RUNTIME" ]; then + # shellcheck disable=SC1090 + source "$HATCH_HOOK_RUNTIME" +else + log() { echo "[uptime-watcher] $1 $2"; } + silent() { echo "[uptime-watcher] silent: $1 $2"; } + wake() { echo "[uptime-watcher] WAKE $1 $2"; } +fi + +# --- identity (per-machine, persistent) --- +ENV_FILE="$HOME/workspace/tunnel/machine.env" +# shellcheck disable=SC1090 +[ -f "$ENV_FILE" ] && . "$ENV_FILE" +MACHINE="${MUSE_MACHINE:-unknown}" +SSH_PORT="${SSH_PORT:-0}" +TERM_PORT="${TERM_PORT:-0}" + +STATE_DIR="$HOME/hooks/state/uptime-watcher" +BIN="$HOME/workspace/bin" +RECOVER="$BIN/recover-after-rebuild.sh" +SUPERVISOR="$BIN/gcp-tunnel-up.sh" +CRON_RESTORE="$BIN/persistent-crontab.sh" +SSH_KEY="$HOME/.ssh/vm_to_gcp" +GCP_HOST="${FD_VM_HOST:-34.139.37.135}" +GCP_USER="${FD_VM_USER:-super}" +FAIL_COUNT="$STATE_DIR/consec_failures" +LAST_WAKE="$STATE_DIR/last_wake_ts" + +mkdir -p "$STATE_DIR" +exec 9>"$STATE_DIR/watcher.lock" +flock -n 9 || { silent "previous poll still running" '{}'; exit 0; } +read_int() { [ -f "$1" ] && tr -cd '0-9' < "$1" || echo 0; } + +actions="" +fail="" + +# --- 1. fresh rebuild? provision --- +if [ ! -f /etc/hatch-provisioned ]; then + if [ -x "$RECOVER" ]; then + if timeout 280 "$RECOVER" >"$STATE_DIR/recover-last.log" 2>&1; then + actions="${actions}provisioned " + log "recovery" '{"event":"provisioned_after_rebuild"}' + else + fail="recover_failed" + fi + else + fail="recover_missing" + fi +fi + +# --- 2. supervisor alive? respawn --- +if [ -z "$fail" ] && ! pgrep -f "workspace/bin/gcp-tunnel-up\.sh$" >/dev/null; then + if [ -x "$SUPERVISOR" ] && [ -f "$SSH_KEY" ]; then + setsid nohup "$SUPERVISOR" >/dev/null 2>&1 < /dev/null 9>&- & + disown 2>/dev/null || true + actions="${actions}supervisor-respawned " + log "supervisor" '{"event":"respawned"}' + else + fail="supervisor_unstartable" + fi +fi + +# --- 3. cron jobs alive? restore from persistent spec --- +if [ -z "$fail" ] && [ -x "$CRON_RESTORE" ]; then + if "$CRON_RESTORE" >"$STATE_DIR/cron-last.log" 2>&1; then + grep -q "reinstalled" "$STATE_DIR/cron-last.log" \ + && actions="${actions}cron-restored " + else + fail="cron_restore_failed" + fi +fi + +# --- 4. VM forward answers? (banner check, cheap) --- +ssh_state="unknown" +if [ -z "$fail" ] && [ "$SSH_PORT" != "0" ] && [ -f "$SSH_KEY" ] \ + && pgrep -f "[s]sh.*${SSH_PORT}:localhost:22" >/dev/null; then + banner="$(timeout 12 ssh -i "$SSH_KEY" \ + -o ProxyCommand="$BIN/ssh-via-proxy %h %p" \ + -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \ + -o ConnectTimeout=8 -o BatchMode=yes \ + "$GCP_USER@$GCP_HOST" \ + "timeout 5 bash -c 'exec 3<>/dev/tcp/127.0.0.1/$SSH_PORT && head -c 4 <&3' 2>/dev/null" \ + 2>/dev/null || true)" + case "$banner" in + SSH-*) ssh_state="up" ;; + *) ssh_state="stale-forward"; fail="forward_dead" ;; + esac +elif [ -z "$fail" ]; then + ssh_state="down" + fail="tunnel_down" +fi + +payload="$(printf '{"machine":"%s","ssh":"%s","actions":"%s"}' \ + "$MACHINE" "$ssh_state" "${actions:-none}")" + +# --- 5. silent ok, or rate-limited wake on persistent failure --- +if [ -z "$fail" ]; then + printf 0 > "$FAIL_COUNT" + silent "uptime watcher poll ok" "$payload" + exit 0 +fi + +count=$(( $(read_int "$FAIL_COUNT") + 1 )) +printf '%s' "$count" > "$FAIL_COUNT" +log "failure" "{\"condition\":\"$fail\",\"consec\":\"$count\"}" +if [ "$count" -ge 2 ]; then + now=$(date +%s); last=$(read_int "$LAST_WAKE") + if [ $(( now - last )) -ge 1800 ]; then + printf '%s' "$now" > "$LAST_WAKE" + wake "$fail" "$payload" + exit 0 + fi +fi +silent "failure $fail ($count) — below wake threshold" "$payload" diff --git a/tests/test_uptime_watcher.py b/tests/test_uptime_watcher.py new file mode 100644 index 0000000..7eae922 --- /dev/null +++ b/tests/test_uptime_watcher.py @@ -0,0 +1,37 @@ +import os +import subprocess +import tempfile +import unittest + +REPO_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +SCRIPT_PATH = os.path.join(REPO_DIR, "cloud-uptime", "uptime-watcher.sh") + +class TestUptimeWatcher(unittest.TestCase): + def test_script_exists_and_executable(self): + self.assertTrue(os.path.exists(SCRIPT_PATH), f"Script missing: {SCRIPT_PATH}") + self.assertTrue(os.access(SCRIPT_PATH, os.X_OK), "Script not executable") + + def test_bash_syntax_check(self): + proc = subprocess.run(["bash", "-n", SCRIPT_PATH], capture_output=True, text=True) + self.assertEqual(proc.returncode, 0, f"Bash syntax error: {proc.stderr}") + + def test_watcher_execution_in_sandbox(self): + with tempfile.TemporaryDirectory() as tmpdir: + hooks_state = os.path.join(tmpdir, "hooks", "state", "uptime-watcher") + os.makedirs(hooks_state, exist_ok=True) + ws_tunnel = os.path.join(tmpdir, "workspace", "tunnel") + os.makedirs(ws_tunnel, exist_ok=True) + + env_file = os.path.join(ws_tunnel, "machine.env") + with open(env_file, "w") as f: + f.write("MUSE_MACHINE=test-node\nSSH_PORT=2224\nTERM_PORT=7681\n") + + env = os.environ.copy() + env["HOME"] = tmpdir + # Running with dry environment should safely exit (fail count tracked) + proc = subprocess.run([SCRIPT_PATH], env=env, capture_output=True, text=True) + # The script exits 0 even on fail unless fatal crash, logging status + self.assertTrue(os.path.exists(os.path.join(hooks_state, "consec_failures"))) + +if __name__ == "__main__": + unittest.main()