2 Commits

3 changed files with 168 additions and 0 deletions
+130
View File
@@ -0,0 +1,130 @@
#!/usr/bin/env bash
# uptime-watcher.sh — simple hatch-hook watcher: spawn/rebuild from spec.
#
# Register as a hatch hook (id `uptime-watcher`, poll 120s, timeout 300s)
# alongside tunnel-keeper. Each poll it guarantees the three things a
# container rebuild destroys:
# 1. provisioning — runs recover-after-rebuild.sh on a fresh root fs
# 2. supervisor — respawns gcp-tunnel-up.sh if it died
# 3. cron jobs — reinstalls crontab from ~/workspace/cron/*.persist
#
# It also verifies the VM-side SSH forward answers a banner, and wakes the
# operator (rate-limited, 30 min) only when something stays broken across
# polls. Silent on success. Safe to run by hand or from cron too.
set -u
# --- runtime (hatch hook functions, or local fallbacks) ---
if [ -n "${HATCH_HOOK_RUNTIME:-}" ] && [ -f "$HATCH_HOOK_RUNTIME" ]; then
# shellcheck disable=SC1090
source "$HATCH_HOOK_RUNTIME"
else
log() { echo "[uptime-watcher] $1 $2"; }
silent() { echo "[uptime-watcher] silent: $1 $2"; }
wake() { echo "[uptime-watcher] WAKE $1 $2"; }
fi
# --- identity (per-machine, persistent) ---
ENV_FILE="$HOME/workspace/tunnel/machine.env"
# shellcheck disable=SC1090
[ -f "$ENV_FILE" ] && . "$ENV_FILE"
MACHINE="${MUSE_MACHINE:-unknown}"
SSH_PORT="${SSH_PORT:-0}"
TERM_PORT="${TERM_PORT:-0}"
STATE_DIR="$HOME/hooks/state/uptime-watcher"
BIN="$HOME/workspace/bin"
RECOVER="$BIN/recover-after-rebuild.sh"
SUPERVISOR="$BIN/gcp-tunnel-up.sh"
CRON_RESTORE="$BIN/persistent-crontab.sh"
SSH_KEY="$HOME/.ssh/vm_to_gcp"
GCP_HOST="${FD_VM_HOST:-34.139.37.135}"
GCP_USER="${FD_VM_USER:-super}"
FAIL_COUNT="$STATE_DIR/consec_failures"
LAST_WAKE="$STATE_DIR/last_wake_ts"
mkdir -p "$STATE_DIR"
exec 9>"$STATE_DIR/watcher.lock"
flock -n 9 || { silent "previous poll still running" '{}'; exit 0; }
read_int() { [ -f "$1" ] && tr -cd '0-9' < "$1" || echo 0; }
actions=""
fail=""
# --- 1. fresh rebuild? provision ---
if [ ! -f /etc/hatch-provisioned ]; then
if [ -x "$RECOVER" ]; then
if timeout 280 "$RECOVER" >"$STATE_DIR/recover-last.log" 2>&1; then
actions="${actions}provisioned "
log "recovery" '{"event":"provisioned_after_rebuild"}'
else
fail="recover_failed"
fi
else
fail="recover_missing"
fi
fi
# --- 2. supervisor alive? respawn ---
if [ -z "$fail" ] && ! pgrep -f "workspace/bin/gcp-tunnel-up\.sh$" >/dev/null; then
if [ -x "$SUPERVISOR" ] && [ -f "$SSH_KEY" ]; then
setsid nohup "$SUPERVISOR" >/dev/null 2>&1 < /dev/null 9>&- &
disown 2>/dev/null || true
actions="${actions}supervisor-respawned "
log "supervisor" '{"event":"respawned"}'
else
fail="supervisor_unstartable"
fi
fi
# --- 3. cron jobs alive? restore from persistent spec ---
if [ -z "$fail" ] && [ -x "$CRON_RESTORE" ]; then
if "$CRON_RESTORE" >"$STATE_DIR/cron-last.log" 2>&1; then
grep -q "reinstalled" "$STATE_DIR/cron-last.log" \
&& actions="${actions}cron-restored "
else
fail="cron_restore_failed"
fi
fi
# --- 4. VM forward answers? (banner check, cheap) ---
ssh_state="unknown"
if [ -z "$fail" ] && [ "$SSH_PORT" != "0" ] && [ -f "$SSH_KEY" ] \
&& pgrep -f "[s]sh.*${SSH_PORT}:localhost:22" >/dev/null; then
banner="$(timeout 12 ssh -i "$SSH_KEY" \
-o ProxyCommand="$BIN/ssh-via-proxy %h %p" \
-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \
-o ConnectTimeout=8 -o BatchMode=yes \
"$GCP_USER@$GCP_HOST" \
"timeout 5 bash -c 'exec 3<>/dev/tcp/127.0.0.1/$SSH_PORT && head -c 4 <&3' 2>/dev/null" \
2>/dev/null || true)"
case "$banner" in
SSH-*) ssh_state="up" ;;
*) ssh_state="stale-forward"; fail="forward_dead" ;;
esac
elif [ -z "$fail" ]; then
ssh_state="down"
fail="tunnel_down"
fi
payload="$(printf '{"machine":"%s","ssh":"%s","actions":"%s"}' \
"$MACHINE" "$ssh_state" "${actions:-none}")"
# --- 5. silent ok, or rate-limited wake on persistent failure ---
if [ -z "$fail" ]; then
printf 0 > "$FAIL_COUNT"
silent "uptime watcher poll ok" "$payload"
exit 0
fi
count=$(( $(read_int "$FAIL_COUNT") + 1 ))
printf '%s' "$count" > "$FAIL_COUNT"
log "failure" "{\"condition\":\"$fail\",\"consec\":\"$count\"}"
if [ "$count" -ge 2 ]; then
now=$(date +%s); last=$(read_int "$LAST_WAKE")
if [ $(( now - last )) -ge 1800 ]; then
printf '%s' "$now" > "$LAST_WAKE"
wake "$fail" "$payload"
exit 0
fi
fi
silent "failure $fail ($count) — below wake threshold" "$payload"
+1
View File
@@ -0,0 +1 @@
ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEn6qqPrW7Vc77pUEBnLRDBF+yX11qyWzDTjZ2+FtL7b def@netvm
+37
View File
@@ -0,0 +1,37 @@
import os
import subprocess
import tempfile
import unittest
REPO_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
SCRIPT_PATH = os.path.join(REPO_DIR, "cloud-uptime", "uptime-watcher.sh")
class TestUptimeWatcher(unittest.TestCase):
def test_script_exists_and_executable(self):
self.assertTrue(os.path.exists(SCRIPT_PATH), f"Script missing: {SCRIPT_PATH}")
self.assertTrue(os.access(SCRIPT_PATH, os.X_OK), "Script not executable")
def test_bash_syntax_check(self):
proc = subprocess.run(["bash", "-n", SCRIPT_PATH], capture_output=True, text=True)
self.assertEqual(proc.returncode, 0, f"Bash syntax error: {proc.stderr}")
def test_watcher_execution_in_sandbox(self):
with tempfile.TemporaryDirectory() as tmpdir:
hooks_state = os.path.join(tmpdir, "hooks", "state", "uptime-watcher")
os.makedirs(hooks_state, exist_ok=True)
ws_tunnel = os.path.join(tmpdir, "workspace", "tunnel")
os.makedirs(ws_tunnel, exist_ok=True)
env_file = os.path.join(ws_tunnel, "machine.env")
with open(env_file, "w") as f:
f.write("MUSE_MACHINE=test-node\nSSH_PORT=2224\nTERM_PORT=7681\n")
env = os.environ.copy()
env["HOME"] = tmpdir
# Running with dry environment should safely exit (fail count tracked)
proc = subprocess.run([SCRIPT_PATH], env=env, capture_output=True, text=True)
# The script exits 0 even on fail unless fatal crash, logging status
self.assertTrue(os.path.exists(os.path.join(hooks_state, "consec_failures")))
if __name__ == "__main__":
unittest.main()