#!/usr/bin/env bash # uptime-watcher.sh — simple hatch-hook watcher: spawn/rebuild from spec. # # Register as a hatch hook (id `uptime-watcher`, poll 120s, timeout 300s) # alongside tunnel-keeper. Each poll it guarantees the three things a # container rebuild destroys: # 1. provisioning — runs recover-after-rebuild.sh on a fresh root fs # 2. supervisor — respawns gcp-tunnel-up.sh if it died # 3. cron jobs — reinstalls crontab from ~/workspace/cron/*.persist # # It also verifies the VM-side SSH forward answers a banner, and wakes the # operator (rate-limited, 30 min) only when something stays broken across # polls. Silent on success. Safe to run by hand or from cron too. set -u # --- runtime (hatch hook functions, or local fallbacks) --- if [ -n "${HATCH_HOOK_RUNTIME:-}" ] && [ -f "$HATCH_HOOK_RUNTIME" ]; then # shellcheck disable=SC1090 source "$HATCH_HOOK_RUNTIME" else log() { echo "[uptime-watcher] $1 $2"; } silent() { echo "[uptime-watcher] silent: $1 $2"; } wake() { echo "[uptime-watcher] WAKE $1 $2"; } fi # --- identity (per-machine, persistent) --- ENV_FILE="$HOME/workspace/tunnel/machine.env" # shellcheck disable=SC1090 [ -f "$ENV_FILE" ] && . "$ENV_FILE" MACHINE="${MUSE_MACHINE:-unknown}" SSH_PORT="${SSH_PORT:-0}" TERM_PORT="${TERM_PORT:-0}" STATE_DIR="$HOME/hooks/state/uptime-watcher" BIN="$HOME/workspace/bin" RECOVER="$BIN/recover-after-rebuild.sh" SUPERVISOR="$BIN/gcp-tunnel-up.sh" CRON_RESTORE="$BIN/persistent-crontab.sh" SSH_KEY="$HOME/.ssh/vm_to_gcp" GCP_HOST="${FD_VM_HOST:-34.139.37.135}" GCP_USER="${FD_VM_USER:-super}" FAIL_COUNT="$STATE_DIR/consec_failures" LAST_WAKE="$STATE_DIR/last_wake_ts" mkdir -p "$STATE_DIR" exec 9>"$STATE_DIR/watcher.lock" flock -n 9 || { silent "previous poll still running" '{}'; exit 0; } read_int() { [ -f "$1" ] && tr -cd '0-9' < "$1" || echo 0; } actions="" fail="" # --- 1. fresh rebuild? provision --- if [ ! -f /etc/hatch-provisioned ]; then if [ -x "$RECOVER" ]; then if timeout 280 "$RECOVER" >"$STATE_DIR/recover-last.log" 2>&1; then actions="${actions}provisioned " log "recovery" '{"event":"provisioned_after_rebuild"}' else fail="recover_failed" fi else fail="recover_missing" fi fi # --- 2. supervisor alive? respawn --- if [ -z "$fail" ] && ! pgrep -f "workspace/bin/gcp-tunnel-up\.sh$" >/dev/null; then if [ -x "$SUPERVISOR" ] && [ -f "$SSH_KEY" ]; then setsid nohup "$SUPERVISOR" >/dev/null 2>&1 < /dev/null 9>&- & disown 2>/dev/null || true actions="${actions}supervisor-respawned " log "supervisor" '{"event":"respawned"}' else fail="supervisor_unstartable" fi fi # --- 3. cron jobs alive? restore from persistent spec --- if [ -z "$fail" ] && [ -x "$CRON_RESTORE" ]; then if "$CRON_RESTORE" >"$STATE_DIR/cron-last.log" 2>&1; then grep -q "reinstalled" "$STATE_DIR/cron-last.log" \ && actions="${actions}cron-restored " else fail="cron_restore_failed" fi fi # --- 4. VM forward answers? (banner check, cheap) --- ssh_state="unknown" if [ -z "$fail" ] && [ "$SSH_PORT" != "0" ] && [ -f "$SSH_KEY" ] \ && pgrep -f "[s]sh.*${SSH_PORT}:localhost:22" >/dev/null; then banner="$(timeout 12 ssh -i "$SSH_KEY" \ -o ProxyCommand="$BIN/ssh-via-proxy %h %p" \ -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \ -o ConnectTimeout=8 -o BatchMode=yes \ "$GCP_USER@$GCP_HOST" \ "timeout 5 bash -c 'exec 3<>/dev/tcp/127.0.0.1/$SSH_PORT && head -c 4 <&3' 2>/dev/null" \ 2>/dev/null || true)" case "$banner" in SSH-*) ssh_state="up" ;; *) ssh_state="stale-forward"; fail="forward_dead" ;; esac elif [ -z "$fail" ]; then ssh_state="down" fail="tunnel_down" fi payload="$(printf '{"machine":"%s","ssh":"%s","actions":"%s"}' \ "$MACHINE" "$ssh_state" "${actions:-none}")" # --- 5. silent ok, or rate-limited wake on persistent failure --- if [ -z "$fail" ]; then printf 0 > "$FAIL_COUNT" silent "uptime watcher poll ok" "$payload" exit 0 fi count=$(( $(read_int "$FAIL_COUNT") + 1 )) printf '%s' "$count" > "$FAIL_COUNT" log "failure" "{\"condition\":\"$fail\",\"consec\":\"$count\"}" if [ "$count" -ge 2 ]; then now=$(date +%s); last=$(read_int "$LAST_WAKE") if [ $(( now - last )) -ge 1800 ]; then printf '%s' "$now" > "$LAST_WAKE" wake "$fail" "$payload" exit 0 fi fi silent "failure $fail ($count) — below wake threshold" "$payload"