Compare commits
8 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| f6dc3233f6 | |||
| f67967550a | |||
| 97f0e4e50e | |||
| 78d22bd950 | |||
| c0113c1ebf | |||
| 69555f809c | |||
| 198603060d | |||
| 87be6ae4fc |
Executable
+60
@@ -0,0 +1,60 @@
|
||||
#!/bin/bash
|
||||
# verify-node-ssh.sh — verify container SSH dial-in readiness across fleet nodes.
|
||||
# Checks from the VM: reverse-tunnel listeners + SSH auth for each node port.
|
||||
#
|
||||
# Port map (docs/OPERATOR-DRIVE-RUNBOOK.md):
|
||||
# muse-main 2224 | muse 2225 | 646 2226 | pip 2227 | opm 2228 | def 2229 | dev 2230
|
||||
#
|
||||
# What it checks per node:
|
||||
# 1. Reverse-tunnel listener on 127.0.0.1:<port> (dark node = no listener)
|
||||
# 2. SSH dial-in with BatchMode (auth failure = authorized_keys perms/key issue)
|
||||
#
|
||||
# Common root causes (see #211):
|
||||
# - sshd requires non-group-writable authorized_keys (must be 600)
|
||||
# - stale /run/nologin blocks logins
|
||||
# - missing id_frontdoor keys on dark nodes
|
||||
#
|
||||
# Usage: run on the VM (super@34.139.37.135), or via:
|
||||
# ssh-vm.sh "bash -s" < verify-node-ssh.sh
|
||||
set -u
|
||||
|
||||
# node:port pairs to check
|
||||
NODES="muse:2225 646:2226 pip:2227 def:2229 dev:2230 muse-main:2224 opm:2228"
|
||||
|
||||
fail=0
|
||||
for pair in $NODES; do
|
||||
node="${pair%%:*}"
|
||||
port="${pair##*:}"
|
||||
|
||||
# 1. listener check
|
||||
if ss -tln 2>/dev/null | grep -q "127.0.0.1:${port} "; then
|
||||
listener="LISTEN"
|
||||
else
|
||||
listener="DARK (no listener)"
|
||||
fi
|
||||
|
||||
# 2. auth check (only if listening)
|
||||
if [ "$listener" = "LISTEN" ]; then
|
||||
out=$(timeout 15 ssh -o StrictHostKeyChecking=no -o BatchMode=yes \
|
||||
-o ConnectTimeout=10 -p "$port" hatch@127.0.0.1 'echo OK' 2>&1)
|
||||
case "$out" in
|
||||
OK) auth="OK" ;;
|
||||
*"Permission denied"*) auth="AUTH-FAIL (check authorized_keys perms/keys)" ;;
|
||||
*"Connection refused"*) auth="REFUSED (tunnel died after listen check)" ;;
|
||||
*) auth="OTHER: $(echo "$out" | head -1 | cut -c1-60)" ;;
|
||||
esac
|
||||
else
|
||||
auth="SKIP"
|
||||
fi
|
||||
|
||||
printf '%-10s port %-5s listener: %-22s auth: %s\n' "$node" "$port" "$listener" "$auth"
|
||||
[ "$listener" = "DARK (no listener)" ] && fail=1
|
||||
case "$auth" in AUTH-FAIL*) fail=1 ;; esac
|
||||
done
|
||||
|
||||
if [ "$fail" -eq 0 ]; then
|
||||
echo "ALL NODES REACHABLE"
|
||||
else
|
||||
echo "ISSUES FOUND (see above)"
|
||||
fi
|
||||
exit "$fail"
|
||||
Executable
+130
@@ -0,0 +1,130 @@
|
||||
#!/usr/bin/env bash
|
||||
# uptime-watcher.sh — simple hatch-hook watcher: spawn/rebuild from spec.
|
||||
#
|
||||
# Register as a hatch hook (id `uptime-watcher`, poll 120s, timeout 300s)
|
||||
# alongside tunnel-keeper. Each poll it guarantees the three things a
|
||||
# container rebuild destroys:
|
||||
# 1. provisioning — runs recover-after-rebuild.sh on a fresh root fs
|
||||
# 2. supervisor — respawns gcp-tunnel-up.sh if it died
|
||||
# 3. cron jobs — reinstalls crontab from ~/workspace/cron/*.persist
|
||||
#
|
||||
# It also verifies the VM-side SSH forward answers a banner, and wakes the
|
||||
# operator (rate-limited, 30 min) only when something stays broken across
|
||||
# polls. Silent on success. Safe to run by hand or from cron too.
|
||||
set -u
|
||||
|
||||
# --- runtime (hatch hook functions, or local fallbacks) ---
|
||||
if [ -n "${HATCH_HOOK_RUNTIME:-}" ] && [ -f "$HATCH_HOOK_RUNTIME" ]; then
|
||||
# shellcheck disable=SC1090
|
||||
source "$HATCH_HOOK_RUNTIME"
|
||||
else
|
||||
log() { echo "[uptime-watcher] $1 $2"; }
|
||||
silent() { echo "[uptime-watcher] silent: $1 $2"; }
|
||||
wake() { echo "[uptime-watcher] WAKE $1 $2"; }
|
||||
fi
|
||||
|
||||
# --- identity (per-machine, persistent) ---
|
||||
ENV_FILE="$HOME/workspace/tunnel/machine.env"
|
||||
# shellcheck disable=SC1090
|
||||
[ -f "$ENV_FILE" ] && . "$ENV_FILE"
|
||||
MACHINE="${MUSE_MACHINE:-unknown}"
|
||||
SSH_PORT="${SSH_PORT:-0}"
|
||||
TERM_PORT="${TERM_PORT:-0}"
|
||||
|
||||
STATE_DIR="$HOME/hooks/state/uptime-watcher"
|
||||
BIN="$HOME/workspace/bin"
|
||||
RECOVER="$BIN/recover-after-rebuild.sh"
|
||||
SUPERVISOR="$BIN/gcp-tunnel-up.sh"
|
||||
CRON_RESTORE="$BIN/persistent-crontab.sh"
|
||||
SSH_KEY="$HOME/.ssh/vm_to_gcp"
|
||||
GCP_HOST="${FD_VM_HOST:-34.139.37.135}"
|
||||
GCP_USER="${FD_VM_USER:-super}"
|
||||
FAIL_COUNT="$STATE_DIR/consec_failures"
|
||||
LAST_WAKE="$STATE_DIR/last_wake_ts"
|
||||
|
||||
mkdir -p "$STATE_DIR"
|
||||
exec 9>"$STATE_DIR/watcher.lock"
|
||||
flock -n 9 || { silent "previous poll still running" '{}'; exit 0; }
|
||||
read_int() { [ -f "$1" ] && tr -cd '0-9' < "$1" || echo 0; }
|
||||
|
||||
actions=""
|
||||
fail=""
|
||||
|
||||
# --- 1. fresh rebuild? provision ---
|
||||
if [ ! -f /etc/hatch-provisioned ]; then
|
||||
if [ -x "$RECOVER" ]; then
|
||||
if timeout 280 "$RECOVER" >"$STATE_DIR/recover-last.log" 2>&1; then
|
||||
actions="${actions}provisioned "
|
||||
log "recovery" '{"event":"provisioned_after_rebuild"}'
|
||||
else
|
||||
fail="recover_failed"
|
||||
fi
|
||||
else
|
||||
fail="recover_missing"
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- 2. supervisor alive? respawn ---
|
||||
if [ -z "$fail" ] && ! pgrep -f "workspace/bin/gcp-tunnel-up\.sh$" >/dev/null; then
|
||||
if [ -x "$SUPERVISOR" ] && [ -f "$SSH_KEY" ]; then
|
||||
setsid nohup "$SUPERVISOR" >/dev/null 2>&1 < /dev/null 9>&- &
|
||||
disown 2>/dev/null || true
|
||||
actions="${actions}supervisor-respawned "
|
||||
log "supervisor" '{"event":"respawned"}'
|
||||
else
|
||||
fail="supervisor_unstartable"
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- 3. cron jobs alive? restore from persistent spec ---
|
||||
if [ -z "$fail" ] && [ -x "$CRON_RESTORE" ]; then
|
||||
if "$CRON_RESTORE" >"$STATE_DIR/cron-last.log" 2>&1; then
|
||||
grep -q "reinstalled" "$STATE_DIR/cron-last.log" \
|
||||
&& actions="${actions}cron-restored "
|
||||
else
|
||||
fail="cron_restore_failed"
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- 4. VM forward answers? (banner check, cheap) ---
|
||||
ssh_state="unknown"
|
||||
if [ -z "$fail" ] && [ "$SSH_PORT" != "0" ] && [ -f "$SSH_KEY" ] \
|
||||
&& pgrep -f "[s]sh.*${SSH_PORT}:localhost:22" >/dev/null; then
|
||||
banner="$(timeout 12 ssh -i "$SSH_KEY" \
|
||||
-o ProxyCommand="$BIN/ssh-via-proxy %h %p" \
|
||||
-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \
|
||||
-o ConnectTimeout=8 -o BatchMode=yes \
|
||||
"$GCP_USER@$GCP_HOST" \
|
||||
"timeout 5 bash -c 'exec 3<>/dev/tcp/127.0.0.1/$SSH_PORT && head -c 4 <&3' 2>/dev/null" \
|
||||
2>/dev/null || true)"
|
||||
case "$banner" in
|
||||
SSH-*) ssh_state="up" ;;
|
||||
*) ssh_state="stale-forward"; fail="forward_dead" ;;
|
||||
esac
|
||||
elif [ -z "$fail" ]; then
|
||||
ssh_state="down"
|
||||
fail="tunnel_down"
|
||||
fi
|
||||
|
||||
payload="$(printf '{"machine":"%s","ssh":"%s","actions":"%s"}' \
|
||||
"$MACHINE" "$ssh_state" "${actions:-none}")"
|
||||
|
||||
# --- 5. silent ok, or rate-limited wake on persistent failure ---
|
||||
if [ -z "$fail" ]; then
|
||||
printf 0 > "$FAIL_COUNT"
|
||||
silent "uptime watcher poll ok" "$payload"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
count=$(( $(read_int "$FAIL_COUNT") + 1 ))
|
||||
printf '%s' "$count" > "$FAIL_COUNT"
|
||||
log "failure" "{\"condition\":\"$fail\",\"consec\":\"$count\"}"
|
||||
if [ "$count" -ge 2 ]; then
|
||||
now=$(date +%s); last=$(read_int "$LAST_WAKE")
|
||||
if [ $(( now - last )) -ge 1800 ]; then
|
||||
printf '%s' "$now" > "$LAST_WAKE"
|
||||
wake "$fail" "$payload"
|
||||
exit 0
|
||||
fi
|
||||
fi
|
||||
silent "failure $fail ($count) — below wake threshold" "$payload"
|
||||
@@ -0,0 +1,38 @@
|
||||
# Ticket #213 verification — SSH key perms and container dial-in (646)
|
||||
|
||||
Date: 2026-10-09 ~22:50 UTC
|
||||
Operator: operator-646 (muse-646-patha)
|
||||
Branch: `dev/646/213-fix-ssh-perms`
|
||||
|
||||
## 1. authorized_keys permissions (port 2226 dial-in)
|
||||
|
||||
- `~/.ssh/authorized_keys` (`/home/hatch/.ssh/authorized_keys`):
|
||||
- before: `600 root:root`
|
||||
- ran `chmod 600 ~/.ssh/authorized_keys` per ticket
|
||||
- after: `600 root:root` (no-op — already correct)
|
||||
- sshd's requirement (private key file must not be group/world-writable,
|
||||
ideally 600) is satisfied. `~/.ssh` itself is `700`.
|
||||
|
||||
## 2. Container sshd
|
||||
|
||||
- `sshd` running (pid 2655, listener, 0 of 10-100 startups).
|
||||
- Listening on `0.0.0.0:22` and `[::]:22`.
|
||||
- `authorized_keys` holds 1 key:
|
||||
- `ssh-ed25519 SHA256:UOeqKF5BehWNmEpBSk53Qhz0Jd9aQXbFO0VKe2AVo8c`
|
||||
(comment `super@bl`) — dial-in identity belongs to super.
|
||||
|
||||
## 3. Reverse tunnel (VM 2226 → container:22)
|
||||
|
||||
- On VM 34.139.37.135 (as dev-operator-646): `127.0.0.1:2226` and
|
||||
`[::1]:2226` are LISTENING — the reverse tunnel is up.
|
||||
- Bind is loopback-only (no GatewayPorts), so dial-in must originate
|
||||
from the VM itself — expected for `ssh -R` forwards.
|
||||
|
||||
## 4. Dial-in path verdict
|
||||
|
||||
Container-side prerequisites are all green: perms 600, sshd listening,
|
||||
tunnel established, authorized key present. The final key-auth step can
|
||||
only be completed by the holder of the `super@bl` private key, so no
|
||||
full loopback auth was attempted from this operator identity.
|
||||
|
||||
Fixes #213
|
||||
@@ -0,0 +1,57 @@
|
||||
# Ticket #215 verification — SSH StrictModes on /home/hatch
|
||||
|
||||
Date: 2026-10-09 ~23:00 UTC
|
||||
Operator: operator-646 (muse-646-patha)
|
||||
Branch: `dev/646/215-strictmodes-fix`
|
||||
|
||||
## Ticket premise
|
||||
|
||||
#215 claims OpenSSH StrictModes rejects public-key auth on port 2226
|
||||
"for user hatch" because `/home/hatch` is `drwxrws---` (group-writable
|
||||
setgid), and asks whether `chmod g-w /home/hatch` or `StrictModes no`
|
||||
permits dial-in.
|
||||
|
||||
## Investigation
|
||||
|
||||
1. **No `hatch` user exists.** `/etc/passwd` has only `root` plus system
|
||||
`nologin` users. The only viable dial-in identity is `root`
|
||||
(`PermitRootLogin without-password`, i.e. pubkey-only).
|
||||
2. **Effective sshd config** (`sshd -T`): `strictmodes yes`,
|
||||
`authorizedkeysfile .ssh/authorized_keys .ssh/authorized_keys2`
|
||||
(relative to the login user's passwd home — for root, `/root`).
|
||||
3. **Root's auth path is StrictModes-clean** and does not include
|
||||
`/home/hatch`:
|
||||
- `/` → `755 root:root`
|
||||
- `/root` → `700 root:root`
|
||||
- `/root/.ssh` → `700 root:root`
|
||||
- `/root/.ssh/authorized_keys` → `600 root:root` (holds 646's
|
||||
`id_ed25519.pub` + `id_frontdoor.pub`, installed by
|
||||
`recover-after-rebuild.sh` §2 — by design)
|
||||
4. **Empirical dial-in test (the decisive check).** From the VM over the
|
||||
live reverse tunnel, with `/home/hatch` still `2770` (group-writable):
|
||||
`ssh -p 2226 root@127.0.0.1` with agent-forwarded `id_frontdoor`
|
||||
→ `DIALIN_OK`, `whoami` → `root`. Public-key dial-in on 2226
|
||||
**works with zero changes**.
|
||||
|
||||
## Verdict
|
||||
|
||||
Neither proposed remediation is required or was applied:
|
||||
|
||||
- `chmod g-w /home/hatch` — unnecessary for SSH (path not consulted);
|
||||
would also alter the setgid shared-directory semantics for no benefit.
|
||||
- `StrictModes no` in sshd config — unnecessary, and would weaken
|
||||
authentication security globally.
|
||||
|
||||
The StrictModes denial described in #215 cannot occur for the actual
|
||||
login path. No sshd reload was needed (no config changed).
|
||||
|
||||
## Adjacent real gap (flagged, not fixed — needs a decision)
|
||||
|
||||
`super@bl`'s ed25519 key (`SHA256:UOeqKF5B…`) lives only in
|
||||
`/home/hatch/.ssh/authorized_keys`, which sshd **never reads** (no
|
||||
`hatch` user exists). If super needs 2226 dial-in, that key must be
|
||||
appended to `/root/.ssh/authorized_keys`. The recover script
|
||||
deliberately installs only 646's own keys there, so this is a
|
||||
provisioning decision for 646/super — left untouched.
|
||||
|
||||
Fixes #215
|
||||
@@ -0,0 +1,37 @@
|
||||
import os
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
REPO_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
SCRIPT_PATH = os.path.join(REPO_DIR, "cloud-uptime", "uptime-watcher.sh")
|
||||
|
||||
class TestUptimeWatcher(unittest.TestCase):
|
||||
def test_script_exists_and_executable(self):
|
||||
self.assertTrue(os.path.exists(SCRIPT_PATH), f"Script missing: {SCRIPT_PATH}")
|
||||
self.assertTrue(os.access(SCRIPT_PATH, os.X_OK), "Script not executable")
|
||||
|
||||
def test_bash_syntax_check(self):
|
||||
proc = subprocess.run(["bash", "-n", SCRIPT_PATH], capture_output=True, text=True)
|
||||
self.assertEqual(proc.returncode, 0, f"Bash syntax error: {proc.stderr}")
|
||||
|
||||
def test_watcher_execution_in_sandbox(self):
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
hooks_state = os.path.join(tmpdir, "hooks", "state", "uptime-watcher")
|
||||
os.makedirs(hooks_state, exist_ok=True)
|
||||
ws_tunnel = os.path.join(tmpdir, "workspace", "tunnel")
|
||||
os.makedirs(ws_tunnel, exist_ok=True)
|
||||
|
||||
env_file = os.path.join(ws_tunnel, "machine.env")
|
||||
with open(env_file, "w") as f:
|
||||
f.write("MUSE_MACHINE=test-node\nSSH_PORT=2224\nTERM_PORT=7681\n")
|
||||
|
||||
env = os.environ.copy()
|
||||
env["HOME"] = tmpdir
|
||||
# Running with dry environment should safely exit (fail count tracked)
|
||||
proc = subprocess.run([SCRIPT_PATH], env=env, capture_output=True, text=True)
|
||||
# The script exits 0 even on fail unless fatal crash, logging status
|
||||
self.assertTrue(os.path.exists(os.path.join(hooks_state, "consec_failures")))
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user