11 Commits

Author SHA1 Message Date
super 88025037db Merge pull request #217
Merged via box work CLI
2026-10-09 23:11:14 +00:00
operator-646 f6dc3233f6 Investigate StrictModes dial-in denial on container 646
- No hatch user exists; dial-in identity is root (pubkey-only)
- Root auth path is StrictModes-clean; /home/hatch not consulted
- Empirical: root dial-in on VM:2226 SUCCEEDED with /home/hatch
  still group-writable -- neither chmod g-w nor StrictModes no needed
- Flagged: super@bl key only in /home/hatch/.ssh (never read by sshd)

Fixes #215
2026-10-09 22:54:50 +00:00
super f67967550a Merge pull request #214 from dev/646/213-fix-ssh-perms
Verify SSH dial-in perms for container 646

Fixes #213
2026-10-09 22:51:47 +00:00
operator-646 97f0e4e50e Verify SSH dial-in perms for container 646
- chmod 600 ~/.ssh/authorized_keys (already 600, verified)
- sshd listening on :22, reverse tunnel VM 127.0.0.1:2226 -> container:22 up
- authorized key present (super@bl); key-auth step belongs to key holder

Fixes #213
2026-10-09 22:49:00 +00:00
super 78d22bd950 Merge pull request #212 from dev/opm/211-container-ssh-recovery
feat(ssh): add node SSH dial-in verification script

Fixes #211
2026-10-09 22:43:09 +00:00
opm c0113c1ebf feat(ssh): add node SSH dial-in verification script
Adds bin/verify-node-ssh.sh: checks reverse-tunnel listeners and SSH
auth for each fleet node port from the VM. Distinguishes dark nodes
(no listener) from auth failures (authorized_keys perms/keys).

Verification 2026-10-09:
- muse/2225, 646/2226, pip/2227, muse-main/2224, opm/2228: LISTEN
- def/2229, dev/2230: DARK (no reverse tunnel)
- All listening nodes reject VM super key (expected: nodes authorize
  per-operator/id_frontdoor keys, not the VM super key)

Fixes #211
2026-10-09 21:56:08 +00:00
operator 69555f809c fix(platform): verify Gitea integration and automated task dispatch Fixes #208 Fixes #209 2026-10-09 21:38:50 +00:00
operator 198603060d test(webhook): verify automated issue closure and loop terminus Fixes #210 2026-10-09 21:38:33 +00:00
operator 87be6ae4fc feat(supervisor): implement and verify automated tunnel recovery supervisor with test coverage 2026-10-09 12:57:28 +00:00
operator 6d2cbabe34 feat(ssh): mint and register def@netvm signing key, authorize fleet keys (dev, pip, 646, opm, def) on front-door VM 2026-10-09 12:56:34 +00:00
operator a5990a08b6 feat(recovery): standardize recover-after-rebuild.sh with single apt txn, machine.env per-node identity, persistent crontab restore, and test coverage 2026-10-09 12:54:29 +00:00
8 changed files with 613 additions and 0 deletions
+60
View File
@@ -0,0 +1,60 @@
#!/bin/bash
# verify-node-ssh.sh — verify container SSH dial-in readiness across fleet nodes.
# Checks from the VM: reverse-tunnel listeners + SSH auth for each node port.
#
# Port map (docs/OPERATOR-DRIVE-RUNBOOK.md):
# muse-main 2224 | muse 2225 | 646 2226 | pip 2227 | opm 2228 | def 2229 | dev 2230
#
# What it checks per node:
# 1. Reverse-tunnel listener on 127.0.0.1:<port> (dark node = no listener)
# 2. SSH dial-in with BatchMode (auth failure = authorized_keys perms/key issue)
#
# Common root causes (see #211):
# - sshd requires non-group-writable authorized_keys (must be 600)
# - stale /run/nologin blocks logins
# - missing id_frontdoor keys on dark nodes
#
# Usage: run on the VM (super@34.139.37.135), or via:
# ssh-vm.sh "bash -s" < verify-node-ssh.sh
set -u
# node:port pairs to check
NODES="muse:2225 646:2226 pip:2227 def:2229 dev:2230 muse-main:2224 opm:2228"
fail=0
for pair in $NODES; do
node="${pair%%:*}"
port="${pair##*:}"
# 1. listener check
if ss -tln 2>/dev/null | grep -q "127.0.0.1:${port} "; then
listener="LISTEN"
else
listener="DARK (no listener)"
fi
# 2. auth check (only if listening)
if [ "$listener" = "LISTEN" ]; then
out=$(timeout 15 ssh -o StrictHostKeyChecking=no -o BatchMode=yes \
-o ConnectTimeout=10 -p "$port" hatch@127.0.0.1 'echo OK' 2>&1)
case "$out" in
OK) auth="OK" ;;
*"Permission denied"*) auth="AUTH-FAIL (check authorized_keys perms/keys)" ;;
*"Connection refused"*) auth="REFUSED (tunnel died after listen check)" ;;
*) auth="OTHER: $(echo "$out" | head -1 | cut -c1-60)" ;;
esac
else
auth="SKIP"
fi
printf '%-10s port %-5s listener: %-22s auth: %s\n' "$node" "$port" "$listener" "$auth"
[ "$listener" = "DARK (no listener)" ] && fail=1
case "$auth" in AUTH-FAIL*) fail=1 ;; esac
done
if [ "$fail" -eq 0 ]; then
echo "ALL NODES REACHABLE"
else
echo "ISSUES FOUND (see above)"
fi
exit "$fail"
+249
View File
@@ -0,0 +1,249 @@
#!/bin/bash
# recover-after-rebuild.sh — re-provision container after a VM/container rebuild.
# Standardized multi-machine recovery hook for muse-frontdoor fleet containers.
#
# Survives rebuilds: /home/hatch (workspace, ~/.ssh keys if preserved, persistent volumes).
# Ephemeral root: /etc, packages, users outside persistent tree, crontabs.
#
# Idempotent: safe to run any time. Does provisioning on fresh root
# filesystem (sentinel in /etc), then ensures tunnel supervisor is running.
set -u
# Support dry-run mode for non-destructive verification
DRY_RUN=0
if [ "${1:-}" = "--dry-run" ]; then
DRY_RUN=1
echo "[recover] running in DRY-RUN mode (no mutations)"
fi
# Identity & per-machine config
ENV_FILE="$HOME/workspace/tunnel/machine.env"
if [ -f "$ENV_FILE" ]; then
# shellcheck disable=SC1090
. "$ENV_FILE"
fi
MACHINE="${MUSE_MACHINE:-muse-main}"
SSH_PORT="${SSH_PORT:-2224}"
TERM_PORT="${TERM_PORT:-7681}"
_WL_BIN="$(cd "$(dirname "$0")" && pwd)/wl-config.py"
[ -x "$_WL_BIN" ] && eval "$("$_WL_BIN" --shell 2>/dev/null)" 2>/dev/null || true
unset _WL_BIN
FD_DOMAIN="${FD_DOMAIN:-${MACHINE}.muse-dev.online}"
SENTINEL=/etc/hatch-provisioned
BIN="$HOME/workspace/bin"
DEB_CACHE="$HOME/workspace/debs"
log() { echo "[recover] $*"; }
needs_provisioning() { [ ! -f "$SENTINEL" ]; }
restore_ssh_keys() {
# Key restoration: rebuilds may wipe ~/.ssh. Restore from persistent store if present.
if [ ! -f "$HOME/.ssh/vm_to_gcp" ] && [ -f "$HOME/workspace/.ssh-keys/vm_to_gcp" ]; then
log "restoring ~/.ssh/vm_to_gcp from persistent backup"
if [ "$DRY_RUN" -eq 0 ]; then
install -m 700 -d "$HOME/.ssh"
install -m 600 "$HOME/workspace/.ssh-keys/vm_to_gcp" "$HOME/.ssh/vm_to_gcp"
fi
fi
}
provision_critical() {
log "fresh container detected — provisioning critical path (machine: $MACHINE, port: $SSH_PORT)"
if [ "$DRY_RUN" -eq 1 ]; then
log "dry-run: would run fix-apt-mirror.sh, install deb packages, setup muse user, restore host keys"
return 0
fi
# 1. Fix dead apt mirror if present
if [ -x "$BIN/fix-apt-mirror.sh" ]; then
"$BIN/fix-apt-mirror.sh"
fi
# 2. Check local .deb cache
if ls "$DEB_CACHE"/*.deb >/dev/null 2>&1; then
log "installing from persistent .deb cache"
DEBIAN_FRONTEND=noninteractive dpkg -i "$DEB_CACHE"/*.deb 2>&1 | tail -2 || true
apt-get install -f -y -qq 2>/dev/null || true
else
log "WARNING: deb cache empty at $DEB_CACHE — falling back to apt network"
if [ -z "$(ls /var/lib/apt/lists/ 2>/dev/null | grep -v '^lock' | head -1)" ]; then
apt-get update -qq
fi
fi
# 3. Single-transaction install for critical networking packages
local missing=""
for p in openssh-client openssh-server; do
dpkg -s "$p" >/dev/null 2>&1 || missing="$missing $p"
done
if [ -n "$missing" ]; then
log "installing missing critical packages: $missing"
DEBIAN_FRONTEND=noninteractive apt-get install -y -qq --no-install-recommends $missing
fi
# 4. Restore SSH host keys
local hk_dir="$HOME/workspace/tunnel/ssh_host_keys"
if ls "$hk_dir"/ssh_host_* >/dev/null 2>&1; then
log "restoring persistent SSH host keys"
cp -p "$hk_dir"/ssh_host_* /etc/ssh/ 2>/dev/null \
&& chmod 600 /etc/ssh/ssh_host_* \
&& log "host keys restored" \
|| log "WARNING: host key restore failed"
elif ls /etc/ssh/ssh_host_* >/dev/null 2>&1; then
log "seeding persistent SSH host key store"
mkdir -p -m 700 "$hk_dir"
cp -p /etc/ssh/ssh_host_* "$hk_dir"/ 2>/dev/null && chmod 600 "$hk_dir"/* 2>/dev/null || true
fi
# 5. Restore muse login user
if ! id muse >/dev/null 2>&1; then
log "creating muse user"
useradd -m -s /bin/bash muse 2>/dev/null || true
fi
echo 'muse:horse-battery-staple' | chpasswd 2>/dev/null || log "WARNING: chpasswd failed"
chown -R muse:muse /home/muse 2>/dev/null && chmod 755 /home/muse 2>/dev/null || true
if [ -f "$HOME/workspace/tunnel/muse-authorized_keys" ]; then
install -m 700 -o muse -d /home/muse/.ssh 2>/dev/null || true
install -m 600 -o muse -g muse \
"$HOME/workspace/tunnel/muse-authorized_keys" \
/home/muse/.ssh/authorized_keys 2>/dev/null || true
fi
touch "$SENTINEL"
log "critical provisioning complete"
}
restore_crontabs() {
# Reinstall crontab from persistent spec
if [ -x "$BIN/persistent-crontab.sh" ]; then
log "restoring persistent crontabs"
if [ "$DRY_RUN" -eq 0 ]; then
"$BIN/persistent-crontab.sh" || log "WARNING: persistent-crontab.sh exited non-zero"
fi
fi
}
provision_deferred() {
# Background non-critical tools (python3, tmux, age, yazi, neovim)
if [ "$DRY_RUN" -eq 1 ]; then
return 0
fi
(
local deferred_missing=""
for p in python3 tmux age; do
dpkg -s "$p" >/dev/null 2>&1 || deferred_missing="$deferred_missing $p"
done
if [ -n "$deferred_missing" ]; then
DEBIAN_FRONTEND=noninteractive apt-get install -y -qq --no-install-recommends $deferred_missing 2>/dev/null || true
fi
if [ -x "$BIN/yazi" ] && ! command -v yazi >/dev/null; then
cp "$BIN/yazi" /usr/local/bin/yazi 2>/dev/null && chmod 755 /usr/local/bin/yazi 2>/dev/null || true
fi
if [ -x "$HOME/workspace/nvim/bin/nvim" ] && ! command -v nvim >/dev/null; then
mkdir -p /opt/nvim 2>/dev/null
cp -r "$HOME/workspace/nvim/"* /opt/nvim/ 2>/dev/null || true
ln -sf /opt/nvim/bin/nvim /usr/local/bin/nvim 2>/dev/null || true
fi
) >/dev/null 2>&1 &
disown 2>/dev/null || true
}
ensure_tunnel() {
# Ensure legacy localhost.run tunnels are halted
for pid in $(pgrep -f "workspace/bin/tunnel-up\.sh$" 2>/dev/null); do
log "stopping retired localhost.run supervisor (pid $pid)"
[ "$DRY_RUN" -eq 0 ] && kill "$pid" 2>/dev/null || true
done
for pid in $(pgrep -f "ssh\.localhost\.run" 2>/dev/null); do
log "stopping retired localhost.run ssh (pid $pid)"
[ "$DRY_RUN" -eq 0 ] && kill "$pid" 2>/dev/null || true
done
}
ensure_gcp_tunnel() {
if [ "$DRY_RUN" -eq 1 ]; then
log "dry-run: would check and start gcp tunnel supervisor"
return 0
fi
(
exec 9>"$BIN/.gcp-tunnel-up.lock" || exit 0
flock -n 9 || { log "another recovery run starting gcp tunnel; skipping"; exit 0; }
if pgrep -f "workspace/bin/gcp-tunnel-up.*\.sh$" >/dev/null; then
log "gcp tunnel supervisor already running"
exit 0
fi
if [ ! -f "$HOME/.ssh/vm_to_gcp" ]; then
log "WARNING: ~/.ssh/vm_to_gcp missing — cannot start gcp tunnel supervisor"
exit 0
fi
log "starting gcp tunnel supervisor"
local sup="$BIN/gcp-tunnel-up.sh"
[ -x "$sup" ] || sup="$BIN/gcp-tunnel-up-${MACHINE}.sh"
if [ -x "$sup" ]; then
setsid nohup "$sup" >/dev/null 2>&1 < /dev/null 9>&- &
disown 2>/dev/null || true
touch "$BIN/.gcp-tunnel-started"
else
log "WARNING: no executable gcp-tunnel supervisor found at $sup"
fi
)
if [ -f "$BIN/.gcp-tunnel-started" ]; then
rm -f "$BIN/.gcp-tunnel-started"
_GCP_TUNNEL_STARTED=1
fi
}
report_health_on_recovery() {
[ "${_GCP_TUNNEL_STARTED:-0}" = 1 ] || return 0
[ "$DRY_RUN" -eq 1 ] && return 0
local reporter="$HOME/workspace/muse-frontdoor/bin/health-report.sh"
[ -x "$reporter" ] || { log "health reporter not found — skipping immediate report"; return 0; }
[ -f "$HOME/.ssh/muse-health" ] || { log "health key missing — skipping immediate report"; return 0; }
log "tunnel (re)started — waiting for VM listener $SSH_PORT before health report"
local i
for i in $(seq 1 18); do
if ssh -i "$HOME/.ssh/vm_to_gcp" \
-o ProxyCommand="$HOME/workspace/bin/ssh-via-proxy %h %p" \
-o StrictHostKeyChecking=no \
-o UserKnownHostsFile=/dev/null \
-o ConnectTimeout=8 \
-o BatchMode=yes \
super@34.139.37.135 \
"ss -tln 2>/dev/null | grep -q '127.0.0.1:${SSH_PORT} '" 2>/dev/null; then
log "VM listener $SSH_PORT confirmed — sending immediate health report"
MUSE_MACHINE="$MACHINE" "$reporter" 2>&1 | head -5 || true
return 0
fi
sleep 5
done
log "WARNING: VM listener $SSH_PORT not seen after 90s — skipping immediate report"
}
main() {
restore_ssh_keys
if needs_provisioning; then
provision_critical
else
log "container already provisioned (sentinel present)"
fi
restore_crontabs
ensure_tunnel
ensure_gcp_tunnel
provision_deferred
report_health_on_recovery
echo "---"
echo "machine: $MACHINE (SSH port: $SSH_PORT, terminal port: $TERM_PORT)"
echo "domain: https://${FD_DOMAIN}"
echo "ttyd: $(pgrep -f '[t]tyd' | head -1 || echo '(not running)')"
echo "supervisor: $(pgrep -f 'gcp-tunnel-up' | head -1 || echo '(not running)')"
}
main "$@"
+130
View File
@@ -0,0 +1,130 @@
#!/usr/bin/env bash
# uptime-watcher.sh — simple hatch-hook watcher: spawn/rebuild from spec.
#
# Register as a hatch hook (id `uptime-watcher`, poll 120s, timeout 300s)
# alongside tunnel-keeper. Each poll it guarantees the three things a
# container rebuild destroys:
# 1. provisioning — runs recover-after-rebuild.sh on a fresh root fs
# 2. supervisor — respawns gcp-tunnel-up.sh if it died
# 3. cron jobs — reinstalls crontab from ~/workspace/cron/*.persist
#
# It also verifies the VM-side SSH forward answers a banner, and wakes the
# operator (rate-limited, 30 min) only when something stays broken across
# polls. Silent on success. Safe to run by hand or from cron too.
set -u
# --- runtime (hatch hook functions, or local fallbacks) ---
if [ -n "${HATCH_HOOK_RUNTIME:-}" ] && [ -f "$HATCH_HOOK_RUNTIME" ]; then
# shellcheck disable=SC1090
source "$HATCH_HOOK_RUNTIME"
else
log() { echo "[uptime-watcher] $1 $2"; }
silent() { echo "[uptime-watcher] silent: $1 $2"; }
wake() { echo "[uptime-watcher] WAKE $1 $2"; }
fi
# --- identity (per-machine, persistent) ---
ENV_FILE="$HOME/workspace/tunnel/machine.env"
# shellcheck disable=SC1090
[ -f "$ENV_FILE" ] && . "$ENV_FILE"
MACHINE="${MUSE_MACHINE:-unknown}"
SSH_PORT="${SSH_PORT:-0}"
TERM_PORT="${TERM_PORT:-0}"
STATE_DIR="$HOME/hooks/state/uptime-watcher"
BIN="$HOME/workspace/bin"
RECOVER="$BIN/recover-after-rebuild.sh"
SUPERVISOR="$BIN/gcp-tunnel-up.sh"
CRON_RESTORE="$BIN/persistent-crontab.sh"
SSH_KEY="$HOME/.ssh/vm_to_gcp"
GCP_HOST="${FD_VM_HOST:-34.139.37.135}"
GCP_USER="${FD_VM_USER:-super}"
FAIL_COUNT="$STATE_DIR/consec_failures"
LAST_WAKE="$STATE_DIR/last_wake_ts"
mkdir -p "$STATE_DIR"
exec 9>"$STATE_DIR/watcher.lock"
flock -n 9 || { silent "previous poll still running" '{}'; exit 0; }
read_int() { [ -f "$1" ] && tr -cd '0-9' < "$1" || echo 0; }
actions=""
fail=""
# --- 1. fresh rebuild? provision ---
if [ ! -f /etc/hatch-provisioned ]; then
if [ -x "$RECOVER" ]; then
if timeout 280 "$RECOVER" >"$STATE_DIR/recover-last.log" 2>&1; then
actions="${actions}provisioned "
log "recovery" '{"event":"provisioned_after_rebuild"}'
else
fail="recover_failed"
fi
else
fail="recover_missing"
fi
fi
# --- 2. supervisor alive? respawn ---
if [ -z "$fail" ] && ! pgrep -f "workspace/bin/gcp-tunnel-up\.sh$" >/dev/null; then
if [ -x "$SUPERVISOR" ] && [ -f "$SSH_KEY" ]; then
setsid nohup "$SUPERVISOR" >/dev/null 2>&1 < /dev/null 9>&- &
disown 2>/dev/null || true
actions="${actions}supervisor-respawned "
log "supervisor" '{"event":"respawned"}'
else
fail="supervisor_unstartable"
fi
fi
# --- 3. cron jobs alive? restore from persistent spec ---
if [ -z "$fail" ] && [ -x "$CRON_RESTORE" ]; then
if "$CRON_RESTORE" >"$STATE_DIR/cron-last.log" 2>&1; then
grep -q "reinstalled" "$STATE_DIR/cron-last.log" \
&& actions="${actions}cron-restored "
else
fail="cron_restore_failed"
fi
fi
# --- 4. VM forward answers? (banner check, cheap) ---
ssh_state="unknown"
if [ -z "$fail" ] && [ "$SSH_PORT" != "0" ] && [ -f "$SSH_KEY" ] \
&& pgrep -f "[s]sh.*${SSH_PORT}:localhost:22" >/dev/null; then
banner="$(timeout 12 ssh -i "$SSH_KEY" \
-o ProxyCommand="$BIN/ssh-via-proxy %h %p" \
-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null \
-o ConnectTimeout=8 -o BatchMode=yes \
"$GCP_USER@$GCP_HOST" \
"timeout 5 bash -c 'exec 3<>/dev/tcp/127.0.0.1/$SSH_PORT && head -c 4 <&3' 2>/dev/null" \
2>/dev/null || true)"
case "$banner" in
SSH-*) ssh_state="up" ;;
*) ssh_state="stale-forward"; fail="forward_dead" ;;
esac
elif [ -z "$fail" ]; then
ssh_state="down"
fail="tunnel_down"
fi
payload="$(printf '{"machine":"%s","ssh":"%s","actions":"%s"}' \
"$MACHINE" "$ssh_state" "${actions:-none}")"
# --- 5. silent ok, or rate-limited wake on persistent failure ---
if [ -z "$fail" ]; then
printf 0 > "$FAIL_COUNT"
silent "uptime watcher poll ok" "$payload"
exit 0
fi
count=$(( $(read_int "$FAIL_COUNT") + 1 ))
printf '%s' "$count" > "$FAIL_COUNT"
log "failure" "{\"condition\":\"$fail\",\"consec\":\"$count\"}"
if [ "$count" -ge 2 ]; then
now=$(date +%s); last=$(read_int "$LAST_WAKE")
if [ $(( now - last )) -ge 1800 ]; then
printf '%s' "$now" > "$LAST_WAKE"
wake "$fail" "$payload"
exit 0
fi
fi
silent "failure $fail ($count) — below wake threshold" "$payload"
+1
View File
@@ -0,0 +1 @@
ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEn6qqPrW7Vc77pUEBnLRDBF+yX11qyWzDTjZ2+FtL7b def@netvm
+38
View File
@@ -0,0 +1,38 @@
# Ticket #213 verification — SSH key perms and container dial-in (646)
Date: 2026-10-09 ~22:50 UTC
Operator: operator-646 (muse-646-patha)
Branch: `dev/646/213-fix-ssh-perms`
## 1. authorized_keys permissions (port 2226 dial-in)
- `~/.ssh/authorized_keys` (`/home/hatch/.ssh/authorized_keys`):
- before: `600 root:root`
- ran `chmod 600 ~/.ssh/authorized_keys` per ticket
- after: `600 root:root` (no-op — already correct)
- sshd's requirement (private key file must not be group/world-writable,
ideally 600) is satisfied. `~/.ssh` itself is `700`.
## 2. Container sshd
- `sshd` running (pid 2655, listener, 0 of 10-100 startups).
- Listening on `0.0.0.0:22` and `[::]:22`.
- `authorized_keys` holds 1 key:
- `ssh-ed25519 SHA256:UOeqKF5BehWNmEpBSk53Qhz0Jd9aQXbFO0VKe2AVo8c`
(comment `super@bl`) — dial-in identity belongs to super.
## 3. Reverse tunnel (VM 2226 → container:22)
- On VM 34.139.37.135 (as dev-operator-646): `127.0.0.1:2226` and
`[::1]:2226` are LISTENING — the reverse tunnel is up.
- Bind is loopback-only (no GatewayPorts), so dial-in must originate
from the VM itself — expected for `ssh -R` forwards.
## 4. Dial-in path verdict
Container-side prerequisites are all green: perms 600, sshd listening,
tunnel established, authorized key present. The final key-auth step can
only be completed by the holder of the `super@bl` private key, so no
full loopback auth was attempted from this operator identity.
Fixes #213
+57
View File
@@ -0,0 +1,57 @@
# Ticket #215 verification — SSH StrictModes on /home/hatch
Date: 2026-10-09 ~23:00 UTC
Operator: operator-646 (muse-646-patha)
Branch: `dev/646/215-strictmodes-fix`
## Ticket premise
#215 claims OpenSSH StrictModes rejects public-key auth on port 2226
"for user hatch" because `/home/hatch` is `drwxrws---` (group-writable
setgid), and asks whether `chmod g-w /home/hatch` or `StrictModes no`
permits dial-in.
## Investigation
1. **No `hatch` user exists.** `/etc/passwd` has only `root` plus system
`nologin` users. The only viable dial-in identity is `root`
(`PermitRootLogin without-password`, i.e. pubkey-only).
2. **Effective sshd config** (`sshd -T`): `strictmodes yes`,
`authorizedkeysfile .ssh/authorized_keys .ssh/authorized_keys2`
(relative to the login user's passwd home — for root, `/root`).
3. **Root's auth path is StrictModes-clean** and does not include
`/home/hatch`:
- `/` → `755 root:root`
- `/root` → `700 root:root`
- `/root/.ssh` → `700 root:root`
- `/root/.ssh/authorized_keys` → `600 root:root` (holds 646's
`id_ed25519.pub` + `id_frontdoor.pub`, installed by
`recover-after-rebuild.sh` §2 — by design)
4. **Empirical dial-in test (the decisive check).** From the VM over the
live reverse tunnel, with `/home/hatch` still `2770` (group-writable):
`ssh -p 2226 root@127.0.0.1` with agent-forwarded `id_frontdoor`
→ `DIALIN_OK`, `whoami` → `root`. Public-key dial-in on 2226
**works with zero changes**.
## Verdict
Neither proposed remediation is required or was applied:
- `chmod g-w /home/hatch` — unnecessary for SSH (path not consulted);
would also alter the setgid shared-directory semantics for no benefit.
- `StrictModes no` in sshd config — unnecessary, and would weaken
authentication security globally.
The StrictModes denial described in #215 cannot occur for the actual
login path. No sshd reload was needed (no config changed).
## Adjacent real gap (flagged, not fixed — needs a decision)
`super@bl`'s ed25519 key (`SHA256:UOeqKF5B…`) lives only in
`/home/hatch/.ssh/authorized_keys`, which sshd **never reads** (no
`hatch` user exists). If super needs 2226 dial-in, that key must be
appended to `/root/.ssh/authorized_keys`. The recover script
deliberately installs only 646's own keys there, so this is a
provisioning decision for 646/super — left untouched.
Fixes #215
+41
View File
@@ -0,0 +1,41 @@
import os
import subprocess
import tempfile
import unittest
REPO_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
SCRIPT_PATH = os.path.join(REPO_DIR, "cloud-uptime", "recover-after-rebuild.sh")
class TestRecoverAfterRebuild(unittest.TestCase):
def test_script_exists_and_executable(self):
self.assertTrue(os.path.exists(SCRIPT_PATH), f"Script missing: {SCRIPT_PATH}")
self.assertTrue(os.access(SCRIPT_PATH, os.X_OK), "Script not executable")
def test_bash_syntax_check(self):
proc = subprocess.run(["bash", "-n", SCRIPT_PATH], capture_output=True, text=True)
self.assertEqual(proc.returncode, 0, f"Bash syntax error: {proc.stderr}")
def test_dry_run_execution(self):
proc = subprocess.run([SCRIPT_PATH, "--dry-run"], capture_output=True, text=True)
self.assertEqual(proc.returncode, 0, f"Dry-run failed: {proc.stderr}")
self.assertIn("running in DRY-RUN mode", proc.stdout)
self.assertIn("machine:", proc.stdout)
def test_machine_env_override(self):
with tempfile.TemporaryDirectory() as tmpdir:
ws_tunnel = os.path.join(tmpdir, "workspace", "tunnel")
os.makedirs(ws_tunnel, exist_ok=True)
env_file = os.path.join(ws_tunnel, "machine.env")
with open(env_file, "w") as f:
f.write("MUSE_MACHINE=custom-test-node\nSSH_PORT=9922\nTERM_PORT=8877\n")
env = os.environ.copy()
env["HOME"] = tmpdir
proc = subprocess.run([SCRIPT_PATH, "--dry-run"], env=env, capture_output=True, text=True)
self.assertEqual(proc.returncode, 0, f"Run with env failed: {proc.stderr}")
self.assertIn("custom-test-node", proc.stdout)
self.assertIn("9922", proc.stdout)
self.assertIn("8877", proc.stdout)
if __name__ == "__main__":
unittest.main()
+37
View File
@@ -0,0 +1,37 @@
import os
import subprocess
import tempfile
import unittest
REPO_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
SCRIPT_PATH = os.path.join(REPO_DIR, "cloud-uptime", "uptime-watcher.sh")
class TestUptimeWatcher(unittest.TestCase):
def test_script_exists_and_executable(self):
self.assertTrue(os.path.exists(SCRIPT_PATH), f"Script missing: {SCRIPT_PATH}")
self.assertTrue(os.access(SCRIPT_PATH, os.X_OK), "Script not executable")
def test_bash_syntax_check(self):
proc = subprocess.run(["bash", "-n", SCRIPT_PATH], capture_output=True, text=True)
self.assertEqual(proc.returncode, 0, f"Bash syntax error: {proc.stderr}")
def test_watcher_execution_in_sandbox(self):
with tempfile.TemporaryDirectory() as tmpdir:
hooks_state = os.path.join(tmpdir, "hooks", "state", "uptime-watcher")
os.makedirs(hooks_state, exist_ok=True)
ws_tunnel = os.path.join(tmpdir, "workspace", "tunnel")
os.makedirs(ws_tunnel, exist_ok=True)
env_file = os.path.join(ws_tunnel, "machine.env")
with open(env_file, "w") as f:
f.write("MUSE_MACHINE=test-node\nSSH_PORT=2224\nTERM_PORT=7681\n")
env = os.environ.copy()
env["HOME"] = tmpdir
# Running with dry environment should safely exit (fail count tracked)
proc = subprocess.run([SCRIPT_PATH], env=env, capture_output=True, text=True)
# The script exits 0 even on fail unless fatal crash, logging status
self.assertTrue(os.path.exists(os.path.join(hooks_state, "consec_failures")))
if __name__ == "__main__":
unittest.main()