Add cdp-latency-check.sh and box cdp-latency action

Proper script replacing the inline-SSH latency monitor. Probes each relay /json/version via pinned ports from netvm-names.sh, outputs name:latency_ms:code per node. box-ctl.py cdp-latency runs it and returns JSON.
This commit is contained in:
operator-main
2026-10-04 18:34:52 +00:00
parent 882dfd8254
commit a76a776a87
2 changed files with 65 additions and 0 deletions
+32
View File
@@ -697,6 +697,35 @@ def act_relay_health():
out(all_healthy, relays=results, healthy=all_healthy)
def act_cdp_latency():
"""Run cdp-latency-check.sh and return per-node latency JSON."""
audit("cdp-latency")
script = BIN / "cdp-latency-check.sh"
try:
r = subprocess.run([str(script)], capture_output=True, text=True, timeout=120)
except subprocess.TimeoutExpired:
fail("LATENCY_TIMEOUT", "cdp-latency-check.sh timed out")
nodes = []
for line in r.stdout.strip().split("\n"):
parts = line.strip().split(":")
if len(parts) != 3:
continue
name, lat, code = parts
entry = {"node": name.strip(), "http_code": code.strip()}
if lat.strip() == "FAIL":
entry["ok"] = False
entry["latency_ms"] = None
else:
try:
entry["latency_ms"] = int(lat.strip())
except ValueError:
entry["latency_ms"] = None
entry["ok"] = code.strip() == "200"
nodes.append(entry)
all_ok = bool(nodes) and all(n["ok"] for n in nodes)
out(all_ok, nodes=nodes)
def act_chrome_errors():
"""Run chrome-error-scan.sh and return per-profile error counts as JSON."""
audit("chrome-errors")
@@ -1101,6 +1130,7 @@ fleet:
chrome-errors per-profile chrome FATAL/crash counts (new since watermark)
watchdog-alerts
relay-health
cdp-latency
timer actions:
timer-list
@@ -1333,6 +1363,8 @@ def main(argv):
act_watchdog_alerts()
elif action == "relay-health":
act_relay_health()
elif action == "cdp-latency":
act_cdp_latency()
elif action == "chrome-errors":
if rest:
fail("BAD_ARGS", "usage: chrome-errors")