{ "agent": "646", "chain_next": null, "description": "Box service + data health \u2014 every 15 minutes (offset 7m)", "followup": { "escalate": "opm", "expect_reply": true, "nudges": 2, "route": "box-health", "timeout": "30m" }, "name": "box-service-health", "on_failure": "alert", "prompt_template": "Box service and data health check.\nJob ID: {job_id}\nTime: {datetime}\n\nVM check: run /srv/box/bin/box-health-check.sh on the VM (dev-operator-646@34.139.37.135) via your operator SSH chain. Expect zero failures: services 3/3 (board.service, caddy.service, box-request-sweeper.timer), data 3/3, HTTP checks all OK. The request-store WARN is known and not a failure.\n\nbl check: the result harvester lives on bl, not on the VM:\n[TOOL service.status {\"unit\": \"response-harvester.timer\"}]\nExpect active.\n\nNote: board.service and caddy.service are inactive on bl by design (they run on the VM) -- do not check them with the tool; the VM health script covers them.\n\nIf healthy, end with [RESULT {job_id}] OK.\nIf any service fails, reply with [RESULT {job_id}] FAIL .", "schedule": "7,22,37,52 * * * *", "sidechat": { "create": true, "name_template": "box-service-health-{datetime}", "reuse_key": "box-service-health" }, "timeout": 600 }