From 59fb6b77a806ab56d0072c8736223d536d6a95dc Mon Sep 17 00:00:00 2001 From: Vuong Hoang Date: Fri, 2 Oct 2026 07:47:17 -0700 Subject: [PATCH] =?UTF-8?q?ops(nh3-dev):=20seat=20healthz=20watchdog=20?= =?UTF-8?q?=E2=80=94=20the=205-day=20pull-only=20outage=20must=20page,=20n?= =?UTF-8?q?ot=20age?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit althing-seat-daemon took an outside SIGTERM on 2026-09-27 and exited 0, so Restart=on-failure never revived it; hermes-gateway sat pull-only for five days with a yellow verdict nobody was watching. Fixed the death mode with a Restart=always drop-in (also on the jekyll twin, same latent bug) and added this watchdog for every other death mode: 5-min timer, alert on the second consecutive non-200 (~10 min sustained), re-alert at most hourly, recovery mail on green, exit 2 distinguishes a broken alarm wire from a down seat (backup-freshness precedent). Lifecycle exercised end-to-end against a dead port before install. --- scripts/install-seat-healthz-timer.sh | 38 +++++++++++ scripts/seat-healthz-alert.sh | 93 +++++++++++++++++++++++++++ 2 files changed, 131 insertions(+) create mode 100755 scripts/install-seat-healthz-timer.sh create mode 100755 scripts/seat-healthz-alert.sh diff --git a/scripts/install-seat-healthz-timer.sh b/scripts/install-seat-healthz-timer.sh new file mode 100755 index 0000000..d931058 --- /dev/null +++ b/scripts/install-seat-healthz-timer.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# install-seat-healthz-timer.sh — install/refresh the seat-healthz watchdog as +# a systemd USER timer on nh3-dev (5-min ticks; alert after 2 consecutive +# failing ticks ~= 10 min sustained). Idempotent; re-run after editing the +# wrapper. Pattern follows install-backup-freshness-timer.sh. +set -euo pipefail +UNIT_DIR="$HOME/.config/systemd/user" +REPO=/home/lkraven/development/eshpfi-management +mkdir -p "$UNIT_DIR" + +cat > "$UNIT_DIR/seat-healthz.service" < "$UNIT_DIR/seat-healthz.timer" < sustained ~10 min), then re-alert at most every +# 60 min while down, and send a RECOVERY line when it goes green. +# +# Send path follows the backup-freshness precedent: send AS hermes-gateway +# TO infra-hermes. Never alert to your own handle (mirror trap), and the +# sender is the monitored unit's handle so the mail names the right thing. +# +# 0 healthy (or alerted-down within cooldown; not our problem this tick) +# 1 DOWN past the grace, alert delivered +# 2 ALERT PATH ITSELF FAILED (the broken-wire class from 2026-09-19 — +# distinguish "the seat is down" from "nobody was told") +# +# --test-alert sends a real message through the real path (positive control). +set -uo pipefail + +URL="${SEAT_HEALTHZ_URL:-http://localhost:8766/healthz}" +STATE_DIR="${SEAT_HEALTHZ_STATE:-$HOME/.cache/seat-healthz}" +POSTBOX="${POSTBOX:-$HOME/.local/bin/postbox}" +export ALTHING_POST_OFFICE="${ALTHING_POST_OFFICE:-http://10.100.50.40:8390}" +RE_ALERT_SECS="${SEAT_REALERT_SECS:-3600}" + +mkdir -p "$STATE_DIR" + +send() { # send SUBJECT BODY -> exit status of postbox + "$POSTBOX" send --handle hermes-gateway --to infra-hermes \ + --subject "$1" --body "$2" >/dev/null 2>&1 +} + +if [ "${1:-}" = "--test-alert" ]; then + send "TEST — seat healthz alarm path" \ + "Positive control from seat-healthz-alert.sh on nh3-dev: this message proves the alarm wire (hermes-gateway -> infra-hermes via the post office) is intact. No action needed." \ + && { echo "test alert delivered"; exit 0; } \ + || { echo "ALERT PATH BROKEN: postbox send failed"; exit 2; } +fi + +# curl prints 000 AND fails on connection-refused, so `|| echo 000` appends a +# second one ("000000" in the alert body). Capture, then normalize instead. +code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 10 "$URL" 2>/dev/null)"; rc=$? +[ "$rc" -ne 0 ] && code=000 + +if [ "$code" = "200" ]; then + if [ -f "$STATE_DIR/alerted" ]; then + down_since="$(cat "$STATE_DIR/alerted" 2>/dev/null || echo '?')" + rm -f "$STATE_DIR/alerted" "$STATE_DIR/lastalert" "$STATE_DIR/failing_since" + send "RECOVERED — seat healthz green again" \ + "healthz on nh3-dev:8766 is 200 again (was failing since $down_since). Check postbox status --handle hermes-gateway for push/reachable before closing." \ + || { echo "ALERT PATH BROKEN on recovery send"; exit 2; } # broken wire still + echo "recovered (was down since $down_since)" + else + echo "healthy" + fi + rm -f "$STATE_DIR/failing" + exit 0 +fi + +# non-200 path +if [ ! -f "$STATE_DIR/failing" ]; then + echo "$code" > "$STATE_DIR/failing" + echo "healthz=$code (first failing tick — grace, no alert)" + exit 0 +fi +echo "$code" > "$STATE_DIR/failing" + +now="$(date +%s)" +if [ -f "$STATE_DIR/lastalert" ] && [ $(( now - $(tr -dc '0-9' < "$STATE_DIR/lastalert") )) -lt "$RE_ALERT_SECS" ]; then + echo "healthz=$code down, within re-alert cooldown" + exit 0 +fi + +ts="$(date -u +%Y-%m-%dT%H:%M:%SZ)" +send "DOWN>10m — seat healthz non-200 (last=$code)" \ + "GET $URL has been non-200 for at least two consecutive 5-min ticks (last code $code). Since 2026-09-27 this daemon has a Restart=always drop-in, so a repeat of the SIGTERM-exit-0 death should self-heal — a sustained failure here means something worse (script crash loop, page unit dead, route wedge). Check: systemctl --user status althing-seat-daemon althing-seat-page; curl -s localhost:8766/api/status. First failing tick seen: $(cat "$STATE_DIR/failing_since" 2>/dev/null || echo "$ts")." \ + || { echo "ALERT PATH BROKEN: postbox send failed while healthz=$code"; exit 2; } + +echo "$(date +%s)" > "$STATE_DIR/lastalert" +[ -f "$STATE_DIR/failing_since" ] || echo "$ts" > "$STATE_DIR/failing_since" +[ -f "$STATE_DIR/alerted" ] || echo "$ts" > "$STATE_DIR/alerted" +echo "ALERTED: healthz=$code sustained past grace" +exit 1