ops(nh3-dev): seat healthz watchdog — the 5-day pull-only outage must page, not age

althing-seat-daemon took an outside SIGTERM on 2026-09-27 and exited 0, so
Restart=on-failure never revived it; hermes-gateway sat pull-only for five
days with a yellow verdict nobody was watching. Fixed the death mode with a
Restart=always drop-in (also on the jekyll twin, same latent bug) and added
this watchdog for every other death mode: 5-min timer, alert on the second
consecutive non-200 (~10 min sustained), re-alert at most hourly, recovery
mail on green, exit 2 distinguishes a broken alarm wire from a down seat
(backup-freshness precedent). Lifecycle exercised end-to-end against a dead
port before install.
This commit is contained in:
vh
2026-10-02 07:47:17 -07:00
parent 902e16630f
commit 59fb6b77a8
2 changed files with 131 additions and 0 deletions
+38
View File
@@ -0,0 +1,38 @@
#!/usr/bin/env bash
# install-seat-healthz-timer.sh — install/refresh the seat-healthz watchdog as
# a systemd USER timer on nh3-dev (5-min ticks; alert after 2 consecutive
# failing ticks ~= 10 min sustained). Idempotent; re-run after editing the
# wrapper. Pattern follows install-backup-freshness-timer.sh.
set -euo pipefail
UNIT_DIR="$HOME/.config/systemd/user"
REPO=/home/lkraven/development/eshpfi-management
mkdir -p "$UNIT_DIR"
cat > "$UNIT_DIR/seat-healthz.service" <<EOF
[Unit]
Description=Seat status-page healthz watchdog + althing alert
After=network-online.target
[Service]
Type=oneshot
Environment=PATH=/home/lkraven/.local/bin:/usr/local/bin:/usr/bin:/bin
Environment=ALTHING_POST_OFFICE=http://10.100.50.40:8390
ExecStart=$REPO/scripts/seat-healthz-alert.sh
EOF
cat > "$UNIT_DIR/seat-healthz.timer" <<EOF
[Unit]
Description=5-min tick for the seat healthz watchdog
[Timer]
OnBootSec=2min
OnUnitActiveSec=5min
RandomizedDelaySec=30
[Install]
WantedBy=timers.target
EOF
systemctl --user daemon-reload
systemctl --user enable --now seat-healthz.timer
systemctl --user list-timers seat-healthz.timer --no-pager | tail -2
+93
View File
@@ -0,0 +1,93 @@
#!/usr/bin/env bash
# seat-healthz-alert.sh — alarm when the seat status page (:8766/healthz) is
# non-200 for more than ~10 minutes, and tell people when it comes back.
#
# WHY (2026-10-02, highseat-dev report 01M3YH4W): althing-seat-daemon.service
# took an outside SIGTERM on 2026-09-27 and its handler exits 0/SUCCESS, so
# Restart=on-failure never brought it back. hermes-gateway sat pull-only for
# FIVE DAYS with a yellow verdict on a Homepage card nobody was watching. The
# Restart=always drop-in fixes that one death mode; this script watches the
# symptom (healthz non-200) so ANY death mode pages instead of aging.
#
# A single bad tick does NOT alert — daemon restarts and herald route-
# reconcile lag blip healthz for up to ~30s and a flapping alarm is an alarm
# people learn to ignore. Alert on the SECOND consecutive failing tick
# (timer runs every 5 min => sustained ~10 min), then re-alert at most every
# 60 min while down, and send a RECOVERY line when it goes green.
#
# Send path follows the backup-freshness precedent: send AS hermes-gateway
# TO infra-hermes. Never alert to your own handle (mirror trap), and the
# sender is the monitored unit's handle so the mail names the right thing.
#
# 0 healthy (or alerted-down within cooldown; not our problem this tick)
# 1 DOWN past the grace, alert delivered
# 2 ALERT PATH ITSELF FAILED (the broken-wire class from 2026-09-19 —
# distinguish "the seat is down" from "nobody was told")
#
# --test-alert sends a real message through the real path (positive control).
set -uo pipefail
URL="${SEAT_HEALTHZ_URL:-http://localhost:8766/healthz}"
STATE_DIR="${SEAT_HEALTHZ_STATE:-$HOME/.cache/seat-healthz}"
POSTBOX="${POSTBOX:-$HOME/.local/bin/postbox}"
export ALTHING_POST_OFFICE="${ALTHING_POST_OFFICE:-http://10.100.50.40:8390}"
RE_ALERT_SECS="${SEAT_REALERT_SECS:-3600}"
mkdir -p "$STATE_DIR"
send() { # send SUBJECT BODY -> exit status of postbox
"$POSTBOX" send --handle hermes-gateway --to infra-hermes \
--subject "$1" --body "$2" >/dev/null 2>&1
}
if [ "${1:-}" = "--test-alert" ]; then
send "TEST — seat healthz alarm path" \
"Positive control from seat-healthz-alert.sh on nh3-dev: this message proves the alarm wire (hermes-gateway -> infra-hermes via the post office) is intact. No action needed." \
&& { echo "test alert delivered"; exit 0; } \
|| { echo "ALERT PATH BROKEN: postbox send failed"; exit 2; }
fi
# curl prints 000 AND fails on connection-refused, so `|| echo 000` appends a
# second one ("000000" in the alert body). Capture, then normalize instead.
code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 10 "$URL" 2>/dev/null)"; rc=$?
[ "$rc" -ne 0 ] && code=000
if [ "$code" = "200" ]; then
if [ -f "$STATE_DIR/alerted" ]; then
down_since="$(cat "$STATE_DIR/alerted" 2>/dev/null || echo '?')"
rm -f "$STATE_DIR/alerted" "$STATE_DIR/lastalert" "$STATE_DIR/failing_since"
send "RECOVERED — seat healthz green again" \
"healthz on nh3-dev:8766 is 200 again (was failing since $down_since). Check postbox status --handle hermes-gateway for push/reachable before closing." \
|| { echo "ALERT PATH BROKEN on recovery send"; exit 2; } # broken wire still
echo "recovered (was down since $down_since)"
else
echo "healthy"
fi
rm -f "$STATE_DIR/failing"
exit 0
fi
# non-200 path
if [ ! -f "$STATE_DIR/failing" ]; then
echo "$code" > "$STATE_DIR/failing"
echo "healthz=$code (first failing tick — grace, no alert)"
exit 0
fi
echo "$code" > "$STATE_DIR/failing"
now="$(date +%s)"
if [ -f "$STATE_DIR/lastalert" ] && [ $(( now - $(tr -dc '0-9' < "$STATE_DIR/lastalert") )) -lt "$RE_ALERT_SECS" ]; then
echo "healthz=$code down, within re-alert cooldown"
exit 0
fi
ts="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
send "DOWN>10m — seat healthz non-200 (last=$code)" \
"GET $URL has been non-200 for at least two consecutive 5-min ticks (last code $code). Since 2026-09-27 this daemon has a Restart=always drop-in, so a repeat of the SIGTERM-exit-0 death should self-heal — a sustained failure here means something worse (script crash loop, page unit dead, route wedge). Check: systemctl --user status althing-seat-daemon althing-seat-page; curl -s localhost:8766/api/status. First failing tick seen: $(cat "$STATE_DIR/failing_since" 2>/dev/null || echo "$ts")." \
|| { echo "ALERT PATH BROKEN: postbox send failed while healthz=$code"; exit 2; }
echo "$(date +%s)" > "$STATE_DIR/lastalert"
[ -f "$STATE_DIR/failing_since" ] || echo "$ts" > "$STATE_DIR/failing_since"
[ -f "$STATE_DIR/alerted" ] || echo "$ts" > "$STATE_DIR/alerted"
echo "ALERTED: healthz=$code sustained past grace"
exit 1