#!/usr/bin/env bash # seat-healthz-alert.sh — alarm when the seat status page (:8766/healthz) is # non-200 for more than ~10 minutes, and tell people when it comes back. # # WHY (2026-10-02, highseat-dev report 01M3YH4W): althing-seat-daemon.service # took an outside SIGTERM on 2026-09-27 and its handler exits 0/SUCCESS, so # Restart=on-failure never brought it back. hermes-gateway sat pull-only for # FIVE DAYS with a yellow verdict on a Homepage card nobody was watching. The # Restart=always drop-in fixes that one death mode; this script watches the # symptom (healthz non-200) so ANY death mode pages instead of aging. # # A single bad tick does NOT alert — daemon restarts and herald route- # reconcile lag blip healthz for up to ~30s and a flapping alarm is an alarm # people learn to ignore. Alert on the SECOND consecutive failing tick # (timer runs every 5 min => sustained ~10 min), then re-alert at most every # 60 min while down, and send a RECOVERY line when it goes green. # # Send path follows the backup-freshness precedent: send AS hermes-gateway # TO infra-hermes. Never alert to your own handle (mirror trap), and the # sender is the monitored unit's handle so the mail names the right thing. # # 0 healthy (or alerted-down within cooldown; not our problem this tick) # 1 DOWN past the grace, alert delivered # 2 ALERT PATH ITSELF FAILED (the broken-wire class from 2026-09-19 — # distinguish "the seat is down" from "nobody was told") # # --test-alert sends a real message through the real path (positive control). set -uo pipefail URL="${SEAT_HEALTHZ_URL:-http://localhost:8766/healthz}" STATE_DIR="${SEAT_HEALTHZ_STATE:-$HOME/.cache/seat-healthz}" POSTBOX="${POSTBOX:-$HOME/.local/bin/postbox}" export ALTHING_POST_OFFICE="${ALTHING_POST_OFFICE:-http://10.100.50.40:8390}" RE_ALERT_SECS="${SEAT_REALERT_SECS:-3600}" mkdir -p "$STATE_DIR" send() { # send SUBJECT BODY -> exit status of postbox "$POSTBOX" send --handle hermes-gateway --to infra-hermes \ --subject "$1" --body "$2" >/dev/null 2>&1 } if [ "${1:-}" = "--test-alert" ]; then send "TEST — seat healthz alarm path" \ "Positive control from seat-healthz-alert.sh on nh3-dev: this message proves the alarm wire (hermes-gateway -> infra-hermes via the post office) is intact. No action needed." \ && { echo "test alert delivered"; exit 0; } \ || { echo "ALERT PATH BROKEN: postbox send failed"; exit 2; } fi # curl prints 000 AND fails on connection-refused, so `|| echo 000` appends a # second one ("000000" in the alert body). Capture, then normalize instead. code="$(curl -s -o /dev/null -w '%{http_code}' --max-time 10 "$URL" 2>/dev/null)"; rc=$? [ "$rc" -ne 0 ] && code=000 if [ "$code" = "200" ]; then if [ -f "$STATE_DIR/alerted" ]; then down_since="$(cat "$STATE_DIR/alerted" 2>/dev/null || echo '?')" rm -f "$STATE_DIR/alerted" "$STATE_DIR/lastalert" "$STATE_DIR/failing_since" send "RECOVERED — seat healthz green again" \ "healthz on nh3-dev:8766 is 200 again (was failing since $down_since). Check postbox status --handle hermes-gateway for push/reachable before closing." \ || { echo "ALERT PATH BROKEN on recovery send"; exit 2; } # broken wire still echo "recovered (was down since $down_since)" else echo "healthy" fi rm -f "$STATE_DIR/failing" exit 0 fi # non-200 path if [ ! -f "$STATE_DIR/failing" ]; then echo "$code" > "$STATE_DIR/failing" echo "healthz=$code (first failing tick — grace, no alert)" exit 0 fi echo "$code" > "$STATE_DIR/failing" now="$(date +%s)" if [ -f "$STATE_DIR/lastalert" ] && [ $(( now - $(tr -dc '0-9' < "$STATE_DIR/lastalert") )) -lt "$RE_ALERT_SECS" ]; then echo "healthz=$code down, within re-alert cooldown" exit 0 fi ts="$(date -u +%Y-%m-%dT%H:%M:%SZ)" send "DOWN>10m — seat healthz non-200 (last=$code)" \ "GET $URL has been non-200 for at least two consecutive 5-min ticks (last code $code). Since 2026-09-27 this daemon has a Restart=always drop-in, so a repeat of the SIGTERM-exit-0 death should self-heal — a sustained failure here means something worse (script crash loop, page unit dead, route wedge). Check: systemctl --user status althing-seat-daemon althing-seat-page; curl -s localhost:8766/api/status. First failing tick seen: $(cat "$STATE_DIR/failing_since" 2>/dev/null || echo "$ts")." \ || { echo "ALERT PATH BROKEN: postbox send failed while healthz=$code"; exit 2; } echo "$(date +%s)" > "$STATE_DIR/lastalert" [ -f "$STATE_DIR/failing_since" ] || echo "$ts" > "$STATE_DIR/failing_since" [ -f "$STATE_DIR/alerted" ] || echo "$ts" > "$STATE_DIR/alerted" echo "ALERTED: healthz=$code sustained past grace" exit 1