From 94899d6fa33c80ccb1f42a980c3df34e7354a2fa Mon Sep 17 00:00:00 2001 From: Vuong Hoang Date: Tue, 22 Sep 2026 00:55:29 -0700 Subject: [PATCH] feat(uptimekuma): normalize names off Homepage, publish the status page, restore the widget MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit NAMES. Homepage already answers "what is this service called", so the monitor name is now that name verbatim -- a second naming authority is how drift starts, and an alert reading "[Uptime Kuma] Beszel hub is DOWN" sends you hunting for a card that does not exist. Only two rows moved (Beszel hub -> Beszel, Dozzle hub -> Dozzle); the " hub" suffixes were mine, not the services'. The remaining mixed case is deliberate and is now documented as such. talk, vor and task-board are lowercase on Homepage and in their own repos; title-casing them here would make this board disagree with both. What actually looked messy was scripts/kuma's own ASCII-ordinal sort, which buried every lowercase name below every capitalised one. Fixed to case-insensitive. ⚠ RENAME SAFETY, which this pass needed and did not have. The seed keys on NAME, so editing a name would have read as a brand-new monitor: added fresh, with the old row orphaned, still checking, still alerting, and holding all the history. `rename_from:` names the old row for one run. Verified: both renamed monitors kept their IDs and all 67 heartbeats. Added with it, an orphan warning for any row on the board the spec no longer names -- because a forgotten monitor keeps paging. Its first cut diffed against the PRE-EDIT snapshot and so cried wolf on its own successful renames; it re-reads the board now. A warning that fires on its own correct work is worse than no warning. STATUS PAGE + WIDGET. The Homepage uptimekuma widget reads a PUBLISHED status page (/api/status-page/), not the admin API -- which is why the widget labels were deliberately absent from the rebuild: a dashboard widget pointed at a 404 is the suspected mechanism behind both of Homepage's unkillable D-state wedges, so shipping one on purpose would have been daft. The page now exists at slug `nethealth` (the pre-rebuild slug, so old references still resolve) and is DECLARED IN monitors.yaml, applied by `kuma seed`. Same principle as the notification channel: a from-scratch rebuild restores the page, the channel and the monitors together, and nothing the widget depends on lives only in Kuma's database. Verified in a browser: "13 SITES UP / 0 SITES DOWN / 100% UPTIME" on the dashboard. ⚠ saveStatusPage calls imgDataUrl.startsWith() unconditionally, so passing null throws and leaves the page CREATED BUT EMPTY -- which reads as success from /api/status-page (200, correct title) while the group list is silently blank. Pass "" instead. Commented at the call site. --- scripts/kuma | 105 +++++++++++++++++++++++++++++--- stacks/uptimekuma/compose.yaml | 21 ++++--- stacks/uptimekuma/monitors.yaml | 26 +++++++- 3 files changed, 136 insertions(+), 16 deletions(-) diff --git a/scripts/kuma b/scripts/kuma index 7fcb95f..1272910 100755 --- a/scripts/kuma +++ b/scripts/kuma @@ -119,6 +119,7 @@ class Kuma: self._monitor_list: dict | None = None self._notification_list: list | None = None + self._status_page_list: dict | None = None @self.sio.on("monitorList") def _on_monitor_list(data): @@ -128,6 +129,10 @@ class Kuma: def _on_notification_list(data): self._notification_list = data or [] + @self.sio.on("statusPageList") + def _on_status_page_list(data): + self._status_page_list = data or {} + def __enter__(self): self.sio.connect(self.url, transports=["websocket"], wait_timeout=20) self._connected = True @@ -172,6 +177,47 @@ class Kuma: raise RuntimeError("no monitorList event within %ss -- not logged in?" % timeout) return self._monitor_list + def status_pages(self, timeout: int = 10) -> list: + deadline = time.time() + timeout + while self._status_page_list is None and time.time() < deadline: + self.sio.sleep(0.2) + return list((self._status_page_list or {}).values()) + + def ensure_status_page(self, slug: str, title: str, groups: list): + """Create the page if absent, then save its config and membership. + + The Homepage `uptimekuma` widget calls /api/status-page/ and + /api/status-page/heartbeat/; without a PUBLISHED page at that slug + the widget polls a 404 forever. That is why the widget labels were held + back when this service was rebuilt -- a dashboard widget pointed at a + dead target is the suspected cause of both of Homepage's unkillable + wedges, so shipping one deliberately would have been daft. + """ + have = {sp.get("slug") for sp in self.status_pages()} + if slug not in have: + self.call("addStatusPage", title, slug) + cfg = { + "slug": slug, "title": title, + "description": "Fleet service layer — is the service actually serving.", + "logo": None, "theme": "dark", "published": True, + "showTags": False, "footerText": None, "customCSS": "", + "showPoweredBy": False, "rssTitle": title, + "showOnlyLastHeartbeat": False, "showCertificateExpiry": False, + "autoRefreshInterval": 300, "domainNameList": [], + "googleAnalyticsId": None, "analyticsId": None, + "analyticsScriptUrl": None, "analyticsType": None, + } + # ⚠ imgDataUrl must be a STRING, not None. The handler calls + # imgDataUrl.startsWith("data:") unconditionally, so null throws + # "Cannot read properties of null" and the page is created but never + # populated -- which looks like success from /api/status-page (200, + # correct title) while the group list is empty. + return self.call("saveStatusPage", (slug, cfg, "", groups), timeout=45) + + def monitors_by_name(self) -> dict: + """Fresh read of the board, keyed by name.""" + return {m.get("name"): m for m in self.monitors().values()} + def delete(self, monitor_id: int): return self.call("deleteMonitor", monitor_id, False) @@ -230,7 +276,10 @@ def cmd_list(args): print("no monitors") return print(f"{'ID':>4} {'NAME':<28} {'TYPE':<6} {'ACT':<4} TARGET") - for m in sorted(mons.values(), key=lambda x: (x.get("name") or "")): + # Case-INSENSITIVE: an ASCII-ordinal sort buries every lowercase + # service name (talk, vor, task-board) below every capitalised one, + # which reads as a messy board when it is really a messy sort. + for m in sorted(mons.values(), key=lambda x: (x.get("name") or "").lower()): tgt = m.get("url") or f"{m.get('hostname','')}:{m.get('port','')}" print(f"{m.get('id'):>4} {(m.get('name') or '')[:28]:<28} " f"{(m.get('type') or '')[:6]:<6} {str(m.get('active')):<4} {tgt[:52]}") @@ -260,18 +309,60 @@ def cmd_seed(args): k.add_notification({**NOTIFICATION_DEFAULTS, **ch}) print(f" + {ch['name']} (channel)") existing = {m.get("name"): m for m in k.monitors().values()} - added = updated = 0 + added = updated = renamed = 0 for m in wanted: - if m["name"] in existing: - merged = {**existing[m["name"]], **m} - k.edit(merged) + spec_m = {kk: vv for kk, vv in m.items() if kk != "rename_from"} + target = existing.get(m["name"]) + + # RENAME SUPPORT, and it is load-bearing rather than a nicety. + # This seed is keyed on NAME, so editing a name in the spec would + # otherwise read as a brand-new monitor: the tool would ADD it and + # leave the old row orphaned, still checking, still alerting, and + # carrying all the history. `rename_from` names the old row for one + # run; drop the line once the rename has landed. + if target is None and m.get("rename_from"): + target = existing.get(m["rename_from"]) + if target is not None: + k.edit({**target, **spec_m}) + renamed += 1 + print(f" > {m['rename_from']} -> {m['name']}") + continue + + if target is not None: + k.edit({**target, **spec_m}) updated += 1 print(f" ~ {m['name']}") else: - k.add(m) + k.add(spec_m) added += 1 print(f" + {m['name']}") - print(f"\n{added} added, {updated} updated, {len(wanted)} in spec") + + # A monitor on the board that the spec no longer names is not silently + # fine -- a forgotten row keeps checking and keeps alerting. + # + # ⚠ RE-READ THE BOARD FIRST. The first cut diffed against `existing`, + # the snapshot taken BEFORE the edits, so every row this run had just + # renamed was still in it under its old name and got reported as an + # orphan that no longer existed. A warning that cries wolf on its own + # successful work is worse than no warning. + spec_names = {m["name"] for m in wanted} + orphans = sorted(n for n in k.monitors_by_name() if n not in spec_names) + if orphans: + print("\n ⚠ on the board but NOT in the spec (left alone, still alerting):") + for o in orphans: + print(f" {o}") + + print(f"\n{added} added, {updated} updated, {renamed} renamed, {len(wanted)} in spec") + + # LAST, because the page references monitor IDs and therefore needs the + # monitors to exist first. + sp = spec.get("status_page") if isinstance(spec, dict) else None + if sp: + board = k.monitors_by_name() + members = [{"id": board[n]["id"]} for n in sorted(board, key=str.lower) if n in board] + k.ensure_status_page(sp["slug"], sp.get("title", "Fleet"), + [{"name": sp.get("group", "Services"), "monitorList": members}]) + print(f" status page /status/{sp['slug']} -> {len(members)} monitors") finally: k.__exit__() diff --git a/stacks/uptimekuma/compose.yaml b/stacks/uptimekuma/compose.yaml index af17af8..fa94e17 100644 --- a/stacks/uptimekuma/compose.yaml +++ b/stacks/uptimekuma/compose.yaml @@ -66,13 +66,20 @@ services: - homepage.description=Service monitoring (fleet) - homepage.href=http://10.250.50.70:3001 - homepage.siteMonitor=http://10.250.50.70:3001 - # ⚠️ NO `homepage.widget.*` LABELS YET, deliberately. The widget needs a - # published status-page slug; on a from-scratch install none exists, so - # the widget would poll a 404 forever. That matters more than usual - # here: homepage widgets pointed at dead targets are the suspected - # mechanism behind BOTH of this dashboard's unkillable D-state wedges - # (see incident_esh_docker_nfs_boot_race, 2026-06-03). Re-add the widget - # labels only once the status page actually exists. + # The widget reads a PUBLISHED status page, not the admin API: + # /api/status-page/ and /api/status-page/heartbeat/. These + # labels were deliberately absent from the 2026-09-21 rebuild until that + # page existed, because a dashboard widget pointed at a 404 is the + # suspected mechanism behind BOTH of Homepage's unkillable D-state wedges + # (incident_esh_docker_nfs_boot_race, 2026-06-03) -- shipping one on + # purpose would have been daft. + # + # The page is defined in monitors.yaml (`status_page:`) and applied by + # `scripts/kuma seed`, so a from-scratch rebuild restores the slug this + # points at. Slug `nethealth` is the one the pre-rebuild instance used. + - homepage.widget.type=uptimekuma + - homepage.widget.url=http://10.250.50.70:3001 + - homepage.widget.slug=nethealth networks: - tnet diff --git a/stacks/uptimekuma/monitors.yaml b/stacks/uptimekuma/monitors.yaml index 9b0c9cd..fd652f6 100644 --- a/stacks/uptimekuma/monitors.yaml +++ b/stacks/uptimekuma/monitors.yaml @@ -13,6 +13,16 @@ # (Beszel's lane: 18 hosts x Status/CPU/Memory/Disk/Temp), and Uptime # Kuma itself (it cannot report its own death — that is Beszel's job). # +# NAMES COME FROM HOMEPAGE, VERBATIM. Homepage already answers "what is this +# service called", and a second naming authority is how drift starts: an alert +# reading "[Uptime Kuma] Beszel hub is DOWN" sends you looking for a card called +# "Beszel hub" that does not exist. So the monitor name IS the dashboard name -- +# which is why this normalisation pass only moved two rows. The remaining mixed +# case (talk, vor, task-board against Gitea, Backrest) is NOT an inconsistency to +# fix: those are the products' own names, lowercase on Homepage and lowercase in +# their own repos. Title-casing them here would make this board disagree with +# both. `rename_from` exists for exactly this operation -- see scripts/kuma. +# # Every URL below was probed before being written here: all returned 200 on # 2026-09-21. A seed that ships red on day one teaches everyone to ignore the # board, which is how you end up with a monitor nobody reads. @@ -30,6 +40,18 @@ # Route: Kuma -> althing-alert-bridge (/kuma) -> postbox -> infra-ops inbox. # Same path Beszel uses, different route, so the subject says which tool spoke: # "[Uptime Kuma] Homepage is DOWN" rather than a Beszel-labelled lie. +# ---- the published status page ----------------------------------------------- +# Exists so Homepage's `uptimekuma` widget has something to read: it calls +# /api/status-page/ and /api/status-page/heartbeat/, and without a +# published page at that slug it polls a 404 forever. The widget labels on +# stacks/uptimekuma/compose.yaml were deliberately held back until this existed. +# Slug kept as `nethealth` -- the same one the pre-rebuild instance used, so any +# bookmark or older reference still resolves. +status_page: + slug: nethealth + title: PFI fleet services + group: Services + notifications: - name: althing (infra-ops) webhookURL: http://10.100.10.50:8096/kuma @@ -78,7 +100,7 @@ monitors: description: fleet voice bench # ---- monitoring + backup: a blind monitor is worse than none ---- - - name: Beszel hub + - name: Beszel url: http://10.250.50.70:8090 description: the host layer; if this is down we are blind to 18 hosts @@ -86,5 +108,5 @@ monitors: url: http://10.250.50.70:9898 description: restic orchestration — a silent backup failure is the expensive kind - - name: Dozzle hub + - name: Dozzle url: http://10.250.50.70:8088