diff --git a/stacks/homepage/README.md b/stacks/homepage/README.md index b6d8e4b..33c029c 100644 --- a/stacks/homepage/README.md +++ b/stacks/homepage/README.md @@ -84,14 +84,21 @@ Three fixes, all in this stack's config except where noted: `settings.yaml`. Check with `GET /api/services`, which prints live per-group counts. -## ⚠ UNRESOLVED — the tab bar disappeared on container recreate +## RESOLVED — the tab bar takes several minutes to appear after a recreate -**Status: open. The dashboard is degraded but usable.** Since the homepage -container was recreated on 2026-08-18, the client render has lost its tab bar, -its wallpaper, and its i18n. Groups render as side-by-side columns instead of -rows, and the search box shows the raw key `search.search`. Every service, -status pill and widget still works — it is a link board without tabs, not a -dead page. +**Status: closed, and the answer is "wait".** After the 2026-08-18 recreate the +client render came up with no tab bar, no wallpaper and no i18n (the search box +read the raw key `search.search`); groups fell back to side-by-side columns. +It restored itself with no further intervention. **A freshly recreated homepage +container needs a few minutes before the client render is whole** — far longer +than the healthcheck takes to report `healthy`, which is the trap: `docker ps` +says the service is up while the page is still visibly wrong. + +**So: after any `docker compose up -d --force-recreate` here, do not judge the +dashboard for at least ~5 minutes, and do not start changing config to chase +it.** Everything below is the evidence trail from doing exactly that, kept +because it rules out four plausible causes and will save the next session the +same hour. **What it is not** — both obvious suspects were tested and cleared: @@ -112,12 +119,13 @@ widget 403s. `GET /api/validate` returns `[]`. A fresh container never renders tabs here regardless of image version, config version, `PUID`/`PGID`, or whether Docker discovery is mounted at all. -**The one thing that did work** was the container that had been up ~12 hours, -and it cannot be reproduced from image + config. The remaining hypothesis is -that its writable layer held state a fresh container does not rebuild — i.e. -something was changed inside the running container by hand at some point and -was never written back to `/opt/docker/conf/homepage`. If that is right, the -fix is to find out what, because the next recreate would have lost it anyway. +**What it actually was: warm-up time.** Every throwaway container in the list +above was judged within ~30 seconds of starting, which is why they all looked +broken — they were all in the same warm-up window, and that consistency read as +a reproduction when it was really the same mistake five times. The live +container recovered on its own once left alone. The lesson is a measurement +discipline, not a config one, and it sits alongside the existing warning that +`docker ps` health and a correct render are different questions. Before/after evidence: `~/booth-data/homepage-cleanup/` on nh3-dev → `http://10.100.10.50:8090/b/homepage-cleanup/` (24h TTL). diff --git a/stacks/homepage/conf/settings.yaml b/stacks/homepage/conf/settings.yaml index c24ec14..583aae9 100644 --- a/stacks/homepage/conf/settings.yaml +++ b/stacks/homepage/conf/settings.yaml @@ -44,6 +44,17 @@ useEqualHeights: true # AI - Image & Media image/video generation + pipelines # AI - Dormant stopped stacks (rollback seats, retired auditions) # +# AI TAB ORDER IS BY CLICKABILITY, NOT BY IMPORTANCE (operator, 2026-08-18). +# Groups render in the order they appear in this block, so the top of the tab +# is prime real estate and it should hold the things you actually open in a +# browser — chat frontends, ComfyUI, the control plane. Most of the model +# seats below them are vLLM API endpoints whose href is a `/docs` page: they +# are worth SEEING (status at a glance) but not worth reaching for, so they +# sink. Order is therefore: +# interactive UIs -> mixed -> API-only seats -> dormant +# If you add an AI group, place it by asking "would I click this?", not by +# how central the service is to the fleet. +# # COLUMN COUNTS ARE NOT A STYLE CHOICE — they are the member count. # `columns: N` lays the group out N-per-row and leaves the remainder of the # last row as dead space. A 1-member group at columns:4 renders one card and @@ -99,7 +110,28 @@ layout: tab: Main style: row columns: 4 - # --- AI tab: the inference fleet, ordered core-models -> support -> apps --- + # --- AI tab: ordered interactive -> API-only -> dormant (see note above) --- + # Things you open: chat frontends, the control plane, the LiteLLM UI. + AI - Gateways & Chat: + icon: mdi-router-network + tab: AI + style: row + columns: 4 + # ComfyUI is a full node editor and Arbo has a real UI — both get clicked. + AI - Image & Media: + icon: mdi-image-multiple + tab: AI + style: row + columns: 2 + # Mixed: YT Voice Clipper has an audition console, Parakeet is an API. + AI - Audio Tools: + icon: mdi-waveform + tab: AI + style: row + columns: 2 + # Below here: model seats whose href is a vLLM `/docs` page. Status at a + # glance is the whole value; you consume these through the gateway, not by + # clicking them. AI - Inference: icon: mdi-brain tab: AI @@ -110,26 +142,11 @@ layout: tab: AI style: row columns: 5 - AI - Gateways & Chat: - icon: mdi-router-network - tab: AI - style: row - columns: 4 AI - Speech (TTS): icon: mdi-account-voice tab: AI style: row columns: 4 - AI - Audio Tools: - icon: mdi-waveform - tab: AI - style: row - columns: 2 - AI - Image & Media: - icon: mdi-image-multiple - tab: AI - style: row - columns: 2 # Stopped stacks kept for rollback / superseded seats / retired auditions. # They stay 'created' (not running) via `docker compose up --no-start`, so # they show here as offline cards and revive with `docker compose start`. diff --git a/stacks/seafile/compose.yaml b/stacks/seafile/compose.yaml new file mode 100644 index 0000000..9cc3bf5 --- /dev/null +++ b/stacks/seafile/compose.yaml @@ -0,0 +1,91 @@ +# seafile — file sync service on ana-docker, fronted by Traefik at +# seafile.phasefinal.com. +# +# ⚠️ RESTART POLICY IS LOAD-BEARING HERE. None of these three services +# declared one until 2026-08-18, which means Docker defaulted them to `no`. +# On 2026-05-06T21:27:45Z the daemon stopped all three within 200ms of each +# other — a daemon restart or host reboot — and because the policy was `no`, +# nothing brought them back. Seafile then sat dead for three months and the +# only trace was an EXITED card on the dashboard that nobody read as an +# outage. The exit code was 255 on all three, which is just what a container +# that ignores SIGTERM reports when the daemon stops it; it is NOT evidence +# of a crash, and reading it as one sends you looking for a bug that does +# not exist. `unless-stopped` on all three is the fix. +# +# Data lives in two local named volumes (`seafile_seafile_db`, +# `seafile_seafile_datastore`) — NOT on ana-nas NFS, so this stack is not +# exposed to the NAS SPOF that affects postgres/rest-server/PBS. +services: + db: + image: mariadb:10.6 + container_name: seafile-mysql + restart: unless-stopped + environment: + - MYSQL_ROOT_PASSWORD=${DB_ROOT_PW} + - MYSQL_LOG_CONSOLE=true + volumes: + - seafile_db:/var/lib/mysql # Requested, specifies the path to MySQL data persistent store. + networks: + - tnet + # Required so seafile can wait for InnoDB to initialize before + # starting seahub. Without it, after a daemon restart seahub races + # mysql, hits "Connection refused", and the container's start.py + # gives up — leaving nginx serving seafile but proxying to a dead + # python backend. Image ships /usr/local/bin/healthcheck.sh. + healthcheck: + test: ["CMD", "healthcheck.sh", "--connect", "--innodb_initialized"] + interval: 5s + timeout: 5s + retries: 30 + start_period: 30s + memcached: + image: memcached:1.6.18 + container_name: seafile-memcached + restart: unless-stopped + entrypoint: memcached -m 256 + networks: + - tnet + seafile: + image: seafileltd/seafile-mc:11.0-latest + container_name: seafile + restart: unless-stopped + ports: + - 9180:80 + volumes: + - seafile_datastore:/shared # Requested, specifies the path to Seafile data persistent store. + environment: + - DB_HOST=db + - DB_ROOT_PASSWD=${DB_ROOT_PW} + - TIME_ZONE=America/Los_Angeles + - SEAFILE_ADMIN_EMAIL=${SEAFILE_ADMIN_EMAIL} + - SEAFILE_ADMIN_PASSWORD=${SEAFILE_ADMIN_PW} + # Long-form depends_on with condition: service_healthy on db. + # Compose waits for the mariadb healthcheck above to pass before + # starting this container, so seahub never sees the + # "Connection refused" race that wedges it on daemon restart. + depends_on: + db: + condition: service_healthy + memcached: + condition: service_started + labels: + - homepage.group=Apps + - homepage.name=SeaFile + - homepage.icon=mdi-sync-circle + - homepage.description=File Sync Service (ana) + - homepage.href=https://seafile.phasefinal.com + - traefik.enable=true + - traefik.http.routers.seafile.tls=true + - traefik.http.routers.seafile.rule=Host(`seafile.phasefinal.com`) + - traefik.http.routers.seafile.tls.certresolver=anaprod + networks: + - tnet + env_file: + - .env +volumes: + seafile_db: null + seafile_datastore: null +networks: + tnet: + name: traefik-net + external: true diff --git a/stacks/searxng/compose.yaml b/stacks/searxng/compose.yaml new file mode 100644 index 0000000..4306e08 --- /dev/null +++ b/stacks/searxng/compose.yaml @@ -0,0 +1,88 @@ +services: + searxng: + image: searxng/searxng:latest + container_name: searxng + restart: unless-stopped + # ------------------------------------------------------------------ + # Port binding — 9996 on all interfaces. + # Change to "127.0.0.1:9996:8080" to restrict to localhost only. + # Traefik handles public routing and TLS via the labels below. + # ------------------------------------------------------------------ + ports: + - 9996:8080 + # ------------------------------------------------------------------ + # Volumes + # Config: settings.yml bind-mounted read-only into the container. + volumes: + - /opt/docker/conf/searxng/searxng-settings.yml:/etc/searxng/settings.yml:ro + # ------------------------------------------------------------------ + # Environment — see https://docs.searxng.org/admin/settings/index.html + # SEARXNG_SECRET — required for cryptographic signing (cookies, etc.) + # BASE_URL — public URL SearXNG reports in pages/RSS/OPDS + # INSTANCE_NAME — shown in the page title / footer + # ------------------------------------------------------------------ + environment: + - SEARXNG_SECRET=${SEARXNG_SECRET} + - BASE_URL=https://searxng.pfi.local/ + - INSTANCE_NAME=SearXNG + # ------------------------------------------------------------------ + # Resource limits — tune for VM 102's available RAM/CPU + # ------------------------------------------------------------------ + deploy: + resources: + limits: + memory: 512M + cpus: "1.0" + reservations: + memory: 128M + # ------------------------------------------------------------------ + # Health check — SearXNG /healthz is the canonical liveness probe. + # + # ⚠️ `--tries=1` MUST keep its `=1`. This read `- --tries` / `- --spider` + # as two separate argv entries until 2026-08-18, and in that form wget + # consumed `--spider` as the VALUE of `--tries` — so spider mode never + # engaged and every probe DOWNLOADED the response to a file instead of + # just checking it. By the time it was caught the container's working + # directory held 295,287 `healthz.N` files, one per probe since April, + # and wget had to scan all of them to pick the next free filename. That + # scan is what intermittently blew the 10s timeout and made the card on + # the dashboard flap UNHEALTHY while the service itself was fine. It was + # self-worsening: every probe made the next one slower. + # + # The junk lived in the container's writable layer (the only volume here + # is the read-only settings mount), so recreating the container cleared + # it. Symptom to watch for if this regresses: `docker exec searxng ls | + # wc -l` climbing, and health log entries reading + # "Health check exceeded timeout (10s)". + # ------------------------------------------------------------------ + healthcheck: + test: + - CMD + - wget + - --no-verbose + - --tries=1 + - --spider + - http://localhost:8080/healthz + interval: 30s + timeout: 10s + retries: 3 + start_period: 15s + networks: + - tnet + labels: + # Traefik configuration — auto-discovery via Docker provider + - traefik.enable=true + - traefik.http.routers.searxng.rule=Host(`searxng.pfi.local`) + - traefik.http.routers.searxng.entrypoints=websecure + - traefik.http.routers.searxng.tls=true + - traefik.http.routers.searxng.service=searxng + - traefik.http.services.searxng.loadbalancer.server.port=8080 + - homepage.group=Apps + - homepage.name=SearXNG + - homepage.icon=si-searxng + - homepage.description=Privacy-respecting meta-search + - homepage.href=http://10.250.50.70:9996 +networks: + tnet: + name: traefik-net + external: true diff --git a/stacks/searxng/conf/searxng-settings.yml b/stacks/searxng/conf/searxng-settings.yml new file mode 100644 index 0000000..9924b30 --- /dev/null +++ b/stacks/searxng/conf/searxng-settings.yml @@ -0,0 +1,73 @@ +# ============================================================================= +# SearXNG Custom Settings — overrides defaults from the container image +# Full reference: https://docs.searxng.org/admin/settings/index.html +# ============================================================================= + +use_default_settings: + engines: + remove: + - wikidata + - ahmia + - torch + - karmasearch + - karmasearch.videos + - brave + - brave.images + - brave.news + - brave.videos + +general: + instance_name: "SearXNG" + instance_about_url: false + contact_url: false + debug: false + # Disable public metrics page (/stats/errors) to reduce attack surface + enable_metrics: false + +search: + safe_search: 0 + # "" disables; "duckduckgo" is the most private working option + autocomplete: "" + default_lang: "auto" + formats: + - html + - json + # 3s is too tight; 8s covers slower engines without hanging the UI + request_timeout: 8.0 + # Ban time after an engine raises a suspended-time exception (default 86400) + ban_time_on_fail: 60 + max_ban_time_on_fail: 600 + +server: + # REQUIRED. Generate with: openssl rand -hex 32 + # Prefer setting SEARXNG_SECRET in docker-compose and letting the entrypoint + # substitute it; hardcoding a real secret here is a leak risk. + #secret_key: "changeme_please_generate_a_secret" + bind_address: "0.0.0.0" + port: 8080 + # Enable ONLY if you ship a limiter.toml AND your proxy forwards X-Real-IP. + # Otherwise you'll get "X-Forwarded-For nor X-Real-IP header is set!" noise. + limiter: false + # Mark as true if instance is internet-facing; tightens some defaults. + public_instance: false + base_url: "https://searxng.pfi.local/" + # Allow only GET to the search endpoint (simpler, works with most clients) + method: "GET" + compression: true + # Set to true if you need image_proxy rewriting for privacy + image_proxy: false + +# Outgoing HTTP pool — tuned for a low-traffic private instance. +# Defaults are fine for most, but these reduce memory use and tighten timeouts. +outgoing: + request_timeout: 6.0 + max_request_timeout: 12.0 + pool_connections: 100 + pool_maxsize: 20 + enable_http2: true + # Uncomment if you want to route outbound traffic via Tor for .onion engines + # proxies: + # all://: + # - socks5h://tor:9050 + +