news-digest: add world + local headlines sections

Two new dense headline rails above the existing reddit/tech cards.
Designed for high-volume "what happened" coverage where the title
is the deliverable — no LLM summarization, ~15 items per section,
6-column-collapsing grid (title / source / time).

Digest pipeline:
  * fetch_miniflux_headlines(category) — flat list per category, dedup
    by lowercased title (different feeds syndicate the same wire stories)
  * 8h look-back window (vs 12h for tech/reddit) since headlines move
    faster
  * cap of 15 per section (DIGEST_MINIFLUX_HEADLINES_MAX)

Frontend:
  * .headline element parallels .item for the hide-button machinery
    (both have data-id, both honored by app.js)
  * dense 3-col layout collapses to 1-col on narrow screens
  * jumpnav now numbers world=01, local=02, reddit=03, tech=04

Setup:
  * seed-headlines.py — one-shot script (lives in the image at
    /app/seed-headlines.py). Creates the World + Local categories in
    miniflux, subscribes a curated feed list, and renames each feed
    to a short display title (BBC vs "BBC News", "LA Times" vs "California").
    Idempotent — reruns only add new feeds.
  * Default world: BBC, NPR, Al Jazeera. Default local: LA Times Local,
    LA Times CA, Voice of OC. (OC Register blocks miniflux; left out.)
  * entrypoint.sh now syncs templates/{style.css,app.js,favicon.svg}
    to /output on container start so frontend asset updates land
    without a manual copy after rebuild.
This commit is contained in:
vh
2026-04-28 11:14:15 -07:00
parent d2ed7671d7
commit 9bdb41ea6a
9 changed files with 391 additions and 13 deletions
+85 -2
View File
@@ -74,6 +74,16 @@ MINIFLUX_TECH_CATEGORY = os.environ.get(
MINIFLUX_HOURS = int(os.environ.get("DIGEST_MINIFLUX_HOURS", "12"))
MINIFLUX_MAX_PER_SOURCE = int(os.environ.get("DIGEST_MINIFLUX_MAX", "8"))
# Headlines (world + local) — high-volume sections, no LLM summarization.
MINIFLUX_WORLD_CATEGORY = os.environ.get(
"DIGEST_MINIFLUX_WORLD_CATEGORY", "World"
)
MINIFLUX_LOCAL_CATEGORY = os.environ.get(
"DIGEST_MINIFLUX_LOCAL_CATEGORY", "Local"
)
MINIFLUX_HEADLINES_HOURS = int(os.environ.get("DIGEST_MINIFLUX_HEADLINES_HOURS", "8"))
MINIFLUX_HEADLINES_MAX = int(os.environ.get("DIGEST_MINIFLUX_HEADLINES_MAX", "15"))
TZ_NAME = os.environ.get("TZ", "America/Los_Angeles")
# ── data shapes ──────────────────────────────────────────────────────
@@ -103,6 +113,16 @@ class Source:
href: str # link to the source's homepage / sub
items: list[Item] = field(default_factory=list)
@dataclass
class Headline:
"""One row in the dense world/local headlines list. Skips
LLM summarization — title is the whole point."""
id: str
title: str
url: str
source: str # display name of the originating feed
posted_at: datetime
# ── http session shared across calls ─────────────────────────────────
S = requests.Session()
@@ -190,6 +210,55 @@ def fetch_miniflux_tech_items() -> list[Source]:
))
return [s for s in by_feed.values() if s.items]
def fetch_miniflux_headlines(category_name: str) -> list[Headline]:
"""Pull recent items from a miniflux category as flat headlines.
Used for high-volume sections (world / local) where headlines move
fast and the volume justifies a dense list rather than the per-source
cards used for tech / reddit. No LLM summarization — the title is
the deliverable. Cross-feed dedup by lowercased title (different
feeds syndicate the same wire stories)."""
cats = miniflux_get("/v1/categories")
cat = next(
(c for c in cats if c["title"].lower() == category_name.lower()),
None,
)
if not cat:
log(f"miniflux: category {category_name!r} not found, skipping")
return []
cutoff = int(
(datetime.now(timezone.utc) - timedelta(hours=MINIFLUX_HEADLINES_HOURS)).timestamp()
)
entries = miniflux_get(
"/v1/entries",
category_id=cat["id"],
published_after=cutoff,
order="published_at",
direction="desc",
limit=200,
)
headlines: list[Headline] = []
seen: set[str] = set()
for e in entries.get("entries", []):
title = (e.get("title") or "(untitled)").strip()
key = title.lower()
if key in seen:
continue
seen.add(key)
feed = e.get("feed") or {}
headlines.append(Headline(
id=_stable_id("headline", str(e["id"])),
title=title,
url=e.get("url", ""),
source=feed.get("title", "?"),
posted_at=_parse_dt(e.get("published_at")),
))
if len(headlines) >= MINIFLUX_HEADLINES_MAX:
break
return headlines
def _parse_dt(s: Optional[str]) -> datetime:
if not s:
return datetime.now(timezone.utc)
@@ -312,6 +381,7 @@ def summarize_source(src: Source) -> None:
# ── render ───────────────────────────────────────────────────────────
def render(reddit_sources: list[Source], tech_sources: list[Source],
world_headlines: list[Headline], local_headlines: list[Headline],
generated_at: datetime) -> str:
env = Environment(
loader=FileSystemLoader(str(TEMPLATE_DIR)),
@@ -328,8 +398,12 @@ def render(reddit_sources: list[Source], tech_sources: list[Source],
return template.render(
reddit_sources=reddit_kept,
tech_sources=tech_kept,
world_headlines=world_headlines,
local_headlines=local_headlines,
reddit_total=sum(len(s.items) for s in reddit_kept),
tech_total=sum(len(s.items) for s in tech_kept),
world_total=len(world_headlines),
local_total=len(local_headlines),
generated_at=generated_at,
edition=edition,
edition_short="AM" if edition == "morning" else "PM",
@@ -439,16 +513,25 @@ def main() -> int:
reddit_sources.append(fetch_reddit_top(sub))
time.sleep(1.5) # gentle to anonymous Reddit
log("phase 3: fetching tech-aggregator items from miniflux")
log("phase 3a: fetching world headlines from miniflux")
world_headlines = fetch_miniflux_headlines(MINIFLUX_WORLD_CATEGORY)
log(f" found {len(world_headlines)} world headlines")
log("phase 3b: fetching local headlines from miniflux")
local_headlines = fetch_miniflux_headlines(MINIFLUX_LOCAL_CATEGORY)
log(f" found {len(local_headlines)} local headlines")
log("phase 3c: fetching tech-aggregator items from miniflux")
tech_sources = fetch_miniflux_tech_items()
log(f" found {len(tech_sources)} non-reddit feeds with recent items")
log("phase 4: summarizing each source via llama-swap")
# Headlines bypass the LLM — title is the whole deliverable.
for src in reddit_sources + tech_sources:
summarize_source(src)
log("phase 5: rendering")
html = render(reddit_sources, tech_sources, now_local)
html = render(reddit_sources, tech_sources, world_headlines, local_headlines, now_local)
write_output(html, now_local)
log("done")
return 0