news-digest: real article summaries + per-desk collapse
Two upgrades to make the digest actually readable:
1) Article-grounded 2-3 sentence summaries (everywhere)
The old prompt got just the title + miniflux's content excerpt,
which for HN/Lobsters/wire feeds is barely more than the title
itself — so summaries paraphrased the title and added nothing.
Now every URL gets fetched and main-content-extracted via
trafilatura on a parallel pre-pass (10 workers, ~15s for ~50
URLs). Extracted text caches to /output/.article-cache.json with
a 7-day TTL so repeat runs in the same window don't re-pull.
Headlines also get summarized now — one batched LLM call per
category (world / local). Rendered as a paragraph below the
title with source + time on the right rail.
Prompt rewrites tell the model to pull names/numbers/places
from the body and explicitly forbid restating the title.
Result: real specifics ("71% saw no pay increase globally",
"third time in less than two weeks", "Islamabad and Moscow
intermediaries") instead of title paraphrase.
2) Per-desk collapse buttons
Chevron next to .desk-count toggles a .is-collapsed class.
Collapsed state is per-device (localStorage by section id) since
collapse is a viewing preference, not content state.
This commit is contained in:
+207
-37
@@ -26,6 +26,7 @@ import os
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
@@ -115,13 +116,13 @@ class Source:
|
||||
|
||||
@dataclass
|
||||
class Headline:
|
||||
"""One row in the dense world/local headlines list. Skips
|
||||
LLM summarization — title is the whole point."""
|
||||
"""One row in the dense world/local headlines list."""
|
||||
id: str
|
||||
title: str
|
||||
url: str
|
||||
source: str # display name of the originating feed
|
||||
posted_at: datetime
|
||||
tldr: str = "" # 2-3 sentence LLM summary of the linked article
|
||||
|
||||
# ── http session shared across calls ─────────────────────────────────
|
||||
|
||||
@@ -131,6 +132,91 @@ S.headers["User-Agent"] = REDDIT_USER_AGENT
|
||||
def log(msg: str) -> None:
|
||||
print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}", flush=True)
|
||||
|
||||
# ── article-text cache ───────────────────────────────────────────────
|
||||
# Most feeds ship just titles + thin excerpts. Real summaries need the
|
||||
# article body, so we fetch + extract with trafilatura. Cache to disk
|
||||
# so re-runs on the same window don't re-pull.
|
||||
ARTICLE_CACHE_PATH = OUTPUT_DIR / ".article-cache.json"
|
||||
ARTICLE_CACHE_TTL_HOURS = 7 * 24 # keep extracted text ~1 week
|
||||
ARTICLE_FETCH_TIMEOUT = 12 # seconds per URL
|
||||
ARTICLE_TEXT_CAP = 4000 # chars; LLM doesn't need more
|
||||
ARTICLE_FETCH_WORKERS = 10 # parallel fetches per warm pass
|
||||
REDDIT_DOMAIN_RE = re.compile(r"^https?://(?:[^/]*\.)?reddit\.com/", re.I)
|
||||
|
||||
|
||||
def article_cache_load() -> dict:
|
||||
if not ARTICLE_CACHE_PATH.exists():
|
||||
return {}
|
||||
try:
|
||||
return json.loads(ARTICLE_CACHE_PATH.read_text())
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def article_cache_save(cache: dict) -> None:
|
||||
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||
tmp = ARTICLE_CACHE_PATH.with_suffix(".json.tmp")
|
||||
tmp.write_text(json.dumps(cache))
|
||||
tmp.rename(ARTICLE_CACHE_PATH)
|
||||
|
||||
|
||||
def fetch_article_text(url: str, cache: dict) -> str:
|
||||
"""Return main-content text for `url`, cached. Empty string on any
|
||||
failure — caller is expected to fall back to the feed body / title.
|
||||
|
||||
Skips reddit.com URLs (callers already have selftext as `body`)
|
||||
and anything that 404s, paywalls, or extracts to less than a
|
||||
paragraph."""
|
||||
if not url or REDDIT_DOMAIN_RE.match(url):
|
||||
return ""
|
||||
key = hashlib.sha1(url.encode("utf-8")).hexdigest()
|
||||
now = int(time.time())
|
||||
cached = cache.get(key)
|
||||
if cached and (now - int(cached.get("ts", 0))) < ARTICLE_CACHE_TTL_HOURS * 3600:
|
||||
return cached.get("text", "")
|
||||
try:
|
||||
import trafilatura
|
||||
downloaded = trafilatura.fetch_url(url)
|
||||
if not downloaded:
|
||||
cache[key] = {"ts": now, "text": ""}
|
||||
return ""
|
||||
text = trafilatura.extract(
|
||||
downloaded,
|
||||
include_comments=False,
|
||||
include_tables=False,
|
||||
no_fallback=False,
|
||||
) or ""
|
||||
text = text.strip()[:ARTICLE_TEXT_CAP]
|
||||
cache[key] = {"ts": now, "text": text}
|
||||
return text
|
||||
except Exception as e:
|
||||
log(f" ! article fetch failed for {url[:80]}: {e!r}")
|
||||
cache[key] = {"ts": now, "text": ""}
|
||||
return ""
|
||||
|
||||
|
||||
def warm_article_cache(urls: Iterable[str], cache: dict) -> None:
|
||||
"""Parallel-prefetch article text for `urls` into `cache`."""
|
||||
pending = []
|
||||
seen_urls: set[str] = set()
|
||||
cutoff = int(time.time()) - ARTICLE_CACHE_TTL_HOURS * 3600
|
||||
for url in urls:
|
||||
if not url or url in seen_urls or REDDIT_DOMAIN_RE.match(url):
|
||||
continue
|
||||
seen_urls.add(url)
|
||||
key = hashlib.sha1(url.encode("utf-8")).hexdigest()
|
||||
cached = cache.get(key)
|
||||
if cached and int(cached.get("ts", 0)) > cutoff:
|
||||
continue
|
||||
pending.append(url)
|
||||
if not pending:
|
||||
return
|
||||
log(f" warming article cache: {len(pending)} URLs ({ARTICLE_FETCH_WORKERS} parallel)")
|
||||
t0 = time.time()
|
||||
with ThreadPoolExecutor(max_workers=ARTICLE_FETCH_WORKERS) as ex:
|
||||
list(ex.map(lambda u: fetch_article_text(u, cache), pending))
|
||||
log(f" done in {time.time() - t0:.1f}s")
|
||||
|
||||
# ── miniflux: discover subreddits + pull tech-aggregator items ───────
|
||||
|
||||
def miniflux_get(path: str, **params) -> Any:
|
||||
@@ -312,16 +398,18 @@ def fetch_reddit_top(sub: str) -> Source:
|
||||
SUMMARIZE_SYSTEM = (
|
||||
"You are a curator producing a tight intelligence briefing for an "
|
||||
"engineer who reads many feeds. You are concise, neutral, and never "
|
||||
"editorialize. You skip pure shitposts and screenshots-without-context."
|
||||
"editorialize. You write summaries grounded in the article body — "
|
||||
"never paraphrase the title back at the reader. You skip pure "
|
||||
"shitposts and screenshots-without-context."
|
||||
)
|
||||
|
||||
SUMMARIZE_USER_TEMPLATE = """Given the {n} posts from {source} below, return a JSON ARRAY where each element has:
|
||||
|
||||
- "id": the post id from the input
|
||||
- "tldr": ONE sentence, 25 words max, capturing the substantive point. Lead with a verb. No "this post discusses". No "a user shares".
|
||||
- "tag": ONE word from {{news, tutorial, release, discussion, question, showcase, drama, meme, other}}
|
||||
- "id": the post id from the input
|
||||
- "tldr": 2-3 sentences (40-80 words) summarizing the SUBSTANCE — what happened, what was announced, what conclusion the author drew. Pull facts, names, numbers from the body. Do NOT restate the title; the reader already sees it. Do NOT begin with "this post" / "the article" / "a user". If the body is too thin to add anything beyond the title, return tldr="".
|
||||
- "tag": ONE word from {{news, tutorial, release, discussion, question, showcase, drama, meme, other}}
|
||||
|
||||
If a post is a pure shitpost / screenshot-without-context / duplicate of an item already in this batch, set "tldr" to "" and "tag" to "skip".
|
||||
If a post is a pure shitpost / screenshot-without-context / duplicate of another item in this batch, set "tldr" to "" and "tag" to "skip".
|
||||
|
||||
Output ONLY the JSON array. No prose, no markdown fence.
|
||||
|
||||
@@ -329,11 +417,60 @@ POSTS:
|
||||
{posts_json}
|
||||
"""
|
||||
|
||||
def summarize_source(src: Source) -> None:
|
||||
HEADLINE_SUMMARIZE_USER_TEMPLATE = """Given the {n} {label} headlines below, return a JSON ARRAY where each element has:
|
||||
|
||||
- "id": the headline id from the input
|
||||
- "tldr": 2-3 sentences (40-80 words) summarizing the article body — who, what, when, where, why. Pull names, numbers, places from the body. Do NOT restate the headline; the reader already sees it. Do NOT editorialize. If the body is too thin (e.g. just the headline rehashed), return tldr="".
|
||||
|
||||
Output ONLY the JSON array. No prose, no markdown fence.
|
||||
|
||||
HEADLINES:
|
||||
{posts_json}
|
||||
"""
|
||||
|
||||
|
||||
def _llm_chat(messages: list[dict], label: str) -> dict[str, dict]:
|
||||
"""Send a chat request and parse the JSON-array reply into a
|
||||
{id: row} map. Returns {} on any failure (caller falls back to
|
||||
raw titles)."""
|
||||
try:
|
||||
r = S.post(
|
||||
f"{LLAMA_SWAP_URL.rstrip('/')}/v1/chat/completions",
|
||||
json={
|
||||
"model": LLAMA_SWAP_MODEL,
|
||||
"messages": messages,
|
||||
"temperature": 0.2,
|
||||
"max_tokens": 4000,
|
||||
},
|
||||
timeout=LLAMA_SWAP_TIMEOUT,
|
||||
)
|
||||
r.raise_for_status()
|
||||
msg = r.json()["choices"][0]["message"]
|
||||
# Extended-thinking models (Qwen3.x) put output in
|
||||
# reasoning_content while content is still streaming. Fall back
|
||||
# so we get something to parse.
|
||||
content = (msg.get("content") or msg.get("reasoning_content") or "").strip()
|
||||
# Some models wrap JSON in ```...``` even when told not to.
|
||||
content = re.sub(r"^```(?:json)?\s*|\s*```$", "", content, flags=re.M).strip()
|
||||
return {x.get("id"): x for x in json.loads(content)}
|
||||
except Exception as e:
|
||||
log(f" ! llm failed for {label}: {e!r}")
|
||||
return {}
|
||||
|
||||
|
||||
def summarize_source(src: Source, cache: dict) -> None:
|
||||
if not src.items:
|
||||
return
|
||||
posts_json = json.dumps([
|
||||
{"id": it.id, "title": it.title, "body": it.body[:600], "url": it.url}
|
||||
{
|
||||
"id": it.id,
|
||||
"title": it.title,
|
||||
# Real article text (cached) wins over feed-shipped excerpt.
|
||||
# Falls back to feed body for self-posts (Reddit selftext)
|
||||
# and any URL where extraction failed.
|
||||
"body": (fetch_article_text(it.url, cache) or it.body or "")[:2500],
|
||||
"url": it.url,
|
||||
}
|
||||
for it in src.items
|
||||
], ensure_ascii=False)
|
||||
user = SUMMARIZE_USER_TEMPLATE.format(
|
||||
@@ -342,31 +479,14 @@ def summarize_source(src: Source) -> None:
|
||||
posts_json=posts_json,
|
||||
)
|
||||
log(f" llm: summarizing {len(src.items)} items from {src.name}")
|
||||
try:
|
||||
r = S.post(
|
||||
f"{LLAMA_SWAP_URL.rstrip('/')}/v1/chat/completions",
|
||||
json={
|
||||
"model": LLAMA_SWAP_MODEL,
|
||||
"messages": [
|
||||
{"role": "system", "content": SUMMARIZE_SYSTEM},
|
||||
{"role": "user", "content": user},
|
||||
],
|
||||
"temperature": 0.2,
|
||||
"max_tokens": 2000,
|
||||
},
|
||||
timeout=LLAMA_SWAP_TIMEOUT,
|
||||
)
|
||||
r.raise_for_status()
|
||||
msg = r.json()["choices"][0]["message"]
|
||||
# Models in extended-thinking mode (e.g. Qwen3.x defaults) put
|
||||
# output in reasoning_content and leave content empty until they
|
||||
# exit thinking — fall back so we get *something* to parse.
|
||||
content = (msg.get("content") or msg.get("reasoning_content") or "").strip()
|
||||
# Some models wrap JSON in ```...``` even when told not to.
|
||||
content = re.sub(r"^```(?:json)?\s*|\s*```$", "", content, flags=re.M).strip()
|
||||
mapped = {x.get("id"): x for x in json.loads(content)}
|
||||
except Exception as e:
|
||||
log(f" ! llm failed for {src.name}: {e!r} — keeping raw titles")
|
||||
mapped = _llm_chat(
|
||||
[
|
||||
{"role": "system", "content": SUMMARIZE_SYSTEM},
|
||||
{"role": "user", "content": user},
|
||||
],
|
||||
src.name,
|
||||
)
|
||||
if not mapped:
|
||||
return
|
||||
|
||||
for it in src.items:
|
||||
@@ -378,6 +498,41 @@ def summarize_source(src: Source) -> None:
|
||||
src.items = [it for it in src.items if it.tag != "skip" and (it.tldr or it.score is None)]
|
||||
log(f" -> {len(src.items)} kept after llm filter")
|
||||
|
||||
|
||||
def summarize_headlines(headlines: list[Headline], label: str, cache: dict) -> None:
|
||||
"""Batch-summarize a headline list in-place. One LLM call for the
|
||||
whole batch. Quietly leaves tldr empty on failure so the dense
|
||||
list still renders (just without summaries)."""
|
||||
if not headlines:
|
||||
return
|
||||
posts_json = json.dumps([
|
||||
{
|
||||
"id": h.id,
|
||||
"title": h.title,
|
||||
"source": h.source,
|
||||
"body": fetch_article_text(h.url, cache)[:2000],
|
||||
}
|
||||
for h in headlines
|
||||
], ensure_ascii=False)
|
||||
user = HEADLINE_SUMMARIZE_USER_TEMPLATE.format(
|
||||
n=len(headlines),
|
||||
label=label,
|
||||
posts_json=posts_json,
|
||||
)
|
||||
log(f" llm: summarizing {len(headlines)} {label} headlines")
|
||||
mapped = _llm_chat(
|
||||
[
|
||||
{"role": "system", "content": SUMMARIZE_SYSTEM},
|
||||
{"role": "user", "content": user},
|
||||
],
|
||||
f"{label} headlines",
|
||||
)
|
||||
if not mapped:
|
||||
return
|
||||
for h in headlines:
|
||||
m = mapped.get(h.id, {})
|
||||
h.tldr = (m.get("tldr") or "").strip()
|
||||
|
||||
# ── render ───────────────────────────────────────────────────────────
|
||||
|
||||
def render(reddit_sources: list[Source], tech_sources: list[Source],
|
||||
@@ -525,10 +680,25 @@ def main() -> int:
|
||||
tech_sources = fetch_miniflux_tech_items()
|
||||
log(f" found {len(tech_sources)} non-reddit feeds with recent items")
|
||||
|
||||
log("phase 4: summarizing each source via llama-swap")
|
||||
# Headlines bypass the LLM — title is the whole deliverable.
|
||||
log("phase 3d: warming article-text cache (parallel)")
|
||||
article_cache = article_cache_load()
|
||||
all_urls: list[str] = []
|
||||
for src in tech_sources + reddit_sources:
|
||||
for it in src.items:
|
||||
all_urls.append(it.url)
|
||||
for h in world_headlines + local_headlines:
|
||||
all_urls.append(h.url)
|
||||
warm_article_cache(all_urls, article_cache)
|
||||
|
||||
log("phase 4a: summarizing reddit + tech sources via llama-swap")
|
||||
for src in reddit_sources + tech_sources:
|
||||
summarize_source(src)
|
||||
summarize_source(src, article_cache)
|
||||
|
||||
log("phase 4b: summarizing world + local headlines via llama-swap")
|
||||
summarize_headlines(world_headlines, "world", article_cache)
|
||||
summarize_headlines(local_headlines, "local", article_cache)
|
||||
|
||||
article_cache_save(article_cache)
|
||||
|
||||
log("phase 5: rendering")
|
||||
html = render(reddit_sources, tech_sources, world_headlines, local_headlines, now_local)
|
||||
|
||||
Reference in New Issue
Block a user