The operator reported that zoomed-in images lose their annotations. That was
never a rendering bug. Three functions independently walked a booth and derived
overlapping subsets of the same facts -- build_gallery (kind, caption, blur,
doc), booth_view_file (kind, doc, image ring) and list_booths (kind counts,
cover) -- and the zoom route's subset was the smallest. Caption resolution lived
inside build_gallery's loop and nowhere else, so there was no code path by which
a caption could reach the zoom template. It was never sent.
booth/items.py is now the one truth: booth_items() returns the full record --
rel, kind, section, caption, blur, doc kind, size -- and the gallery, the zoom
view, the doc view and the index all read it. Patching view.html would have
fixed the symptom for images and left the next surface starting from the same
missing truth.
Two things fall out of the consolidation:
- the index and the booth page now agree on what an item IS. list_booths
counted every non-dot file, so an A/B pair with two caption sidecars read
as 4 items on the index and showed 2 tiles when you opened it.
- "section" (the item's subfolder) is computed and carried but nothing renders
it yet. That is deliberate: it is U7's whole input, and shipping the field
now makes U7 a template change rather than a resolver change.
Doc bodies are NOT rendered by the resolver -- the index touches every booth on
every page load, and rendering every markdown file in every booth would be the
price of that convenience. render_doc_body is a separate step for the one
surface that inlines them; an invariant test monkeypatches it to raise and
loads the index.
Verified beyond the suite, because this repo has shipped two dead controls that
every test passed: the caption was measured in a real browser at 1280x41 px,
visible, with elementFromPoint at its centre returning the caption itself.
layout-probe reports all controls hittable across index, gallery, zoom and doc.
192 tests pass (173 before, 19 new).
Contract: docs/contracts/u1_item_record.contract.md
245 lines
8.2 KiB
Python
245 lines
8.2 KiB
Python
"""The item record — ONE resolver for what is in a booth.
|
|
|
|
Before this module, three functions independently walked a booth and derived
|
|
overlapping subsets of the same facts: `build_gallery` (kind, caption, blur,
|
|
doc), `booth_view_file` (kind, doc, image ring) and `list_booths` (kind counts,
|
|
cover thumb). The zoom route's subset was the smallest, and the fact it lacked
|
|
was the caption — so an annotated image lost its annotation at exactly the size
|
|
where the annotation is most readable.
|
|
|
|
That was never a rendering bug. It was three readers of one truth. This module
|
|
is the one truth; every surface reads its record and derives nothing itself.
|
|
|
|
See docs/contracts/u1_item_record.contract.md.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import Sequence
|
|
from urllib.parse import quote
|
|
|
|
try: # optional: markdown rendering degrades to raw text without it
|
|
import markdown as _markdown
|
|
except ImportError: # pragma: no cover
|
|
_markdown = None
|
|
|
|
from booth.asks import is_answer_file, is_ask_file
|
|
|
|
# Browser-playable media buckets. Anything else renders as a download link.
|
|
IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif", ".svg", ".bmp"}
|
|
VIDEO_EXTS = {".webm", ".mp4", ".ogv", ".m4v", ".mov"}
|
|
AUDIO_EXTS = {".mp3", ".wav", ".ogg", ".oga", ".flac", ".m4a", ".opus", ".aac"}
|
|
|
|
# Loose text docs that render as a readable page rather than a download.
|
|
MARKDOWN_EXTS = {".md", ".markdown", ".mdown"}
|
|
TEXT_EXTS = {".txt", ".text", ".log"}
|
|
|
|
CAPTION_MAX = 800 # chars of a sidecar .txt caption we render
|
|
DOC_MAX_BYTES = 2 * 1024 * 1024 # above this, a doc is handed back raw, not rendered
|
|
|
|
BLUR_FILE = ".blurred"
|
|
|
|
|
|
def classify(name: str) -> str:
|
|
"""image | video | audio | other, by extension."""
|
|
ext = Path(name).suffix.lower()
|
|
if ext in IMAGE_EXTS:
|
|
return "image"
|
|
if ext in VIDEO_EXTS:
|
|
return "video"
|
|
if ext in AUDIO_EXTS:
|
|
return "audio"
|
|
return "other"
|
|
|
|
|
|
def doc_kind(name: str) -> str | None:
|
|
"""'markdown' | 'text' | None — a booth file viewable as a readable page."""
|
|
ext = Path(name).suffix.lower()
|
|
if ext in MARKDOWN_EXTS:
|
|
return "markdown"
|
|
if ext in TEXT_EXTS:
|
|
return "text"
|
|
return None
|
|
|
|
|
|
def render_doc(text: str, kind: str) -> tuple[str, bool]:
|
|
"""(rendered, is_html). Markdown → HTML (fenced code, tables, sane lists);
|
|
plain text — or markdown when the lib is unavailable — → raw text for <pre>.
|
|
|
|
Text is returned RAW on purpose: the template escapes it inside <pre>, and
|
|
pre-escaping here would double-encode under Jinja autoescape.
|
|
"""
|
|
if kind == "markdown" and _markdown is not None:
|
|
html = _markdown.markdown(text, extensions=["fenced_code", "tables", "sane_lists"])
|
|
return html, True
|
|
return text, False
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Item:
|
|
"""One renderable file in a booth, with every fact any surface needs.
|
|
|
|
`rel` is the identity — the booth-relative POSIX path. Marks (U2) attach to
|
|
it, blur is keyed by it, and the zoom route resolves by it.
|
|
"""
|
|
|
|
rel: str
|
|
url: str
|
|
kind: str
|
|
section: str | None
|
|
caption: str | None
|
|
blurred: bool
|
|
doc: str | None
|
|
size: int
|
|
|
|
|
|
def read_blurred(booth: Path) -> set[str]:
|
|
"""Blurred item paths for a booth. Missing file -> empty set."""
|
|
try:
|
|
text = (booth / BLUR_FILE).read_text()
|
|
except (OSError, UnicodeDecodeError):
|
|
return set()
|
|
return {ln.strip() for ln in text.splitlines() if ln.strip()}
|
|
|
|
|
|
def _section_of(rel: str) -> str | None:
|
|
"""The item's parent directory relative to the booth; None at the root.
|
|
|
|
Derived, never stored. This is the whole input to the navigation fix (U7):
|
|
the structure a poster already created on disk, which `rglob` has been
|
|
flattening into one wall at render time.
|
|
"""
|
|
parent = Path(rel).parent
|
|
return None if str(parent) == "." else parent.as_posix()
|
|
|
|
|
|
def _resolve_captions(by_rel: dict[str, Path]) -> tuple[dict[str, str], set[str]]:
|
|
"""(caption-by-rel, rels consumed as sidecars).
|
|
|
|
Two forms, in this precedence, preserved from the original gallery:
|
|
1. `<file>.txt` — `a.png.txt` captions `a.png`
|
|
2. `<stem>.txt` beside a same-stem MEDIA sibling — `a.txt` captions
|
|
`a.png`, but NOT `a.bin` (the `classify != "other"` guard, so a stray
|
|
`data.txt` next to `data.bin` stays an item of its own)
|
|
|
|
The sibling scan runs in sorted order rather than filesystem order: when two
|
|
media files share a stem in one directory (`a.png` and `a.webm`), the
|
|
original picked whichever `rglob` happened to yield first. Same rule, now
|
|
deterministic.
|
|
"""
|
|
caption: dict[str, str] = {}
|
|
sidecars: set[str] = set()
|
|
|
|
for rel in sorted(by_rel):
|
|
if not rel.lower().endswith(".txt"):
|
|
continue
|
|
p = by_rel[rel]
|
|
target = None
|
|
|
|
base_full = rel[:-4] # "a.png.txt" -> "a.png"
|
|
if base_full in by_rel:
|
|
target = base_full
|
|
else:
|
|
parent = str(Path(rel).parent)
|
|
stem = Path(rel).stem
|
|
for q_rel in sorted(by_rel):
|
|
if q_rel == rel:
|
|
continue
|
|
q = by_rel[q_rel]
|
|
if (
|
|
str(Path(q_rel).parent) == parent
|
|
and Path(q_rel).stem == stem
|
|
and classify(q.name) != "other"
|
|
):
|
|
target = q_rel
|
|
break
|
|
|
|
if target is not None:
|
|
try:
|
|
caption[target] = p.read_text(errors="replace").strip()[:CAPTION_MAX]
|
|
except OSError:
|
|
pass
|
|
sidecars.add(rel)
|
|
|
|
return caption, sidecars
|
|
|
|
|
|
def booth_items(booth: Path) -> list[Item]:
|
|
"""Every renderable file in a booth, sorted by relative path.
|
|
|
|
Excluded: dotfiles, `*.ask.json` / `*.answer.json` (they render as the asks
|
|
panel, not as tiles), and any file consumed as another item's caption.
|
|
|
|
Doc BODIES are deliberately not rendered here. The index calls this once per
|
|
booth to count items and pick a cover; rendering every doc in every booth on
|
|
every page load would be the cost of that convenience. `render_doc_body` is
|
|
the separate step, for the one consumer that needs it.
|
|
"""
|
|
by_rel: dict[str, Path] = {}
|
|
for p in booth.rglob("*"):
|
|
if not p.is_file() or p.name.startswith("."):
|
|
continue
|
|
if is_ask_file(p.name) or is_answer_file(p.name):
|
|
continue
|
|
by_rel[p.relative_to(booth).as_posix()] = p
|
|
|
|
caption, sidecars = _resolve_captions(by_rel)
|
|
blurred = read_blurred(booth) # ONE read per call, not one per item
|
|
|
|
items: list[Item] = []
|
|
for rel in sorted(by_rel):
|
|
if rel in sidecars:
|
|
continue
|
|
p = by_rel[rel]
|
|
try:
|
|
size = p.stat().st_size
|
|
except OSError:
|
|
size = 0
|
|
items.append(
|
|
Item(
|
|
rel=rel,
|
|
url=quote(rel, safe="/"),
|
|
kind=classify(p.name),
|
|
section=_section_of(rel),
|
|
caption=caption.get(rel),
|
|
blurred=rel in blurred,
|
|
doc=doc_kind(p.name),
|
|
size=size,
|
|
)
|
|
)
|
|
return items
|
|
|
|
|
|
def image_chain(items: Sequence[Item]) -> list[str]:
|
|
"""The rels of the image items, in order — the zoom view's prev/next ring.
|
|
|
|
Replaces `booth_image_names`, which walked the tree a second time to derive
|
|
what the item list already knows.
|
|
"""
|
|
return [it.rel for it in items if it.kind == "image"]
|
|
|
|
|
|
def find_item(items: Sequence[Item], rel: str) -> Item | None:
|
|
"""The record for one rel, or None — the zoom/doc route's entry point."""
|
|
for it in items:
|
|
if it.rel == rel:
|
|
return it
|
|
return None
|
|
|
|
|
|
def render_doc_body(booth: Path, item: Item) -> tuple[str, bool] | None:
|
|
"""(body, is_html) for a doc item under DOC_MAX_BYTES, else None.
|
|
|
|
None means "do not inline this": either it is not a doc, or it is a log big
|
|
enough that inlining it into every page render is the wrong trade.
|
|
"""
|
|
if item.doc is None or item.size > DOC_MAX_BYTES:
|
|
return None
|
|
try:
|
|
text = (booth / item.rel).read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
return None
|
|
return render_doc(text, item.doc)
|