Files
booth/booth/items.py
T
vh 4cfbce5109 fix(blur): .blurred round-trips any rel, and one writer serves both surfaces
The heid bug-hunt on r2b merge 1 found the /blur route stripping `f` before
writing, so the form for " a.png" blurred its neighbour "a.png". The route was
only half of it: `.blurred` was one stripped rel per line, so no writer could
store a rel with a leading space or a newline, whatever the route did.
Operator-ruled 2026-09-23 ("fix the blur").

- booth/blur.py (new, stdlib-only): read_blurred / set_blurred / BLUR_FILE.
  `.blurred` is now a JSON array in sorted order, the `.seen` shape: opened
  O_NOFOLLOW | O_NONBLOCK with an S_ISREG check and a 1 MiB cap, so a planted
  symlink is refused and a FIFO can no longer hang every Desk render (the old
  read_text() blocked on one). Writes go through mkstemp + os.replace. The
  legacy line format is still READ, so the 6 live line-format files keep their
  blur until their next write upgrades them. Measured before the change: 42
  live rels, none with edge whitespace, so the defect had no live victims.
- The route no longer strips `f`.
- scripts/booth `blur`/`unblur` go through booth.blur.set_blurred instead of
  their own grep/printf line writer. Two writers of one format is how the
  formats drift, and after this change the shell writer would have appended a
  line to a JSON array. Every path is checked before anything is written.
- Item.blurred_self (appended to the record): the item's own blur, resolved in
  booth_items from the same read as `blurred`. It replaces build_gallery's
  second read_blurred, which a write between the two reads could split
  (invariant 3). app.py no longer reads blur state at all, and a test asserts
  it.

Names stay importable from booth.app and booth.items (invariant 4). blur joins
test_stdlib_only. test_cli's per-item-survives test now reads through the reader
rather than asserting the old byte format. The r2b contract and its mutation
row follow blurred_self onto the record. tests/mutations/blur_storage.toml
proves 12 falsifiers by running the change each forbids.

Not in this change, and still ours: the "off"-means-ON idiom drift between
/blur, /blurbooth and /flag (forms only ever send 0/1), and the CLI's
`.blurbooth` touch following a symlink where the service no longer does.
2026-09-23 22:05:18 -07:00

449 lines
18 KiB
Python

"""The item record — ONE resolver for what is in a booth.
Before this module, three functions independently walked a booth and derived
overlapping subsets of the same facts: `build_gallery` (kind, caption, blur,
doc), `booth_view_file` (kind, doc, image ring) and `list_booths` (kind counts,
cover thumb). The zoom route's subset was the smallest, and the fact it lacked
was the caption — so an annotated image lost its annotation at exactly the size
where the annotation is most readable.
That was never a rendering bug. It was three readers of one truth. This module
is the one truth; every surface reads its record and derives nothing itself.
See docs/contracts/u1_item_record.contract.md.
"""
from __future__ import annotations
import json
import os
import re
import stat
from dataclasses import dataclass
from pathlib import Path
from typing import Sequence
from urllib.parse import quote
try: # optional: markdown rendering degrades to raw text without it
import markdown as _markdown
except ImportError: # pragma: no cover
_markdown = None
from booth.asks import is_answer_file, is_ask_file
from booth.blur import BLUR_FILE, read_blurred # noqa: F401 (re-exported)
from booth.thumbs import wants_thumb
# Browser-playable media buckets. Anything else renders as a download link.
IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif", ".svg", ".bmp"}
VIDEO_EXTS = {".webm", ".mp4", ".ogv", ".m4v", ".mov"}
AUDIO_EXTS = {".mp3", ".wav", ".ogg", ".oga", ".flac", ".m4a", ".opus", ".aac"}
# Loose text docs that render as a readable page rather than a download.
MARKDOWN_EXTS = {".md", ".markdown", ".mdown"}
TEXT_EXTS = {".txt", ".text", ".log"}
CAPTION_MAX = 800 # chars of a sidecar .txt caption we render
DOC_MAX_BYTES = 2 * 1024 * 1024 # above this, a doc is handed back raw, not rendered
# `BLUR_FILE` and `read_blurred` live in booth/blur.py (stdlib-only, so the CLI
# shares the reader and the writer) and are re-exported from here.
# Booth-level blur: the whole booth is fogged, agent-set at post time or
# toggled by the operator. A MARKER, deliberately not JSON like `.seen` —
# `.seen` is JSON because it holds rels that must round-trip exactly, and a
# boolean has nothing to round-trip. It matches `.forever`, which is the other
# whole-booth flag, so the two read the same way.
BOOTH_BLUR_FILE = ".blurbooth"
# What booth-level blur applies to. Audio has nothing to hide from a glance.
BLURRABLE_KINDS = {"image", "video"}
def classify(name: str) -> str:
"""image | video | audio | other, by extension."""
ext = Path(name).suffix.lower()
if ext in IMAGE_EXTS:
return "image"
if ext in VIDEO_EXTS:
return "video"
if ext in AUDIO_EXTS:
return "audio"
return "other"
def doc_kind(name: str) -> str | None:
"""'markdown' | 'text' | None — a booth file viewable as a readable page."""
ext = Path(name).suffix.lower()
if ext in MARKDOWN_EXTS:
return "markdown"
if ext in TEXT_EXTS:
return "text"
return None
def render_doc(text: str, kind: str) -> tuple[str, bool]:
"""(rendered, is_html). Markdown → HTML (fenced code, tables, sane lists);
plain text — or markdown when the lib is unavailable — → raw text for <pre>.
Text is returned RAW on purpose: the template escapes it inside <pre>, and
pre-escaping here would double-encode under Jinja autoescape.
"""
if kind == "markdown" and _markdown is not None:
html = _markdown.markdown(text, extensions=["fenced_code", "tables", "sane_lists"])
return html, True
return text, False
@dataclass(frozen=True)
class Item:
"""One renderable file in a booth, with every fact any surface needs.
`rel` is the identity — the booth-relative POSIX path. Marks (U2) attach to
it, blur is keyed by it, and the zoom route resolves by it.
"""
rel: str
url: str
kind: str
section: str | None
group: str | None
caption: str | None
blurred: bool
doc: str | None
size: int
# R2 C1: the 1-based position in `booth_items` order over ALL items — the
# number the operator means by "the third one". Set in the resolver loop
# and nowhere else (INV-1). APPENDED, never inserted: a mid-dataclass field
# is a positional-construction break.
ordinal: int
# The tile's image source, or None when the original IS the right source
# (vector, video, a type Pillow cannot open, or an image already tile-sized).
# Derived HERE so no template reasons about `kind` to decide — INV-1, which
# is the caption bug in a new field.
thumb: str | None
# The item's OWN per-item blur, apart from the booth's fog: the per-item
# control toggles only this, so it must not offer an un-blur the booth flag
# would override (r2b D2b). From the SAME read as `blurred` — it used to be
# a second `read_blurred` in build_gallery, and a write between the two
# reads could split them (invariant 3). APPENDED, like `ordinal`.
blurred_self: bool
# R2 C2: which items have been looked at full size. UI state, not judgment —
# never exposed to sessions, holds nothing. One viewer: this records WHAT was
# seen, never who saw it.
SEEN_FILE = ".seen"
# A seen marker bigger than this is not one this service wrote: a JSON array of
# every rel in a 270-item booth is a few KB.
SEEN_MAX_BYTES = 1 << 20
def read_seen(booth: Path) -> set[str]:
"""Rels seen at full size (R2 C2). A JSON array of strings, because a rel
may hold a leading space or a newline and must round-trip exactly.
NEVER RAISES and NEVER BLOCKS. Any fleet session can write into a booth,
so the marker may be planted: it is opened without following a link and
without blocking (a FIFO with no writer), refused unless it is a regular
file of sane size, and anything unreadable or malformed reads as nothing
seen — a damaged marker costs the tape its memory, never the page.
"""
try:
fd = os.open(booth / SEEN_FILE, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK)
except OSError:
return set()
try:
st = os.fstat(fd)
if not stat.S_ISREG(st.st_mode) or st.st_size > SEEN_MAX_BYTES:
return set()
raw = os.read(fd, SEEN_MAX_BYTES + 1)
except OSError:
return set()
finally:
os.close(fd)
try:
data = json.loads(raw.decode("utf-8"))
except (UnicodeDecodeError, ValueError, RecursionError):
# RecursionError: a deeply nested array (`[[[[...`) blows the parser's
# stack, and it is neither a ValueError nor an OSError — the same hole
# marks.py, manifest.py and benches.py already close.
return set()
if not isinstance(data, list):
return set()
return {r for r in data if isinstance(r, str)}
def is_booth_blurred(booth: Path) -> bool:
"""Whether the WHOLE booth is blurred.
`lstat`, not `exists()`, and an unreadable answer counts as BLURRED —
the same shape as `is_kept` with the safety inverted, and the inversion is
the point. `is_kept` fails toward keeping because a failed read must not
authorize a delete; this fails toward HIDING, because a failed read must not
reveal something the poster asked to fog. Both directions are "the failure
does not cause the loss".
A SYMLINK counts, dangling or not: somebody put it there to mean blur.
Composes with `.blurred`, never overrides it — turning booth blur off must
not erase an agent's per-item choice, and an override would need a per-item
"unblurred" exception list, which is state nobody can see.
"""
try:
(booth / BOOTH_BLUR_FILE).lstat()
return True
except FileNotFoundError:
return False
except OSError:
return True # cannot tell -> fog it; see above
def _section_of(rel: str) -> str | None:
"""The item's parent directory relative to the booth; None at the root.
Derived, never stored. This is the whole input to the navigation fix (U7):
the structure a poster already created on disk, which `rglob` has been
flattening into one wall at render time.
"""
parent = Path(rel).parent
return None if str(parent) == "." else parent.as_posix()
# One separator run between name segments. A filename is the only grouping
# signal the live booths actually carry: 0 of 11 galleries have a subdirectory.
_SEG = re.compile(r"[-_. ]+")
def _group_of(rel: str) -> str | None:
"""The grouping key for an item, or None when it has none.
THE RULE, in one line: **the first separator-delimited segment of the
basename's stem — with a trailing digit run stripped only when the stem has
no separator at all.** `00-sheet-c1-market-noon.png` -> `00`;
`m-c1-market-noon-9401.png` -> `m`; `flag-rear.png` -> `flag`;
`ac01.png` -> `ac` (no separator, so the digits are the separator);
`v30-seed8302.png` -> `v30` (separator present, so `v30` survives and does
not merge with `v35`, which is the axis that booth is about).
None for a stem with nothing before the digits -- `01.png` has no prefix to
group on, and inventing one would file every numbered render under the
empty string.
⚠ THIS IS NOT THE RULE THE CONTRACT FIRST NAMED. `strip ONE trailing run of
digits` was measured against the live set on 2026-09-22 and yields 24 groups
for sindra-bakeoff's 40 images and 27 for sindra's 30 -- a rail with one row
per tile. The contract's own table claimed 5 and 1 for those two booths;
neither reproduces under the rule it states beside them. The rewritten table
carries the re-measurement.
Derived HERE and nowhere else (INV-1). A route body that re-derived it would
be the caption bug in a new field.
"""
stem = Path(rel).stem # basename without its last suffix; `a.tar.gz` -> `a.tar`
segs = _SEG.split(stem)
if len(segs) == 1:
return re.sub(r"\d+$", "", stem) or None
return segs[0] or None
def _resolve_captions(by_rel: dict[str, Path]) -> tuple[dict[str, str], set[str]]:
"""(caption-by-rel, rels consumed as sidecars).
Two forms, in this precedence, preserved from the original gallery:
1. `<file>.txt` — `a.png.txt` captions `a.png`
2. `<stem>.txt` beside a same-stem MEDIA sibling — `a.txt` captions
`a.png`, but NOT `a.bin` (the `classify != "other"` guard, so a stray
`data.txt` next to `data.bin` stays an item of its own)
The sibling scan runs in sorted order rather than filesystem order: when two
media files share a stem in one directory (`a.png` and `a.webm`), the
original picked whichever `rglob` happened to yield first. Same rule, now
deterministic.
"""
caption: dict[str, str] = {}
sidecars: set[str] = set()
for rel in sorted(by_rel):
if not rel.lower().endswith(".txt"):
continue
p = by_rel[rel]
target = None
base_full = rel[:-4] # "a.png.txt" -> "a.png"
if base_full in by_rel:
target = base_full
else:
parent = str(Path(rel).parent)
stem = Path(rel).stem
for q_rel in sorted(by_rel):
if q_rel == rel:
continue
q = by_rel[q_rel]
if (
str(Path(q_rel).parent) == parent
and Path(q_rel).stem == stem
and classify(q.name) != "other"
):
target = q_rel
break
if target is not None:
try:
# BOUNDED AT THE READ. `read_text()` pulled the whole sidecar
# into memory before the slice trimmed it, so a pathological
# file was a MemoryError — which the OSError handler below does
# not catch — rather than a missing caption.
#
# Deliberately NOT bounded by st_size: a FIFO reports 0 and a
# bound that trusts it inherits what it does not mean, which is
# the hang in persistent-memory.d/2026-09-22-size-cap-opened-a-hang.md.
# The factor of 4 is UTF-8's worst case, so CAPTION_MAX
# characters always survive the byte bound.
with p.open("r", errors="replace") as fh:
caption[target] = fh.read(CAPTION_MAX * 4).strip()[:CAPTION_MAX]
except OSError:
pass
sidecars.add(rel)
return caption, sidecars
def booth_items(booth: Path) -> list[Item]:
"""Every renderable file in a booth, sorted by relative path.
Excluded: dotfiles, `*.ask.json` / `*.answer.json` (they render as the asks
panel, not as tiles), and any file consumed as another item's caption.
Doc BODIES are deliberately not rendered here. The index calls this once per
booth to count items and pick a cover; rendering every doc in every booth on
every page load would be the cost of that convenience. `render_doc_body` is
the separate step, for the one consumer that needs it.
"""
by_rel: dict[str, Path] = {}
for p in booth.rglob("*"):
try:
# `is_file` swallows a missing entry but PROPAGATES EACCES: a
# directory that lists but cannot be searched made every stat under
# it raise out of here, and `list_booths` calls this for every
# booth — one such folder took down the index for all of them.
# An entry nobody can stat is not a renderable file. (design-dev)
if not p.is_file():
continue
except OSError:
continue
# ⚠ EVERY path component, not just the filename. `p.name.startswith(".")`
# tested only the leaf, so `.thumbs/a.png` (name `a.png`) sailed through
# as a gallery item — and CLAUDE.md invariant 2 promises a dotfile costs
# nothing in item counts, galleries or zips. That promise was true only
# at the top level until the `.thumbs/` cache made it matter.
#
# BOTH guards, not either: they were written independently for different
# failures and the merge that kept one would have quietly dropped the
# other.
if any(part.startswith(".") for part in p.relative_to(booth).parts):
continue
continue
if is_ask_file(p.name) or is_answer_file(p.name):
continue
rel = p.relative_to(booth).as_posix()
try:
quote(rel, safe="/")
except UnicodeEncodeError:
# A non-UTF-8 filename reaches CPython as a surrogate escape, and
# `quote` raises on it. This used to happen at Item construction,
# OUTSIDE any per-item handler — so one 0xff byte in one filename
# took out that booth's page AND the index for every booth, because
# `list_booths` calls this too. The repo's posture is that a damaged
# file costs its own tile and never the page.
#
# Skipped rather than rescued: a name that cannot be percent-encoded
# cannot be linked, served or zipped either, so there is no item to
# render. Found by the heid bug-hunt panel (hulda), 2026-09-22.
continue
by_rel[rel] = p
caption, sidecars = _resolve_captions(by_rel)
blurred = read_blurred(booth) # ONE read per call, not one per item
booth_blur = is_booth_blurred(booth) # likewise: one stat, not one per item
items: list[Item] = []
for rel in sorted(by_rel):
if rel in sidecars:
continue
p = by_rel[rel]
try:
size = p.stat().st_size
except OSError:
size = 0
kind = classify(p.name)
items.append(
Item(
rel=rel,
url=quote(rel, safe="/"),
kind=kind,
section=_section_of(rel),
group=_group_of(rel),
caption=caption.get(rel),
# Booth blur COMPOSES with the per-item set. Resolved HERE so
# every surface inherits it for free — Desk strip, tiles, tray,
# filmstrip, stage all already read `Item.blurred` and none of
# them learns about the booth flag (INV-1).
blurred=rel in blurred or (booth_blur and kind in BLURRABLE_KINDS),
doc=doc_kind(p.name),
size=size,
# Counted over items that RENDER: a caption sidecar or a name
# the quote() guard skipped takes no number, so the numbers
# stay contiguous over what the operator can see.
ordinal=len(items) + 1,
thumb=(quote(rel, safe='/') + '?thumb=1') if wants_thumb(rel) else None,
blurred_self=rel in blurred,
)
)
return items
def image_chain(items: Sequence[Item]) -> list[str]:
"""The rels of the image items, in order — the zoom view's prev/next ring.
Replaces `booth_image_names`, which walked the tree a second time to derive
what the item list already knows.
"""
return [it.rel for it in items if it.kind == "image"]
# R2 C2: what the review route steps through. ONE LINE: the item order
# filtered to media. It is a declared change to the zoom-ring rule, which was
# images only: a listening set is reviewed the same way a picture set is.
REVIEW_KINDS = ("image", "video", "audio")
def review_chain(items: Sequence[Item]) -> list[str]:
"""The rels of the media items, in item order — the review's prev/next ring,
its filmstrip and its tape."""
return [it.rel for it in items if it.kind in REVIEW_KINDS]
def find_item(items: Sequence[Item], rel: str) -> Item | None:
"""The record for one rel, or None — the zoom/doc route's entry point."""
for it in items:
if it.rel == rel:
return it
return None
def render_doc_body(booth: Path, item: Item) -> tuple[str, bool] | None:
"""(body, is_html) for a doc item under DOC_MAX_BYTES, else None.
None means "do not inline this": either it is not a doc, or it is a log big
enough that inlining it into every page render is the wrong trade.
"""
if item.doc is None or item.size > DOC_MAX_BYTES:
return None
try:
text = (booth / item.rel).read_text(encoding="utf-8", errors="replace")
except OSError:
return None
return render_doc(text, item.doc)