The heid bug-hunt on r2b merge 1 found the /blur route stripping `f` before
writing, so the form for " a.png" blurred its neighbour "a.png". The route was
only half of it: `.blurred` was one stripped rel per line, so no writer could
store a rel with a leading space or a newline, whatever the route did.
Operator-ruled 2026-09-23 ("fix the blur").
- booth/blur.py (new, stdlib-only): read_blurred / set_blurred / BLUR_FILE.
`.blurred` is now a JSON array in sorted order, the `.seen` shape: opened
O_NOFOLLOW | O_NONBLOCK with an S_ISREG check and a 1 MiB cap, so a planted
symlink is refused and a FIFO can no longer hang every Desk render (the old
read_text() blocked on one). Writes go through mkstemp + os.replace. The
legacy line format is still READ, so the 6 live line-format files keep their
blur until their next write upgrades them. Measured before the change: 42
live rels, none with edge whitespace, so the defect had no live victims.
- The route no longer strips `f`.
- scripts/booth `blur`/`unblur` go through booth.blur.set_blurred instead of
their own grep/printf line writer. Two writers of one format is how the
formats drift, and after this change the shell writer would have appended a
line to a JSON array. Every path is checked before anything is written.
- Item.blurred_self (appended to the record): the item's own blur, resolved in
booth_items from the same read as `blurred`. It replaces build_gallery's
second read_blurred, which a write between the two reads could split
(invariant 3). app.py no longer reads blur state at all, and a test asserts
it.
Names stay importable from booth.app and booth.items (invariant 4). blur joins
test_stdlib_only. test_cli's per-item-survives test now reads through the reader
rather than asserting the old byte format. The r2b contract and its mutation
row follow blurred_self onto the record. tests/mutations/blur_storage.toml
proves 12 falsifiers by running the change each forbids.
Not in this change, and still ours: the "off"-means-ON idiom drift between
/blur, /blurbooth and /flag (forms only ever send 0/1), and the CLI's
`.blurbooth` touch following a symlink where the service no longer does.
449 lines
18 KiB
Python
449 lines
18 KiB
Python
"""The item record — ONE resolver for what is in a booth.
|
|
|
|
Before this module, three functions independently walked a booth and derived
|
|
overlapping subsets of the same facts: `build_gallery` (kind, caption, blur,
|
|
doc), `booth_view_file` (kind, doc, image ring) and `list_booths` (kind counts,
|
|
cover thumb). The zoom route's subset was the smallest, and the fact it lacked
|
|
was the caption — so an annotated image lost its annotation at exactly the size
|
|
where the annotation is most readable.
|
|
|
|
That was never a rendering bug. It was three readers of one truth. This module
|
|
is the one truth; every surface reads its record and derives nothing itself.
|
|
|
|
See docs/contracts/u1_item_record.contract.md.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
import re
|
|
import stat
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
from typing import Sequence
|
|
from urllib.parse import quote
|
|
|
|
try: # optional: markdown rendering degrades to raw text without it
|
|
import markdown as _markdown
|
|
except ImportError: # pragma: no cover
|
|
_markdown = None
|
|
|
|
from booth.asks import is_answer_file, is_ask_file
|
|
from booth.blur import BLUR_FILE, read_blurred # noqa: F401 (re-exported)
|
|
from booth.thumbs import wants_thumb
|
|
|
|
# Browser-playable media buckets. Anything else renders as a download link.
|
|
IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".avif", ".svg", ".bmp"}
|
|
VIDEO_EXTS = {".webm", ".mp4", ".ogv", ".m4v", ".mov"}
|
|
AUDIO_EXTS = {".mp3", ".wav", ".ogg", ".oga", ".flac", ".m4a", ".opus", ".aac"}
|
|
|
|
# Loose text docs that render as a readable page rather than a download.
|
|
MARKDOWN_EXTS = {".md", ".markdown", ".mdown"}
|
|
TEXT_EXTS = {".txt", ".text", ".log"}
|
|
|
|
CAPTION_MAX = 800 # chars of a sidecar .txt caption we render
|
|
DOC_MAX_BYTES = 2 * 1024 * 1024 # above this, a doc is handed back raw, not rendered
|
|
|
|
# `BLUR_FILE` and `read_blurred` live in booth/blur.py (stdlib-only, so the CLI
|
|
# shares the reader and the writer) and are re-exported from here.
|
|
|
|
# Booth-level blur: the whole booth is fogged, agent-set at post time or
|
|
# toggled by the operator. A MARKER, deliberately not JSON like `.seen` —
|
|
# `.seen` is JSON because it holds rels that must round-trip exactly, and a
|
|
# boolean has nothing to round-trip. It matches `.forever`, which is the other
|
|
# whole-booth flag, so the two read the same way.
|
|
BOOTH_BLUR_FILE = ".blurbooth"
|
|
|
|
# What booth-level blur applies to. Audio has nothing to hide from a glance.
|
|
BLURRABLE_KINDS = {"image", "video"}
|
|
|
|
|
|
def classify(name: str) -> str:
|
|
"""image | video | audio | other, by extension."""
|
|
ext = Path(name).suffix.lower()
|
|
if ext in IMAGE_EXTS:
|
|
return "image"
|
|
if ext in VIDEO_EXTS:
|
|
return "video"
|
|
if ext in AUDIO_EXTS:
|
|
return "audio"
|
|
return "other"
|
|
|
|
|
|
def doc_kind(name: str) -> str | None:
|
|
"""'markdown' | 'text' | None — a booth file viewable as a readable page."""
|
|
ext = Path(name).suffix.lower()
|
|
if ext in MARKDOWN_EXTS:
|
|
return "markdown"
|
|
if ext in TEXT_EXTS:
|
|
return "text"
|
|
return None
|
|
|
|
|
|
def render_doc(text: str, kind: str) -> tuple[str, bool]:
|
|
"""(rendered, is_html). Markdown → HTML (fenced code, tables, sane lists);
|
|
plain text — or markdown when the lib is unavailable — → raw text for <pre>.
|
|
|
|
Text is returned RAW on purpose: the template escapes it inside <pre>, and
|
|
pre-escaping here would double-encode under Jinja autoescape.
|
|
"""
|
|
if kind == "markdown" and _markdown is not None:
|
|
html = _markdown.markdown(text, extensions=["fenced_code", "tables", "sane_lists"])
|
|
return html, True
|
|
return text, False
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Item:
|
|
"""One renderable file in a booth, with every fact any surface needs.
|
|
|
|
`rel` is the identity — the booth-relative POSIX path. Marks (U2) attach to
|
|
it, blur is keyed by it, and the zoom route resolves by it.
|
|
"""
|
|
|
|
rel: str
|
|
url: str
|
|
kind: str
|
|
section: str | None
|
|
group: str | None
|
|
caption: str | None
|
|
blurred: bool
|
|
doc: str | None
|
|
size: int
|
|
# R2 C1: the 1-based position in `booth_items` order over ALL items — the
|
|
# number the operator means by "the third one". Set in the resolver loop
|
|
# and nowhere else (INV-1). APPENDED, never inserted: a mid-dataclass field
|
|
# is a positional-construction break.
|
|
ordinal: int
|
|
# The tile's image source, or None when the original IS the right source
|
|
# (vector, video, a type Pillow cannot open, or an image already tile-sized).
|
|
# Derived HERE so no template reasons about `kind` to decide — INV-1, which
|
|
# is the caption bug in a new field.
|
|
thumb: str | None
|
|
# The item's OWN per-item blur, apart from the booth's fog: the per-item
|
|
# control toggles only this, so it must not offer an un-blur the booth flag
|
|
# would override (r2b D2b). From the SAME read as `blurred` — it used to be
|
|
# a second `read_blurred` in build_gallery, and a write between the two
|
|
# reads could split them (invariant 3). APPENDED, like `ordinal`.
|
|
blurred_self: bool
|
|
|
|
|
|
# R2 C2: which items have been looked at full size. UI state, not judgment —
|
|
# never exposed to sessions, holds nothing. One viewer: this records WHAT was
|
|
# seen, never who saw it.
|
|
SEEN_FILE = ".seen"
|
|
|
|
|
|
# A seen marker bigger than this is not one this service wrote: a JSON array of
|
|
# every rel in a 270-item booth is a few KB.
|
|
SEEN_MAX_BYTES = 1 << 20
|
|
|
|
|
|
def read_seen(booth: Path) -> set[str]:
|
|
"""Rels seen at full size (R2 C2). A JSON array of strings, because a rel
|
|
may hold a leading space or a newline and must round-trip exactly.
|
|
|
|
NEVER RAISES and NEVER BLOCKS. Any fleet session can write into a booth,
|
|
so the marker may be planted: it is opened without following a link and
|
|
without blocking (a FIFO with no writer), refused unless it is a regular
|
|
file of sane size, and anything unreadable or malformed reads as nothing
|
|
seen — a damaged marker costs the tape its memory, never the page.
|
|
"""
|
|
try:
|
|
fd = os.open(booth / SEEN_FILE, os.O_RDONLY | os.O_NOFOLLOW | os.O_NONBLOCK)
|
|
except OSError:
|
|
return set()
|
|
try:
|
|
st = os.fstat(fd)
|
|
if not stat.S_ISREG(st.st_mode) or st.st_size > SEEN_MAX_BYTES:
|
|
return set()
|
|
raw = os.read(fd, SEEN_MAX_BYTES + 1)
|
|
except OSError:
|
|
return set()
|
|
finally:
|
|
os.close(fd)
|
|
try:
|
|
data = json.loads(raw.decode("utf-8"))
|
|
except (UnicodeDecodeError, ValueError, RecursionError):
|
|
# RecursionError: a deeply nested array (`[[[[...`) blows the parser's
|
|
# stack, and it is neither a ValueError nor an OSError — the same hole
|
|
# marks.py, manifest.py and benches.py already close.
|
|
return set()
|
|
if not isinstance(data, list):
|
|
return set()
|
|
return {r for r in data if isinstance(r, str)}
|
|
|
|
|
|
def is_booth_blurred(booth: Path) -> bool:
|
|
"""Whether the WHOLE booth is blurred.
|
|
|
|
`lstat`, not `exists()`, and an unreadable answer counts as BLURRED —
|
|
the same shape as `is_kept` with the safety inverted, and the inversion is
|
|
the point. `is_kept` fails toward keeping because a failed read must not
|
|
authorize a delete; this fails toward HIDING, because a failed read must not
|
|
reveal something the poster asked to fog. Both directions are "the failure
|
|
does not cause the loss".
|
|
|
|
A SYMLINK counts, dangling or not: somebody put it there to mean blur.
|
|
|
|
Composes with `.blurred`, never overrides it — turning booth blur off must
|
|
not erase an agent's per-item choice, and an override would need a per-item
|
|
"unblurred" exception list, which is state nobody can see.
|
|
"""
|
|
try:
|
|
(booth / BOOTH_BLUR_FILE).lstat()
|
|
return True
|
|
except FileNotFoundError:
|
|
return False
|
|
except OSError:
|
|
return True # cannot tell -> fog it; see above
|
|
|
|
|
|
def _section_of(rel: str) -> str | None:
|
|
"""The item's parent directory relative to the booth; None at the root.
|
|
|
|
Derived, never stored. This is the whole input to the navigation fix (U7):
|
|
the structure a poster already created on disk, which `rglob` has been
|
|
flattening into one wall at render time.
|
|
"""
|
|
parent = Path(rel).parent
|
|
return None if str(parent) == "." else parent.as_posix()
|
|
|
|
|
|
# One separator run between name segments. A filename is the only grouping
|
|
# signal the live booths actually carry: 0 of 11 galleries have a subdirectory.
|
|
_SEG = re.compile(r"[-_. ]+")
|
|
|
|
|
|
def _group_of(rel: str) -> str | None:
|
|
"""The grouping key for an item, or None when it has none.
|
|
|
|
THE RULE, in one line: **the first separator-delimited segment of the
|
|
basename's stem — with a trailing digit run stripped only when the stem has
|
|
no separator at all.** `00-sheet-c1-market-noon.png` -> `00`;
|
|
`m-c1-market-noon-9401.png` -> `m`; `flag-rear.png` -> `flag`;
|
|
`ac01.png` -> `ac` (no separator, so the digits are the separator);
|
|
`v30-seed8302.png` -> `v30` (separator present, so `v30` survives and does
|
|
not merge with `v35`, which is the axis that booth is about).
|
|
|
|
None for a stem with nothing before the digits -- `01.png` has no prefix to
|
|
group on, and inventing one would file every numbered render under the
|
|
empty string.
|
|
|
|
⚠ THIS IS NOT THE RULE THE CONTRACT FIRST NAMED. `strip ONE trailing run of
|
|
digits` was measured against the live set on 2026-09-22 and yields 24 groups
|
|
for sindra-bakeoff's 40 images and 27 for sindra's 30 -- a rail with one row
|
|
per tile. The contract's own table claimed 5 and 1 for those two booths;
|
|
neither reproduces under the rule it states beside them. The rewritten table
|
|
carries the re-measurement.
|
|
|
|
Derived HERE and nowhere else (INV-1). A route body that re-derived it would
|
|
be the caption bug in a new field.
|
|
"""
|
|
stem = Path(rel).stem # basename without its last suffix; `a.tar.gz` -> `a.tar`
|
|
segs = _SEG.split(stem)
|
|
if len(segs) == 1:
|
|
return re.sub(r"\d+$", "", stem) or None
|
|
return segs[0] or None
|
|
|
|
|
|
def _resolve_captions(by_rel: dict[str, Path]) -> tuple[dict[str, str], set[str]]:
|
|
"""(caption-by-rel, rels consumed as sidecars).
|
|
|
|
Two forms, in this precedence, preserved from the original gallery:
|
|
1. `<file>.txt` — `a.png.txt` captions `a.png`
|
|
2. `<stem>.txt` beside a same-stem MEDIA sibling — `a.txt` captions
|
|
`a.png`, but NOT `a.bin` (the `classify != "other"` guard, so a stray
|
|
`data.txt` next to `data.bin` stays an item of its own)
|
|
|
|
The sibling scan runs in sorted order rather than filesystem order: when two
|
|
media files share a stem in one directory (`a.png` and `a.webm`), the
|
|
original picked whichever `rglob` happened to yield first. Same rule, now
|
|
deterministic.
|
|
"""
|
|
caption: dict[str, str] = {}
|
|
sidecars: set[str] = set()
|
|
|
|
for rel in sorted(by_rel):
|
|
if not rel.lower().endswith(".txt"):
|
|
continue
|
|
p = by_rel[rel]
|
|
target = None
|
|
|
|
base_full = rel[:-4] # "a.png.txt" -> "a.png"
|
|
if base_full in by_rel:
|
|
target = base_full
|
|
else:
|
|
parent = str(Path(rel).parent)
|
|
stem = Path(rel).stem
|
|
for q_rel in sorted(by_rel):
|
|
if q_rel == rel:
|
|
continue
|
|
q = by_rel[q_rel]
|
|
if (
|
|
str(Path(q_rel).parent) == parent
|
|
and Path(q_rel).stem == stem
|
|
and classify(q.name) != "other"
|
|
):
|
|
target = q_rel
|
|
break
|
|
|
|
if target is not None:
|
|
try:
|
|
# BOUNDED AT THE READ. `read_text()` pulled the whole sidecar
|
|
# into memory before the slice trimmed it, so a pathological
|
|
# file was a MemoryError — which the OSError handler below does
|
|
# not catch — rather than a missing caption.
|
|
#
|
|
# Deliberately NOT bounded by st_size: a FIFO reports 0 and a
|
|
# bound that trusts it inherits what it does not mean, which is
|
|
# the hang in persistent-memory.d/2026-09-22-size-cap-opened-a-hang.md.
|
|
# The factor of 4 is UTF-8's worst case, so CAPTION_MAX
|
|
# characters always survive the byte bound.
|
|
with p.open("r", errors="replace") as fh:
|
|
caption[target] = fh.read(CAPTION_MAX * 4).strip()[:CAPTION_MAX]
|
|
except OSError:
|
|
pass
|
|
sidecars.add(rel)
|
|
|
|
return caption, sidecars
|
|
|
|
|
|
def booth_items(booth: Path) -> list[Item]:
|
|
"""Every renderable file in a booth, sorted by relative path.
|
|
|
|
Excluded: dotfiles, `*.ask.json` / `*.answer.json` (they render as the asks
|
|
panel, not as tiles), and any file consumed as another item's caption.
|
|
|
|
Doc BODIES are deliberately not rendered here. The index calls this once per
|
|
booth to count items and pick a cover; rendering every doc in every booth on
|
|
every page load would be the cost of that convenience. `render_doc_body` is
|
|
the separate step, for the one consumer that needs it.
|
|
"""
|
|
by_rel: dict[str, Path] = {}
|
|
for p in booth.rglob("*"):
|
|
try:
|
|
# `is_file` swallows a missing entry but PROPAGATES EACCES: a
|
|
# directory that lists but cannot be searched made every stat under
|
|
# it raise out of here, and `list_booths` calls this for every
|
|
# booth — one such folder took down the index for all of them.
|
|
# An entry nobody can stat is not a renderable file. (design-dev)
|
|
if not p.is_file():
|
|
continue
|
|
except OSError:
|
|
continue
|
|
# ⚠ EVERY path component, not just the filename. `p.name.startswith(".")`
|
|
# tested only the leaf, so `.thumbs/a.png` (name `a.png`) sailed through
|
|
# as a gallery item — and CLAUDE.md invariant 2 promises a dotfile costs
|
|
# nothing in item counts, galleries or zips. That promise was true only
|
|
# at the top level until the `.thumbs/` cache made it matter.
|
|
#
|
|
# BOTH guards, not either: they were written independently for different
|
|
# failures and the merge that kept one would have quietly dropped the
|
|
# other.
|
|
if any(part.startswith(".") for part in p.relative_to(booth).parts):
|
|
continue
|
|
continue
|
|
if is_ask_file(p.name) or is_answer_file(p.name):
|
|
continue
|
|
rel = p.relative_to(booth).as_posix()
|
|
try:
|
|
quote(rel, safe="/")
|
|
except UnicodeEncodeError:
|
|
# A non-UTF-8 filename reaches CPython as a surrogate escape, and
|
|
# `quote` raises on it. This used to happen at Item construction,
|
|
# OUTSIDE any per-item handler — so one 0xff byte in one filename
|
|
# took out that booth's page AND the index for every booth, because
|
|
# `list_booths` calls this too. The repo's posture is that a damaged
|
|
# file costs its own tile and never the page.
|
|
#
|
|
# Skipped rather than rescued: a name that cannot be percent-encoded
|
|
# cannot be linked, served or zipped either, so there is no item to
|
|
# render. Found by the heid bug-hunt panel (hulda), 2026-09-22.
|
|
continue
|
|
by_rel[rel] = p
|
|
|
|
caption, sidecars = _resolve_captions(by_rel)
|
|
blurred = read_blurred(booth) # ONE read per call, not one per item
|
|
booth_blur = is_booth_blurred(booth) # likewise: one stat, not one per item
|
|
|
|
items: list[Item] = []
|
|
for rel in sorted(by_rel):
|
|
if rel in sidecars:
|
|
continue
|
|
p = by_rel[rel]
|
|
try:
|
|
size = p.stat().st_size
|
|
except OSError:
|
|
size = 0
|
|
kind = classify(p.name)
|
|
items.append(
|
|
Item(
|
|
rel=rel,
|
|
url=quote(rel, safe="/"),
|
|
kind=kind,
|
|
section=_section_of(rel),
|
|
group=_group_of(rel),
|
|
caption=caption.get(rel),
|
|
# Booth blur COMPOSES with the per-item set. Resolved HERE so
|
|
# every surface inherits it for free — Desk strip, tiles, tray,
|
|
# filmstrip, stage all already read `Item.blurred` and none of
|
|
# them learns about the booth flag (INV-1).
|
|
blurred=rel in blurred or (booth_blur and kind in BLURRABLE_KINDS),
|
|
doc=doc_kind(p.name),
|
|
size=size,
|
|
# Counted over items that RENDER: a caption sidecar or a name
|
|
# the quote() guard skipped takes no number, so the numbers
|
|
# stay contiguous over what the operator can see.
|
|
ordinal=len(items) + 1,
|
|
thumb=(quote(rel, safe='/') + '?thumb=1') if wants_thumb(rel) else None,
|
|
blurred_self=rel in blurred,
|
|
)
|
|
)
|
|
return items
|
|
|
|
|
|
def image_chain(items: Sequence[Item]) -> list[str]:
|
|
"""The rels of the image items, in order — the zoom view's prev/next ring.
|
|
|
|
Replaces `booth_image_names`, which walked the tree a second time to derive
|
|
what the item list already knows.
|
|
"""
|
|
return [it.rel for it in items if it.kind == "image"]
|
|
|
|
|
|
# R2 C2: what the review route steps through. ONE LINE: the item order
|
|
# filtered to media. It is a declared change to the zoom-ring rule, which was
|
|
# images only: a listening set is reviewed the same way a picture set is.
|
|
REVIEW_KINDS = ("image", "video", "audio")
|
|
|
|
|
|
def review_chain(items: Sequence[Item]) -> list[str]:
|
|
"""The rels of the media items, in item order — the review's prev/next ring,
|
|
its filmstrip and its tape."""
|
|
return [it.rel for it in items if it.kind in REVIEW_KINDS]
|
|
|
|
|
|
def find_item(items: Sequence[Item], rel: str) -> Item | None:
|
|
"""The record for one rel, or None — the zoom/doc route's entry point."""
|
|
for it in items:
|
|
if it.rel == rel:
|
|
return it
|
|
return None
|
|
|
|
|
|
def render_doc_body(booth: Path, item: Item) -> tuple[str, bool] | None:
|
|
"""(body, is_html) for a doc item under DOC_MAX_BYTES, else None.
|
|
|
|
None means "do not inline this": either it is not a doc, or it is a log big
|
|
enough that inlining it into every page render is the wrong trade.
|
|
"""
|
|
if item.doc is None or item.size > DOC_MAX_BYTES:
|
|
return None
|
|
try:
|
|
text = (booth / item.rel).read_text(encoding="utf-8", errors="replace")
|
|
except OSError:
|
|
return None
|
|
return render_doc(text, item.doc)
|