Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1b394dde18 | ||
|
|
e702be4e1a | ||
|
|
c5ac49356f |
+1
-1
@@ -1,7 +1,7 @@
|
||||
# The Booth — roadmap
|
||||
|
||||
Design: [`docs/design/information-architecture.md`](docs/design/information-architecture.md).
|
||||
Current version: `0.6.0` (U1 through U6 landed; extracted from eshpfi 2026-09-21).
|
||||
Current version: `0.6.1` (U1 through U6 landed; extracted from eshpfi 2026-09-21).
|
||||
|
||||
## v1 target
|
||||
|
||||
|
||||
+13
-3
@@ -1204,10 +1204,20 @@ def create_app(
|
||||
return _pick_fragments(name, mark)
|
||||
except Exception as exc: # noqa: BLE001 - deliberate
|
||||
broken = replace(mark, error=f"this question could not be rendered: {exc}")
|
||||
try:
|
||||
whole = str(_frag.whole(broken, ask_form_id(mark.id),
|
||||
quote(name, safe="")))
|
||||
except Exception: # noqa: BLE001 - deliberate
|
||||
# THE HANDLER MUST SURVIVE THE FAILURE IT IS HANDLING. The
|
||||
# fallback re-rendered through the SAME macro module that had
|
||||
# just raised, so when `whole` itself was the broken thing this
|
||||
# guard re-raised and took the report anyway — a guard that
|
||||
# only works when the failure is somewhere else. Found while
|
||||
# building a falsifier for the guard: the falsifier tripped it.
|
||||
# Plain text, escaped by the caller, no macro involved.
|
||||
whole = ""
|
||||
return {"id": mark.id, "error": broken.error,
|
||||
"whole": str(_frag.whole(broken, ask_form_id(mark.id),
|
||||
quote(name, safe=""))),
|
||||
"submit": "", "questions": []}
|
||||
"whole": whole, "submit": "", "questions": []}
|
||||
|
||||
@app.get("/b/{name}/embed.json")
|
||||
def booth_embed_json(name: str):
|
||||
|
||||
@@ -488,6 +488,40 @@ def _hydrate(entry: dict) -> Mark:
|
||||
norm = normalize_ask(decl, mid)
|
||||
except AskError as exc:
|
||||
return Mark(**base, declaration=decl, answer=answer, error=str(exc))
|
||||
# THE ANSWER'S SHAPE IS VALIDATED HERE, at the ONE boundary every
|
||||
# surface crosses — not at the three render sites that happen to draw
|
||||
# it today, and not defensively in the template, which would hide that
|
||||
# anything is wrong.
|
||||
#
|
||||
# `{"answer": {"answers": [], "notes": ""}}` is well-formed JSON with a
|
||||
# wrong-shaped value. It passed `_entry_type_error`, passed the
|
||||
# `isinstance(answer, dict)` check above, and `marks_for` and
|
||||
# `hold_read` both reported the mark HEALTHY with no read error — and
|
||||
# then `_ask_inline.html` did `a.answer.answers.get(q.key)`, Jinja asked
|
||||
# a LIST for `.get`, and the gallery page and the marks page returned
|
||||
# 500. Measured at 42ea67f, so it predates U3; U3 guarded only its own
|
||||
# surface with `_safe_fragments` and left these two by scope.
|
||||
#
|
||||
# This is the v0.2.2 lesson finished rather than half-done. That outage
|
||||
# was a file that could not be PARSED and the reader was made lenient;
|
||||
# this one parses perfectly and breaks one layer further in, at render,
|
||||
# where no leniency exists. `read_error` was answering a narrower
|
||||
# question than every caller assumed.
|
||||
#
|
||||
# ONLY the multi case is checked, because only the multi case indexes:
|
||||
# a single-question pick's answer IS the record, with no `answers` key
|
||||
# to get wrong. Requiring one unconditionally would break every single
|
||||
# pick, which is the direction a too-eager guard fails in.
|
||||
if norm["multi"] and isinstance(answer, dict) and \
|
||||
not isinstance(answer.get("answers"), dict):
|
||||
return Mark(**base, declaration=decl, answer=None,
|
||||
prompt=norm["prompt"], title=norm["title"],
|
||||
multi=norm["multi"], questions=norm["questions"],
|
||||
options=norm.get("options", []),
|
||||
notes_enabled=norm["notes"], notes_label=norm["notes_label"],
|
||||
error="this pick's answer is stored in a shape the page "
|
||||
"cannot render; the answer was dropped and the "
|
||||
"question is unanswered")
|
||||
return Mark(
|
||||
**base,
|
||||
declaration=decl,
|
||||
|
||||
@@ -436,7 +436,7 @@ arms flagged the staleness themselves.
|
||||
| **BH-2** | **A submit anchor inside the author's own `<form>` loses ours** — the HTML parser drops a nested form outright. Every control's `form=` then points at nothing, and the code recorded the pick as submitted so the tail added no fallback. The operator fills it in and the button does nothing. | 1 | **Genuine add.** A submit anchor counts as submitted only if the form actually survived (`hasForm`); otherwise the tail supplies one at body level, where no form encloses it. |
|
||||
| **BH-3** | **A broken pick's diagnostic never rendered from a submit-only anchor.** An errored pick's `submit` is empty; mounting that and marking it placed made the tail skip it, so the "broken ask" box vanished from the one surface built to show it. | 3 of 4 | **Genuine add.** A submit anchor for an errored pick is left alone, exactly as an anchor naming no mark is, and the tail mounts the diagnostic. |
|
||||
| **BH-4** | **An author's own element can hijack the chip.** `<section id="bk-ask-winner-background">` satisfies any id-prefix rule — the hyphen boundary from CR-7 included. | 4 of 4 | **Genuine add, and it supersedes CR-7's fix.** The chip now searches only the elements THIS SCRIPT MOUNTED, which is the identity the deleted `bk-ask-<id>-top` anchor used to guarantee, and takes the earliest of those by `compareDocumentPosition`. |
|
||||
| **BH-5** | **No error boundary around fragment rendering.** A `.marks.json` that is well-formed JSON with a wrong-shaped `answer` hydrates with no error and then raises in the macro. | 1, `needs-repro` | **Genuine add — reproduced before building for it.** `_safe_fragments` returns a per-mark error record, the same leniency `_hydrate_safe` applies one layer down. ⚠ **The gallery and marks pages still 500 on it, and that is PRE-EXISTING** — measured at `42ea67f`. Out of scope here and recorded rather than quietly widened: `persistent-memory.d/2026-09-22-a-wrong-shaped-answer-500s-the-gallery.md`. |
|
||||
| **BH-5** | **No error boundary around fragment rendering.** A `.marks.json` that is well-formed JSON with a wrong-shaped `answer` hydrates with no error and then raises in the macro. | 1, `needs-repro` | **Genuine add — reproduced before building for it.** `_safe_fragments` returns a per-mark error record, the same leniency `_hydrate_safe` applies one layer down. ⚠ **The gallery and marks pages still 500 on it, and that is PRE-EXISTING** — measured at `42ea67f`. Out of scope here and recorded rather than quietly widened: `persistent-memory.d/2026-09-22-a-wrong-shaped-answer-500s-the-gallery.md`. **CLOSED 2026-09-22**, after U6, at the hydration boundary rather than by a third copy of this guard — so `_safe_fragments` no longer has a reachable natural trigger and is now a pure backstop, falsified synthetically. Hardening the falsifier found that this guard's own fallback re-rendered through the macro module that had just raised, so it re-raised whenever `whole` was the broken thing; fixed in the same pass. |
|
||||
| **BH-6** | Prototype pollution in the placement maps (`toString` as a mark id, `constructor` as a question key). | 1 | **Already fixed this round** as CR-13, from the code-review panel. Two panels, two lenses, the same defect independently — the strongest signal of the evening that the lenses are not redundant. |
|
||||
| **BH-7** | Bare-substring declaration suppresses the chrome. | 4 of 4 | **Already fixed** as CR-3, before the reply landed. |
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# A wrong-shaped answer 500s the gallery and the marks page — PRE-EXISTING, NOT U3
|
||||
# A wrong-shaped answer 500s the gallery and the marks page — CLOSED 2026-09-22
|
||||
|
||||
_2026-09-22 · booth_
|
||||
|
||||
@@ -72,3 +72,49 @@ whose closing comment points back here.
|
||||
Related: [[2026-09-21-marks-write-wiped-judgment]],
|
||||
[[2026-09-22-lenient-reader-blast-radius]],
|
||||
[[2026-09-22-u3-declared-embed-seam-landed]].
|
||||
|
||||
|
||||
---
|
||||
|
||||
## CLOSED — 2026-09-22, after U6, at option (1)
|
||||
|
||||
Fixed in `_hydrate`, the option this entry argued for: **one predicate, one
|
||||
place, every surface inherits it.** The operator was asked three times where the
|
||||
guard belonged and did not answer; the placement was taken under the stated
|
||||
assumption, and it is cheap to move if he disagrees — the whole fix is one
|
||||
condition in one function.
|
||||
|
||||
**Only the MULTI case is checked**, because only the multi case indexes: a
|
||||
single-question pick's answer IS the record, with no `answers` key to get wrong.
|
||||
Requiring one unconditionally would break every single pick — the direction a
|
||||
too-eager guard fails in, and it has its own test.
|
||||
|
||||
Measured before and after, on the gallery booth (no `index.html`):
|
||||
|
||||
before /b/g/ 500 /b/g/marks 500 / 200 /healthz 200
|
||||
after /b/g/ 200 /b/g/marks 200 / 200 /healthz 200
|
||||
and the error is VISIBLE on the page, and the booth's
|
||||
OTHER, healthy pick still renders
|
||||
|
||||
**Two things fell out of it that are worth more than the fix.**
|
||||
|
||||
1. **`_safe_fragments` lost its natural trigger.** Probed every wrong answer
|
||||
shape reachable from a `.marks.json`: `answers` as a list, a string or null
|
||||
all become hydration errors now, and a wrong-typed VALUE inside `answers`
|
||||
renders without raising because Jinja absorbs attribute access on a
|
||||
non-mapping. So U3's guard is now a pure backstop with **no reachable
|
||||
natural input**. Its test was rewritten to a synthetic trigger that says so —
|
||||
patching the shared macro module through `app.state.templates` — rather than
|
||||
left asserting a path nothing reaches. An untested guard and a guard tested
|
||||
by an unreachable input are the same thing.
|
||||
|
||||
2. **The guard's own handler could not survive the failure it was handling.**
|
||||
Building that falsifier tripped it: `_safe_fragments` caught a raising
|
||||
`_pick_fragments` and then rebuilt the broken-ask box **through the same
|
||||
macro module that had just raised**, so when `whole` itself was broken the
|
||||
handler re-raised and took the whole report. Fixed, with its own test. Found
|
||||
by accident, which is the usual way.
|
||||
|
||||
Both new falsifiers were **verified RED against their defeating change** rather
|
||||
than assumed — the discipline from [[2026-09-22-vacuous-falsifiers]], applied to
|
||||
the fix for the entry that names it.
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
# Three cold panels on one unit, and what each lens could only see alone
|
||||
|
||||
_2026-09-22 · booth_
|
||||
|
||||
U6 ran all three `/heid*` gates plus two in-session passes. **Every one of the
|
||||
five found something the others structurally could not**, which is the
|
||||
strongest evidence this repo has for running them all rather than picking one.
|
||||
|
||||
## The scoreboard
|
||||
|
||||
| gate | when | found |
|
||||
|---|---|---|
|
||||
| **seam review** (in-session, sibling-aware) | before code | **3 real contract defects** — a claim about a sibling test that was false, `resolve_booth` named as a per-row predicate when it RAISES 404, and silence on percent-encoding |
|
||||
| **adversarial self-pass** (in-session) | during | **4 defects** — a FIFO hang, `unquote` leaking control characters, a fail-closed-by-accident guard, a stranded scratch file |
|
||||
| **`/heid-contract-review`** (4 arms) | parallel | **the import/apply selection gap, 4-of-4** — plus per-field cap semantics, and two passages of the document contradicting each other |
|
||||
| **`/heid-code-review`** (4 arms) | parallel | **3 surface-drift findings 4-of-4**, an IPv6 identity bug, and **a falsifier that could not fail** |
|
||||
| **`/heid-bug-hunt`** (4 arms) | parallel | a `<div>` inside a `<span>`, a symlink disagreement, an append outside its lock |
|
||||
|
||||
## The three findings worth remembering
|
||||
|
||||
**1. The highest-value finding was a MISSING FEATURE, and the paraphrase lens
|
||||
found it.** `bench import --apply` registered every candidate while the same
|
||||
contract said ~14 of 35 were bookmarks that must stay on the board. The dry-run
|
||||
report existed *because* the decision is not mechanizable — and then `--apply`
|
||||
ignored it. A code-vs-contract lens cannot see this: the code matched the
|
||||
contract. Only reading the contract *as prose*, for what it promises a human,
|
||||
surfaces "these two sentences cannot both be satisfied."
|
||||
|
||||
**2. A falsifier that could not fail, again.** INV-4's tie-break test went
|
||||
through the registry, and `_write_all` serializes with `sort_keys=True` — so
|
||||
both insertion orders came back off disk already id-sorted, and removing the
|
||||
tie-break left the test green. Same class as the five vacuous U4 falsifiers.
|
||||
**We ran a vacuity pass and still shipped one**; a cold reader caught it. See
|
||||
[[2026-09-22-vacuous-falsifiers]].
|
||||
|
||||
**3. The single sharpest line came from a cross-module memory no new-module
|
||||
review could have.** Three bug-hunt arms independently noted that **this repo
|
||||
had already paid for the `RecursionError` class in `marks.py`, with a test
|
||||
documenting it — and the new module re-introduced the unguarded parse.** No
|
||||
amount of reading `benches.py` in isolation surfaces that.
|
||||
|
||||
## Complementarity, measured in both directions on one diff
|
||||
|
||||
The bug-hunt panel found **three live defects the in-session pass missed** — all
|
||||
three invisible to any test (a layout nesting, a symlink disagreement, a
|
||||
lock-ordering race). The in-session pass had **already closed three of that
|
||||
panel's four convergent findings** before the reply landed. Neither substitutes
|
||||
for the other, and this round is the cleanest specimen of it so far.
|
||||
|
||||
**One finding was declined**, with reasoning recorded in the contract: on a host
|
||||
where `booth.links` cannot be imported, `booth link` now refuses every URL
|
||||
rather than only booth ones. A guard that fails open is not a guard, and that
|
||||
state is a broken install where most of the CLI is equally broken.
|
||||
@@ -58,8 +58,10 @@ SR-4 and SR-5 were **verified rather than assumed**: both `list_booths` and
|
||||
a dot, so the registry is safe from the sweeper by two guards, not one. Had
|
||||
either been absent the design would have eaten its own registry on tick one.
|
||||
|
||||
## Still open at the time of writing
|
||||
## How it closed
|
||||
|
||||
Both cold gates are IN FLIGHT — contract review `01M35BWCJ806MT75NA630Y4WFH`,
|
||||
code review `01M35CK8YKEKMV7T15JXEF6A8N`. The bug-hunt has not run. **Committed
|
||||
but NOT tagged**, per the v0.2.0 lesson: if a gate is outstanding, the tag waits.
|
||||
All three cold gates came back and were folded in full, with exactly one finding
|
||||
declined. Released as `v0.6.0` — see [[2026-09-22-u6-benches-released]] and
|
||||
[[2026-09-22-three-cold-panels-on-one-unit]]. The tag waited for the gates, per
|
||||
the v0.2.0 lesson, and that sequencing was right: the panels produced ten code
|
||||
fixes after this entry was first written.
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
# U6 released as v0.6.0 — benches, and the number that was two defects
|
||||
|
||||
_2026-09-22 · booth_
|
||||
|
||||
**The sixth of seven v1 units. Only U7 remains.** 444 → 607 tests. Tagged
|
||||
`v0.6.0` (minor, operator-approved). **NOT PUSHED** — push is his call.
|
||||
|
||||
## What shipped
|
||||
|
||||
- **`booth/benches.py`** — stdlib-only AND sibling-free. `Bench`,
|
||||
`normalize_bench_url` (the identity), a lenient `read_benches` on the render
|
||||
path and a strict `_load_strict` on the write path, `mkstemp` + `fsync` +
|
||||
`os.replace` under an flock, and `order_benches` with a stated total order
|
||||
`(state rank, name casefolded, id)`.
|
||||
- **`links.booth_target`** — ONE predicate for "is this a booth URL",
|
||||
host-agnostic, path-shaped, percent-decoding, control-character-rejecting,
|
||||
never raising. Three callers: the CLI refusal, the board's dead marker,
|
||||
`bench import`.
|
||||
- **`booth link` refuses** a booth URL (naming `booth new --why`) and a
|
||||
credentialed one, writing nothing in either case.
|
||||
- **The board marks dead rows** — 161 of 221 live. Removal stays the operator's
|
||||
two clicks through the bulk control that already existed. Nothing deletes.
|
||||
- **`booth bench add|ls|state|rm|import`**. `--apply` REQUIRES the ids.
|
||||
|
||||
## The decision that shaped the unit, and it was measured
|
||||
|
||||
**The design doc's headline "69% rot" was two defects wearing one number**, and
|
||||
splitting them is what made the unit the right size — see
|
||||
[[2026-09-22-one-number-was-two-defects]]. 178 of 221 rows are booth
|
||||
announcements (156 already dead) whose *cause* U5 had already closed; only 8 are
|
||||
the bench re-post the registry fixes. A unit scoped off the unsplit number would
|
||||
have built the registry, declared victory, and left 178 rows rotting.
|
||||
|
||||
**Identity is the FULL normalized URL, not the origin**, and that was measured
|
||||
rather than chosen: origin identity merges eight distinct gitea repositories
|
||||
into one row, three unrelated HuggingFace model cards into one, and the two LRPG
|
||||
surfaces on `10.100.10.50:8321` — *the design doc's own example of two real
|
||||
benches* — into one. It destroys more than it deduplicates.
|
||||
|
||||
**`booth link` is NOT deprecated**, against the design doc's plan. Roughly 14 of
|
||||
the 35 distinct non-booth targets are reference bookmarks (repos, model cards,
|
||||
docs) for which the board is the right and only home. Deprecating it would have
|
||||
evicted a third of its live content. The IA doc is corrected.
|
||||
|
||||
## The gates
|
||||
|
||||
All four closed, and every one paid — see
|
||||
[[2026-09-22-three-cold-panels-on-one-unit]]. Contract review
|
||||
`01M35BWCJ806MT75NA630Y4WFH`, code review `01M35CK8YKEKMV7T15JXEF6A8N`, bug hunt
|
||||
`01M35CRRK2RTVWWF1BN09AFQG3`, one consolidated reply sent to heid at
|
||||
`01M35FY8QZTB9E5VR4WXDSBGEV`.
|
||||
|
||||
## Live evidence, unplanned
|
||||
|
||||
The sweeper ran mid-session: **23 booths → 19**, and dead board rows went
|
||||
**156 → 161 in about fifteen minutes**. The defect compounding in real time
|
||||
while the fix was being built — which is the argument for U6-before-U7 playing
|
||||
out on its own.
|
||||
+57
-97
@@ -19,109 +19,69 @@ loop it turned out to actually be.
|
||||
|
||||
_As of 2026-09-22:_
|
||||
|
||||
- **v1 is gated on seven units** in `ROADMAP.md`, dependency-ordered
|
||||
**U1 → U2 → {U3, U4, U5} → U7**, with **U6 independent**.
|
||||
- **U1, U2, U3, U4 and U5 are landed — the whole middle tier is closed.** U1
|
||||
`ce598b3`; U2 `c7f9437` → `v0.2.0`, `5e41108` → `v0.2.1`, `026a1fc` →
|
||||
`v0.2.2`; U5 `c015a91` + `95beede` → `v0.3.0`; U4 `c3a97c1` → `v0.4.0`.
|
||||
**U3 landed 2026-09-22 and released as `v0.5.0`** — 444 tests green
|
||||
(410 → 444), deployed and verified live, 23/23 booth pages 200, and each of
|
||||
the four verbatim booths served at exactly +46 bytes, which is
|
||||
`len(EMBED_SCRIPT_TAG)` — one append, nothing else. `87e2c53` is the unit,
|
||||
`5c20e2f` the panel fixes, `7996fbd` the release. **PUSHED AND DEPLOYED**
|
||||
2026-09-22 on the operator's word — and it was **the first push of this
|
||||
repo's history**: `main` was 26 commits ahead of `origin/main`
|
||||
(`ce598b3..7996fbd`), so `v0.2.0` through `v0.5.0` all reached
|
||||
`git@gitea.phasefinal.com:vh/booth.git` in the same motion. `main` tracks
|
||||
`origin/main` clean now; a future session can assume a remote exists, which
|
||||
no earlier one could.
|
||||
- **U4 released as `v0.4.0`** (operator approved the minor on 2026-09-22).
|
||||
`c3a97c1` is the unit; the release commit carries the pre-existing fixes the
|
||||
bug-hunt panel surfaced in touched files. The tag waited for the last gate to
|
||||
close, per the `v0.2.0` lesson — see Tried and abandoned.
|
||||
- ⚠ **The 17 consuming handles are NOT being told** that `keep` no longer means
|
||||
"waiting on an answer" — operator decision, 2026-09-22, no broadcast. This is
|
||||
deliberate and it CHANGES HOW THE 2026-10-06 RE-COUNT READS: the hold rides
|
||||
for free, but not-pressing-`keep` has to be learned, so a flat `.forever` rate
|
||||
does not falsify anything. Read its entry before measuring.
|
||||
- **U6 LANDED 2026-09-22 — ONE UNIT LEFT TO v1.** Benches: a registry keyed by
|
||||
normalized URL, `booth link` refusing a booth URL, dead rows marked on the
|
||||
board, and a non-destructive `bench import`. 444 → 555 tests, deployed and
|
||||
verified live (23/23 booths 200, 156 of 221 rows marked dead — matching an
|
||||
independent pre-implementation count exactly). ⚠ **COMMITTED BUT NOT TAGGED
|
||||
AND NOT RELEASED**: both cold gates were still in flight at commit time
|
||||
(contract review `01M35BWCJ806MT75NA630Y4WFH`, code review
|
||||
`01M35CK8YKEKMV7T15JXEF6A8N`) and the bug-hunt had not run. Per the v0.2.0
|
||||
lesson, the tag waits for the gates. → `persistent-memory.d/2026-09-22-u6-benches-landed.md`
|
||||
- ⚠ **U6 WAS CHOSEN WITHOUT THE OPERATOR ANSWERING.** He set an autonomous goal
|
||||
("hydrate and land next unit stated in handoff") and the handoff named no
|
||||
unit. The session recommended U6 on measured grounds (its defect compounds —
|
||||
145 → 156 dead rows in a day — while U7's is dormant, and U6 had no
|
||||
unresolved design questions) and proceeded rather than blocking. **The scope
|
||||
call is still his to reverse**; nothing is pushed and nothing is tagged.
|
||||
- **U7 IS THE LAST UNIT, and its premise degraded again.** ⚠ Read
|
||||
- **NOTHING IS IN FLIGHT.** U6 (benches) landed, all four review gates closed,
|
||||
**released as `v0.6.0`** and deployed. Tree clean at `3296a86`, 607 tests
|
||||
green, 19/19 booths 200 live. ⚠ **NOT PUSHED** — push is the operator's call
|
||||
and he did not give it this session; `main` is ahead of `origin/main`.
|
||||
→ `persistent-memory.d/2026-09-22-u6-benches-released.md`
|
||||
- **v1 is gated on seven units. SIX ARE LANDED. U7 IS THE LAST ONE.** U1
|
||||
`ce598b3`; U2 → `v0.2.0`/`v0.2.1`/`v0.2.2`; U5 → `v0.3.0`; U4 → `v0.4.0`;
|
||||
U3 → `v0.5.0`; U6 `1c3ce5d` → `v0.6.0`.
|
||||
- ⚠ **Before starting U7, read
|
||||
`persistent-memory.d/2026-09-21-u7-section-premise-half-wrong.md` AND
|
||||
re-count first. On 2026-09-22 the four large booths U7 was sized against
|
||||
(`pancake-v3-full`/`pancake-v4-full` at 270 items, `sindra20-engines`,
|
||||
`sindra-finalists`) had ALL been swept. Largest live booth is `miranda-is` at
|
||||
**92 items, flat**. Two of 23 booths have subfolders and **both are reports**.
|
||||
Sections buy close to nothing; the rail, filters and grid keyboard are the
|
||||
unit.
|
||||
- **U3's tier was MINOR and the operator approved it** (2026-09-22). The
|
||||
argument that settled it, recorded because the tie-break rule says patch: a
|
||||
capability arrived AND one left — the verbatim path gained a declared public
|
||||
API (`<script src="/_booth/embed.js" defer>`) and lost no-JavaScript
|
||||
operation. That asymmetry is what made it not a tie.
|
||||
- **ALL FOUR U3 GATES ARE CLOSED.** In-session seam review (5 findings, SR-2 a
|
||||
real payload-shape bug); `/heid-contract-review`
|
||||
(`01M351WKV666D681SSRNY7D7X6`, 12 findings, 10 adopted, 2 already settled by
|
||||
the seam review while it was in flight, 1 declined);
|
||||
`/heid-code-review` (`01M352RXV1ZET566KV73C7TSB8`, 3 more vacuous falsifiers
|
||||
+ the prototype-pollution bug); `/heid-bug-hunt`
|
||||
(`01M352TPCSN52G6NGJ07T5WSGY`, 5 net-new, incl. the byte-exactness break).
|
||||
**Seven of the adopted findings were CODE fixes, not wording** — the cold
|
||||
gates were not ceremony on this unit. The **seam review ran in-session and is
|
||||
folded in** — five findings as a table at the end of the U3 contract, and SR-2
|
||||
was a real payload-shape bug the cold panel structurally could not see. U4's three and U5's three are all closed
|
||||
(`01M34VX0SH23Y3VC92E7GM4S70`, `01M34WAFJC3RTERFYBBZJN1SVG`,
|
||||
`01M34Y2R0RAJRSN36Q8K4KAB36`; `01M340PNVRS21HPASZT38PXQPN`,
|
||||
`01M341E9XAPZEFBSPK9HPGAM0S`, `01M343SXX27Z47C3STXXRC7M42`).
|
||||
- ⚠ **ONE DEFECT IS OPEN, FOUND BUT DELIBERATELY NOT FIXED, AND IT HAS NO
|
||||
TRACKING SURFACE YET.** A `.marks.json` that is well-formed JSON with a
|
||||
wrong-shaped `answer` 500s **the gallery page and the marks page** —
|
||||
reproduced, and measured at `42ea67f` so it PREDATES U3. U3 guarded its own
|
||||
surface (`_safe_fragments`) and left those two alone rather than widening the
|
||||
unit; the gallery is named out of scope in the U3 contract. **The operator was
|
||||
asked where the guard belongs and has not answered** — `_hydrate` (session
|
||||
recommendation: one predicate, one place, every surface inherits it), per
|
||||
render site, or the template. No issue filed. Read
|
||||
re-count the booths first.** Its premise has degraded twice over: every booth
|
||||
that needs navigation is FLAT, and on 2026-09-22 the four large booths it was
|
||||
sized against (`pancake-v3-full`/`pancake-v4-full` at 270 items,
|
||||
`sindra20-engines`, `sindra-finalists`) had ALL been swept. Largest live booth
|
||||
is `miranda-is` at 92 items. Two of 19 booths have subfolders and both are
|
||||
reports. Sections buy close to nothing; the rail, filters and grid keyboard
|
||||
are the unit.
|
||||
- **THE LAST OPEN DEFECT IS CLOSED.** The wrong-shaped `answer` that 500'd the
|
||||
gallery and marks pages (pre-existing, measured at `42ea67f`) is fixed at
|
||||
`_hydrate` — the placement the session recommended three times and the
|
||||
operator never ruled on, **taken under a stated assumption and cheap to move**
|
||||
(one condition in one function) if he disagrees. Measured before/after: both
|
||||
pages 500 → 200, error visible, the booth's other pick untouched. Two things
|
||||
fell out of it that matter more than the fix — U3's `_safe_fragments` lost its
|
||||
natural trigger and is now a synthetically-falsified backstop, and that guard's
|
||||
own handler could not survive the failure it was handling. Read
|
||||
`persistent-memory.d/2026-09-22-a-wrong-shaped-answer-500s-the-gallery.md`
|
||||
before touching marks rendering anywhere.
|
||||
- **Two dated predictions are pending and must not be forgotten.** U5's adoption
|
||||
re-measure on **2026-09-29** (two counts, see its entry — already at 3 of 24
|
||||
announced and 2 with a `why`, all from peers told nothing), and the `.forever`
|
||||
re-count **on or after 2026-10-06**, a fortnight after U4 landed, which is
|
||||
U4's success criterion. ⚠ Only 4 booths carry marks at all, so the hold's live
|
||||
blast radius is small and the prediction rests on both halves of U4 — see its
|
||||
entry for what a null result would and would not mean.
|
||||
- **FOUR methodology proposals sit with the operator, all UNTRACKED BY OPERATOR
|
||||
CHOICE** (no issue, no ticket — they are `/heid*` skill changes, not this
|
||||
repo's work, and are recorded here only so they are not lost). Three are from
|
||||
the U5 round: reshaping the paraphrase gate toward a drift-check for
|
||||
narrative-heavy contracts, a standing "green-tests-prove-nothing" direction
|
||||
for the code-review gate, and regin's table-vs-signature consistency pass.
|
||||
The fourth is new and is the one with evidence behind it: a **contract-time
|
||||
VACUITY PASS** — for each invariant, name a change that defeats it and check
|
||||
the test goes red. Regin and Kimi proposed it independently on the U4
|
||||
paraphrase round; the code-review panel then showed five of seven U4
|
||||
falsifiers were vacuous, and heid rates that the strongest single data point
|
||||
for it so far. See `persistent-memory.d/2026-09-22-vacuous-falsifiers.md`.
|
||||
- The booth set churns hard: 26 → 24 → 25 across the last two sessions as the
|
||||
sweeper ran. Re-count rather than trusting any number written here.
|
||||
- ⚠ **TWO OPERATOR DECISIONS ARE OUTSTANDING AND BOTH ARE DELIBERATELY NOT
|
||||
DONE.** (1) The single althing note to the 17 handles about `booth link`
|
||||
refusing booth URLs — gated as multi-recipient, drafted nowhere, NOT SENT.
|
||||
(2) Seeding the bench registry from the board — he said "no seeding yet", so
|
||||
`booth bench import --apply` has NOT been run against live data and
|
||||
`.benches.json` does not exist in `~/booth-data`.
|
||||
- ⚠ **THE 17 CONSUMING HANDLES WERE NEVER TOLD that `keep` stopped meaning
|
||||
"waiting on an answer"** — operator decision 2026-09-22, no broadcast, and it
|
||||
still stands. **This CHANGES HOW THE 2026-10-06 RE-COUNT READS**: the hold
|
||||
rides for free but not-pressing-`keep` has to be learned, so a flat `.forever`
|
||||
rate does NOT falsify the diagnosis. Read its entry before measuring.
|
||||
- **A remote exists and `main` is AHEAD of it.** `origin` is
|
||||
`git@gitea.phasefinal.com:vh/booth.git`; the first push of this repo's history
|
||||
was 2026-09-22 (26 commits, `v0.2.0`–`v0.5.0` in one motion). As of this
|
||||
snapshot `main` is **8 commits ahead of `origin/main`** — the whole of U6
|
||||
including `v0.6.0`. Pushing is the operator's call.
|
||||
- **Two dated predictions are pending and must not be run early.** U5's adoption
|
||||
re-measure on **2026-09-29**; the `.forever` re-count **on or after
|
||||
2026-10-06**. Before the second, read
|
||||
`persistent-memory.d/2026-09-22-no-notice-and-what-it-does-to-the-prediction.md`.
|
||||
- **FIVE methodology proposals sit with the operator, untracked by his choice**
|
||||
— four from earlier rounds plus Kimi's new one: promote "the falsifiable test
|
||||
is weaker than the invariant it guards" to its own ambiguity class in
|
||||
`/heid-contract-review`. It now has two data points in this repo (five of
|
||||
seven U4 falsifiers vacuous; U6 shipped a tie-break falsifier that could not
|
||||
fail). They are `/heid*` skill changes, not this repo's work.
|
||||
- The booth set churns hard: 26 → 24 → 25 → 23 → **19** across five sessions.
|
||||
Re-count rather than trusting any number written here.
|
||||
|
||||
## Recent decisions
|
||||
|
||||
- `[2026-09-22]` **U6 landed — three surfaces, three jobs, one predicate** — the seam review caught three real contract defects incl. a per-row `resolve_booth` that would have 404'd the board; NOT TAGGED, gates in flight → `persistent-memory.d/2026-09-22-u6-benches-landed.md`
|
||||
- `[2026-09-22]` **The last open defect closed, and building its falsifier found another** — the wrong-shaped answer fixed at `_hydrate`; `_safe_fragments` lost its natural trigger and its handler could not survive the failure it handled → `persistent-memory.d/2026-09-22-a-wrong-shaped-answer-500s-the-gallery.md`
|
||||
- `[2026-09-22]` **U6 released as `v0.6.0` — benches, and the number that was two defects** — six of seven v1 units landed, NOT PUSHED → `persistent-memory.d/2026-09-22-u6-benches-released.md`
|
||||
- `[2026-09-22]` **Three cold panels on one unit, and what each lens could only see alone** — READ BEFORE DECIDING TO SKIP A GATE; all five passes found something the others structurally could not → `persistent-memory.d/2026-09-22-three-cold-panels-on-one-unit.md`
|
||||
- `[2026-09-22]` **U6 landed — three surfaces, three jobs, one predicate** — the seam review caught three real contract defects incl. a per-row `resolve_booth` that would have 404'd the board → `persistent-memory.d/2026-09-22-u6-benches-landed.md`
|
||||
- `[2026-09-22]` **The 69% link-board rot was two defects wearing one number** — READ BEFORE SCOPING ANY LINK-BOARD WORK; U5 closed the larger half and full-URL-vs-origin identity is a measured call → `persistent-memory.d/2026-09-22-one-number-was-two-defects.md`
|
||||
- `[2026-09-22]` **U3 landed — the page declares the seam, the Booth mounts into it** — ten regexes against author HTML replaced by a substring test and a `+` → `persistent-memory.d/2026-09-22-u3-declared-embed-seam-landed.md`
|
||||
- `[2026-09-22]` **A wrong-shaped answer 500s the gallery and the marks page** — PRE-EXISTING (measured at `42ea67f`), NOT U3; the v0.2.2 lesson is only half-implemented → `persistent-memory.d/2026-09-22-a-wrong-shaped-answer-500s-the-gallery.md`
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "booth"
|
||||
version = "0.6.0"
|
||||
version = "0.6.1"
|
||||
description = "The Booth — a dead-simple standing web server that scans a data dir of drop-folders and renders each as an ephemeral media 'booth' (image/webm/audio auto-gallery, or a folder's own index.html verbatim). Also accepts browser/curl uploads for pickup under a human-readable id. 24h TTL, then the folder is wiped. Fleet tool for CC sessions to surface A/B and smoke results to the operator."
|
||||
requires-python = ">=3.11"
|
||||
dependencies = [
|
||||
|
||||
+95
-6
@@ -307,18 +307,107 @@ def test_a_wrongly_shaped_answer_costs_its_pick_not_the_report(client):
|
||||
r = c.get("/b/b/embed.json")
|
||||
assert r.status_code == 200
|
||||
by = {m["id"]: m for m in r.json()["marks"]}
|
||||
assert by["batch"]["error"] and "could not be rendered" in by["batch"]["error"]
|
||||
# THE PROMISE, NOT THE LAYER. This used to pin the string
|
||||
# `_safe_fragments` produces ("could not be rendered"), which made the test
|
||||
# an assertion about WHICH guard fired. As of the `_hydrate` answer-shape
|
||||
# check, this input is caught one layer earlier and never reaches
|
||||
# `_pick_fragments` at all — the endpoint's promise is unchanged and the
|
||||
# error is better (it names what is wrong with the stored answer instead of
|
||||
# reporting a render failure), so the assertion moved to the promise.
|
||||
# `_safe_fragments` is still the backstop and is still falsified, by
|
||||
# `test_safe_fragments_still_catches_what_hydration_cannot` below.
|
||||
assert by["batch"]["error"], "a wrong-shaped answer reported no error"
|
||||
assert "broken ask" in by["batch"]["whole"]
|
||||
# and the booth's other pick is untouched — one bad entry costs one entry
|
||||
assert by["healthy"]["error"] is None
|
||||
assert "Which render wins?" in by["healthy"]["whole"]
|
||||
assert c.get("/b/b/").status_code == 200
|
||||
|
||||
# ⚠ THE GALLERY AND MARKS PAGES STILL 500 ON THIS ENTRY, and that is NOT
|
||||
# U3's doing — measured at 42ea67f, the commit before this unit. They render
|
||||
# the same macro without this guard. Out of scope here (the gallery is named
|
||||
# out of scope in the contract) and recorded rather than quietly widened:
|
||||
# see persistent-memory.d/2026-09-22-a-wrong-shaped-answer-500s-the-gallery.md
|
||||
# ⚠ THE GALLERY AND MARKS PAGES USED TO 500 ON THIS ENTRY, and that was NOT
|
||||
# U3's doing — measured at 42ea67f, the commit before that unit. CLOSED
|
||||
# 2026-09-22 at the hydration boundary rather than by a third copy of this
|
||||
# guard: see tests/test_marks.py
|
||||
# ::test_a_wrong_shaped_answer_is_an_error_at_hydration_not_a_500 and
|
||||
# persistent-memory.d/2026-09-22-a-wrong-shaped-answer-500s-the-gallery.md
|
||||
|
||||
|
||||
def test_safe_fragments_still_catches_what_hydration_cannot(client):
|
||||
"""U3's `_safe_fragments` guard, kept falsifiable after `_hydrate` took its
|
||||
natural trigger away.
|
||||
|
||||
The answer-shape check in `_hydrate` now catches every wrong answer shape
|
||||
reachable from a `.marks.json` — probed 2026-09-22: `answers` as a list, a
|
||||
string or null all become hydration errors, and a wrong-typed VALUE inside
|
||||
`answers` renders without raising, because Jinja absorbs attribute access
|
||||
on a non-mapping. **No natural input reaches `_safe_fragments` by this
|
||||
route any more**, and a test that kept pretending one did would assert
|
||||
nothing — which is the failure this suite has now paid for twice.
|
||||
|
||||
So the trigger is synthetic and says so: the shared `_ask_inline` macro
|
||||
module is made to raise. `_pick_fragments` resolves `whole` off that object
|
||||
per call, and `create_app` stashes the environment on `app.state`, so this
|
||||
reaches the very object the closure captured. What it pins is the guard
|
||||
itself — one raising pick costs that pick, never the report.
|
||||
|
||||
Defeating change: removing the try/except in `_safe_fragments`, under which
|
||||
this returns 500.
|
||||
"""
|
||||
c, data = client
|
||||
b = data / "b"
|
||||
b.mkdir(parents=True, exist_ok=True)
|
||||
declare_pick(b, "batch", {"prompt": "Which?", "options": ["x", "y"]})
|
||||
(b / "index.html").write_text(DECLARED)
|
||||
|
||||
frag = c.app.state.templates.env.get_template("_ask_inline.html").module
|
||||
real_submit = frag.submit
|
||||
|
||||
def explode(*a, **k):
|
||||
raise RuntimeError("synthetic render failure")
|
||||
|
||||
object.__setattr__(frag, "submit", explode)
|
||||
try:
|
||||
assert frag.submit is explode, "the patch did not take; this test is vacuous"
|
||||
r = c.get("/b/b/embed.json")
|
||||
assert r.status_code == 200, "a raising fragment renderer took the whole report"
|
||||
by = {m["id"]: m for m in r.json()["marks"]}
|
||||
assert by["batch"]["error"] and "could not be rendered" in by["batch"]["error"]
|
||||
assert by["batch"]["whole"], "the fallback rendered nothing at all"
|
||||
finally:
|
||||
object.__setattr__(frag, "submit", real_submit)
|
||||
|
||||
# and the guard is not sticky — with the macro restored, the pick is fine
|
||||
assert c.get("/b/b/embed.json").json()["marks"][0]["error"] is None
|
||||
|
||||
|
||||
def test_the_handler_survives_the_failure_it_is_handling(client):
|
||||
"""`_safe_fragments` caught a raising `_pick_fragments` and then rebuilt the
|
||||
broken-ask box THROUGH THE SAME MACRO MODULE that had just raised. So when
|
||||
`whole` itself was the broken thing, the handler re-raised and took the
|
||||
whole report — a guard that only worked when the failure was somewhere
|
||||
else.
|
||||
|
||||
Found by accident: the first draft of the falsifier above patched `whole`,
|
||||
and the guard failed rather than caught. Defeating change: removing the
|
||||
inner try/except, under which this returns 500."""
|
||||
c, data = client
|
||||
b = data / "b"
|
||||
b.mkdir(parents=True, exist_ok=True)
|
||||
declare_pick(b, "batch", {"prompt": "Which?", "options": ["x", "y"]})
|
||||
(b / "index.html").write_text(DECLARED)
|
||||
|
||||
frag = c.app.state.templates.env.get_template("_ask_inline.html").module
|
||||
real_whole = frag.whole
|
||||
|
||||
def explode(*a, **k):
|
||||
raise RuntimeError("even the fallback macro is broken")
|
||||
|
||||
object.__setattr__(frag, "whole", explode)
|
||||
try:
|
||||
r = c.get("/b/b/embed.json")
|
||||
assert r.status_code == 200, "the handler re-raised through the broken macro"
|
||||
assert r.json()["marks"][0]["error"]
|
||||
finally:
|
||||
object.__setattr__(frag, "whole", real_whole)
|
||||
|
||||
|
||||
def test_no_regex_touches_author_html():
|
||||
|
||||
@@ -1374,3 +1374,57 @@ def test_a_clock_restore_that_fails_does_not_take_the_route_down(tmp_path):
|
||||
|
||||
assert (booth / MARKS_LOCK).exists()
|
||||
assert [m.target for m in marks_for(booth)] == ["a.png"]
|
||||
|
||||
|
||||
# ---- the wrong-shaped answer, closed at the hydration boundary --------------
|
||||
|
||||
|
||||
def test_a_wrong_shaped_answer_is_an_error_at_hydration_not_a_500(tmp_path):
|
||||
"""A `.marks.json` that is well-formed JSON with a wrong-shaped `answer`
|
||||
passed every reader and then raised in the TEMPLATE: `_hydrate` checked only
|
||||
that `answer` was a dict, never that `answer["answers"]` was one, so
|
||||
`marks_for` and `hold_read` both reported the mark healthy with no read
|
||||
error — and `_ask_inline.html` asked a list for `.get`.
|
||||
|
||||
Measured at `42ea67f`, so it predates U3. U3 guarded its own surface with
|
||||
`_safe_fragments` and left the gallery and marks pages alone by scope. This
|
||||
closes it at the boundary the rest of the module already argues for: ONE
|
||||
predicate, ONE place, every surface inherits it.
|
||||
|
||||
Defeating change: restoring the bare `isinstance(answer, dict)` check —
|
||||
under which `error` is None here and both pages 500.
|
||||
"""
|
||||
declare_pick(tmp_path, "batch", {"title": "T", "questions": [
|
||||
{"key": "r1", "prompt": "A?", "options": ["x", "y"]},
|
||||
{"key": "r2", "prompt": "B?", "options": ["x", "y"]}]})
|
||||
raw = json.loads((tmp_path / ".marks.json").read_text())
|
||||
for e in raw["marks"]:
|
||||
if e["id"] == "batch":
|
||||
e["answer"] = {"answers": [], "notes": ""}
|
||||
(tmp_path / ".marks.json").write_text(json.dumps(raw))
|
||||
|
||||
mark = {m.id: m for m in marks_for(tmp_path)}["batch"]
|
||||
assert mark.error, "a wrong-shaped answer hydrated as healthy"
|
||||
assert "answer" in mark.error
|
||||
# AND the mark is not silently emptied — the declaration survives, so the
|
||||
# operator can still see WHICH question broke rather than a bare error.
|
||||
assert mark.declaration is not None
|
||||
|
||||
|
||||
def test_a_healthy_multi_answer_still_hydrates(tmp_path):
|
||||
"""The other direction, so the guard cannot be satisfied by rejecting
|
||||
everything. Defeating change: requiring `answers` unconditionally, which
|
||||
would break every single-question pick."""
|
||||
declare_pick(tmp_path, "multi", {"title": "T", "questions": [
|
||||
{"key": "r1", "prompt": "A?", "options": ["x", "y"]}]})
|
||||
declare_pick(tmp_path, "single", {"prompt": "Which?", "options": ["x", "y"]})
|
||||
raw = json.loads((tmp_path / ".marks.json").read_text())
|
||||
for e in raw["marks"]:
|
||||
if e["id"] == "multi":
|
||||
e["answer"] = {"answers": {"r1": {"choice": "x", "notes": ""}}, "notes": ""}
|
||||
if e["id"] == "single":
|
||||
e["answer"] = {"choice": "x", "notes": ""}
|
||||
(tmp_path / ".marks.json").write_text(json.dumps(raw))
|
||||
by = {m.id: m for m in marks_for(tmp_path)}
|
||||
assert by["multi"].error is None, by["multi"].error
|
||||
assert by["single"].error is None, by["single"].error
|
||||
|
||||
Reference in New Issue
Block a user