fix(docs): a posted doc cannot run script on the Booth's origin

Found by design-dev's impeccable run and confirmed at source. Python-Markdown
passes raw HTML through, and doc.html and booth.html render the result |safe.
A <script> in any session's .md ran on the Booth's origin, and a contract that
quoted <pre> opened a real one and swallowed the rest of the doc.

Operator ruling: escape raw HTML (not an allowlist).
- render_doc deregisters Python-Markdown's block and inline HTML processors,
  so raw HTML reaches the serializer as text and is escaped there. Fenced and
  inline code are unchanged.
- Every link href in a doc goes through links.is_safe_href after
  browser-style decoding. Python-Markdown keeps character references in
  attributes, so `java&#115;cript:` reached the browser as `javascript:`.
- is_safe_href reads a backslash as a slash, as a browser does in an http(s)
  URL: `/\evil.test` is `//evil.test`. This also closes the hole on the
  link board.
- A render that raises falls back to raw text, which the template escapes.

Two of 19 live .md files render differently. One is a contract losing the
quoted <pre> that swallowed it. The other is links.md, which renders as a
board, not through render_doc.

heid bug-hunt panel (4/4): the core claim held. Its two concrete edges (the
backslash twin, the unbounded render) are fixed here. Table
tests/mutations/doc_html.toml: 9/9 proved. Suite 951 -> 975.
This commit is contained in:
vh
2026-09-28 10:37:36 -07:00
parent 50bfc7b4ec
commit 190a75a0e1
8 changed files with 361 additions and 6 deletions
+9
View File
@@ -151,6 +151,15 @@ a crash mid-write cannot truncate a file into a shorter — and therefore quiete
template escapes it inside `<pre>`, and pre-escaping here double-encodes under
Jinja autoescape.
⚠ **The markdown case is the one `|safe` render in the repo, so it carries its
own escaping.** Raw HTML in a doc is escaped to text (the block and inline HTML
processors are deregistered), and every link href goes through
`links.is_safe_href` after browser-style decoding, where `java&#115;cript:` is
`javascript:` (and a backslash reads as a slash, so `/\evil.test` is
off-origin). A render that raises falls back to raw text, which the template
escapes. Until 2026-09-28 a posted `.md` could run script on the Booth's
origin. Anything else that renders author text `|safe` inherits these rules.
### 6. Every ordered collection has a stated, deterministic order
Operator directive, 2026-09-21. Not "usually stable" and not "whatever `rglob`
+71 -4
View File
@@ -17,6 +17,7 @@ from __future__ import annotations
import json
import os
import html as _html
import re
import stat
from dataclasses import dataclass
@@ -31,6 +32,7 @@ except ImportError: # pragma: no cover
from booth.asks import is_answer_file, is_ask_file
from booth.blur import BLUR_FILE, read_blurred # noqa: F401 (re-exported)
from booth.links import is_safe_href
from booth.thumbs import wants_thumb
# Browser-playable media buckets. Anything else renders as a download link.
@@ -81,16 +83,81 @@ def doc_kind(name: str) -> str | None:
return None
# What a browser ignores in a URL before it reads the scheme: ASCII tab, LF and
# CR anywhere, and C0 controls or space at either end (WHATWG URL parsing).
# Python 3.13's urlsplit, which `is_safe_href` calls, drops the same characters
# itself, so no test here can see these two go; they are stated anyway, because
# the guard's correctness should not rest on one stdlib release's cleanup.
_URL_DROPPED = str.maketrans("", "", "\t\n\r")
_URL_TRIMMED = "".join(map(chr, range(0x21)))
def _browser_href(raw: str) -> str:
"""An href as the browser will act on it: markdown's `&` placeholder put
back, character references decoded ONCE (the browser decodes an attribute
value once), then the characters URL parsing drops. `java&#115;cript:` is
`javascript:` to a browser, and a scheme test that skips this is blind to
it."""
s = raw.replace(_markdown.util.AMP_SUBSTITUTE, "&")
return _html.unescape(s).translate(_URL_DROPPED).strip(_URL_TRIMMED)
if _markdown is not None:
class _UnsafeHrefs(_markdown.treeprocessors.Treeprocessor):
"""Drops every link href `links.is_safe_href` would refuse — the ONE
predicate for "may this be a clickable link on the Booth's origin", the
board's since 2026-09-23. The link keeps its words; it just goes
nowhere. Runs last, after markdown has finished writing hrefs.
`a@href` ONLY, stated so nobody reads more into it: an `img@src` of
`javascript:` or `data:text/html` is inert in every current browser, and
a `data:image/...` picture is a legitimate thing for a doc to carry."""
def run(self, root):
for el in root.iter("a"):
href = el.get("href")
if href is not None and not is_safe_href(_browser_href(href)):
del el.attrib["href"]
def _markdown_renderer():
"""A Markdown instance that treats raw HTML as TEXT.
Python-Markdown passes raw HTML through, and doc.html / booth.html render
the result `|safe` — so a `<script>` in any session's `.md` ran on the
Booth's origin, and a contract that merely QUOTED `<pre>` opened a real one
and swallowed the rest of the doc (design-dev's impeccable run, 2026-09-28).
Operator ruling: ESCAPE raw HTML, not an allowlist; the live docs that carry
tags mean the literal tag. With the block and inline HTML processors gone,
`<` reaches the serializer as text and is escaped there. Fenced and inline
code are untouched: they never went through either processor.
"""
md = _markdown.Markdown(extensions=["fenced_code", "tables", "sane_lists"])
md.preprocessors.deregister("html_block")
md.inlinePatterns.deregister("html")
md.treeprocessors.register(_UnsafeHrefs(md), "booth_unsafe_hrefs", -10)
return md
def render_doc(text: str, kind: str) -> tuple[str, bool]:
"""(rendered, is_html). Markdown → HTML (fenced code, tables, sane lists);
plain text — or markdown when the lib is unavailable — → raw text for <pre>.
"""(rendered, is_html). Markdown → HTML (fenced code, tables, sane lists),
with raw HTML ESCAPED and unsafe link hrefs dropped (see
`_markdown_renderer`); plain text — or markdown when the lib is unavailable
— → raw text for <pre>.
Text is returned RAW on purpose: the template escapes it inside <pre>, and
pre-escaping here would double-encode under Jinja autoescape.
"""
if kind == "markdown" and _markdown is not None:
html = _markdown.markdown(text, extensions=["fenced_code", "tables", "sane_lists"])
return html, True
# BOUNDED, like every other reader of author content here: a doc that
# makes the renderer raise — deep nesting, or a markdown upgrade that
# renames the processors deregistered above — costs that doc its
# formatting and falls back to raw text, which the template escapes.
# It never raises out of the page (heid bug-hunt, 3 of 4 arms).
try:
return _markdown_renderer().convert(text), True
except Exception: # noqa: BLE001 - deliberate
return text, False
return text, False
+6 -1
View File
@@ -80,9 +80,14 @@ def is_safe_href(url: str) -> bool:
NEVER RAISES: a board row is arbitrary agent-written text and a predicate
that raises on one row takes the whole page.
A backslash is read as a SLASH first, because a browser does that in an
http(s) URL: `/\\evil.test` is `//evil.test` to it, and passed here as a
relative path until 2026-09-28 (heid bug-hunt, groa). This predicate also
guards every link in a markdown doc (`booth.items`).
"""
try:
parts = urlsplit((url or "").strip())
parts = urlsplit((url or "").strip().replace("\\", "/"))
except (ValueError, UnicodeDecodeError):
return False
# Scheme-relative (`//evil.test/x`) parses with an EMPTY scheme and a netloc,
@@ -0,0 +1,38 @@
# 2026-09-28 — A posted doc could run script on the Booth's origin
**Found by design-dev's impeccable run** (the whole-surface audit Prime asked
for, report booth `booth-antislop`), confirmed at source by booth-dev:
Python-Markdown passes raw HTML through and `doc.html` / `booth.html` render it
`|safe`. Any session's `.md` could carry a `<script>`; a contract that merely
QUOTED `<pre>` opened a real one and swallowed the rest of the doc.
**Operator ruling (Prime, in this session: "A"; and in design-dev's): ESCAPE
raw HTML, not an allowlist** — the live docs that carry tags mean the literal
tag, and an allowlist would still turn a quoted `<pre>` into a real one.
Measured before shipping: 2 of 19 live `.md` files render differently; one is
`booth-redesign/03-u1-item-record.contract.md` losing exactly the swallowing
`<pre>`, the other is `links/links.md`, which renders as a board and never
through `render_doc`.
**Found while fixing it, same class:** markdown link hrefs were never checked,
and Python-Markdown keeps character references in attributes, so
`[x](java&#115;cript:...)` reached the browser as `javascript:`. Every doc href
now goes through `links.is_safe_href` after browser-style decoding
(`_browser_href`). mailto autolinks lose their href as a result; accepted.
**The heid panel (4/4, thread `01M3MFG07JCAJTQXRBC1GKS2FG`) said the core
claim holds** and found its edges: `/\evil.test` passed `is_safe_href` as a
relative path although a browser reads it as `//evil.test` (fixed in the ONE
predicate, so the board is closed too); no bound around the render (fixed:
a raising render falls back to escaped raw text, which also covers a markdown
upgrade renaming the deregistered processors — the tests would go red on that
upgrade). Declined: emphasis still applying inside quoted HTML (`**x**` in a
quoted attribute renders bold) — that is prose getting markdown; quote in a
code span for byte-literal. Accepted: `img@src` unguarded (inert in current
browsers; data: images are legitimate), `html.unescape` as a superset of
attribute decoding (over-refuses at worst).
**Guards not claimed as falsifiers:** the tab/CR/LF drop and C0 trim in
`_browser_href`, because Python 3.13's urlsplit does the same; the
AMP_SUBSTITUTE restore, reachable only through automail (always `mailto:`).
Table: `tests/mutations/doc_html.toml`, 9/9 proved.
+4 -1
View File
@@ -1,6 +1,6 @@
# Persistent memory — booth
_Last updated: 2026-09-27_
_Last updated: 2026-09-28_
> **Always check for `/tmp/booth-dev-handoff.md`** — if it exists and its
> `Written:` stamp is under 8 hours old, read it (it carries the in-flight
@@ -19,6 +19,8 @@ loop it turned out to actually be.
_As of 2026-09-27:_
- ✅ **DOCS CAN NO LONGER RUN SCRIPT** (2026-09-28, Prime ruled "escape"): raw HTML in a `.md` renders as text, doc hrefs go through `is_safe_href`, the board's backslash twin of `//host` is closed. design-dev's anti-slop fix slices (`design-dev/antislop-sN`, Prime said GO in design-dev's session) arrive one ref at a time for booth-dev's gate.
→ `persistent-memory.d/2026-09-28-a-posted-doc-could-run-script.md`
- ✅ **ONE SUBMIT SAVES EVERY ASK ON A PAGE** (2026-09-27, Prime's bug via
infra-ops, thread `01M3JED397G1SZH7580PCXNVVA`). Client-side on both
surfaces (embed.js, base.html's in-place script), no server change; a
@@ -87,6 +89,7 @@ _As of 2026-09-27:_
## Recent decisions
- `[2026-09-28]` ✅ **A posted doc could run script; raw HTML is now escaped and doc hrefs guarded** — Prime ruled ESCAPE; READ BEFORE RENDERING ANY AUTHOR TEXT `|safe` or touching `links.is_safe_href`, which now guards docs too → `persistent-memory.d/2026-09-28-a-posted-doc-could-run-script.md`
- `[2026-09-27]` ✅ **One submit saves every ask on the page; the heid panel found the async window** — READ BEFORE TOUCHING THE SUBMIT PATH OF embed.js OR base.html: a batch reads forms at the press while the page stays live, and only a test that acts inside the flight can see it → `persistent-memory.d/2026-09-27-one-submit-saves-every-ask.md`
- `[2026-09-24]` ⏸ **The upload route's three lifecycle gaps: DEFERRED** — the pickup-id `mkdir` sits outside the try (a FileExistsError race), `rmtree(ignore_errors=True)` hides its own failure, and `except Exception` misses CancelledError. All three are rare; the operator was told and merged without them. Tracked in `225ba32`'s commit message.
- `[2026-09-24]` ✅ **Upload names: two crashes found, then two holes in the fix** — READ BEFORE WRITING A SANITISER: drop everything droppable FIRST, then apply the structural rules; a NUL test through httpx `files=` proves nothing → `persistent-memory.d/2026-09-24-upload-names-two-crashes-then-two-holes.md`
+93
View File
@@ -0,0 +1,93 @@
# A posted doc cannot run code (2026-09-28). design-dev's impeccable run found
# raw HTML passing through Python-Markdown into a `|safe` render; operator
# ruling: ESCAPE it. Found while fixing it: markdown link hrefs, where an
# entity-encoded `java&#115;cript:` passes any scheme test that does not decode
# it first. Every row is a change tests/test_items.py claims to forbid.
#
# NOT here, on purpose: the tab/CR/LF drop and the C0 trim in `_browser_href`.
# Python 3.13's urlsplit, under `is_safe_href`, drops the same characters, so no
# test can see them go. They stay as a statement of browser semantics, and are
# not claimed as proven falsifiers.
unit = "doc html"
[[mutation]]
label = "block-level raw HTML passes through (a <script> block runs)"
file = "booth/items.py"
test = "tests/test_items.py::test_raw_html_in_a_doc_is_text_never_markup"
old = '''
md.preprocessors.deregister("html_block")'''
new = ''''''
[[mutation]]
label = "inline raw HTML passes through (an <img onerror> runs)"
file = "booth/items.py"
test = "tests/test_items.py::test_raw_html_in_a_doc_is_text_never_markup"
old = '''
md.inlinePatterns.deregister("html")'''
new = ''''''
[[mutation]]
label = "no href guard at all (javascript: links stay clickable)"
file = "booth/items.py"
test = "tests/test_items.py::test_a_link_that_would_run_code_keeps_its_text_and_loses_its_href"
old = '''
md.treeprocessors.register(_UnsafeHrefs(md), "booth_unsafe_hrefs", -10)'''
new = ''''''
[[mutation]]
label = "the href guard runs before markdown has written any link"
file = "booth/items.py"
test = "tests/test_items.py::test_a_link_that_would_run_code_keeps_its_text_and_loses_its_href"
old = '''
md.treeprocessors.register(_UnsafeHrefs(md), "booth_unsafe_hrefs", -10)'''
new = '''
md.treeprocessors.register(_UnsafeHrefs(md), "booth_unsafe_hrefs", 30)'''
[[mutation]]
label = "the scheme test reads the raw attribute (java&#115;cript: passes)"
file = "booth/items.py"
test = "tests/test_items.py::test_a_link_that_would_run_code_keeps_its_text_and_loses_its_href"
old = '''
return _html.unescape(s).translate(_URL_DROPPED).strip(_URL_TRIMMED)'''
new = '''
return s.translate(_URL_DROPPED).strip(_URL_TRIMMED)'''
[[mutation]]
label = "the guard drops every href, ordinary links included"
file = "booth/items.py"
test = "tests/test_items.py::test_ordinary_links_survive"
old = '''
if href is not None and not is_safe_href(_browser_href(href)):'''
new = '''
if href is not None:'''
[[mutation]]
label = "a backslash is not read as a slash (/\\evil.test passes as a relative path) — docs"
file = "booth/links.py"
test = "tests/test_items.py::test_a_link_that_leaves_the_origin_by_backslash_is_refused"
old = '''
parts = urlsplit((url or "").strip().replace("\\", "/"))'''
new = '''
parts = urlsplit((url or "").strip())'''
[[mutation]]
label = "a backslash is not read as a slash — the board"
file = "booth/links.py"
test = "tests/test_booth.py::test_the_link_board_refuses_the_backslash_twin_of_protocol_relative"
old = '''
parts = urlsplit((url or "").strip().replace("\\", "/"))'''
new = '''
parts = urlsplit((url or "").strip())'''
[[mutation]]
label = "the render is unbounded (a renderer failure raises out of the page)"
file = "booth/items.py"
test = "tests/test_items.py::test_a_renderer_failure_costs_the_doc_its_formatting_never_the_page"
old = '''
try:
return _markdown_renderer().convert(text), True
except Exception: # noqa: BLE001 - deliberate
return text, False'''
new = '''
return _markdown_renderer().convert(text), True'''
+16
View File
@@ -1695,6 +1695,22 @@ def test_the_link_board_refuses_to_render_a_script_href(tmp_path):
assert "evil.test" in html, "the refused row vanished instead of being shown inert"
def test_the_link_board_refuses_the_backslash_twin_of_protocol_relative(tmp_path):
"""heid bug-hunt 2026-09-28, groa: `//evil.test` was refused and
`/\\evil.test` was not, but a browser reads a backslash as a slash in an
http(s) URL. Same predicate as the docs, same hole, closed once."""
b = tmp_path / "links"
b.mkdir()
b.joinpath("links.md").write_text(
"- [twin](/\\evil.test/x) <sub>· rogue · 2026-09-28 10:00</sub>\n"
"- [legitimate](https://ok.test/r) <sub>· fine · 2026-09-28 10:01</sub>\n"
)
c = TestClient(create_app(tmp_path, ttl_hours=24, start_sweeper=False))
html = c.get("/b/links/").text
assert 'href="https://ok.test/r"' in html
assert 'href="/\\evil.test' not in html
def test_the_board_delete_dialog_cannot_be_rewritten_by_a_link_row(tmp_path):
"""A board row's description and URL come from any of seventeen agent
handles, and they are pasted into a `confirm()` dialog — which is the text
+124
View File
@@ -406,3 +406,127 @@ def test_a_dot_directory_hides_its_whole_subtree(tmp_path):
assert [i.rel for i in booth_items(b)] == ["real.png"]
assert zipfile.ZipFile(io.BytesIO(zip_booth(b))).namelist() == ["real.png"]
# ---- a posted doc cannot run code (2026-09-28) --------------------------------
#
# design-dev's impeccable run found it and it held at source: Python-Markdown
# passes raw HTML through, and doc.html / booth.html render the result `|safe`,
# so a `<script>` in any agent's `.md` ran on the Booth's origin, and a contract
# that merely QUOTED `<pre>` opened a real one and swallowed the rest of the
# doc. Operator ruling: ESCAPE raw HTML, not an allowlist — the live docs that
# carry tags mean the literal tag. The same class, found while fixing it: a
# markdown link's href is not HTML-escaped either, and an entity-encoded
# `java&#115;cript:` sails past any scheme test that does not decode it first.
from html.parser import HTMLParser
from booth.items import render_doc
class _Scan(HTMLParser):
"""What a BROWSER would see: tags as parsed, attribute values decoded."""
def __init__(self):
super().__init__(convert_charrefs=True)
self.tags, self.hrefs = [], []
def handle_starttag(self, tag, attrs):
self.tags.append(tag)
for k, v in attrs:
if k == "href":
self.hrefs.append(v)
def _scan(md_text):
html, is_html = render_doc(md_text, "markdown")
assert is_html
s = _Scan()
s.feed(html)
return html, s
def _navigates_to_script(href):
# the browser drops tab/CR/LF anywhere and C0-or-space at the ends
bare = "".join(ch for ch in href if ch not in "\t\r\n").strip("".join(map(chr, range(0x21))))
return bare.lower().startswith(("javascript:", "vbscript:", "data:"))
@pytest.mark.parametrize("src", [
"<script>alert(1)</script>\n\nafter",
"inline <img src=x onerror=alert(1)> here",
"<iframe src=//evil.test></iframe>",
"| a |\n|---|\n| <svg onload=alert(1)> |",
])
def test_raw_html_in_a_doc_is_text_never_markup(src):
html, s = _scan(src)
assert not {"script", "img", "iframe", "svg"} & set(s.tags), (s.tags, html)
assert "&lt;" in html
def test_a_quoted_pre_is_shown_not_opened_and_the_doc_goes_on():
html, s = _scan("a contract that says <pre> opens one\n\nnext paragraph")
assert "pre" not in s.tags, html
assert "<p>next paragraph</p>" in html
def test_fenced_code_is_still_a_code_block():
"""Positive control: escaping raw HTML must not cost the code block."""
html, s = _scan("```html\n<script>x</script>\n```")
assert s.tags == ["pre", "code"], html
assert "&lt;script&gt;" in html
def test_inline_html_a_doc_meant_is_now_literal_text():
"""The declared cost of the ruling: <sub> and <details> show as tags."""
html, s = _scan("H<sub>2</sub>O")
assert "sub" not in s.tags and "&lt;sub&gt;" in html
@pytest.mark.parametrize("dest", [
"[x](javascript:alert(1))",
"[x](JaVaScRiPt:alert(1))",
"[x](java&#115;cript:alert(1))",
"[x](javascript&colon;alert(1))",
"[x](&#106;avascript:alert`1`)",
"[x](java&Tab;script:alert`1`)",
"[x](javascript&#x3a;alert`1`)",
"[x](<java\tscript:alert(1)>)",
"[x](\x01javascript:alert`1`)",
"[r]: javascript:alert`1`\n\n[go][r]",
"[x](data:text/html,<script>alert(1)</script>)",
])
def test_a_link_that_would_run_code_keeps_its_text_and_loses_its_href(dest):
html, s = _scan(dest)
assert not any(_navigates_to_script(h) for h in s.hrefs), (s.hrefs, html)
assert "a" in s.tags, html # the words are still there
def test_ordinary_links_survive():
html, s = _scan("[a](https://example.com) [b](http://x.test/p) [c](other.md) [d](#frag)")
assert s.hrefs == ["https://example.com", "http://x.test/p", "other.md", "#frag"], html
# Markdown reads `\\` as an escaped backslash, so FOUR in the source put two
# in the href: `\\evil.test`, which a browser reads as `//evil.test`.
@pytest.mark.parametrize("dest", ["[x](//evil.test/p)", "[x](/\\evil.test/p)", "[x](\\\\\\\\evil.test/p)"])
def test_a_link_that_leaves_the_origin_by_backslash_is_refused(dest):
"""heid bug-hunt, groa: `is_safe_href` refused `//host` but not its
backslash twin — a browser reads `\\` as `/` in an http(s) URL, so
`/\\evil.test` is `//evil.test`."""
html, s = _scan(dest)
assert s.hrefs == [], (s.hrefs, html)
def test_a_renderer_failure_costs_the_doc_its_formatting_never_the_page(monkeypatch):
"""heid bug-hunt, 3 of 4 arms: nothing bounded the render. A doc that
makes Python-Markdown raise — deep nesting, or an upgrade that renames the
processors this module deregisters — must fall back to escaped raw text,
never raise out of the page."""
import booth.items as items_mod
def boom():
raise RecursionError("too deep")
monkeypatch.setattr(items_mod, "_markdown_renderer", boom)
assert items_mod.render_doc("# t\n\n<script>x</script>", "markdown") == \
("# t\n\n<script>x</script>", False)