diff --git a/CLAUDE.md b/CLAUDE.md index 5aea5a3..4e14f85 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -151,6 +151,15 @@ a crash mid-write cannot truncate a file into a shorter — and therefore quiete template escapes it inside `
`, and pre-escaping here double-encodes under
Jinja autoescape.
+⚠ **The markdown case is the one `|safe` render in the repo, so it carries its
+own escaping.** Raw HTML in a doc is escaped to text (the block and inline HTML
+processors are deregistered), and every link href goes through
+`links.is_safe_href` after browser-style decoding, where `javascript:` is
+`javascript:` (and a backslash reads as a slash, so `/\evil.test` is
+off-origin). A render that raises falls back to raw text, which the template
+escapes. Until 2026-09-28 a posted `.md` could run script on the Booth's
+origin. Anything else that renders author text `|safe` inherits these rules.
+
### 6. Every ordered collection has a stated, deterministic order
Operator directive, 2026-09-21. Not "usually stable" and not "whatever `rglob`
diff --git a/booth/items.py b/booth/items.py
index e7145f6..5c11fe6 100644
--- a/booth/items.py
+++ b/booth/items.py
@@ -17,6 +17,7 @@ from __future__ import annotations
import json
import os
+import html as _html
import re
import stat
from dataclasses import dataclass
@@ -31,6 +32,7 @@ except ImportError: # pragma: no cover
from booth.asks import is_answer_file, is_ask_file
from booth.blur import BLUR_FILE, read_blurred # noqa: F401 (re-exported)
+from booth.links import is_safe_href
from booth.thumbs import wants_thumb
# Browser-playable media buckets. Anything else renders as a download link.
@@ -81,16 +83,81 @@ def doc_kind(name: str) -> str | None:
return None
+# What a browser ignores in a URL before it reads the scheme: ASCII tab, LF and
+# CR anywhere, and C0 controls or space at either end (WHATWG URL parsing).
+# Python 3.13's urlsplit, which `is_safe_href` calls, drops the same characters
+# itself, so no test here can see these two go; they are stated anyway, because
+# the guard's correctness should not rest on one stdlib release's cleanup.
+_URL_DROPPED = str.maketrans("", "", "\t\n\r")
+_URL_TRIMMED = "".join(map(chr, range(0x21)))
+
+
+def _browser_href(raw: str) -> str:
+ """An href as the browser will act on it: markdown's `&` placeholder put
+ back, character references decoded ONCE (the browser decodes an attribute
+ value once), then the characters URL parsing drops. `javascript:` is
+ `javascript:` to a browser, and a scheme test that skips this is blind to
+ it."""
+ s = raw.replace(_markdown.util.AMP_SUBSTITUTE, "&")
+ return _html.unescape(s).translate(_URL_DROPPED).strip(_URL_TRIMMED)
+
+
+if _markdown is not None:
+ class _UnsafeHrefs(_markdown.treeprocessors.Treeprocessor):
+ """Drops every link href `links.is_safe_href` would refuse — the ONE
+ predicate for "may this be a clickable link on the Booth's origin", the
+ board's since 2026-09-23. The link keeps its words; it just goes
+ nowhere. Runs last, after markdown has finished writing hrefs.
+
+ `a@href` ONLY, stated so nobody reads more into it: an `img@src` of
+ `javascript:` or `data:text/html` is inert in every current browser, and
+ a `data:image/...` picture is a legitimate thing for a doc to carry."""
+
+ def run(self, root):
+ for el in root.iter("a"):
+ href = el.get("href")
+ if href is not None and not is_safe_href(_browser_href(href)):
+ del el.attrib["href"]
+
+
+def _markdown_renderer():
+ """A Markdown instance that treats raw HTML as TEXT.
+
+ Python-Markdown passes raw HTML through, and doc.html / booth.html render
+ the result `|safe` — so a `\n\nafter",
+ "inline
here",
+ "",
+ "| a |\n|---|\n|