diff --git a/CLAUDE.md b/CLAUDE.md index 5aea5a3..4e14f85 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -151,6 +151,15 @@ a crash mid-write cannot truncate a file into a shorter — and therefore quiete template escapes it inside `
`, and pre-escaping here double-encodes under
 Jinja autoescape.
 
+⚠ **The markdown case is the one `|safe` render in the repo, so it carries its
+own escaping.** Raw HTML in a doc is escaped to text (the block and inline HTML
+processors are deregistered), and every link href goes through
+`links.is_safe_href` after browser-style decoding, where `javascript:` is
+`javascript:` (and a backslash reads as a slash, so `/\evil.test` is
+off-origin). A render that raises falls back to raw text, which the template
+escapes. Until 2026-09-28 a posted `.md` could run script on the Booth's
+origin. Anything else that renders author text `|safe` inherits these rules.
+
 ### 6. Every ordered collection has a stated, deterministic order
 
 Operator directive, 2026-09-21. Not "usually stable" and not "whatever `rglob`
diff --git a/booth/items.py b/booth/items.py
index e7145f6..5c11fe6 100644
--- a/booth/items.py
+++ b/booth/items.py
@@ -17,6 +17,7 @@ from __future__ import annotations
 
 import json
 import os
+import html as _html
 import re
 import stat
 from dataclasses import dataclass
@@ -31,6 +32,7 @@ except ImportError:  # pragma: no cover
 
 from booth.asks import is_answer_file, is_ask_file
 from booth.blur import BLUR_FILE, read_blurred  # noqa: F401 (re-exported)
+from booth.links import is_safe_href
 from booth.thumbs import wants_thumb
 
 # Browser-playable media buckets. Anything else renders as a download link.
@@ -81,16 +83,81 @@ def doc_kind(name: str) -> str | None:
     return None
 
 
+# What a browser ignores in a URL before it reads the scheme: ASCII tab, LF and
+# CR anywhere, and C0 controls or space at either end (WHATWG URL parsing).
+# Python 3.13's urlsplit, which `is_safe_href` calls, drops the same characters
+# itself, so no test here can see these two go; they are stated anyway, because
+# the guard's correctness should not rest on one stdlib release's cleanup.
+_URL_DROPPED = str.maketrans("", "", "\t\n\r")
+_URL_TRIMMED = "".join(map(chr, range(0x21)))
+
+
+def _browser_href(raw: str) -> str:
+    """An href as the browser will act on it: markdown's `&` placeholder put
+    back, character references decoded ONCE (the browser decodes an attribute
+    value once), then the characters URL parsing drops. `javascript:` is
+    `javascript:` to a browser, and a scheme test that skips this is blind to
+    it."""
+    s = raw.replace(_markdown.util.AMP_SUBSTITUTE, "&")
+    return _html.unescape(s).translate(_URL_DROPPED).strip(_URL_TRIMMED)
+
+
+if _markdown is not None:
+    class _UnsafeHrefs(_markdown.treeprocessors.Treeprocessor):
+        """Drops every link href `links.is_safe_href` would refuse — the ONE
+        predicate for "may this be a clickable link on the Booth's origin", the
+        board's since 2026-09-23. The link keeps its words; it just goes
+        nowhere. Runs last, after markdown has finished writing hrefs.
+
+        `a@href` ONLY, stated so nobody reads more into it: an `img@src` of
+        `javascript:` or `data:text/html` is inert in every current browser, and
+        a `data:image/...` picture is a legitimate thing for a doc to carry."""
+
+        def run(self, root):
+            for el in root.iter("a"):
+                href = el.get("href")
+                if href is not None and not is_safe_href(_browser_href(href)):
+                    del el.attrib["href"]
+
+
+def _markdown_renderer():
+    """A Markdown instance that treats raw HTML as TEXT.
+
+    Python-Markdown passes raw HTML through, and doc.html / booth.html render
+    the result `|safe` — so a `\n\nafter",
+    "inline  here",
+    "",
+    "| a |\n|---|\n|  |",
+])
+def test_raw_html_in_a_doc_is_text_never_markup(src):
+    html, s = _scan(src)
+    assert not {"script", "img", "iframe", "svg"} & set(s.tags), (s.tags, html)
+    assert "<" in html
+
+
+def test_a_quoted_pre_is_shown_not_opened_and_the_doc_goes_on():
+    html, s = _scan("a contract that says 
 opens one\n\nnext paragraph")
+    assert "pre" not in s.tags, html
+    assert "

next paragraph

" in html + + +def test_fenced_code_is_still_a_code_block(): + """Positive control: escaping raw HTML must not cost the code block.""" + html, s = _scan("```html\n\n```") + assert s.tags == ["pre", "code"], html + assert "<script>" in html + + +def test_inline_html_a_doc_meant_is_now_literal_text(): + """The declared cost of the ruling: and
show as tags.""" + html, s = _scan("H2O") + assert "sub" not in s.tags and "<sub>" in html + + +@pytest.mark.parametrize("dest", [ + "[x](javascript:alert(1))", + "[x](JaVaScRiPt:alert(1))", + "[x](javascript:alert(1))", + "[x](javascript:alert(1))", + "[x](javascript:alert`1`)", + "[x](java script:alert`1`)", + "[x](javascript:alert`1`)", + "[x]()", + "[x](\x01javascript:alert`1`)", + "[r]: javascript:alert`1`\n\n[go][r]", + "[x](data:text/html,)", +]) +def test_a_link_that_would_run_code_keeps_its_text_and_loses_its_href(dest): + html, s = _scan(dest) + assert not any(_navigates_to_script(h) for h in s.hrefs), (s.hrefs, html) + assert "a" in s.tags, html # the words are still there + + +def test_ordinary_links_survive(): + html, s = _scan("[a](https://example.com) [b](http://x.test/p) [c](other.md) [d](#frag)") + assert s.hrefs == ["https://example.com", "http://x.test/p", "other.md", "#frag"], html + + +# Markdown reads `\\` as an escaped backslash, so FOUR in the source put two +# in the href: `\\evil.test`, which a browser reads as `//evil.test`. +@pytest.mark.parametrize("dest", ["[x](//evil.test/p)", "[x](/\\evil.test/p)", "[x](\\\\\\\\evil.test/p)"]) +def test_a_link_that_leaves_the_origin_by_backslash_is_refused(dest): + """heid bug-hunt, groa: `is_safe_href` refused `//host` but not its + backslash twin — a browser reads `\\` as `/` in an http(s) URL, so + `/\\evil.test` is `//evil.test`.""" + html, s = _scan(dest) + assert s.hrefs == [], (s.hrefs, html) + + +def test_a_renderer_failure_costs_the_doc_its_formatting_never_the_page(monkeypatch): + """heid bug-hunt, 3 of 4 arms: nothing bounded the render. A doc that + makes Python-Markdown raise — deep nesting, or an upgrade that renames the + processors this module deregisters — must fall back to escaped raw text, + never raise out of the page.""" + import booth.items as items_mod + + def boom(): + raise RecursionError("too deep") + monkeypatch.setattr(items_mod, "_markdown_renderer", boom) + assert items_mod.render_doc("# t\n\n", "markdown") == \ + ("# t\n\n", False)