fix(docs): a posted doc cannot run script on the Booth's origin
Found by design-dev's impeccable run and confirmed at source. Python-Markdown passes raw HTML through, and doc.html and booth.html render the result |safe. A <script> in any session's .md ran on the Booth's origin, and a contract that quoted <pre> opened a real one and swallowed the rest of the doc. Operator ruling: escape raw HTML (not an allowlist). - render_doc deregisters Python-Markdown's block and inline HTML processors, so raw HTML reaches the serializer as text and is escaped there. Fenced and inline code are unchanged. - Every link href in a doc goes through links.is_safe_href after browser-style decoding. Python-Markdown keeps character references in attributes, so `javascript:` reached the browser as `javascript:`. - is_safe_href reads a backslash as a slash, as a browser does in an http(s) URL: `/\evil.test` is `//evil.test`. This also closes the hole on the link board. - A render that raises falls back to raw text, which the template escapes. Two of 19 live .md files render differently. One is a contract losing the quoted <pre> that swallowed it. The other is links.md, which renders as a board, not through render_doc. heid bug-hunt panel (4/4): the core claim held. Its two concrete edges (the backslash twin, the unbounded render) are fixed here. Table tests/mutations/doc_html.toml: 9/9 proved. Suite 951 -> 975.
This commit is contained in:
@@ -406,3 +406,127 @@ def test_a_dot_directory_hides_its_whole_subtree(tmp_path):
|
||||
|
||||
assert [i.rel for i in booth_items(b)] == ["real.png"]
|
||||
assert zipfile.ZipFile(io.BytesIO(zip_booth(b))).namelist() == ["real.png"]
|
||||
|
||||
|
||||
# ---- a posted doc cannot run code (2026-09-28) --------------------------------
|
||||
#
|
||||
# design-dev's impeccable run found it and it held at source: Python-Markdown
|
||||
# passes raw HTML through, and doc.html / booth.html render the result `|safe`,
|
||||
# so a `<script>` in any agent's `.md` ran on the Booth's origin, and a contract
|
||||
# that merely QUOTED `<pre>` opened a real one and swallowed the rest of the
|
||||
# doc. Operator ruling: ESCAPE raw HTML, not an allowlist — the live docs that
|
||||
# carry tags mean the literal tag. The same class, found while fixing it: a
|
||||
# markdown link's href is not HTML-escaped either, and an entity-encoded
|
||||
# `javascript:` sails past any scheme test that does not decode it first.
|
||||
|
||||
from html.parser import HTMLParser
|
||||
|
||||
from booth.items import render_doc
|
||||
|
||||
|
||||
class _Scan(HTMLParser):
|
||||
"""What a BROWSER would see: tags as parsed, attribute values decoded."""
|
||||
|
||||
def __init__(self):
|
||||
super().__init__(convert_charrefs=True)
|
||||
self.tags, self.hrefs = [], []
|
||||
|
||||
def handle_starttag(self, tag, attrs):
|
||||
self.tags.append(tag)
|
||||
for k, v in attrs:
|
||||
if k == "href":
|
||||
self.hrefs.append(v)
|
||||
|
||||
|
||||
def _scan(md_text):
|
||||
html, is_html = render_doc(md_text, "markdown")
|
||||
assert is_html
|
||||
s = _Scan()
|
||||
s.feed(html)
|
||||
return html, s
|
||||
|
||||
|
||||
def _navigates_to_script(href):
|
||||
# the browser drops tab/CR/LF anywhere and C0-or-space at the ends
|
||||
bare = "".join(ch for ch in href if ch not in "\t\r\n").strip("".join(map(chr, range(0x21))))
|
||||
return bare.lower().startswith(("javascript:", "vbscript:", "data:"))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("src", [
|
||||
"<script>alert(1)</script>\n\nafter",
|
||||
"inline <img src=x onerror=alert(1)> here",
|
||||
"<iframe src=//evil.test></iframe>",
|
||||
"| a |\n|---|\n| <svg onload=alert(1)> |",
|
||||
])
|
||||
def test_raw_html_in_a_doc_is_text_never_markup(src):
|
||||
html, s = _scan(src)
|
||||
assert not {"script", "img", "iframe", "svg"} & set(s.tags), (s.tags, html)
|
||||
assert "<" in html
|
||||
|
||||
|
||||
def test_a_quoted_pre_is_shown_not_opened_and_the_doc_goes_on():
|
||||
html, s = _scan("a contract that says <pre> opens one\n\nnext paragraph")
|
||||
assert "pre" not in s.tags, html
|
||||
assert "<p>next paragraph</p>" in html
|
||||
|
||||
|
||||
def test_fenced_code_is_still_a_code_block():
|
||||
"""Positive control: escaping raw HTML must not cost the code block."""
|
||||
html, s = _scan("```html\n<script>x</script>\n```")
|
||||
assert s.tags == ["pre", "code"], html
|
||||
assert "<script>" in html
|
||||
|
||||
|
||||
def test_inline_html_a_doc_meant_is_now_literal_text():
|
||||
"""The declared cost of the ruling: <sub> and <details> show as tags."""
|
||||
html, s = _scan("H<sub>2</sub>O")
|
||||
assert "sub" not in s.tags and "<sub>" in html
|
||||
|
||||
|
||||
@pytest.mark.parametrize("dest", [
|
||||
"[x](javascript:alert(1))",
|
||||
"[x](JaVaScRiPt:alert(1))",
|
||||
"[x](javascript:alert(1))",
|
||||
"[x](javascript:alert(1))",
|
||||
"[x](javascript:alert`1`)",
|
||||
"[x](java	script:alert`1`)",
|
||||
"[x](javascript:alert`1`)",
|
||||
"[x](<java\tscript:alert(1)>)",
|
||||
"[x](\x01javascript:alert`1`)",
|
||||
"[r]: javascript:alert`1`\n\n[go][r]",
|
||||
"[x](data:text/html,<script>alert(1)</script>)",
|
||||
])
|
||||
def test_a_link_that_would_run_code_keeps_its_text_and_loses_its_href(dest):
|
||||
html, s = _scan(dest)
|
||||
assert not any(_navigates_to_script(h) for h in s.hrefs), (s.hrefs, html)
|
||||
assert "a" in s.tags, html # the words are still there
|
||||
|
||||
|
||||
def test_ordinary_links_survive():
|
||||
html, s = _scan("[a](https://example.com) [b](http://x.test/p) [c](other.md) [d](#frag)")
|
||||
assert s.hrefs == ["https://example.com", "http://x.test/p", "other.md", "#frag"], html
|
||||
|
||||
|
||||
# Markdown reads `\\` as an escaped backslash, so FOUR in the source put two
|
||||
# in the href: `\\evil.test`, which a browser reads as `//evil.test`.
|
||||
@pytest.mark.parametrize("dest", ["[x](//evil.test/p)", "[x](/\\evil.test/p)", "[x](\\\\\\\\evil.test/p)"])
|
||||
def test_a_link_that_leaves_the_origin_by_backslash_is_refused(dest):
|
||||
"""heid bug-hunt, groa: `is_safe_href` refused `//host` but not its
|
||||
backslash twin — a browser reads `\\` as `/` in an http(s) URL, so
|
||||
`/\\evil.test` is `//evil.test`."""
|
||||
html, s = _scan(dest)
|
||||
assert s.hrefs == [], (s.hrefs, html)
|
||||
|
||||
|
||||
def test_a_renderer_failure_costs_the_doc_its_formatting_never_the_page(monkeypatch):
|
||||
"""heid bug-hunt, 3 of 4 arms: nothing bounded the render. A doc that
|
||||
makes Python-Markdown raise — deep nesting, or an upgrade that renames the
|
||||
processors this module deregisters — must fall back to escaped raw text,
|
||||
never raise out of the page."""
|
||||
import booth.items as items_mod
|
||||
|
||||
def boom():
|
||||
raise RecursionError("too deep")
|
||||
monkeypatch.setattr(items_mod, "_markdown_renderer", boom)
|
||||
assert items_mod.render_doc("# t\n\n<script>x</script>", "markdown") == \
|
||||
("# t\n\n<script>x</script>", False)
|
||||
|
||||
Reference in New Issue
Block a user