From 190a75a0e103a0724f08be96e74c92a97c8b28b5 Mon Sep 17 00:00:00 2001 From: Vuong Hoang Date: Mon, 28 Sep 2026 10:37:36 -0700 Subject: [PATCH] fix(docs): a posted doc cannot run script on the Booth's origin Found by design-dev's impeccable run and confirmed at source. Python-Markdown passes raw HTML through, and doc.html and booth.html render the result |safe. A \n\nafter", + "inline here", + "", + "| a |\n|---|\n| |", +]) +def test_raw_html_in_a_doc_is_text_never_markup(src): + html, s = _scan(src) + assert not {"script", "img", "iframe", "svg"} & set(s.tags), (s.tags, html) + assert "<" in html + + +def test_a_quoted_pre_is_shown_not_opened_and_the_doc_goes_on(): + html, s = _scan("a contract that says
 opens one\n\nnext paragraph")
+    assert "pre" not in s.tags, html
+    assert "

next paragraph

" in html + + +def test_fenced_code_is_still_a_code_block(): + """Positive control: escaping raw HTML must not cost the code block.""" + html, s = _scan("```html\n\n```") + assert s.tags == ["pre", "code"], html + assert "<script>" in html + + +def test_inline_html_a_doc_meant_is_now_literal_text(): + """The declared cost of the ruling: and
show as tags.""" + html, s = _scan("H2O") + assert "sub" not in s.tags and "<sub>" in html + + +@pytest.mark.parametrize("dest", [ + "[x](javascript:alert(1))", + "[x](JaVaScRiPt:alert(1))", + "[x](javascript:alert(1))", + "[x](javascript:alert(1))", + "[x](javascript:alert`1`)", + "[x](java script:alert`1`)", + "[x](javascript:alert`1`)", + "[x]()", + "[x](\x01javascript:alert`1`)", + "[r]: javascript:alert`1`\n\n[go][r]", + "[x](data:text/html,)", +]) +def test_a_link_that_would_run_code_keeps_its_text_and_loses_its_href(dest): + html, s = _scan(dest) + assert not any(_navigates_to_script(h) for h in s.hrefs), (s.hrefs, html) + assert "a" in s.tags, html # the words are still there + + +def test_ordinary_links_survive(): + html, s = _scan("[a](https://example.com) [b](http://x.test/p) [c](other.md) [d](#frag)") + assert s.hrefs == ["https://example.com", "http://x.test/p", "other.md", "#frag"], html + + +# Markdown reads `\\` as an escaped backslash, so FOUR in the source put two +# in the href: `\\evil.test`, which a browser reads as `//evil.test`. +@pytest.mark.parametrize("dest", ["[x](//evil.test/p)", "[x](/\\evil.test/p)", "[x](\\\\\\\\evil.test/p)"]) +def test_a_link_that_leaves_the_origin_by_backslash_is_refused(dest): + """heid bug-hunt, groa: `is_safe_href` refused `//host` but not its + backslash twin — a browser reads `\\` as `/` in an http(s) URL, so + `/\\evil.test` is `//evil.test`.""" + html, s = _scan(dest) + assert s.hrefs == [], (s.hrefs, html) + + +def test_a_renderer_failure_costs_the_doc_its_formatting_never_the_page(monkeypatch): + """heid bug-hunt, 3 of 4 arms: nothing bounded the render. A doc that + makes Python-Markdown raise — deep nesting, or an upgrade that renames the + processors this module deregisters — must fall back to escaped raw text, + never raise out of the page.""" + import booth.items as items_mod + + def boom(): + raise RecursionError("too deep") + monkeypatch.setattr(items_mod, "_markdown_renderer", boom) + assert items_mod.render_doc("# t\n\n", "markdown") == \ + ("# t\n\n", False)