From 3948511b2a0e85c4adec6af5b85ef796eca68bc7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Wed, 7 Oct 2026 13:27:25 -0700 Subject: [PATCH 1/2] =?UTF-8?q?=F0=9F=90=9B=20fix(markdown):=20keep=20bloc?= =?UTF-8?q?k=20content=20in=20headings?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit to_markdown let a block inside a heading break out of it.

x

became an empty "### " followed by a paragraph, and in

a

b

c

the text after the block left the heading too. A Markdown heading is one line, so blocks inside it now flatten onto that line, the way link text already flattens them. The end of a flattened block reads as a space as well as its start, so words on either side of a block stay apart in headings and link text alike. --- docs/changelog/1214.bugfix.rst | 2 ++ src/turbohtml/_c/serialize/markdown.c | 15 +++++++++++---- tests/serialize/test_markdown.py | 12 ++++++++++-- 3 files changed, 23 insertions(+), 6 deletions(-) create mode 100644 docs/changelog/1214.bugfix.rst diff --git a/docs/changelog/1214.bugfix.rst b/docs/changelog/1214.bugfix.rst new file mode 100644 index 000000000..91c7640b9 --- /dev/null +++ b/docs/changelog/1214.bugfix.rst @@ -0,0 +1,2 @@ +``to_markdown`` keeps block content inside a heading on the heading line and reads a block edge in link text or a +heading as a space. diff --git a/src/turbohtml/_c/serialize/markdown.c b/src/turbohtml/_c/serialize/markdown.c index 1233d23ed..b5d10dfee 100644 --- a/src/turbohtml/_c/serialize/markdown.c +++ b/src/turbohtml/_c/serialize/markdown.c @@ -163,6 +163,7 @@ enum md_leave { MD_LEAVE_TABLE, MD_LEAVE_BLOCKQUOTE, /* mark a quote that opened no block of its own */ MD_LEAVE_BLOCK_END, /* the block inside an inline element is done */ + MD_LEAVE_SPACE, /* a flattened block's end reads as a space */ }; enum md_table_phase { @@ -1751,7 +1752,7 @@ static void md_render_inline_tag(md_ctx *ctx, th_node *node) { return; } if (is_md_block(atom)) { - if (!ctx->inline_only) { + if (!ctx->inline_only && !ctx->in_heading) { if (!ctx->in_cell) { /* the content after the block belongs to a new block, so note where it ends; a cell flattens its blocks onto its one line instead */ @@ -1765,10 +1766,13 @@ static void md_render_inline_tag(md_ctx *ctx, th_node *node) { md_enter_cell_flat(ctx, node, -1); return; } - /* inside link text a block cannot open its own line (a blank line would - split the CommonMark link), so it flattens to inline; its boundary still - reads as a space so adjacent words never fuse */ + /* inside link text or a heading a block cannot open its own line (a blank + line would split the CommonMark link, and a heading is one line), so it + flattens to inline; its boundary still reads as a space so adjacent words + never fuse */ ctx->space_pending = 1; + md_push(ctx, node, MD_WALK_INLINE, MD_LEAVE_SPACE); + return; } md_push(ctx, node, MD_WALK_INLINE, MD_LEAVE_NONE); } @@ -3223,6 +3227,9 @@ static void md_leave(md_ctx *ctx) { case MD_LEAVE_BLOCK_END: ctx->block_ended = 1; break; + case MD_LEAVE_SPACE: + ctx->space_pending = 1; + break; case MD_LEAVE_BLOCKQUOTE: /* a leading quote writes its marker only with its first block; CommonMark reads a lone ">" as an empty quote, so keep one when no block came */ diff --git a/tests/serialize/test_markdown.py b/tests/serialize/test_markdown.py index 32767e4d9..5c8e0975b 100644 --- a/tests/serialize/test_markdown.py +++ b/tests/serialize/test_markdown.py @@ -62,15 +62,23 @@ def md(html: str) -> str: pytest.param("

a

", "### a", id="heading-trailing-break-dropped"), pytest.param( "

x
y

z
w

", - "# \n\n## x y\n\nz w", + "# x y z w", id="nested-heading-keeps-break-as-space", ), + pytest.param("

x

", "### x", id="heading-paragraph-stays-in-heading"), + pytest.param("

a

b

c

", "## a b c", id="heading-block-edges-read-as-spaces"), + pytest.param("

a
b

", "## a b", id="heading-sibling-blocks-join"), ], ) def test_headings(html: str, expected: str) -> None: assert md(html) == expected +def test_setext_heading_keeps_block_content() -> None: + config: Final = Markdown(headings=Markdown.Headings(style="setext")) + assert parse_fragment("

a

b

").to_markdown(config) == "a b\n===" + + @pytest.mark.parametrize( ("html", "expected"), [ @@ -272,7 +280,7 @@ def test_code(html: str, expected: str) -> None: id="emphasis-adjacent-in-mtext-html", ), pytest.param( - "
ab
x
", "[*a*bx](h)", id="emphasis-adjacent-in-link-html" + "
ab
x
", "[*a*b x](h)", id="emphasis-adjacent-in-link-html" ), pytest.param("

;i

", ";i", id="emphasis-close-punct-before-letter"), pytest.param("

i;

", "i;", id="emphasis-open-letter-before-punct"), From 3cfb39f1c168b6cfe452ff27a030a6b2fd46ce16 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Wed, 7 Oct 2026 16:58:50 -0700 Subject: [PATCH 2/2] =?UTF-8?q?=F0=9F=90=9B=20fix(fuzz):=20compare=20headi?= =?UTF-8?q?ngs=20as=20one=20line?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Markdown meaning check compared the blocks nested in a heading as blocks. A CommonMark heading holds one line of inline content, so no conversion can keep those blocks, and the flattened heading this branch writes still failed the check. The check now flattens heading content onto one line, reading each block edge and each
as a space, which is the heading a reader rebuilds from the converted Markdown. --- ...test_fuzz_markdown_structure_generation.py | 3 +++ tools/fuzz/markdown_structure_generators.py | 22 +++++++++++++++++++ 2 files changed, 25 insertions(+) diff --git a/tests/test_fuzz_markdown_structure_generation.py b/tests/test_fuzz_markdown_structure_generation.py index 683d05c17..da4643211 100644 --- a/tests/test_fuzz_markdown_structure_generation.py +++ b/tests/test_fuzz_markdown_structure_generation.py @@ -184,6 +184,9 @@ def test_markdown_unsupported_element() -> None: pytest.param("
a

b
", id="empty-block-breaks-code-line"), pytest.param("x
", id="foster-parented-emphasis"), pytest.param("a
b
", id="foster-parented-before-rows"), + pytest.param("

x

", id="heading-paragraph"), + pytest.param("

a

b

c

", id="heading-text-around-block"), + pytest.param("

a
b

", id="heading-sibling-blocks"), ], ) def test_markdown_supported_html_meaning(markup: str) -> None: diff --git a/tools/fuzz/markdown_structure_generators.py b/tools/fuzz/markdown_structure_generators.py index 44b340c96..9ec3ae9ad 100644 --- a/tools/fuzz/markdown_structure_generators.py +++ b/tools/fuzz/markdown_structure_generators.py @@ -227,9 +227,30 @@ def _record(tag: str, element: _TreeNode, children: tuple[_Meaning, ...]) -> tup # quote and item content is a flow of blocks; a tight item's single paragraph renders bare (CommonMark 5.1-5.3), # so items compare with every inline run wrapped return ((tag, (), _flow(children)),) + if tag in _HEADINGS: + # a heading is one line of inline content (CommonMark 4.2), so its blocks and breaks flatten onto it as spaces + return ((tag, (), _line(children)),) return ((tag, _attributes(tag, element.attrib), children),) +def _line(records: tuple[_Meaning, ...]) -> tuple[_Meaning, ...]: + flat: Final[list[_Meaning]] = [] + for record in records: + if record[0] in _FLOW_INLINE and record[0] != "br": + flat.append(record) + elif record[0] == "br" or record == _BOUNDARY: + flat.append(("#text", (("value", " "),), ())) + else: + flat.extend((("#text", (("value", " "),), ()), *_line(record[2]), ("#text", (("value", " "),), ()))) + merged: Final[list[_Meaning]] = [] + for record in flat: + if record[0] == "#text" and merged and merged[-1][0] == "#text": + merged[-1] = ("#text", (("value", merged[-1][1][0][1] + record[1][0][1]),), ()) + else: + merged.append(record) + return _spacing(tuple(merged), preserve=False, inline=False) + + def _emphasis(tag: str, children: tuple[_Meaning, ...]) -> tuple[_Meaning, ...]: if all(record[0] == "#text" and not record[1][0][1].strip(" \t\n\f\r") for record in children): return children # a delimiter run cannot wrap whitespace alone (CommonMark 6.2) @@ -581,6 +602,7 @@ def _html_grammar() -> Grammar: _EMPHASIS: Final = frozenset({"strong", "em", "s"}) _BLOCKS: Final = frozenset({"p", "ul", "ol", "table"}) _BOUNDARY: Final[_Meaning] = ("#block", (), ()) +_HEADINGS: Final = frozenset({"h1", "h2", "h3", "h4", "h5", "h6"}) # the WHATWG content models of the containers whose Markdown syntax can hold nothing else (4.4.5-4.4.8, 4.9) _CONTENT_MODEL: Final[dict[str, frozenset[str]]] = { "ul": frozenset({"li", "script"}),