diff --git a/docs/changelog/1214.bugfix.rst b/docs/changelog/1214.bugfix.rst new file mode 100644 index 000000000..91c7640b9 --- /dev/null +++ b/docs/changelog/1214.bugfix.rst @@ -0,0 +1,2 @@ +``to_markdown`` keeps block content inside a heading on the heading line and reads a block edge in link text or a +heading as a space. diff --git a/src/turbohtml/_c/serialize/markdown.c b/src/turbohtml/_c/serialize/markdown.c index 1233d23ed..b5d10dfee 100644 --- a/src/turbohtml/_c/serialize/markdown.c +++ b/src/turbohtml/_c/serialize/markdown.c @@ -163,6 +163,7 @@ enum md_leave { MD_LEAVE_TABLE, MD_LEAVE_BLOCKQUOTE, /* mark a quote that opened no block of its own */ MD_LEAVE_BLOCK_END, /* the block inside an inline element is done */ + MD_LEAVE_SPACE, /* a flattened block's end reads as a space */ }; enum md_table_phase { @@ -1751,7 +1752,7 @@ static void md_render_inline_tag(md_ctx *ctx, th_node *node) { return; } if (is_md_block(atom)) { - if (!ctx->inline_only) { + if (!ctx->inline_only && !ctx->in_heading) { if (!ctx->in_cell) { /* the content after the block belongs to a new block, so note where it ends; a cell flattens its blocks onto its one line instead */ @@ -1765,10 +1766,13 @@ static void md_render_inline_tag(md_ctx *ctx, th_node *node) { md_enter_cell_flat(ctx, node, -1); return; } - /* inside link text a block cannot open its own line (a blank line would - split the CommonMark link), so it flattens to inline; its boundary still - reads as a space so adjacent words never fuse */ + /* inside link text or a heading a block cannot open its own line (a blank + line would split the CommonMark link, and a heading is one line), so it + flattens to inline; its boundary still reads as a space so adjacent words + never fuse */ ctx->space_pending = 1; + md_push(ctx, node, MD_WALK_INLINE, MD_LEAVE_SPACE); + return; } md_push(ctx, node, MD_WALK_INLINE, MD_LEAVE_NONE); } @@ -3223,6 +3227,9 @@ static void md_leave(md_ctx *ctx) { case MD_LEAVE_BLOCK_END: ctx->block_ended = 1; break; + case MD_LEAVE_SPACE: + ctx->space_pending = 1; + break; case MD_LEAVE_BLOCKQUOTE: /* a leading quote writes its marker only with its first block; CommonMark reads a lone ">" as an empty quote, so keep one when no block came */ diff --git a/tests/serialize/test_markdown.py b/tests/serialize/test_markdown.py index 32767e4d9..5c8e0975b 100644 --- a/tests/serialize/test_markdown.py +++ b/tests/serialize/test_markdown.py @@ -62,15 +62,23 @@ def md(html: str) -> str: pytest.param("

a

", "### a", id="heading-trailing-break-dropped"), pytest.param( "

x
y

z
w

", - "# \n\n## x y\n\nz w", + "# x y z w", id="nested-heading-keeps-break-as-space", ), + pytest.param("

x

", "### x", id="heading-paragraph-stays-in-heading"), + pytest.param("

a

b

c

", "## a b c", id="heading-block-edges-read-as-spaces"), + pytest.param("

a
b

", "## a b", id="heading-sibling-blocks-join"), ], ) def test_headings(html: str, expected: str) -> None: assert md(html) == expected +def test_setext_heading_keeps_block_content() -> None: + config: Final = Markdown(headings=Markdown.Headings(style="setext")) + assert parse_fragment("

a

b

").to_markdown(config) == "a b\n===" + + @pytest.mark.parametrize( ("html", "expected"), [ @@ -272,7 +280,7 @@ def test_code(html: str, expected: str) -> None: id="emphasis-adjacent-in-mtext-html", ), pytest.param( - "
ab
x
", "[*a*bx](h)", id="emphasis-adjacent-in-link-html" + "
ab
x
", "[*a*b x](h)", id="emphasis-adjacent-in-link-html" ), pytest.param("

;i

", ";i", id="emphasis-close-punct-before-letter"), pytest.param("

i;

", "i;", id="emphasis-open-letter-before-punct"), diff --git a/tests/test_fuzz_markdown_structure_generation.py b/tests/test_fuzz_markdown_structure_generation.py index 683d05c17..da4643211 100644 --- a/tests/test_fuzz_markdown_structure_generation.py +++ b/tests/test_fuzz_markdown_structure_generation.py @@ -184,6 +184,9 @@ def test_markdown_unsupported_element() -> None: pytest.param("
a

b
", id="empty-block-breaks-code-line"), pytest.param("x
", id="foster-parented-emphasis"), pytest.param("a
b
", id="foster-parented-before-rows"), + pytest.param("

x

", id="heading-paragraph"), + pytest.param("

a

b

c

", id="heading-text-around-block"), + pytest.param("

a
b

", id="heading-sibling-blocks"), ], ) def test_markdown_supported_html_meaning(markup: str) -> None: diff --git a/tools/fuzz/markdown_structure_generators.py b/tools/fuzz/markdown_structure_generators.py index 44b340c96..9ec3ae9ad 100644 --- a/tools/fuzz/markdown_structure_generators.py +++ b/tools/fuzz/markdown_structure_generators.py @@ -227,9 +227,30 @@ def _record(tag: str, element: _TreeNode, children: tuple[_Meaning, ...]) -> tup # quote and item content is a flow of blocks; a tight item's single paragraph renders bare (CommonMark 5.1-5.3), # so items compare with every inline run wrapped return ((tag, (), _flow(children)),) + if tag in _HEADINGS: + # a heading is one line of inline content (CommonMark 4.2), so its blocks and breaks flatten onto it as spaces + return ((tag, (), _line(children)),) return ((tag, _attributes(tag, element.attrib), children),) +def _line(records: tuple[_Meaning, ...]) -> tuple[_Meaning, ...]: + flat: Final[list[_Meaning]] = [] + for record in records: + if record[0] in _FLOW_INLINE and record[0] != "br": + flat.append(record) + elif record[0] == "br" or record == _BOUNDARY: + flat.append(("#text", (("value", " "),), ())) + else: + flat.extend((("#text", (("value", " "),), ()), *_line(record[2]), ("#text", (("value", " "),), ()))) + merged: Final[list[_Meaning]] = [] + for record in flat: + if record[0] == "#text" and merged and merged[-1][0] == "#text": + merged[-1] = ("#text", (("value", merged[-1][1][0][1] + record[1][0][1]),), ()) + else: + merged.append(record) + return _spacing(tuple(merged), preserve=False, inline=False) + + def _emphasis(tag: str, children: tuple[_Meaning, ...]) -> tuple[_Meaning, ...]: if all(record[0] == "#text" and not record[1][0][1].strip(" \t\n\f\r") for record in children): return children # a delimiter run cannot wrap whitespace alone (CommonMark 6.2) @@ -581,6 +602,7 @@ def _html_grammar() -> Grammar: _EMPHASIS: Final = frozenset({"strong", "em", "s"}) _BLOCKS: Final = frozenset({"p", "ul", "ol", "table"}) _BOUNDARY: Final[_Meaning] = ("#block", (), ()) +_HEADINGS: Final = frozenset({"h1", "h2", "h3", "h4", "h5", "h6"}) # the WHATWG content models of the containers whose Markdown syntax can hold nothing else (4.4.5-4.4.8, 4.9) _CONTENT_MODEL: Final[dict[str, frozenset[str]]] = { "ul": frozenset({"li", "script"}),