Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions docs/changelog/1214.bugfix.rst
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
``to_markdown`` keeps block content inside a heading on the heading line and reads a block edge in link text or a
heading as a space.
15 changes: 11 additions & 4 deletions src/turbohtml/_c/serialize/markdown.c
Original file line number Diff line number Diff line change
Expand Up @@ -163,6 +163,7 @@ enum md_leave {
MD_LEAVE_TABLE,
MD_LEAVE_BLOCKQUOTE, /* mark a quote that opened no block of its own */
MD_LEAVE_BLOCK_END, /* the block inside an inline element is done */
MD_LEAVE_SPACE, /* a flattened block's end reads as a space */
};

enum md_table_phase {
Expand Down Expand Up @@ -1751,7 +1752,7 @@ static void md_render_inline_tag(md_ctx *ctx, th_node *node) {
return;
}
if (is_md_block(atom)) {
if (!ctx->inline_only) {
if (!ctx->inline_only && !ctx->in_heading) {
if (!ctx->in_cell) {
/* the content after the block belongs to a new block, so note where it ends;
a cell flattens its blocks onto its one line instead */
Expand All @@ -1765,10 +1766,13 @@ static void md_render_inline_tag(md_ctx *ctx, th_node *node) {
md_enter_cell_flat(ctx, node, -1);
return;
}
/* inside link text a block cannot open its own line (a blank line would
split the CommonMark link), so it flattens to inline; its boundary still
reads as a space so adjacent words never fuse */
/* inside link text or a heading a block cannot open its own line (a blank
line would split the CommonMark link, and a heading is one line), so it
flattens to inline; its boundary still reads as a space so adjacent words
never fuse */
ctx->space_pending = 1;
md_push(ctx, node, MD_WALK_INLINE, MD_LEAVE_SPACE);
return;
}
md_push(ctx, node, MD_WALK_INLINE, MD_LEAVE_NONE);
}
Expand Down Expand Up @@ -3223,6 +3227,9 @@ static void md_leave(md_ctx *ctx) {
case MD_LEAVE_BLOCK_END:
ctx->block_ended = 1;
break;
case MD_LEAVE_SPACE:
ctx->space_pending = 1;
break;
case MD_LEAVE_BLOCKQUOTE:
/* a leading quote writes its marker only with its first block; CommonMark
reads a lone ">" as an empty quote, so keep one when no block came */
Expand Down
12 changes: 10 additions & 2 deletions tests/serialize/test_markdown.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,15 +62,23 @@ def md(html: str) -> str:
pytest.param("<h3>a<br></h3>", "### a", id="heading-trailing-break-dropped"),
pytest.param(
"<h1><div><h2>x<br>y</h2>z<br>w</div></h1>",
"# \n\n## x y\n\nz w",
"# x y z w",
id="nested-heading-keeps-break-as-space",
),
pytest.param("<h3><p>x</p></h3>", "### x", id="heading-paragraph-stays-in-heading"),
pytest.param("<h2>a<p>b</p>c</h2>", "## a b c", id="heading-block-edges-read-as-spaces"),
pytest.param("<h2><div>a</div><div>b</div></h2>", "## a b", id="heading-sibling-blocks-join"),
],
)
def test_headings(html: str, expected: str) -> None:
assert md(html) == expected


def test_setext_heading_keeps_block_content() -> None:
config: Final = Markdown(headings=Markdown.Headings(style="setext"))
assert parse_fragment("<h1>a<p>b</p></h1>").to_markdown(config) == "a b\n==="


@pytest.mark.parametrize(
("html", "expected"),
[
Expand Down Expand Up @@ -272,7 +280,7 @@ def test_code(html: str, expected: str) -> None:
id="emphasis-adjacent-in-mtext-html",
),
pytest.param(
"<a href='h'><div><i>a</i><i>b</i></div>x</a>", "[*a*<em>b</em>x](h)", id="emphasis-adjacent-in-link-html"
"<a href='h'><div><i>a</i><i>b</i></div>x</a>", "[*a*<em>b</em> x](h)", id="emphasis-adjacent-in-link-html"
),
pytest.param("<p><i>;</i>i</p>", "<em>;</em>i", id="emphasis-close-punct-before-letter"),
pytest.param("<p>i<i>;</i></p>", "i<em>;</em>", id="emphasis-open-letter-before-punct"),
Expand Down
3 changes: 3 additions & 0 deletions tests/test_fuzz_markdown_structure_generation.py
Original file line number Diff line number Diff line change
Expand Up @@ -184,6 +184,9 @@ def test_markdown_unsupported_element() -> None:
pytest.param("<pre>a<p></p>b</pre>", id="empty-block-breaks-code-line"),
pytest.param("<table><b>x</b></table>", id="foster-parented-emphasis"),
pytest.param("<table><em>a</em><tr><td>b</td></tr></table>", id="foster-parented-before-rows"),
pytest.param("<h3><p>x</p></h3>", id="heading-paragraph"),
pytest.param("<h2>a<p>b</p>c</h2>", id="heading-text-around-block"),
pytest.param("<h2><div>a</div><div>b</div></h2>", id="heading-sibling-blocks"),
],
)
def test_markdown_supported_html_meaning(markup: str) -> None:
Expand Down
22 changes: 22 additions & 0 deletions tools/fuzz/markdown_structure_generators.py
Original file line number Diff line number Diff line change
Expand Up @@ -227,9 +227,30 @@ def _record(tag: str, element: _TreeNode, children: tuple[_Meaning, ...]) -> tup
# quote and item content is a flow of blocks; a tight item's single paragraph renders bare (CommonMark 5.1-5.3),
# so items compare with every inline run wrapped
return ((tag, (), _flow(children)),)
if tag in _HEADINGS:
# a heading is one line of inline content (CommonMark 4.2), so its blocks and breaks flatten onto it as spaces
return ((tag, (), _line(children)),)
return ((tag, _attributes(tag, element.attrib), children),)


def _line(records: tuple[_Meaning, ...]) -> tuple[_Meaning, ...]:
flat: Final[list[_Meaning]] = []
for record in records:
if record[0] in _FLOW_INLINE and record[0] != "br":
flat.append(record)
elif record[0] == "br" or record == _BOUNDARY:
flat.append(("#text", (("value", " "),), ()))
else:
flat.extend((("#text", (("value", " "),), ()), *_line(record[2]), ("#text", (("value", " "),), ())))
merged: Final[list[_Meaning]] = []
for record in flat:
if record[0] == "#text" and merged and merged[-1][0] == "#text":
merged[-1] = ("#text", (("value", merged[-1][1][0][1] + record[1][0][1]),), ())
else:
merged.append(record)
return _spacing(tuple(merged), preserve=False, inline=False)


def _emphasis(tag: str, children: tuple[_Meaning, ...]) -> tuple[_Meaning, ...]:
if all(record[0] == "#text" and not record[1][0][1].strip(" \t\n\f\r") for record in children):
return children # a delimiter run cannot wrap whitespace alone (CommonMark 6.2)
Expand Down Expand Up @@ -581,6 +602,7 @@ def _html_grammar() -> Grammar:
_EMPHASIS: Final = frozenset({"strong", "em", "s"})
_BLOCKS: Final = frozenset({"p", "ul", "ol", "table"})
_BOUNDARY: Final[_Meaning] = ("#block", (), ())
_HEADINGS: Final = frozenset({"h1", "h2", "h3", "h4", "h5", "h6"})
# the WHATWG content models of the containers whose Markdown syntax can hold nothing else (4.4.5-4.4.8, 4.9)
_CONTENT_MODEL: Final[dict[str, frozenset[str]]] = {
"ul": frozenset({"li", "script"}),
Expand Down
Loading