From 32e53ca213a4152707951a7ed19647c1f185a659 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Thu, 8 Oct 2026 01:48:44 -0700 Subject: [PATCH] =?UTF-8?q?=F0=9F=90=9B=20fix(markdown):=20drop=20tabs=20a?= =?UTF-8?q?nd=20newlines=20from=20URLs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A destination cannot hold a line ending (CommonMark 6.3), so an href with a newline came out as literal text. The URL parser removes every tab and newline anyway (URL Standard 4.4), so the destination writer leaves them out and the URL stays the same. --- docs/changelog/1237.bugfix.rst | 1 + src/turbohtml/_c/serialize/markdown.c | 46 ++++++++++++++++++++++++--- tests/serialize/test_markdown.py | 5 +++ 3 files changed, 48 insertions(+), 4 deletions(-) create mode 100644 docs/changelog/1237.bugfix.rst diff --git a/docs/changelog/1237.bugfix.rst b/docs/changelog/1237.bugfix.rst new file mode 100644 index 000000000..aaa916380 --- /dev/null +++ b/docs/changelog/1237.bugfix.rst @@ -0,0 +1 @@ +``to_markdown`` drops tabs and line breaks from link and image destinations. diff --git a/src/turbohtml/_c/serialize/markdown.c b/src/turbohtml/_c/serialize/markdown.c index b5d10dfee..c5fc02cab 100644 --- a/src/turbohtml/_c/serialize/markdown.c +++ b/src/turbohtml/_c/serialize/markdown.c @@ -1223,12 +1223,24 @@ static void md_emit_code_span(md_ctx *ctx, th_node *node) { PyMem_Free(content.data); } +static void md_emit_url_without_breaks(md_ctx *ctx, const char *base, const Py_UCS4 *url, Py_ssize_t len); + +/* What a destination character means to md_emit_url, so one table load per character + answers every test the layout pass makes. */ +enum { MD_URL_SPACE = 1, MD_URL_LT = 2, MD_URL_OPEN = 4, MD_URL_CLOSE = 8, MD_URL_BREAK = 16 }; +static const uint8_t MD_URL[128] = { + ['\t'] = MD_URL_BREAK, ['\n'] = MD_URL_BREAK, ['\r'] = MD_URL_BREAK, [' '] = MD_URL_SPACE, + ['<'] = MD_URL_LT, ['('] = MD_URL_OPEN, [')'] = MD_URL_CLOSE, +}; + /* Write a link destination (CommonMark 6.3) after an optional base prefix. A bare destination cannot hold a space or start with `<`, and takes parentheses only in balanced pairs; anything else goes in the `<...>` form, which takes any parenthesis but no unescaped angle bracket. Parentheses stay bare while they balance, so a `wiki/Foo_(bar)` URL reads as written. A backslash, and a `&` that - would decode as a character reference, are escaped in either form. */ + would decode as a character reference, are escaped in either form. A tab or line + break is left out: no destination form holds a line ending (CommonMark 6.3), and the + URL parser removes every tab and newline anyway (URL Standard 4.4). */ static void md_emit_url(md_ctx *ctx, const char *base, const Py_UCS4 *url, Py_ssize_t len) { /* an empty destination is spelled "<>": bare, a following title or a reference definition's line end would be read in its place (CommonMark 4.7, 6.3) */ @@ -1236,11 +1248,19 @@ static void md_emit_url(md_ctx *ctx, const char *base, const Py_UCS4 *url, Py_ss int depth = 0; int unbalanced = 0; for (Py_ssize_t index = 0; index < len; index++) { - if (url[index] == ' ' || (index == 0 && *base == '\0' && url[index] == '<')) { + uint8_t kind = url[index] < 128 ? MD_URL[url[index]] : 0; + if (kind == 0) { + continue; + } + if (kind & MD_URL_BREAK) { + md_emit_url_without_breaks(ctx, base, url, len); + return; + } + if ((kind & MD_URL_SPACE) || ((kind & MD_URL_LT) && index == 0 && *base == '\0')) { angle = 1; - } else if (url[index] == '(') { + } else if (kind & MD_URL_OPEN) { depth++; - } else if (url[index] == ')') { + } else if (kind & MD_URL_CLOSE) { unbalanced |= depth == 0; depth -= depth > 0; } @@ -1263,6 +1283,24 @@ static void md_emit_url(md_ctx *ctx, const char *base, const Py_UCS4 *url, Py_ss } } +/* Write a destination that holds a tab or line break without them. Only such a rare + destination pays for the copy, made so that a reference split by a break is seen. */ +static void md_emit_url_without_breaks(md_ctx *ctx, const char *base, const Py_UCS4 *url, Py_ssize_t len) { + Py_UCS4 *kept = PyMem_Malloc((size_t)len * sizeof(Py_UCS4)); + if (kept == NULL) { /* GCOVR_EXCL_BR_LINE: allocation failure cannot be forced from a test */ + ctx->out.failed = 1; /* GCOVR_EXCL_LINE: allocation-failure path */ + return; /* GCOVR_EXCL_LINE: allocation-failure path */ + } + Py_ssize_t count = 0; + for (Py_ssize_t index = 0; index < len; index++) { + if (url[index] >= 128 || !(MD_URL[url[index]] & MD_URL_BREAK)) { + kept[count++] = url[index]; + } + } + md_emit_url(ctx, base, kept, count); + PyMem_Free(kept); +} + /* Write a link/image title inside its `"..."` delimiters: a `"` would close the title early and a `\` would escape the next character, so both are backslashed. */ static void md_emit_title(md_ctx *ctx, const Py_UCS4 *title, Py_ssize_t len) { diff --git a/tests/serialize/test_markdown.py b/tests/serialize/test_markdown.py index 5c8e0975b..afcf62a50 100644 --- a/tests/serialize/test_markdown.py +++ b/tests/serialize/test_markdown.py @@ -930,6 +930,11 @@ def test_escaping_none(html: str, options: Markdown, expected: str) -> None: 'http://x/\x7f', "[http://x/\x7f](http://x/\x7f)", id="no-autolink-del" ), pytest.param('http://x/y', "", id="autolink"), + pytest.param('x', "[x](ab)", id="newline-dropped"), + pytest.param('x', "[x](ab)", id="tab-dropped"), + pytest.param('', "![](ie)", id="carriage-return-dropped"), + pytest.param('t', "[t](éx)", id="newline-dropped-after-non-ascii"), + pytest.param('x', "[x](<\\<>)", id="leading-angle-after-dropped-newline"), ], ) def test_link_destination(html: str, expected: str) -> None: