From ccb02da0635ab509e81ea4d2c219e17be6ddb050 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Thu, 1 Oct 2026 09:37:40 -0700 Subject: [PATCH] =?UTF-8?q?=F0=9F=94=92=20fix(url):=20attribute=20hosts=20?= =?UTF-8?q?the=20WHATWG=20way?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Resolve the URL/host layer to the WHATWG URL host parser so the host turbohtml attributes matches the one a browser connects to. - End a special-scheme authority at a backslash in url_split, parse_ref, the relative join, and the sanitizer media-host scan, so evil.example\@good.example is read as evil.example. - Parse an IPv4 host in all four notations and emit dotted-decimal, and percent-decode the host before domain-to-ASCII. - Canonicalize an IPv6 literal with zero-run compression. - resolve_links joins through the C _url_join instead of stdlib urljoin. This fixes a media_hosts allowlist bypass and host-confusion that defeated external_only scoping and blocklist/SSRF checks keyed on the extraction helpers' output. --- docs/changelog/960.bugfix.rst | 4 + docs/how-to/links.rst | 7 + docs/reference/clean.rst | 7 +- src/turbohtml/_c/clean/sanitize.c | 28 ++- src/turbohtml/_c/core/common.h | 9 + src/turbohtml/_c/extract/links.c | 19 +- src/turbohtml/_c/url/clean.c | 48 +--- src/turbohtml/_c/url/url.c | 356 +++++++++++++++++++++++++++++- src/turbohtml/_c/url/url.h | 5 + tests/clean/test_sanitize.py | 17 ++ tests/extract/test_links.py | 27 +++ tests/url/test_clean.py | 100 +++++++++ tests/url/test_url.py | 39 ++++ 13 files changed, 599 insertions(+), 67 deletions(-) create mode 100644 docs/changelog/960.bugfix.rst diff --git a/docs/changelog/960.bugfix.rst b/docs/changelog/960.bugfix.rst new file mode 100644 index 000000000..ea8c4dcd8 --- /dev/null +++ b/docs/changelog/960.bugfix.rst @@ -0,0 +1,4 @@ +Attribute URL hosts the way a browser resolves them in ``normalize_url``, ``clean_url``, ``extract_links``, +``resolve_links``, and the sanitizer's ``media_hosts`` allowlist: a special-scheme authority ends at a backslash, an +IPv4 address is read in all four notations and emitted dotted-decimal, the host is percent-decoded, and an IPv6 literal +is zero-compressed. diff --git a/docs/how-to/links.rst b/docs/how-to/links.rst index 05e27da9d..4af0e7233 100644 --- a/docs/how-to/links.rst +++ b/docs/how-to/links.rst @@ -46,6 +46,13 @@ with :func:`turbohtml.extract.normalize_url`. It applies the WHATWG URL standard https://example.org/page?a=1&b=2 +The host is attributed the way a browser resolves it, so a link-safety or SSRF check keyed on the result sees the host a +browser fetches: a special-scheme authority ends at a backslash (``http://a\@b/`` has host ``a``), an IPv4 address is +read in decimal, octal, hexadecimal, and short forms and re-emitted dotted-decimal (``http://127.1/`` becomes +``http://127.0.0.1/``), the host is percent-decoded, and an IPv6 literal is zero-compressed (``[0:0:0:0:0:0:0:1]`` +becomes ``[::1]``). An IPv6 literal with an embedded-IPv4 tail (``[::ffff:1.2.3.4]``) keeps its given spelling. The same +host parsing backs ``extract_links(external_only=True)`` and :meth:`~turbohtml.Node.resolve_links`. + For URLs scraped out of markup, :func:`turbohtml.extract.clean_url` first scrubs HTML damage (stray whitespace, ``&``, a truncating quote) and answers ``None`` for anything that is not a fetchable web URL, so a scraping pipeline can filter and normalize in one call. :class:`turbohtml.extract.UrlCleaning` carries the knobs: a strict query-parameter diff --git a/docs/reference/clean.rst b/docs/reference/clean.rst index 85809e182..a5ebc4e54 100644 --- a/docs/reference/clean.rst +++ b/docs/reference/clean.rst @@ -66,9 +66,10 @@ drops the declaration holding it. ``Policy.attribute_filter`` replacements and ``Policy.set_attributes`` additions pass through the mandatory safety checks before serialization. These checks remove event handlers, disallowed URL, ``srcset`` and meta refresh schemes, -unsafe CSS, media hosts outside ``media_hosts``, and values outside ``attribute_values``. Template stripping and -named-property isolation run on the final values. The sanitizer ASCII-lowercases HTML names created by these rules -before the checks. +unsafe CSS, media hosts outside ``media_hosts`` (matched against the host a browser resolves, so an authority ending at +a backslash cannot smuggle an off-allowlist host past the check), and values outside ``attribute_values``. Template +stripping and named-property isolation run on the final values. The sanitizer ASCII-lowercases HTML names created by +these rules before the checks. ``Policy.transform_tags`` renames elements during the same walk, sanitize-html's ``transformTags``. Key it by source tag: map to a bare string to rename, or to a :class:`Transform` to rename and add attributes. The rename runs *before* diff --git a/src/turbohtml/_c/clean/sanitize.c b/src/turbohtml/_c/clean/sanitize.c index 9d2c1dcaf..a5d972423 100644 --- a/src/turbohtml/_c/clean/sanitize.c +++ b/src/turbohtml/_c/clean/sanitize.c @@ -552,31 +552,41 @@ static int is_media_host_tag(uint16_t atom) { } } -/* The authority marker bytes that end a URL host: a path, query, or fragment. */ +/* The bytes that end a URL authority: a path, query, or fragment delimiter, plus the '\' a browser treats as a + separator for a special scheme (WHATWG authority state, https://url.spec.whatwg.org/#authority-state). A real host + never contains '\', so ending the authority at one only tightens the host the allowlist sees, closing the + `evil.example\@good.example` userinfo trick a browser resolves to evil.example. */ static int ends_authority(Py_UCS4 c) { switch (c) { case '/': case '?': case '#': + case '\\': return 1; default: return 0; } } -/* Locate the authority host of a URL value: the host after "scheme://" or a protocol-relative "//". The authority is - bounded here without preprocessing the value (the WHATWG tab/newline stripping a browser applies is intentionally not - done, so an obfuscated host never masquerades as an allowlisted one), then th_url_authority -- the same decomposition - url_split runs -- splits off any "userinfo@" and ":port" and reports the host span and its literal kind. Sets - *start,*end to the host span and *kind to the host literal, returns 0, or returns -1 when the URL carries no - authority (a relative or opaque src, which has no host to match). */ +/* A URL authority opener slash: '/', or the '\' a browser treats alike for a special scheme. */ +static int authority_slash(Py_UCS4 c) { + return c == '/' || c == '\\'; +} + +/* Locate the authority host of a URL value: the host after "scheme://" or a protocol-relative "//" (with '\' accepted + like '/', as a browser does for a special scheme). The authority is bounded here without preprocessing the value (the + WHATWG tab/newline stripping a browser applies is intentionally not done, so an obfuscated host never masquerades as + an allowlisted one), then th_url_authority -- the same decomposition url_split runs -- splits off any "userinfo@" and + ":port" and reports the host span and its literal kind. Sets *start,*end to the host span and *kind to the host + literal, returns 0, or returns -1 when the URL carries no authority (a relative or opaque src, which has no host to + match). */ static int url_host_span(const Py_UCS4 *value, Py_ssize_t len, Py_ssize_t *start, Py_ssize_t *end, int *kind) { Py_ssize_t authority = -1; - if (len >= 2 && value[0] == '/' && value[1] == '/') { + if (len >= 2 && authority_slash(value[0]) && authority_slash(value[1])) { authority = 2; /* protocol-relative //host/path */ } for (Py_ssize_t index = 0; authority < 0 && index + 2 < len; index++) { - if (value[index] == ':' && value[index + 1] == '/' && value[index + 2] == '/') { + if (value[index] == ':' && authority_slash(value[index + 1]) && authority_slash(value[index + 2])) { authority = index + 3; /* scheme://host */ } } diff --git a/src/turbohtml/_c/core/common.h b/src/turbohtml/_c/core/common.h index afd37fd07..baf4eaeba 100644 --- a/src/turbohtml/_c/core/common.h +++ b/src/turbohtml/_c/core/common.h @@ -197,6 +197,15 @@ PyObject *turbohtml_url_language_matches(PyObject *module, PyObject *args); PyObject *th_url_to_ascii(PyObject *host); PyObject *turbohtml_url_to_ascii(PyObject *module, PyObject *arg); +/* Implemented in url/url.c. th_url_host_canonical runs the rest of the WHATWG host parser over the bracket-stripped + host span url_split reports (https://url.spec.whatwg.org/#concept-host-parser): a bracketed IPv6 literal (kind + TH_HOST_IPV6) is parsed and serialized with zero-run compression; any other host is percent-decoded, run through + domain-to-ASCII, and -- when it ends in a number -- parsed as IPv4 and re-serialized in dotted-decimal, so two + spellings of one address compare equal. The host is a borrowed str, the result a new str (brackets dropped for IPv6); + NULL with an error only on allocation failure. A host that is not valid for its form falls back to its lowercased + spelling, the advisory behavior normalize_url already takes for an unencodable label. */ +PyObject *th_url_host_canonical(PyObject *host, int kind); + /* Implemented in url/registrable.c. _registrable_domain(host) returns a lowercased host's registrable domain (eTLD+1) from the shipped IANA and Public Suffix List tables, the site boundary behind extract_links(external_only=True). METH_O. */ diff --git a/src/turbohtml/_c/extract/links.c b/src/turbohtml/_c/extract/links.c index 07bcfdcc1..10ae61ea0 100644 --- a/src/turbohtml/_c/extract/links.c +++ b/src/turbohtml/_c/extract/links.c @@ -4,8 +4,8 @@ This file is the walk that locates every link-bearing location and, for the rewrite path, splices a replacement back in place. The genuinely new capability over iterating by hand is the URLs embedded in CSS url()/@import (in a style attribute and in