diff --git a/docs/changelog/960.bugfix.rst b/docs/changelog/960.bugfix.rst new file mode 100644 index 000000000..ea8c4dcd8 --- /dev/null +++ b/docs/changelog/960.bugfix.rst @@ -0,0 +1,4 @@ +Attribute URL hosts the way a browser resolves them in ``normalize_url``, ``clean_url``, ``extract_links``, +``resolve_links``, and the sanitizer's ``media_hosts`` allowlist: a special-scheme authority ends at a backslash, an +IPv4 address is read in all four notations and emitted dotted-decimal, the host is percent-decoded, and an IPv6 literal +is zero-compressed. diff --git a/docs/how-to/links.rst b/docs/how-to/links.rst index 05e27da9d..4af0e7233 100644 --- a/docs/how-to/links.rst +++ b/docs/how-to/links.rst @@ -46,6 +46,13 @@ with :func:`turbohtml.extract.normalize_url`. It applies the WHATWG URL standard https://example.org/page?a=1&b=2 +The host is attributed the way a browser resolves it, so a link-safety or SSRF check keyed on the result sees the host a +browser fetches: a special-scheme authority ends at a backslash (``http://a\@b/`` has host ``a``), an IPv4 address is +read in decimal, octal, hexadecimal, and short forms and re-emitted dotted-decimal (``http://127.1/`` becomes +``http://127.0.0.1/``), the host is percent-decoded, and an IPv6 literal is zero-compressed (``[0:0:0:0:0:0:0:1]`` +becomes ``[::1]``). An IPv6 literal with an embedded-IPv4 tail (``[::ffff:1.2.3.4]``) keeps its given spelling. The same +host parsing backs ``extract_links(external_only=True)`` and :meth:`~turbohtml.Node.resolve_links`. + For URLs scraped out of markup, :func:`turbohtml.extract.clean_url` first scrubs HTML damage (stray whitespace, ``&``, a truncating quote) and answers ``None`` for anything that is not a fetchable web URL, so a scraping pipeline can filter and normalize in one call. :class:`turbohtml.extract.UrlCleaning` carries the knobs: a strict query-parameter diff --git a/docs/reference/clean.rst b/docs/reference/clean.rst index 85809e182..a5ebc4e54 100644 --- a/docs/reference/clean.rst +++ b/docs/reference/clean.rst @@ -66,9 +66,10 @@ drops the declaration holding it. ``Policy.attribute_filter`` replacements and ``Policy.set_attributes`` additions pass through the mandatory safety checks before serialization. These checks remove event handlers, disallowed URL, ``srcset`` and meta refresh schemes, -unsafe CSS, media hosts outside ``media_hosts``, and values outside ``attribute_values``. Template stripping and -named-property isolation run on the final values. The sanitizer ASCII-lowercases HTML names created by these rules -before the checks. +unsafe CSS, media hosts outside ``media_hosts`` (matched against the host a browser resolves, so an authority ending at +a backslash cannot smuggle an off-allowlist host past the check), and values outside ``attribute_values``. Template +stripping and named-property isolation run on the final values. The sanitizer ASCII-lowercases HTML names created by +these rules before the checks. ``Policy.transform_tags`` renames elements during the same walk, sanitize-html's ``transformTags``. Key it by source tag: map to a bare string to rename, or to a :class:`Transform` to rename and add attributes. The rename runs *before* diff --git a/src/turbohtml/_c/clean/sanitize.c b/src/turbohtml/_c/clean/sanitize.c index 9d2c1dcaf..a5d972423 100644 --- a/src/turbohtml/_c/clean/sanitize.c +++ b/src/turbohtml/_c/clean/sanitize.c @@ -552,31 +552,41 @@ static int is_media_host_tag(uint16_t atom) { } } -/* The authority marker bytes that end a URL host: a path, query, or fragment. */ +/* The bytes that end a URL authority: a path, query, or fragment delimiter, plus the '\' a browser treats as a + separator for a special scheme (WHATWG authority state, https://url.spec.whatwg.org/#authority-state). A real host + never contains '\', so ending the authority at one only tightens the host the allowlist sees, closing the + `evil.example\@good.example` userinfo trick a browser resolves to evil.example. */ static int ends_authority(Py_UCS4 c) { switch (c) { case '/': case '?': case '#': + case '\\': return 1; default: return 0; } } -/* Locate the authority host of a URL value: the host after "scheme://" or a protocol-relative "//". The authority is - bounded here without preprocessing the value (the WHATWG tab/newline stripping a browser applies is intentionally not - done, so an obfuscated host never masquerades as an allowlisted one), then th_url_authority -- the same decomposition - url_split runs -- splits off any "userinfo@" and ":port" and reports the host span and its literal kind. Sets - *start,*end to the host span and *kind to the host literal, returns 0, or returns -1 when the URL carries no - authority (a relative or opaque src, which has no host to match). */ +/* A URL authority opener slash: '/', or the '\' a browser treats alike for a special scheme. */ +static int authority_slash(Py_UCS4 c) { + return c == '/' || c == '\\'; +} + +/* Locate the authority host of a URL value: the host after "scheme://" or a protocol-relative "//" (with '\' accepted + like '/', as a browser does for a special scheme). The authority is bounded here without preprocessing the value (the + WHATWG tab/newline stripping a browser applies is intentionally not done, so an obfuscated host never masquerades as + an allowlisted one), then th_url_authority -- the same decomposition url_split runs -- splits off any "userinfo@" and + ":port" and reports the host span and its literal kind. Sets *start,*end to the host span and *kind to the host + literal, returns 0, or returns -1 when the URL carries no authority (a relative or opaque src, which has no host to + match). */ static int url_host_span(const Py_UCS4 *value, Py_ssize_t len, Py_ssize_t *start, Py_ssize_t *end, int *kind) { Py_ssize_t authority = -1; - if (len >= 2 && value[0] == '/' && value[1] == '/') { + if (len >= 2 && authority_slash(value[0]) && authority_slash(value[1])) { authority = 2; /* protocol-relative //host/path */ } for (Py_ssize_t index = 0; authority < 0 && index + 2 < len; index++) { - if (value[index] == ':' && value[index + 1] == '/' && value[index + 2] == '/') { + if (value[index] == ':' && authority_slash(value[index + 1]) && authority_slash(value[index + 2])) { authority = index + 3; /* scheme://host */ } } diff --git a/src/turbohtml/_c/core/common.h b/src/turbohtml/_c/core/common.h index afd37fd07..baf4eaeba 100644 --- a/src/turbohtml/_c/core/common.h +++ b/src/turbohtml/_c/core/common.h @@ -197,6 +197,15 @@ PyObject *turbohtml_url_language_matches(PyObject *module, PyObject *args); PyObject *th_url_to_ascii(PyObject *host); PyObject *turbohtml_url_to_ascii(PyObject *module, PyObject *arg); +/* Implemented in url/url.c. th_url_host_canonical runs the rest of the WHATWG host parser over the bracket-stripped + host span url_split reports (https://url.spec.whatwg.org/#concept-host-parser): a bracketed IPv6 literal (kind + TH_HOST_IPV6) is parsed and serialized with zero-run compression; any other host is percent-decoded, run through + domain-to-ASCII, and -- when it ends in a number -- parsed as IPv4 and re-serialized in dotted-decimal, so two + spellings of one address compare equal. The host is a borrowed str, the result a new str (brackets dropped for IPv6); + NULL with an error only on allocation failure. A host that is not valid for its form falls back to its lowercased + spelling, the advisory behavior normalize_url already takes for an unencodable label. */ +PyObject *th_url_host_canonical(PyObject *host, int kind); + /* Implemented in url/registrable.c. _registrable_domain(host) returns a lowercased host's registrable domain (eTLD+1) from the shipped IANA and Public Suffix List tables, the site boundary behind extract_links(external_only=True). METH_O. */ diff --git a/src/turbohtml/_c/extract/links.c b/src/turbohtml/_c/extract/links.c index 07bcfdcc1..10ae61ea0 100644 --- a/src/turbohtml/_c/extract/links.c +++ b/src/turbohtml/_c/extract/links.c @@ -4,8 +4,8 @@ This file is the walk that locates every link-bearing location and, for the rewrite path, splices a replacement back in place. The genuinely new capability over iterating by hand is the URLs embedded in CSS url()/@import (in a style attribute and in