Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs/changelog/1193.bugfix.rst
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Reject forbidden characters after URL host decoding and IDNA mapping.
1 change: 1 addition & 0 deletions docs/changelog/1193.feature.rst
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Check URL split/recomposition and reparse invariants through public normalization.
8 changes: 4 additions & 4 deletions src/turbohtml/_c/url/clean.c
Original file line number Diff line number Diff line change
Expand Up @@ -174,8 +174,8 @@ static PyObject *port_suffix(const th_url_parts *parts) {
/* The authority rebuilt from its WHATWG-canonical host and port, keeping userinfo verbatim. */
static PyObject *normalize_netloc(const th_url_parts *parts, int web_only) {
PyObject *canonical = th_url_host_canonical(parts->part[TH_URL_HOST], parts->kind);
if (canonical == NULL) { /* GCOVR_EXCL_BR_LINE: the host parse only fails on allocation failure */
return NULL; /* GCOVR_EXCL_LINE: allocation-failure path */
if (canonical == NULL) {
return NULL;
}
PyObject *host = parts->kind == TH_HOST_IPV6 ? th_str_format("[%U]", canonical) : Py_NewRef(canonical);
Py_DECREF(canonical);
Expand Down Expand Up @@ -485,8 +485,8 @@ static PyObject *site_of(PyObject *url) {
}
PyObject *host = th_url_host_canonical(parts.part[TH_URL_HOST], parts.kind);
th_url_parts_clear(&parts);
if (host == NULL) { /* GCOVR_EXCL_BR_LINE: the host parse only fails on allocation failure */
return NULL; /* GCOVR_EXCL_LINE: allocation-failure path */
if (host == NULL) {
return NULL;
}
PyObject *site = turbohtml_registrable_domain(NULL, host);
Py_DECREF(host);
Expand Down
26 changes: 22 additions & 4 deletions src/turbohtml/_c/url/url.c
Original file line number Diff line number Diff line change
Expand Up @@ -309,9 +309,9 @@ PyObject *th_url_host_canonical(PyObject *host, int kind) {
Py_DECREF(ascii);
return ipv4;
}
if (PyErr_Occurred()) { /* GCOVR_EXCL_BR_LINE: maybe_ipv4 only errors on the excluded allocation path */
Py_DECREF(ascii); /* GCOVR_EXCL_LINE: allocation-failure path */
return NULL; /* GCOVR_EXCL_LINE */
if (PyErr_Occurred()) {
Py_DECREF(ascii);
return NULL;
}
return ascii;
}
Expand Down Expand Up @@ -459,14 +459,32 @@ static PyObject *domain_to_ascii(PyObject *host) {
}

/* The dotted-decimal form of `ascii` read by the WHATWG IPv4 parser (https://url.spec.whatwg.org/#concept-ipv4-parser),
or NULL with no error set when it is not an address, so the caller keeps the domain. */
or NULL with no error set when it is not an address, so the caller keeps the domain. Reject forbidden domain
characters before a serialization can reinterpret them as component delimiters. */
static PyObject *maybe_ipv4(PyObject *ascii) {
static const unsigned char FORBIDDEN_DOMAIN[128] = {
1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1,
1, 0, 0, 1, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 1, 0, 1, 1,
1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 0,
0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 1,
};
Py_ssize_t len = PyUnicode_GET_LENGTH(ascii);
if (len == 0) {
return NULL;
}
int kind = PyUnicode_KIND(ascii);
const void *data = PyUnicode_DATA(ascii);
for (Py_ssize_t index = 0; index < len; index++) {
Py_UCS4 codepoint = PyUnicode_READ(kind, data, index);
if (codepoint < 0x80 && FORBIDDEN_DOMAIN[codepoint]) {
PyErr_SetString(PyExc_ValueError, "host contains a forbidden domain code point");
return NULL;
}
}
Py_UCS4 first = PyUnicode_READ(kind, data, 0);
if (first < '0' || first > '9') {
return NULL;
}
Py_UCS4 *cp = PyMem_Malloc((size_t)len * sizeof(Py_UCS4));
if (cp == NULL) { /* GCOVR_EXCL_BR_LINE: allocation failure cannot be forced from a test */
return PyErr_NoMemory(); /* GCOVR_EXCL_LINE: allocation-failure path */
Expand Down
7 changes: 4 additions & 3 deletions src/turbohtml/extract/_urls.py
Original file line number Diff line number Diff line change
Expand Up @@ -76,7 +76,7 @@ class UrlCleaning:
fragment can address content (``#page2``, text fragments).
:param language: an ISO 639-1 code; :func:`clean_url` and :func:`extract_links` then reject URLs whose language
markers (a leading path segment such as ``/de/``, a ``lang``/``language`` query parameter, or an anchor's
``hreflang``) point at another language. :func:`normalize_url` never rejects, so it ignores this field.
``hreflang``) point at another language. :func:`normalize_url` ignores this field.
:param query_allow: when set, keep only these query parameters (matched case-insensitively against the decoded
name), the ``w3lib.url.url_query_cleaner`` keep-list; a listed parameter survives even when it is a known
tracker. Mutually exclusive with ``strict``, which is itself an allowlist.
Expand Down Expand Up @@ -141,7 +141,7 @@ def normalize_url(url: str, options: UrlCleaning | None = None, /) -> str:
and a fragment shaped like a query string is scrubbed the same way. Unlike ``courlan``, repeated slashes are kept
(the spec preserves them) and punycode is the output form, not the input form.

A Unicode host longer than 16384 code points, or holding a code point UTS #46 disallows in a domain (a C0/C1
A Unicode host longer than 16384 code points, or holding a code point UTS #46 disallows in a domain (a C1
control, a noncharacter), keeps its lowercased Unicode form instead of punycode. The cap bounds the domain-to-ASCII
step, whose combining-mark reorder and punycode encoder are quadratic in the host length. The WHATWG standard
leaves overlong labels undefined (`whatwg/url#824 <https://github.com/whatwg/url/issues/824>`_), and DNS caps a
Expand All @@ -152,7 +152,8 @@ def normalize_url(url: str, options: UrlCleaning | None = None, /) -> str:
:returns: the normalized URL.
:raises TypeError: if ``url`` is not a ``str``.
:raises ValueError: if the URL cannot be split into components (e.g. an unclosed IPv6 bracket) or carries a
character that cannot be percent-encoded (a lone surrogate).
forbidden domain character after host decoding or IDNA mapping, or a character that cannot be percent-encoded
(a lone surrogate).
"""
return _url_normalize(url, *_knobs(options or _DEFAULT))

Expand Down
111 changes: 111 additions & 0 deletions tests/test_fuzz_url_reparse.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
from __future__ import annotations

from typing import TYPE_CHECKING

import pytest
from fuzz.round_trip_oracles import ORACLES, OutOfScopeError, main, url_reparse_check

from turbohtml.extract import normalize_url

if TYPE_CHECKING:
from pathlib import Path

from pytest_mock import MockerFixture


@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param("HTTPS://Example.test/a?b=2&a=1#f", "https://example.test/a?a=1&b=2#f", id="canonical-components"),
pytest.param("https:a", "https:a", id="special-without-authority"),
pytest.param("file:///tmp/x", "file:///tmp/x", id="empty-authority"),
pytest.param("file:/tmp/x", "file:///tmp/x", id="file-without-authority"),
pytest.param("//host/a?q=1#f", "//host/a?q=1#f", id="scheme-relative"),
pytest.param("a/b?x=1#f", "a/b?x=1#f", id="relative"),
pytest.param("mailto:a@b.test", "mailto:a@b.test", id="opaque"),
pytest.param("http://example.test/?#", "http://example.test/", id="empty-delimiters"),
pytest.param("https://example.test/#?", "https://example.test/#?", id="query-mark-in-fragment"),
],
)
def test_url_reparse_public_components(source: str, expected: str) -> None:
assert (normalize_url(source), url_reparse_check(source)) == (expected, None)


@pytest.mark.parametrize(
"normalized",
[
pytest.param("https://example.test/?", id="bare-query"),
pytest.param("https://example.test/#", id="bare-fragment"),
pytest.param("file:///x?#", id="all-delimiters"),
],
)
def test_url_reparse_keeps_optional_delimiters(normalized: str) -> None:
assert url_reparse_check("raw", lambda _text: normalized) is None


@pytest.mark.parametrize(
"source",
[
pytest.param("https://[broken/", id="target-rejection"),
pytest.param("https://[bad]/", id="reference-bracket-rejection"),
],
)
def test_url_reparse_rejects_unsupported_domain(source: str) -> None:
with pytest.raises(OutOfScopeError):
url_reparse_check(source)


@pytest.mark.parametrize(
"output",
[
pytest.param("HTTPS://example.test/", id="stable-upper-scheme"),
pytest.param(" https://example.test/", id="leading-space"),
pytest.param("https://exam\tple.test/", id="embedded-tab"),
],
)
def test_url_reparse_detects_noncanonical_split(output: str) -> None:
assert (result := url_reparse_check("raw", lambda _text: output)) is not None
assert result.startswith("split/recompose changes normalized URL")


def test_url_reparse_detects_changed_second_result() -> None:
assert (
result := url_reparse_check(
"raw", lambda text: "https://example.test/" if text == "raw" else "https://changed.test/"
)
) is not None
assert result.startswith("reparse changes normalized URL")


def test_url_reparse_detects_rejection_of_recomposed_output() -> None:
def reject(text: str) -> str:
if text == "raw":
return "https://example.test/"
msg = "rejected"
raise ValueError(msg)

assert url_reparse_check("raw", reject) == "rejects recomposed output"


def test_url_reparse_controls_discriminate() -> None:
assert ORACLES["url-split-reparse"].controls() == {
"stable uppercased scheme": True,
"changes after recomposition": True,
"rejects recomposed output": True,
}


def test_url_reparse_cli_is_registered(
mocker: MockerFixture, tmp_path: Path, capsys: pytest.CaptureFixture[str]
) -> None:
seeds = tmp_path / "tools/fuzz-data/wpt/url/resources/urltestdata.json"
seeds.parent.mkdir(parents=True)
seeds.write_text('[{"input": "HTTP://Example.COM:80/"}]', encoding="utf-8")
corpus = tmp_path / "tools/fuzz/corpus/url"
corpus.mkdir(parents=True)
(corpus / "url.txt").write_text("https://seed.example/", encoding="utf-8")
mocker.patch("fuzz.round_trip_oracles._ROOT", tmp_path)
assert (
main(["--oracle", "url-split-reparse", "--minutes", "0", "--crash-dir", str(tmp_path)]),
capsys.readouterr().out.splitlines()[-1],
) == (0, f"0 finding(s) written to {tmp_path}")
2 changes: 1 addition & 1 deletion tests/url/test_clean.py
Original file line number Diff line number Diff line change
Expand Up @@ -109,7 +109,7 @@ def test_clean_accepts_colon_hosts(url: str) -> None:
pytest.param(
"HTTP://u\x00:pw@LOCALHOST:80/a/../x", "http://u\x00:pw@localhost/x", id="nul-before-userinfo-colon"
),
pytest.param("HTTP://a\x00b.EXAMPLE:80/a/../x", "http://a\x00b.example/x", id="nul-before-host-dot"),
pytest.param("HTTP://a\x00b.EXAMPLE:80/a/../x", None, id="nul-before-host-dot"),
pytest.param("HTTP://:8000/a/../x", None, id="empty-host"),
],
)
Expand Down
77 changes: 77 additions & 0 deletions tests/url/test_normalize_host.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,77 @@
from __future__ import annotations

from typing import Final

import pytest

from turbohtml.extract import clean_url, extract_links, normalize_url


@pytest.mark.parametrize(
"host",
[
pytest.param(f"a%{code:02X}b.example", id=f"encoded-{code:02x}")
for code in (*range(0x21), 0x23, 0x25, 0x2F, 0x3A, 0x3C, 0x3E, 0x3F, 0x40, 0x5B, 0x5C, 0x5D, 0x5E, 0x7C, 0x7F)
]
+ [
pytest.param("\uff05\uff10\uff10", id="mapped-null-escape"),
pytest.param("\uff05\uff14\uff11", id="mapped-letter-escape"),
pytest.param("a\uff03b.example", id="mapped-fragment-delimiter"),
pytest.param("a\uff1ab.example", id="mapped-port-delimiter"),
pytest.param("a\uff20b.example", id="mapped-userinfo-delimiter"),
],
)
def test_normalize_rejects_forbidden_domain(host: str) -> None:
with pytest.raises(ValueError, match="host contains a forbidden domain code point"):
normalize_url(f"http://{host}/")


@pytest.mark.parametrize(
"source",
[
pytest.param("//\uff05\uff10\uff10", id="mapped-null"),
pytest.param("http://o%23", id="fragment"),
pytest.param("http://o%3F", id="query"),
pytest.param("//\uff05\uff14\uff11", id="mapped-escape"),
pytest.param("file://%3A", id="port"),
pytest.param("//%09t", id="tab"),
],
)
def test_normalize_rejects_saved_reparse_findings(source: str) -> None:
with pytest.raises(ValueError, match="host contains a forbidden domain code point"):
normalize_url(source)


@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param("http://good%2eexample/", "http://good.example/", id="decoded-dot"),
pytest.param("http://%41.example/", "http://a.example/", id="decoded-letter"),
pytest.param("http://caf%C3%A9.example/", "http://xn--caf-dma.example/", id="decoded-unicode"),
pytest.param("http://a\u0085b.example/", "http://a\u0085b.example/", id="retained-c1-fallback"),
pytest.param("http://a\u2028b.example/", "http://a\u2028b.example/", id="retained-disallowed-fallback"),
pytest.param("file:///", "file:///", id="empty-file-host"),
pytest.param("http://[::1]/", "http://[::1]/", id="ipv6"),
pytest.param("http://1g/", "http://1g/", id="numeric-prefix-fallback"),
pytest.param("http://-a/", "http://-a/", id="leading-hyphen-domain"),
],
)
def test_normalize_host_valid_and_fallback_forms(source: str, expected: str) -> None:
normalized: Final = normalize_url(source)
assert (normalized, normalize_url(normalized)) == (expected, expected)


@pytest.mark.parametrize(
"source",
[
pytest.param("http://a%23b.example/", id="decoded-fragment"),
pytest.param("http://a\uff05\uff14\uff11.example/", id="mapped-escape"),
],
)
def test_clean_rejects_forbidden_domain(source: str) -> None:
assert clean_url(source) is None


def test_external_links_reject_forbidden_base_domain() -> None:
with pytest.raises(ValueError, match="host contains a forbidden domain code point"):
extract_links("<a href='https://valid.example/'>link</a>", "http://a%23b.example/", external_only=True)
43 changes: 43 additions & 0 deletions tools/fuzz/round_trip_oracles.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,7 @@
from itertools import pairwise, starmap
from pathlib import Path
from typing import TYPE_CHECKING, Final
from urllib.parse import urlsplit

from fuzz.css_custom_oracles import (
UnsupportedCssCustomCaseError,
Expand Down Expand Up @@ -147,6 +148,7 @@

if TYPE_CHECKING:
from collections.abc import Callable, Iterable, Sequence
from urllib.parse import SplitResult

__all__ = [
"MAX_INPUT",
Expand All @@ -171,6 +173,7 @@
"selector_entry_check",
"span_check",
"style_check",
"url_reparse_check",
"xml_check",
"xpath_entry_check",
]
Expand Down Expand Up @@ -561,6 +564,34 @@ def normalize_url_check(text: str, normalize: Callable[[str], str] = normalize_u
return fixpoint_check(text, normalize, numeric=False)


def url_reparse_check(text: str, normalize: Callable[[str], str] = normalize_url) -> str | None:
"""Preserve delimiter presence when checking serialization through a reference split."""
try:
normalized: Final = normalize(text)
parts: Final = urlsplit(normalized)
except ValueError as error:
raise OutOfScopeError(str(error)) from error
recomposed: Final = _url_recompose(parts, normalized)
if recomposed != normalized:
return f"split/recompose changes normalized URL: {_divergence(normalized, recomposed)}"
try:
reparsed: Final = normalize(recomposed)
except ValueError:
return "rejects recomposed output"
return None if reparsed == normalized else f"reparse changes normalized URL: {_divergence(normalized, reparsed)}"


def _url_recompose(parts: SplitResult, text: str) -> str:
body: Final = text[len(parts.scheme) + 1 :] if parts.scheme else text
return "".join((
f"{parts.scheme}:" if parts.scheme else "",
f"//{parts.netloc}" if body.startswith("//") else "",
parts.path,
f"?{parts.query}" if "?" in body.partition("#")[0] else "",
f"#{parts.fragment}" if "#" in body else "",
))


def clean_url_check(text: str, clean: Callable[[str], str | None] = clean_url) -> str | None:
"""Skip rejected inputs; flag rejection of a cleaned URL."""
try:
Expand Down Expand Up @@ -2188,6 +2219,17 @@ def _normalize_url_controls() -> dict[str, bool]:
}


def _url_reparse_controls() -> dict[str, bool]:
return {
"stable uppercased scheme": url_reparse_check("raw", lambda _text: "HTTPS://example.test/") is not None,
"changes after recomposition": url_reparse_check(
"raw", lambda text: "https://example.test/" if text == "raw" else "https://changed.test/"
)
is not None,
"rejects recomposed output": url_reparse_check("raw", _reject_url_output) is not None,
}


def _reject_url_output(text: str) -> str:
if text == "raw":
return "https://example.org/"
Expand Down Expand Up @@ -2408,6 +2450,7 @@ def _xml_literal(case: str) -> str | None:
"normalize-url-fixpoint": Oracle(
normalize_url_check, _generate_url, _seeds_url, _normalize_url_controls, Floor(100, 0.5)
),
"url-split-reparse": Oracle(url_reparse_check, _generate_url, _seeds_url, _url_reparse_controls, Floor(100, 0.5)),
"idna-host": Oracle(idna_host_check, _generate_idna_host, _idna_host_seeds, _idna_host_controls, Floor(100, 0.95)),
"idna-nfc": Oracle(_idna_nfc_check, idna_nfc_generate, idna_nfc_seeds, idna_nfc_controls, Floor(100, 1)),
"clean-url-fixpoint": Oracle(clean_url_check, _generate_url, _seeds_url, _clean_url_controls, Floor(100, 0.25)),
Expand Down
Loading