diff --git a/tests/tokenizer/test_xml.py b/tests/tokenizer/test_xml.py index bb96feacb..cef5c9bec 100644 --- a/tests/tokenizer/test_xml.py +++ b/tests/tokenizer/test_xml.py @@ -1,7 +1,6 @@ from __future__ import annotations -import time -from typing import TYPE_CHECKING, Final, cast +from typing import Final, cast import pytest from bench.operations import INPUTS @@ -19,9 +18,6 @@ parse_xml, ) -if TYPE_CHECKING: - from collections.abc import Callable - def elements(node: Element) -> list[Element]: return [child for child in node if isinstance(child, Element)] @@ -698,30 +694,6 @@ def _plain_repeated(count: int) -> str: return "" + f"" * 2 + "" # the second element re-reads already-interned names -def _min_parse_seconds(markup: str) -> float: - best = float("inf") - for _ in range(5): # the floor over repeats drops scheduler noise, which only ever inflates a sample - start = time.perf_counter() - parse_xml(markup) - best = min(best, time.perf_counter() - start) - return best - - -@pytest.mark.parametrize( - "build", - [ - pytest.param(_namespaced_distinct, id="expanded-name"), - pytest.param(_plain_repeated, id="raw-name"), - ], -) -def test_xml_duplicate_detection_scales_linearly(build: Callable[[int], str]) -> None: - base: Final = _min_parse_seconds(build(10_000)) - quadrupled: Final = _min_parse_seconds(build(40_000)) - # Linear work quadruples with the input; the reverted nested scan is ~16x. 8x sits two-fold - # below the quadratic floor and two-fold above the linear one, so jitter cannot flip it. - assert quadrupled < base * 8 - - @pytest.mark.parametrize("count", [pytest.param(2000, id="grows-and-rehashes")]) def test_xml_many_distinct_namespaced_attributes_parse(count: int) -> None: root: Final = parse_xml(_namespaced_distinct(count)).find("r") @@ -823,7 +795,7 @@ def test_round_trip_against_lxml() -> None: assert [node.tag.split(":")[-1] for node in (ours, *ours.descendants) if isinstance(node, Element)] == their_locals -@pytest.mark.parametrize(("index", "count"), [(0, 128), (1, 1)], ids=["many", "single"]) +@pytest.mark.parametrize(("index", "count"), [(0, 128), (1, 1), (2, 1_000)], ids=["many", "single", "wide"]) @pytest.mark.oracle def test_lxml_namespace_benchmark_output(index: int, count: int) -> None: etree: Final = pytest.importorskip("lxml.etree", exc_type=ImportError) diff --git a/tests/url/test_idna.py b/tests/url/test_idna.py index cad1c8b9e..0cc2e4587 100644 --- a/tests/url/test_idna.py +++ b/tests/url/test_idna.py @@ -13,8 +13,8 @@ from __future__ import annotations import re -import time from pathlib import Path +from typing import Final import pytest @@ -347,13 +347,11 @@ def test_host_at_the_input_cap_still_encodes() -> None: assert _url_to_ascii("\u3400" * 16384).startswith("xn--") -def test_oversize_host_is_rejected_before_the_quadratic_passes_run() -> None: - """A host past the cap raises at once, in bounded time, instead of driving the O(n^2) reorder and punycode loops.""" - host = "\u0316\u0301" * 60000 - start = time.process_time() - with pytest.raises(ValueError, match="exceeds the IDNA input limit of 16384"): - _url_to_ascii(host) - assert time.process_time() - start < 2.0 +def test_oversize_host_keeps_its_unicode_form() -> None: + # past the 16384-code-point cap the host skips the O(n^2) reorder and punycode passes and keeps its Unicode form; + # U+0316 then U+0301 alternate combining classes 220 and 230, the input an insertion-sort reorder is O(n^2) on + url: Final = "http://" + "\u0316\u0301" * 60000 + "/" + assert normalize_url(url) == url @pytest.mark.parametrize( diff --git a/tools/bench/ci.py b/tools/bench/ci.py index 4ed63ca11..777393fd9 100644 --- a/tools/bench/ci.py +++ b/tools/bench/ci.py @@ -240,6 +240,7 @@ def _book_html() -> str: "parse-xml-text-references": ("parse-xml-text", 1), "parse-xml-text-tiny": ("parse-xml-text", 2), "parse-xml-prefixes-single": ("parse-xml-prefixes", 1), + "parse-xml-prefixes-wide": ("parse-xml-prefixes", 2), "query-closest-distinct": ("query-closest", 1), "query-parents-distinct": ("query-parents", 1), "query-siblings-single": ("query-siblings", 1), diff --git a/tools/bench/operations.py b/tools/bench/operations.py index 720a0d504..959cf78fd 100644 --- a/tools/bench/operations.py +++ b/tools/bench/operations.py @@ -1929,7 +1929,7 @@ def _encoding_result_cases() -> tuple[tuple[str, bytes], ...]: + " ".join(f'p{index}:value="x"' for index in range(count)) + "/>", ) - for count in (128, 1) + for count in (128, 1, 1_000) ), "parse-xml-names": lambda: ( (