From 52a476cbdb5d8cd88b60e08cb0aa3d69e36df718 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Tue, 6 Oct 2026 20:53:02 -0700 Subject: [PATCH 1/2] =?UTF-8?q?=E2=9C=A8=20feat(fuzz):=20bound=20XML=20gra?= =?UTF-8?q?mmar=20expansion?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit XML lexical choices emit no DOM nodes but still need a finite expansion budget. Reserve node and step costs together, and keep literal XML tree expectations beside materialized corpus inputs. Refs #1018 --- docs/changelog/51018.feature.rst | 1 + pyproject.toml | 2 +- tests/test_fuzz_structure_generation.py | 2 +- tests/test_fuzz_xml_structure_generation.py | 210 ++++++++++ tools/fuzz/round_trip_oracles.py | 9 + tools/fuzz/structure_generators.py | 76 +++- tools/fuzz/xml_structure_generators.py | 420 ++++++++++++++++++++ 7 files changed, 699 insertions(+), 21 deletions(-) create mode 100644 docs/changelog/51018.feature.rst create mode 100644 tests/test_fuzz_xml_structure_generation.py create mode 100644 tools/fuzz/xml_structure_generators.py diff --git a/docs/changelog/51018.feature.rst b/docs/changelog/51018.feature.rst new file mode 100644 index 00000000..3a966273 --- /dev/null +++ b/docs/changelog/51018.feature.rst @@ -0,0 +1 @@ +Add bounded XML structure generation with materialized corpus bytes and separate node and expansion budgets. diff --git a/pyproject.toml b/pyproject.toml index 3bab168f..73aa84f1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -307,7 +307,7 @@ lint.per-file-ignores."tests/**/*.py" = [ "import-private-name", # the conformance harness drives the private _html._tokenize_states hook "private-member-access", # same private hook ] -lint.per-file-ignores."tests/{test_fuzz_structure_generation,test_fuzz_html_structure_generation}.py" = [ +lint.per-file-ignores."tests/{test_fuzz_structure_generation,test_fuzz_html_structure_generation,test_fuzz_xml_structure_generation}.py" = [ "suspicious-non-cryptographic-random-usage", # deterministic grammar seeds reproduce public parser cases ] lint.per-file-ignores."tools/**/*.py" = [ diff --git a/tests/test_fuzz_structure_generation.py b/tests/test_fuzz_structure_generation.py index 2643a850..4471073f 100644 --- a/tests/test_fuzz_structure_generation.py +++ b/tests/test_fuzz_structure_generation.py @@ -98,7 +98,7 @@ def test_generation_rejects_unknown_production(grammar: Grammar) -> None: "unique", id="duplicate-name", ), - pytest.param((Production("zero", "root", (b"x",), 0, "test"),), "root", "positive", id="zero-cost"), + pytest.param((Production("negative", "root", (b"x",), -1, "test"),), "root", "nonnegative", id="negative-cost"), pytest.param( (Production("negative-height", "root", (b"x",), 1, "test", -1),), "root", diff --git a/tests/test_fuzz_xml_structure_generation.py b/tests/test_fuzz_xml_structure_generation.py new file mode 100644 index 00000000..248d8279 --- /dev/null +++ b/tests/test_fuzz_xml_structure_generation.py @@ -0,0 +1,210 @@ +from __future__ import annotations + +import random +from dataclasses import replace +from typing import TYPE_CHECKING, Final + +import pytest +from fuzz.structure_generators import ( + BudgetError, + Generated, + GenerationBudget, + Production, + ProductionFloorError, + Reference, + assert_production_floors, + compile_grammar, + generate, + generation_sweep, +) +from fuzz.xml_structure_generators import ( + main, + xml_document_check, + xml_expected, + xml_generate, + xml_grammar, + xml_raw_generate, + xml_raw_inputs, + xml_snapshot, + xml_source_controls, + xml_source_generate, + xml_source_seeds, + xml_structure_check, +) + +from turbohtml import Html, HTMLParseError, parse_xml + +if TYPE_CHECKING: + from pathlib import Path + + +_CASES: Final = generation_sweep(xml_grammar(), budget=GenerationBudget(30, 120)) + + +@pytest.mark.parametrize("case", _CASES, ids=[production.name for production in xml_grammar().productions]) +def test_xml_structure_literals(case: Generated) -> None: + assert xml_structure_check(case) is None + + +@pytest.mark.parametrize("case", _CASES, ids=[production.name for production in xml_grammar().productions]) +def test_xml_structure_materialized_budget(case: Generated) -> None: + root: Final = parse_xml(case.data.decode()) + nodes: Final = (root, *root.descendants) + assert (len(nodes), max(len(tuple(node.ancestors)) for node in nodes)) == (case.nodes, case.depth) + + +def test_xml_structure_independent_parent_literals() -> None: + case: Final = generate(xml_grammar(), random.Random(0), GenerationBudget(4, 6), force="xml:leaf:0") + assert xml_expected(case) == ( + ("document", "", "", (), -1), + ("element:html", "Root", "", (("id", "x"),), 0), + ("element:html", "Leaf", "", (), 1), + ("element:html", "Ref", "", (("target", "#x"),), 1), + ) + + +def test_xml_structure_changed_input() -> None: + case: Final = replace(_CASES[0], data=_CASES[0].data.replace(b"Root", b"Wrong")) + assert xml_structure_check(case) == "parsed XML differs from production literals" + + +def test_xml_structure_changed_serialization() -> None: + assert ( + xml_structure_check( + _CASES[0], serialize=lambda node: node.serialize(Html(xml=True)).replace('id="x"', 'id="wrong"') + ) + == "serialized XML differs from production literals" + ) + + +@pytest.mark.parametrize("seed", range(64)) +def test_xml_structure_random_bounds(seed: int) -> None: + case: Final = xml_generate(random.Random(seed), 12, steps=60) + assert (xml_structure_check(case), case.nodes <= 12, len(case.productions) <= 60) == (None, True, True) + + +@pytest.mark.parametrize("case", xml_raw_inputs(), ids=[case.productions[0] for case in xml_raw_inputs()]) +def test_xml_structure_raw_rejection(case: Generated) -> None: + with pytest.raises((UnicodeError, HTMLParseError, ValueError)): + parse_xml(case.data.decode("utf-8")) + + +@pytest.mark.parametrize("raw", [False, True], ids=["valid", "raw"]) +def test_xml_structure_materialized_corpus(tmp_path: Path, *, raw: bool) -> None: + cases: Final = xml_raw_inputs() if raw else _CASES + assert main(["--output", str(tmp_path), *(["--raw"] if raw else [])]) == 0 + assert {path.read_bytes() for path in tmp_path.iterdir()} == {case.data for case in cases} + + +def test_generation_zero_node_forced_path() -> None: + grammar: Final = compile_grammar( + ( + Production("root", "root", (b""), 1, "test"), + Production("empty", "attribute", (b"",), 0, "test"), + Production("attribute", "attribute", (b' a="x"',), 0, "test"), + ), + "root", + ) + assert generate(grammar, random.Random(0), GenerationBudget(1, 2), force="attribute").data == b'' + + +def test_generation_zero_node_cycle_terminates() -> None: + grammar: Final = compile_grammar( + ( + Production("cycle", "root", (b"x", Reference("root", 0)), 0, "test"), + Production("leaf", "root", (b"z",), 0, "test"), + ), + "root", + ) + case: Final = generate(grammar, random.Random(0), GenerationBudget(0, 2), force="cycle") + assert (case.data, case.nodes, case.productions) == (b"xz", 0, ("cycle", "leaf")) + + +def test_generation_step_reservation_for_siblings() -> None: + grammar: Final = compile_grammar( + ( + Production("root", "root", (Reference("child", 0), Reference("child", 0)), 1, "test"), + Production("cycle", "child", (b"x", Reference("child", 0)), 0, "test"), + Production("leaf", "child", (b"z",), 0, "test"), + ), + "root", + ) + case: Final = generate(grammar, random.Random(0), GenerationBudget(1, 4), force="cycle") + assert (case.data, case.nodes, len(case.productions)) == (b"xzz", 1, 4) + + +@pytest.mark.parametrize("budget", [GenerationBudget(3, 120), GenerationBudget(30, 5)], ids=["nodes", "steps"]) +def test_xml_structure_insufficient_budget(budget: GenerationBudget) -> None: + with pytest.raises(BudgetError, match="cannot complete"): + generate(xml_grammar(), random.Random(0), budget) + + +def test_xml_structure_snapshot_literal() -> None: + assert xml_snapshot(parse_xml("")) == ( + ("document", "", "", (), -1), + ("element:html", "Root", "", (), 0), + ) + + +def test_xml_document_seed_consumers() -> None: + assert [xml_document_check(source) for source in xml_source_seeds()] == [None] * len(_CASES) + + +def test_xml_document_generated_consumer() -> None: + assert xml_document_check(xml_source_generate(random.Random(4))) is None + + +def test_xml_document_failure_controls() -> None: + assert xml_source_controls() == {"dropped attribute": True, "changed parent": True} + + +def test_xml_document_custom_serializer() -> None: + assert xml_document_check("", lambda node: node.serialize(Html(xml=True))) is None + + +def test_xml_structure_custom_serializer() -> None: + assert xml_structure_check(_CASES[0], serialize=lambda node: node.serialize(Html(xml=True))) is None + + +@pytest.mark.parametrize("arguments", [["--count", "-1"], ["--budget", "0"]], ids=["count", "budget"]) +def test_xml_structure_cli_errors(tmp_path: Path, arguments: list[str]) -> None: + with pytest.raises(SystemExit) as error: + main(["--output", str(tmp_path), *arguments]) + assert error.value.code == 2 + + +def test_xml_structure_random_corpus(tmp_path: Path) -> None: + assert main(["--output", str(tmp_path), "--count", "1", "--seed", "4"]) == 0 + assert {path.read_bytes() for path in tmp_path.iterdir()} == {xml_generate(random.Random(4)).data} + + +def test_xml_structure_disabled_production() -> None: + name: Final = xml_grammar().productions[-1].name + with pytest.raises(ProductionFloorError, match="production floors missed"): + assert_production_floors( + (case for case in _CASES if name not in case.productions), + (production.name for production in xml_grammar().productions), + ) + + +def test_xml_raw_generate_literal() -> None: + assert xml_raw_generate(random.Random(0), 4) == Generated(bytes.fromhex("4c09c2"), 0, 0, ("xml:raw:bytes",), ()) + + +def test_xml_raw_generate_empty_budget() -> None: + assert xml_raw_generate(random.Random(0), 0).data == b"" + + +def test_xml_raw_generate_negative_budget() -> None: + with pytest.raises(BudgetError, match="byte budget must be nonnegative"): + xml_raw_generate(random.Random(0), -1) + + +def test_xml_raw_random_corpus(tmp_path: Path) -> None: + assert main(["--output", str(tmp_path), "--raw", "--count", "1", "--byte-budget", "4"]) == 0 + assert {path.read_bytes() for path in tmp_path.iterdir()} == {bytes.fromhex("4c09c2")} + + +def test_xml_structure_identifier_lookup() -> None: + root: Final = parse_xml(_CASES[0].data.decode()) + assert [(node.tag, dict(node.attrs)) for node in root.select("#x")] == [("Root", {"id": "x"})] diff --git a/tools/fuzz/round_trip_oracles.py b/tools/fuzz/round_trip_oracles.py index 2bf765a9..aa95b7a1 100644 --- a/tools/fuzz/round_trip_oracles.py +++ b/tools/fuzz/round_trip_oracles.py @@ -114,6 +114,12 @@ xml_island_generate, xml_island_seeds, ) +from fuzz.xml_structure_generators import ( + xml_document_check, + xml_source_controls, + xml_source_generate, + xml_source_seeds, +) from markdown_it import MarkdownIt from typing_extensions import override @@ -2437,6 +2443,9 @@ def _xml_literal(case: str) -> str | None: "html-list-grammar": Oracle( _html_list_check, html_list_generate, html_list_seeds, html_list_controls, Floor(24, 1) ), + "xml-document-grammar": Oracle( + xml_document_check, xml_source_generate, xml_source_seeds, xml_source_controls, Floor(60, 1) + ), "xml-fixpoint": Oracle(xml_check, _generate_html, _seeds_html, _xml_controls, Floor(500, 0.95)), "css-fixpoint": Oracle(_css_fixpoint, _generate_css, _seeds_css, _css_controls, Floor(500, 0.95)), "js-fixpoint": Oracle(_js_fixpoint, _generate_js, _seeds_js, _js_controls, Floor(300, 0.6)), diff --git a/tools/fuzz/structure_generators.py b/tools/fuzz/structure_generators.py index 96ec65ff..93e6d737 100644 --- a/tools/fuzz/structure_generators.py +++ b/tools/fuzz/structure_generators.py @@ -17,13 +17,13 @@ def main(argv: Sequence[str] | None = None) -> int: """Keep corpus files as materialized inputs rather than oracle mode strings.""" - parser = argparse.ArgumentParser() + parser: Final = argparse.ArgumentParser() parser.add_argument("--output", type=Path, required=True) parser.add_argument("--count", type=int, default=64) parser.add_argument("--budget", type=int, default=30) parser.add_argument("--seed", type=int, default=0) parser.add_argument("--sweep", action="store_true") - args = parser.parse_args(argv) + args: Final = parser.parse_args(argv) if args.count < 1: parser.error("count must be positive") try: @@ -45,8 +45,8 @@ def compile_grammar(productions: Sequence[Production], root: str) -> Grammar: rules: Final[dict[str, list[Production]]] = {} names: Final[set[str]] = set() for production in productions: - if production.name in names or production.nodes < 1 or production.height < 0: - msg = "production names must be unique, node costs positive, and heights nonnegative" + if production.name in names or production.nodes < 0 or production.height < 0: + msg = "production names must be unique, node costs nonnegative, and heights nonnegative" raise GrammarError(msg) names.add(production.name) rules.setdefault(production.symbol, []).append(production) @@ -143,11 +143,22 @@ def _route_cost(grammar: Grammar, route: tuple[tuple[Production, int], ...]) -> def generate( - grammar: Grammar, rng: random.Random, budget: int = 30, *, leaf: bool = False, force: str | None = None + grammar: Grammar, + rng: random.Random, + budget: int | GenerationBudget = 30, + *, + leaf: bool = False, + force: str | None = None, ) -> Generated: - """Reserve minimum sibling costs so a selected branch cannot starve later children.""" + """Reserve the canonical minimum-node derivation and its expansion steps.""" route: Final = _route(grammar, force) if force is not None else () - if budget < (_route_cost(grammar, route)[0] if route else grammar.minimum[grammar.root][0]): + remaining = ( + (budget.nodes, budget.steps) + if isinstance(budget, GenerationBudget) + else (budget, budget * max(cost[1] for cost in grammar.minimum.values())) + ) + required: Final = _route_cost(grammar, route) if route else grammar.minimum[grammar.root] + if any(available < needed for available, needed in zip(remaining, required, strict=True)): msg = "budget cannot complete the root" raise BudgetError(msg) pending: Final[list[tuple[bytes | Reference | Identifier, int, bool, tuple[tuple[Production, int], ...]]]] = [ @@ -156,7 +167,6 @@ def generate( output: Final[list[bytes]] = [] fired: Final[list[str]] = [] bindings: Final = _Bindings({}, set(), set()) - remaining = budget depth = 0 while pending: part, level, minimal, path = pending.pop() @@ -165,14 +175,12 @@ def generate( elif isinstance(part, Identifier): output.append(_identifier(part, bindings)) else: - available: Final = remaining - sum( - _route_cost(grammar, route)[0] if route else grammar.minimum[item.symbol][0] - for item, _, _, route in pending - if isinstance(item, Reference) + chosen, index, minimal = _choose( + grammar, rng, (part, level, minimal, path), _available(grammar, pending, remaining) ) - chosen, index, minimal = _choose(grammar, rng, (part, level, minimal, path), available) - remaining -= chosen.nodes - depth = max(depth, level + chosen.height) + remaining = remaining[0] - chosen.nodes, remaining[1] - 1 + if chosen.nodes: + depth = max(depth, level + chosen.height) fired.append(chosen.name) pending.extend( ( @@ -186,7 +194,26 @@ def generate( if missing := bindings.references - bindings.definitions: msg = f"undefined identifiers: {sorted(missing)}" raise GrammarError(msg) - return Generated(b"".join(output), budget - remaining, depth, tuple(fired), tuple(sorted(bindings.values.items()))) + return Generated( + b"".join(output), + (budget.nodes if isinstance(budget, GenerationBudget) else budget) - remaining[0], + depth, + tuple(fired), + tuple(sorted(bindings.values.items())), + ) + + +def _available( + grammar: Grammar, + pending: list[tuple[bytes | Reference | Identifier, int, bool, tuple[tuple[Production, int], ...]]], + remaining: tuple[int, int], +) -> tuple[int, int]: + reserved: Final = [ + _route_cost(grammar, route) if route else grammar.minimum[item.symbol] + for item, _, _, route in pending + if isinstance(item, Reference) + ] + return remaining[0] - sum(cost[0] for cost in reserved), remaining[1] - sum(cost[1] for cost in reserved) def _identifier(part: Identifier, bindings: _Bindings) -> bytes: @@ -209,20 +236,22 @@ def _choose( grammar: Grammar, rng: random.Random, frame: tuple[Reference, int, bool, tuple[tuple[Production, int], ...]], - available: int, + available: tuple[int, int], ) -> tuple[Production, int, bool]: part, level, minimal, path = frame if path: return path[0][0], path[0][1], True eligible: Final = [ - production for production in grammar.rules[part.symbol] if _cost(production, grammar.minimum)[0] <= available + production + for production in grammar.rules[part.symbol] + if all(cost <= limit for cost, limit in zip(_cost(production, grammar.minimum), available, strict=True)) ] if minimal or rng.random() >= 0.75 ** (level + 1): return min(eligible, key=lambda production: _cost(production, grammar.minimum)), -1, True return rng.choice(eligible), -1, False -def generation_sweep(grammar: Grammar, minimum: int = 1, budget: int = 30) -> tuple[Generated, ...]: +def generation_sweep(grammar: Grammar, minimum: int = 1, budget: int | GenerationBudget = 30) -> tuple[Generated, ...]: """Force each declared production through a minimum-cost root path.""" if minimum < 1: msg = "production floors must be positive" @@ -265,6 +294,14 @@ class _Bindings: references: set[str] +@dataclass(frozen=True) +class GenerationBudget: + """Bound lexical recursion separately from materialized DOM nodes.""" + + nodes: int + steps: int + + @dataclass(frozen=True) class Production: """Charge the nodes emitted by fixed markup, including parser-created wrappers.""" @@ -472,6 +509,7 @@ def _attribute(name: str, value: str, index: int) -> tuple[tuple[bytes | Referen __all__ = [ "BudgetError", "Generated", + "GenerationBudget", "Grammar", "GrammarError", "Identifier", diff --git a/tools/fuzz/xml_structure_generators.py b/tools/fuzz/xml_structure_generators.py new file mode 100644 index 00000000..a187fa57 --- /dev/null +++ b/tools/fuzz/xml_structure_generators.py @@ -0,0 +1,420 @@ +"""XML lexical choices consume expansion steps without inventing DOM nodes.""" + +from __future__ import annotations + +import argparse +import random +from dataclasses import dataclass +from pathlib import Path +from typing import TYPE_CHECKING, Final + +from fuzz.structure_generators import ( + BudgetError, + Generated, + GenerationBudget, + Grammar, + Identifier, + Production, + ProductionFloorError, + Reference, + compile_grammar, + generate, + generation_sweep, + write_corpus, +) + +from turbohtml import CData, Comment, Doctype, Element, Html, ProcessingInstruction, Text, parse_xml + +if TYPE_CHECKING: + from collections.abc import Callable, Iterator, Sequence + + from turbohtml import Document, Node + +XmlRecord = tuple[str, str, str, tuple[tuple[str, str], ...], int] + + +def main(argv: Sequence[str] | None = None) -> int: + """Keep corpus files independent of parser acceptance.""" + parser: Final = argparse.ArgumentParser() + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--budget", type=int, default=30) + parser.add_argument("--steps", type=int, default=120) + parser.add_argument("--raw", action="store_true") + parser.add_argument("--count", type=int, default=0) + parser.add_argument("--seed", type=int, default=0) + parser.add_argument("--byte-budget", type=int, default=64, help="maximum random raw input length") + args: Final = parser.parse_args(argv) + if args.count < 0: + parser.error("count must be nonnegative") + try: + cases: Final = ( + tuple(xml_raw_generate(random.Random(args.seed + index), args.byte_budget) for index in range(args.count)) + if args.raw and args.count + else xml_raw_inputs() + if args.raw + else tuple( + xml_generate(random.Random(args.seed + index), args.budget, steps=args.steps) + for index in range(args.count) + ) + if args.count + else generation_sweep(_GRAMMAR, budget=GenerationBudget(args.budget, args.steps)) + ) + except (BudgetError, ProductionFloorError) as error: + parser.error(str(error)) + write_corpus(cases, args.output) + return 0 + + +def xml_generate(rng: random.Random, budget: int = 30, *, steps: int = 120) -> Generated: + """Keep well-formed XML and raw-byte rejection seeds in distinct profiles.""" + return generate(_GRAMMAR, rng, GenerationBudget(budget, steps)) + + +def xml_grammar() -> Grammar: + """Retain production provenance for corpus audits.""" + return _GRAMMAR + + +def xml_expected(case: Generated) -> tuple[XmlRecord, ...]: + """Use literal production semantics independently of either parse.""" + records: Final[list[XmlRecord]] = [] + _expect(iter(case.productions), -1, records) + return tuple(records) + + +def _expect(trace: Iterator[str], parent: int, records: list[XmlRecord]) -> None: + positions: Final[dict[str, int]] = {} + for item in _RULES[next(trace)].expected: + owner: Final = parent if not item.parent else positions[item.parent] + if isinstance(item, _Child): + _expect(trace, owner, records) + else: + positions[item.label] = len(records) + records.append((item.kind, item.name, item.data, item.attrs, owner)) + + +def xml_structure_check( + case: Generated, + *, + read: Callable[[str], Document] = parse_xml, + serialize: Callable[[Node], str] | None = None, +) -> str | None: + """Require literal names, content, parents and bindings before comparing serialization.""" + expected: Final = xml_expected(case) + root: Final = read(case.data.decode("utf-8")) + if xml_snapshot(root) != expected: + return "parsed XML differs from production literals" + printed: Final = root.serialize(Html(xml=True)) if serialize is None else serialize(root) + if xml_snapshot(read(printed)) != expected: + return "serialized XML differs from production literals" + return None + + +def xml_snapshot(root: Document) -> tuple[XmlRecord, ...]: + """Keep parent ordinals stable when wrapper identities differ.""" + nodes: Final = [root, *root.descendants] + records: Final[list[XmlRecord]] = [] + for node in nodes: + parent: Final = -1 if node.parent is None else nodes.index(node.parent) + if isinstance(node, Element): + records.append(( + f"element:{node.namespace.value}", + node.tag, + "", + tuple(sorted((name, str(value)) for name, value in node.attrs.items())), + parent, + )) + elif isinstance(node, Doctype): + records.append(( + "doctype", + node.name, + "", + tuple( + (name, value) + for name, value in (("public", node.public_id), ("system", node.system_id)) + if value is not None + ), + parent, + )) + elif isinstance(node, ProcessingInstruction): + records.append(("pi", node.target, node.data, (), parent)) + elif isinstance(node, (Text, Comment, CData)): + records.append((type(node).__name__.lower(), "", node.data, (), parent)) + else: + records.append(("document", "", "", (), parent)) + return tuple(records) + + +def xml_document_check(markup: str, serialize: Callable[[Node], str] | None = None) -> str | None: + """Compare XML input semantics rather than reparsing an HTML interpretation.""" + root: Final = parse_xml(markup) + expected: Final = xml_snapshot(root) + printed: Final = root.serialize(Html(xml=True)) if serialize is None else serialize(root) + return None if xml_snapshot(parse_xml(printed)) == expected else "XML serialization changes the document" + + +def xml_source_generate(rng: random.Random) -> str: + """Use XML documents so grammar cases reach the XML parser.""" + return xml_generate(rng).data.decode("utf-8") + + +def xml_source_seeds() -> list[str]: + """Force production coverage so random choices cannot leave alternatives untested.""" + return [case.data.decode("utf-8") for case in generation_sweep(_GRAMMAR, budget=GenerationBudget(30, 120))] + + +def xml_source_controls() -> dict[str, bool]: + """Make oracle no-ops fail on data loss and reparenting.""" + return { + "dropped attribute": xml_document_check( + '', + lambda node: node.serialize(Html(xml=True)).replace(' key="value"', ""), + ) + is not None, + "changed parent": xml_document_check( + "", + lambda _node: "", + ) + is not None, + } + + +def xml_raw_generate(rng: random.Random, budget: int = 64) -> Generated: + """Bound raw input length without claiming a DOM node cost.""" + if budget < 0: + msg = "byte budget must be nonnegative" + raise BudgetError(msg) + return Generated(rng.randbytes(rng.randint(0, budget)), 0, 0, ("xml:raw:bytes",), ()) + + +def xml_raw_inputs() -> tuple[Generated, ...]: + """Preserve invalid UTF-8 and XML delimiters as bounded byte inputs.""" + return tuple(Generated(source, 0, 0, (f"xml:raw:{index}",), ()) for index, source in enumerate(_RAW)) + + +@dataclass(frozen=True) +class _Literal: + kind: str + name: str = "" + data: str = "" + attrs: tuple[tuple[str, str], ...] = () + parent: str = "" + label: str = "" + + +@dataclass(frozen=True) +class _Child: + parent: str = "" + + +@dataclass(frozen=True) +class _Rule: + production: Production + expected: tuple[_Literal | _Child, ...] + + +def _rules() -> dict[str, _Rule]: + rules: Final = [ + _Rule( + Production( + "xml:document", + "document", + ( + Reference("declaration", 0), + Reference("misc", 1), + Reference("doctype", 1), + b'', + Reference("node", 2), + b'', + Reference("misc", 1), + ), + 3, + _STANDARD, + 2, + ), + ( + _Literal("document", label="document"), + _Child("document"), + _Child("document"), + _Child("document"), + _Literal("element:html", "Root", attrs=(("id", "x"),), parent="document", label="root"), + _Child("root"), + _Literal("element:html", "Ref", attrs=(("target", "#x"),), parent="root"), + _Child("document"), + ), + ), + _Rule( + Production( + "xml:nested", "node", (b"", Reference("node"), Reference("node"), b""), 1, _STANDARD + ), + (_Literal("element:html", "Group", label="group"), _Child("group"), _Child("group")), + ), + ] + for index, declaration in enumerate(_DECLARATIONS): + rules.append(_Rule(Production(f"xml:declaration:{index}", "declaration", (declaration,), 0, _STANDARD), ())) + for index, (source, expected) in enumerate(_MISC): + rules.append(_Rule(Production(f"xml:misc:{index}", "misc", (source,), len(expected), _STANDARD), expected)) + for index, (source, attrs) in enumerate(_DOCTYPES): + rules.append( + _Rule( + Production(f"xml:doctype:{index}", "doctype", (source,), int(bool(source)), _COMPETITOR), + (_Literal("doctype", "Root", attrs=attrs),) if source else (), + ) + ) + for index, (source, expected) in enumerate(_LEAVES): + rules.append( + _Rule( + Production(f"xml:leaf:{index}", "node", (source,), len(expected), _STANDARD, int(len(expected) > 1)), + expected, + ) + ) + return {rule.production.name: rule for rule in rules} + + +_STANDARD: Final = "https://www.w3.org/TR/2008/REC-xml-20081126/" +_COMPETITOR: Final = "https://github.com/GNOME/libxml2/blob/c43dc98d27ac315a48d93dbd399c6c22cf7125b1/fuzz/xml.dict" +_DECLARATIONS: Final = ( + b"", + b'', + b"", + b'', + b"", + b'', + b'', +) +_MISC: Final = ( + (b"", ()), + (b" \t\r\n", ()), + (b"", (_Literal("comment", data="before"),)), + (b"", (_Literal("pi", "probe", "data"),)), + (b"", (_Literal("comment", data="before"), _Literal("pi", "probe", "data"))), +) +_DOCTYPES: Final = ( + (b"", ()), + (b"", ()), + (b'', (("system", "root.dtd"),)), + (b'', (("public", "public-id"), ("system", "root.dtd"))), + (b"]>", ()), +) +_LEAVES: Final = ( + (b"", (_Literal("element:html", "Leaf"),)), + (b"", (_Literal("element:html", "Leaf"),)), + *[ + ( + f"<{name}>x".encode(), + (_Literal("element:html", name, label="leaf"), _Literal("text", data="x", parent="leaf")), + ) + for name in ("Leaf", "leaf", "_name", "name-1", "name.1", "é", "水") + ], + *[ + ( + f"{encoded}".encode(), + (_Literal("element:html", "Leaf", label="leaf"), _Literal("text", data=decoded, parent="leaf")), + ) + for encoded, decoded in ( + ("x", "x"), + ("a\tb\r\nc\rd", "a\tb\nc\nd"), + ("&<>"'", "&<>\"'"), + (" ", "\t\n\r"), + ("A😀", "A😀"), + ("é水😀", "é水😀"), + ) + ], + *[ + ( + f"".encode(), + (_Literal("element:html", "Leaf", attrs=(("value", decoded),)),), + ) + for quote in ("'", '"') + for encoded, decoded in ( + ("", ""), + ("a b", "a b"), + (""'", "\"'"), + ("a\tb\r\nc\rd", "a b c d"), + ("&<", "&<"), + (" ", "\t\n\r"), + ("é水😀", "é水😀"), + ) + ], + ( + b"ab", + ( + _Literal("element:html", "Leaf", label="leaf"), + _Literal("text", data="a", parent="leaf"), + _Literal("element:html", "Inner", parent="leaf"), + _Literal("text", data="b", parent="leaf"), + ), + ), + ( + b"", + (_Literal("element:html", "Leaf", label="leaf"), _Literal("comment", data="inside", parent="leaf")), + ), + ( + b"", + (_Literal("element:html", "Leaf", label="leaf"), _Literal("pi", "probe", "data", parent="leaf")), + ), + *[ + ( + f"".encode(), + (_Literal("element:html", "Leaf", label="leaf"), _Literal("cdata", data=text, parent="leaf")), + ) + for text in ("", "x", "<&", "é水😀") + ], + ( + b'', + (_Literal("element:html", "p:Leaf", attrs=(("p:key", "value"), ("xmlns:p", "urn:p"))),), + ), + (b'', (_Literal("element:html", "Leaf", attrs=(("xmlns", "urn:d"),)),)), + (b'', (_Literal("element:html", "Leaf", attrs=(("xml:lang", "en"),)),)), + ( + b'', + ( + _Literal("element:html", "Leaf", attrs=(("xmlns", "urn:d"),), label="leaf"), + _Literal("element:html", "Inner", attrs=(("xmlns", ""),), parent="leaf"), + ), + ), + ( + b'', + ( + _Literal("element:html", "p:Leaf", attrs=(("xmlns:p", "urn:a"),), label="leaf"), + _Literal("element:html", "p:Inner", attrs=(("xmlns:p", "urn:b"),), parent="leaf"), + ), + ), +) +_RAW: Final = ( + b"", + b"\xff", + b"\xc0\xaf", + b"<", + b"", + b"tail", + b"&unknown;", + b"�", + b"", + b"", +) +_RULES: Final = _rules() +_GRAMMAR: Final = compile_grammar(tuple(rule.production for rule in _RULES.values()), "document") + +__all__ = [ + "XmlRecord", + "main", + "xml_document_check", + "xml_expected", + "xml_generate", + "xml_grammar", + "xml_raw_generate", + "xml_raw_inputs", + "xml_snapshot", + "xml_source_controls", + "xml_source_generate", + "xml_source_seeds", + "xml_structure_check", +] + +if __name__ == "__main__": + raise SystemExit(main()) From d61ee7e6b3d0875e620a4105ee1d8554a4c6a26c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Tue, 6 Oct 2026 20:53:39 -0700 Subject: [PATCH 2/2] =?UTF-8?q?=F0=9F=93=9D=20docs(changelog):=20link=20XM?= =?UTF-8?q?L=20grammar=20PR?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/changelog/{51018.feature.rst => 1196.feature.rst} | 0 1 file changed, 0 insertions(+), 0 deletions(-) rename docs/changelog/{51018.feature.rst => 1196.feature.rst} (100%) diff --git a/docs/changelog/51018.feature.rst b/docs/changelog/1196.feature.rst similarity index 100% rename from docs/changelog/51018.feature.rst rename to docs/changelog/1196.feature.rst