From 537636429bc53928e28a909093d55ec65ac4344d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Tue, 6 Oct 2026 20:56:02 -0700 Subject: [PATCH 1/3] =?UTF-8?q?=F0=9F=90=9B=20fix(schema):=20compare=20ref?= =?UTF-8?q?erence=20outcomes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Compare inline XSLT and schema outcomes with the reference engines. Repair seed-tree references so mutations exercise compiler and validation behavior. Reject undefined RELAX NG references before removing unused definitions and remove foreign annotations on the compiler's private schema tree. Refs #1016 --- docs/changelog/1016-reference.bugfix.rst | 1 + docs/changelog/1016-reference.feature.rst | 1 + src/turbohtml/_c/validate/relaxng.h | 76 +++++++- tests/test_fuzz_rng_inline.py | 152 +++++++++++++++ tests/test_fuzz_schema_mutation.py | 161 ++++++++++++++++ tests/test_fuzz_xsd_inline.py | 2 +- tests/test_fuzz_xslt_inline.py | 130 +++++++++++++ tests/validate/test_relaxng.py | 12 +- tests/validate/test_relaxng_references.py | 115 ++++++++++++ tools/fuzz/rng_inline.py | 184 ++++++++++++++++++ tools/fuzz/schema_mutation.py | 128 +++++++++++++ tools/fuzz/xsd_inline.py | 42 ++++- tools/fuzz/xslt_inline.py | 217 ++++++++++++++++++++++ 13 files changed, 1200 insertions(+), 21 deletions(-) create mode 100644 docs/changelog/1016-reference.bugfix.rst create mode 100644 docs/changelog/1016-reference.feature.rst create mode 100644 tests/test_fuzz_rng_inline.py create mode 100644 tests/test_fuzz_schema_mutation.py create mode 100644 tests/test_fuzz_xslt_inline.py create mode 100644 tests/validate/test_relaxng_references.py create mode 100644 tools/fuzz/rng_inline.py create mode 100644 tools/fuzz/schema_mutation.py create mode 100644 tools/fuzz/xslt_inline.py diff --git a/docs/changelog/1016-reference.bugfix.rst b/docs/changelog/1016-reference.bugfix.rst new file mode 100644 index 000000000..21fcbb324 --- /dev/null +++ b/docs/changelog/1016-reference.bugfix.rst @@ -0,0 +1 @@ +Reject undefined RELAX NG references and ignore foreign annotation subtrees during compilation. diff --git a/docs/changelog/1016-reference.feature.rst b/docs/changelog/1016-reference.feature.rst new file mode 100644 index 000000000..c5da55b12 --- /dev/null +++ b/docs/changelog/1016-reference.feature.rst @@ -0,0 +1 @@ +Compare inline XSLT and schema outcomes with libxml2 and libxslt, including repaired seed-tree mutations. diff --git a/src/turbohtml/_c/validate/relaxng.h b/src/turbohtml/_c/validate/relaxng.h index 2179fd58d..c6f8ba08e 100644 --- a/src/turbohtml/_c/validate/relaxng.h +++ b/src/turbohtml/_c/validate/relaxng.h @@ -709,6 +709,7 @@ static pattern *rng_build(th_schema *schema, th_node *node) { node_pat->def_index = (int)index; return node_pat; } + PyErr_SetString(PyExc_ValueError, "RELAX NG has no matching define"); return schema->p_notallowed; } return schema->p_notallowed; @@ -1105,6 +1106,8 @@ static pattern *rng_child_element(valctx *ctx, pattern *p, th_node *element) { /* ---- compile & entry ---- */ static int rng_scan(th_schema *schema, th_node *container, int depth); +static int rng_check_unused_refs(th_schema *schema, th_node *container); +static void rng_prune_annotations(th_schema *schema, th_node *container); static int rng_compile(th_schema *schema) { qname root_name = schema_direct_qname(schema, schema->root); @@ -1112,6 +1115,7 @@ static int rng_compile(th_schema *schema) { PyErr_SetString(PyExc_ValueError, "RELAX NG schema root must use the structure namespace"); return 0; } + rng_prune_annotations(schema, schema->root); th_tree *tree = schema->tree; schema->p_empty = pat_new(schema, P_EMPTY); schema->p_notallowed = pat_new(schema, P_NOTALLOWED); @@ -1144,10 +1148,65 @@ static int rng_compile(th_schema *schema) { if (rng_scan(schema, start, 0) < 0) { return 0; } + for (Py_ssize_t index = 0; index < schema->defines.len; index++) { + def_entry *entry = &schema->defines.items[index]; + if (entry->cycle_depth != -1) { + continue; + } + if (rng_check_unused_refs(schema, entry->first) < 0) { + return 0; + } + for (def_part *part = entry->extra; part != NULL; part = part->next) { + if (rng_check_unused_refs(schema, part->node) < 0) { + return 0; + } + } + } schema->start = rng_build_children(schema, start, NULL); return 1; } +/* Section 4.1 removes annotation subtrees before pattern and name-class construction. */ +static void rng_prune_annotations(th_schema *schema, th_node *container) { + th_node *child = container->first_child; + while (child != NULL) { + th_node *next = child->next_sibling; + if (child->type == TH_NODE_ELEMENT) { + qname name = schema_direct_qname(schema, child); + if (!u_eq_ascii(name.uri, name.uri_len, RNG_NS)) { + th_node_remove(child); + } else { + rng_prune_annotations(schema, child); + } + } + child = next; + } +} + +/* Section 4.18 resolves names before 4.19 removes unused definitions. */ +static int rng_check_unused_refs(th_schema *schema, th_node *container) { + for (th_node *child = container->first_child; child != NULL; child = child->next_sibling) { + if (child->type != TH_NODE_ELEMENT) { + continue; + } + if (is_schema_el(schema, child, RNG_NS, "ref")) { + const th_node_attr *name = attr_exact(schema->tree, child, "name", 4); + if (name == NULL) { + PyErr_SetString(PyExc_ValueError, "RELAX NG is missing the required name attribute"); + return -1; + } + if (def_find(&schema->defines, name->value, name->value_len) < 0) { + PyErr_SetString(PyExc_ValueError, "RELAX NG has no matching define"); + return -1; + } + } + if (rng_check_unused_refs(schema, child) < 0) { + return -1; + } + } + return 0; +} + static int rng_scan_node(th_schema *schema, th_node *node, int depth); /* Checks 4.10, 4.19 and 7.4 on reachable patterns only: an unreachable is never built, so it can neither @@ -1166,11 +1225,11 @@ static int rng_scan(th_schema *schema, th_node *container, int depth) { enum { RNG_SCAN_OTHER, RNG_SCAN_REF, RNG_SCAN_INTERLEAVE, RNG_SCAN_ELEMENT }; -static int rng_scan_kind(const th_schema *schema, th_node *node); +static int rng_scan_kind(th_node *node); static int rng_scan_define(th_schema *schema, Py_ssize_t def_index, int depth); static int rng_scan_node(th_schema *schema, th_node *node, int depth) { - switch (rng_scan_kind(schema, node)) { + switch (rng_scan_kind(node)) { case RNG_SCAN_REF: { const th_node_attr *name = attr_exact(schema->tree, node, "name", 4); if (name == NULL) { @@ -1178,7 +1237,11 @@ static int rng_scan_node(th_schema *schema, th_node *node, int depth) { return -1; } Py_ssize_t index = def_find(&schema->defines, name->value, name->value_len); - return index >= 0 ? rng_scan_define(schema, index, depth) : 0; + if (index < 0) { + PyErr_SetString(PyExc_ValueError, "RELAX NG has no matching define"); + return -1; + } + return rng_scan_define(schema, index, depth); } case RNG_SCAN_INTERLEAVE: return rng_check_interleave_node(schema, node) < 0 ? -1 : rng_scan(schema, node, depth); @@ -1189,7 +1252,7 @@ static int rng_scan_node(th_schema *schema, th_node *node, int depth) { } } -static int rng_scan_kind(const th_schema *schema, th_node *node) { +static int rng_scan_kind(th_node *node) { const Py_UCS4 *local, *prefix; Py_ssize_t local_len = 0, prefix_len = 0; split_prefix(node->text, node->text_len, &local, &local_len, &prefix, &prefix_len); @@ -1201,10 +1264,9 @@ static int rng_scan_kind(const th_schema *schema, th_node *node) { } else if (u_eq_ascii(local, local_len, "element")) { kind = RNG_SCAN_ELEMENT; } else { - return RNG_SCAN_OTHER; /* most nodes are not a restriction keyword: skip the namespace resolution */ + return RNG_SCAN_OTHER; } - qname name = schema_direct_qname(schema, node); - return u_eq_ascii(name.uri, name.uri_len, RNG_NS) ? kind : RNG_SCAN_OTHER; + return kind; } /* RELAX NG 4.19: meeting a define again at the element depth where its expansion began is a ref loop with no diff --git a/tests/test_fuzz_rng_inline.py b/tests/test_fuzz_rng_inline.py new file mode 100644 index 000000000..167d12256 --- /dev/null +++ b/tests/test_fuzz_rng_inline.py @@ -0,0 +1,152 @@ +from __future__ import annotations + +import json +from typing import TYPE_CHECKING, Final + +import pytest + +if TYPE_CHECKING: + from types import ModuleType + + +@pytest.fixture +def engine() -> ModuleType: + return pytest.importorskip("fuzz.rng_inline", exc_type=ImportError) + + +@pytest.mark.oracle +@pytest.mark.parametrize("seed", range(8)) +def test_rng_inline_generated_labels(engine: ModuleType, seed: int) -> None: + case: Final = engine.generate(seed) + assert [(row.engine, row.phase, row.expected, row.actual) for row in engine.compare(case)] == [ + ("turbohtml", "compilation", case.compiles, case.compiles), + ("libxml2", "compilation", case.compiles, case.compiles), + *[ + (name, "validation", expected, expected) + for _, expected in case.documents + for name in ("turbohtml", "libxml2") + ], + ] + + +@pytest.mark.oracle +@pytest.mark.parametrize("schema", ["<", '']) +def test_rng_inline_compile_verdicts(engine: ModuleType, schema: str) -> None: + expected: Final = False + case: Final = engine.Case(0, schema, expected, ()) + assert [(row.engine, row.phase, row.expected, row.actual) for row in engine.compare(case)] == [ + ("turbohtml", "compilation", expected, expected), + ("libxml2", "compilation", expected, expected), + ] + + +@pytest.mark.oracle +def test_rng_inline_unavailable_documents_remain_findings(engine: ModuleType) -> None: + case: Final = engine.Case(0, "<", compiles=False, documents=(("", True),)) + assert [(row.engine, row.phase, row.expected, row.actual) for row in engine.compare(case)] == [ + ("turbohtml", "compilation", False, False), + ("libxml2", "compilation", False, False), + ("turbohtml", "validation", True, None), + ("libxml2", "validation", True, None), + ] + + +@pytest.mark.oracle +def test_rng_inline_negative_control_changes_public_verdict(engine: ModuleType) -> None: + case: Final = engine.generate(0) + assert [ + (row.engine, row.phase, row.expected, row.actual) for row in engine.compare(case, negative_control=True) + ] == [ + ("turbohtml", "compilation", True, True), + ("libxml2", "compilation", True, True), + ("turbohtml", "validation", True, False), + ("libxml2", "validation", True, True), + ("turbohtml", "validation", True, False), + ("libxml2", "validation", True, True), + ("turbohtml", "validation", False, True), + ("libxml2", "validation", False, False), + ] + + +@pytest.mark.oracle +@pytest.mark.parametrize( + ("arguments", "exit_status", "findings"), + [ + pytest.param(["--cases", "8"], 0, 0, id="agreement"), + pytest.param(["--cases", "8", "--negative-control"], 1, 18, id="wrong-verdict"), + ], +) +def test_rng_inline_cli( + engine: ModuleType, + capsys: pytest.CaptureFixture[str], + arguments: list[str], + exit_status: int, + findings: int, +) -> None: + assert engine.main(arguments) == exit_status + rows: Final = [json.loads(line) for line in capsys.readouterr().out.splitlines()] + assert rows[-1] == {"summary": {"cases": 8, "rows": 52, "findings": findings}} + assert {(row["engine"], row["phase"]) for row in rows[:-1]} == { + ("turbohtml", "compilation"), + ("libxml2", "compilation"), + ("turbohtml", "validation"), + ("libxml2", "validation"), + } + + +@pytest.mark.oracle +@pytest.mark.parametrize("count", ["0", "257", "-1"]) +def test_rng_inline_cli_rejects_unbounded_cases(engine: ModuleType, count: str) -> None: + with pytest.raises(SystemExit, match="2"): + engine.main(["--cases", count]) + + +@pytest.mark.oracle +@pytest.mark.parametrize( + ("pattern", "documents"), + [ + pytest.param("", (("", True), ("wrong", False)), id="group"), + pytest.param( + 'ok{annotation}', + (("ok", True), ("", False)), + id="choice", + ), + pytest.param( + "{annotation}", + (("", True), ("wrong", False)), + id="interleave", + ), + pytest.param( + '{annotation}', + (('', True), ("", False)), + id="attribute-default-text", + ), + ], +) +def test_rng_inline_foreign_annotations_preserve_pattern_verdicts( + engine: ModuleType, + pattern: str, + documents: tuple[tuple[str, bool], ...], +) -> None: + annotation: Final = '' + body: Final = pattern.replace("{annotation}", annotation) + schema: Final = ( + '' + f'{body}{annotation}' + ) + case: Final = engine.Case(0, schema, compiles=True, documents=documents) + assert [(row.engine, row.phase, row.expected, row.actual) for row in engine.compare(case)] == [ + ("turbohtml", "compilation", True, True), + ("libxml2", "compilation", True, True), + *[(name, "validation", expected, expected) for _, expected in documents for name in ("turbohtml", "libxml2")], + ] + + +@pytest.mark.oracle +def test_rng_inline_foreign_annotation_preserves_explicit_name_class(engine: ModuleType) -> None: + schema: Final = ( + '' + 'v' + ) + case: Final = engine.Case(0, schema, compiles=True, documents=(("", True), ("", False))) + assert [row.actual == row.expected for row in engine.compare(case)] == [True] * 6 diff --git a/tests/test_fuzz_schema_mutation.py b/tests/test_fuzz_schema_mutation.py new file mode 100644 index 000000000..c051bf09a --- /dev/null +++ b/tests/test_fuzz_schema_mutation.py @@ -0,0 +1,161 @@ +from __future__ import annotations + +from typing import TYPE_CHECKING, Final + +import pytest + +if TYPE_CHECKING: + from types import ModuleType + + +@pytest.fixture +def engines() -> tuple[ModuleType, ModuleType, ModuleType, ModuleType, ModuleType]: + return ( + pytest.importorskip("fuzz.schema_mutation", exc_type=ImportError), + pytest.importorskip("fuzz.xslt_inline", exc_type=ImportError), + pytest.importorskip("fuzz.xsd_inline", exc_type=ImportError), + pytest.importorskip("fuzz.rng_inline", exc_type=ImportError), + pytest.importorskip("lxml.etree", exc_type=ImportError), + ) + + +@pytest.mark.oracle +@pytest.mark.parametrize( + ("index", "count"), + [pytest.param(1, 10, id="xslt"), pytest.param(2, 6, id="xsd"), pytest.param(3, 6, id="rng")], +) +@pytest.mark.parametrize("broken", [False, True], ids=["repaired", "broken"]) +def test_schema_mutation_registered_cli( + engines: tuple[ModuleType, ModuleType, ModuleType, ModuleType, ModuleType], + capsys: pytest.CaptureFixture[str], + index: int, + count: int, + *, + broken: bool, +) -> None: + arguments = ["--cases", str(count), "--mutate"] + if broken: + arguments.append("--broken-references") + assert engines[index].main(arguments) == 0 + assert '"findings": 0' in capsys.readouterr().out + + +@pytest.mark.oracle +def test_schema_mutation_xpath_literals_and_declarations( + engines: tuple[ModuleType, ModuleType, ModuleType, ModuleType, ModuleType], +) -> None: + root: Final = ( + '' + '' + '' + 'template' + '' + ) + donor: Final = root.replace("recipient", "donor").replace( + '', + """ + """, + ) + mutated: Final = engines[0].mutate(root, (donor,), 0) + assert str(engines[4].XSLT(engines[4].fromstring(mutated.encode()))(engines[4].fromstring(b""))) == ( + "$literalkepttemplate" + ) + + +@pytest.mark.oracle +@pytest.mark.parametrize( + ("source", "message"), + [ + pytest.param("", "only inline", id="unsupported"), + pytest.param( + '', + "root template", + id="missing-template", + ), + pytest.param( + '', + "global element", + id="missing-element", + ), + pytest.param( + '', + "start pattern", + id="missing-start", + ), + ], +) +def test_schema_mutation_rejects_missing_seed_contract( + engines: tuple[ModuleType, ModuleType, ModuleType, ModuleType, ModuleType], + source: str, + message: str, +) -> None: + with pytest.raises(ValueError, match=message): + engines[0].mutate(source, (source,), 0) + + +@pytest.mark.oracle +@pytest.mark.parametrize("index", [2, 3], ids=["xsd", "rng"]) +def test_schema_mutation_cli_requires_mutation_for_broken_references( + engines: tuple[ModuleType, ModuleType, ModuleType, ModuleType, ModuleType], + index: int, +) -> None: + with pytest.raises(SystemExit, match="2"): + engines[index].main(["--broken-references"]) + + +@pytest.mark.oracle +def test_schema_mutation_rng_missing_recipient_definition_is_not_invented( + engines: tuple[ModuleType, ModuleType, ModuleType, ModuleType, ModuleType], +) -> None: + source: Final = ( + '' + '' + ) + mutated: Final = engines[0].mutate(source, (engines[3].generate(0).schema,), 0) + case: Final = engines[3].Case(0, mutated, compiles=False, documents=()) + assert [(row.engine, row.expected, row.actual) for row in engines[3].compare(case)] == [ + ("turbohtml", False, False), + ("libxml2", False, False), + ] + + +@pytest.mark.oracle +@pytest.mark.parametrize( + ("index", "source"), + [ + pytest.param( + 1, + '' + 'word', + id="x", + ), + pytest.param(2, '', id="xs"), + pytest.param( + 3, + '', + id="rng", + ), + ], +) +def test_schema_mutation_binds_dictionary_prefixes( + engines: tuple[ModuleType, ModuleType, ModuleType, ModuleType, ModuleType], + index: int, + source: str, +) -> None: + mutated: Final = engines[0].mutate(source, (source,), 0) + reference: Final = engines[4].fromstring(mutated.encode()) + if index == 1: + assert str(engines[4].XSLT(reference)(engines[4].fromstring(b""))) == "word" + else: + validator: Final = engines[4].XMLSchema(reference) if index == 2 else engines[4].RelaxNG(reference) + assert validator.validate(engines[4].fromstring(b"")) + + +@pytest.mark.oracle +@pytest.mark.parametrize("source", ["", ""], ids=["empty", "multiple"]) +def test_schema_mutation_requires_one_dictionary_root( + engines: tuple[ModuleType, ModuleType, ModuleType, ModuleType, ModuleType], + source: str, +) -> None: + with pytest.raises(ValueError, match="one root element"): + engines[0].mutate(source, (source,), 0) diff --git a/tests/test_fuzz_xsd_inline.py b/tests/test_fuzz_xsd_inline.py index 31681ae6e..7ea869ef8 100644 --- a/tests/test_fuzz_xsd_inline.py +++ b/tests/test_fuzz_xsd_inline.py @@ -15,7 +15,7 @@ def engine() -> ModuleType: @pytest.mark.oracle -@pytest.mark.parametrize("seed", range(8)) +@pytest.mark.parametrize("seed", range(9)) def test_xsd_inline_generated_labels(engine: ModuleType, seed: int) -> None: case: Final = engine.generate(seed) assert [(row.engine, row.phase, row.expected, row.actual) for row in engine.compare(case)] == [ diff --git a/tests/test_fuzz_xslt_inline.py b/tests/test_fuzz_xslt_inline.py new file mode 100644 index 000000000..7f863905f --- /dev/null +++ b/tests/test_fuzz_xslt_inline.py @@ -0,0 +1,130 @@ +from __future__ import annotations + +import json +from typing import TYPE_CHECKING, Final + +import pytest + +if TYPE_CHECKING: + from types import ModuleType + + +@pytest.fixture +def engines() -> tuple[ModuleType, ModuleType]: + return ( + pytest.importorskip("fuzz.xslt_inline", exc_type=ImportError), + pytest.importorskip("fuzz.schema_mutation", exc_type=ImportError), + ) + + +@pytest.mark.oracle +@pytest.mark.parametrize("seed", range(12)) +def test_xslt_inline_generated_phase_labels(engines: tuple[ModuleType, ModuleType], seed: int) -> None: + engine: Final = engines[0] + case: Final = engine.generate(seed) + rows: Final = list(engine.compare(case)) + assert [(row.engine, row.phase, row.actual == row.expected) for row in rows] == [ + (name, phase, True) + for name in ("turbohtml", "libxslt") + for phase in (("compilation", "application", "result") if case.applies else ("compilation", "application")) + ] + + +@pytest.mark.oracle +@pytest.mark.parametrize("seed", [0, 7]) +def test_xslt_inline_wrong_result_is_a_finding(engines: tuple[ModuleType, ModuleType], seed: int) -> None: + engine: Final = engines[0] + assert [ + (row.engine, row.phase) + for row in engine.compare(engine.generate(seed), negative_control=True) + if row.actual != row.expected + ] == [ + ("turbohtml", "result"), + ] + + +@pytest.mark.oracle +@pytest.mark.parametrize( + ("body", "output"), + [ + pytest.param( + ' alpha beta ', + ' alpha beta ', + id="expanded-names", + ), + pytest.param( + 'notenow', + "", + id="comment-and-instruction", + ), + ], +) +def test_xslt_inline_xml_content_is_preserved(engines: tuple[ModuleType, ModuleType], body: str, output: str) -> None: + engine: Final = engines[0] + stylesheet: Final = ( + '' + '' + body + "" + ) + case: Final = engine.Case(0, stylesheet, "", "xml", compiles=True, applies=True, output=output) + assert [(row.engine, row.phase, row.actual == row.expected) for row in engine.compare(case)] == [ + (name, phase, True) for name in ("turbohtml", "libxslt") for phase in ("compilation", "application", "result") + ] + + +@pytest.mark.oracle +def test_xslt_inline_malformed_stylesheet_retains_unavailable_application( + engines: tuple[ModuleType, ModuleType], +) -> None: + engine: Final = engines[0] + case: Final = engine.Case(0, "<", "", "text", compiles=False, applies=False, output=None) + assert [(row.engine, row.phase, row.expected, row.actual) for row in engine.compare(case)] == [ + ("turbohtml", "compilation", False, False), + ("turbohtml", "application", None, None), + ("libxslt", "compilation", False, False), + ("libxslt", "application", None, None), + ] + + +@pytest.mark.oracle +@pytest.mark.parametrize( + ("arguments", "expected"), + [ + pytest.param([], (0, 0, 20), id="labelled-families"), + pytest.param(["--negative-control"], (1, 10, 20), id="wrong-result"), + pytest.param(["--mutate"], (0, 0, 24), id="repaired-subtrees"), + pytest.param(["--mutate", "--broken-references"], (0, 0, 0), id="broken-references"), + ], +) +def test_xslt_inline_cli_outcome_counts( + engines: tuple[ModuleType, ModuleType], + capsys: pytest.CaptureFixture[str], + arguments: list[str], + expected: tuple[int, int, int], +) -> None: + exit_status, findings, results = expected + assert engines[0].main(arguments) == exit_status + rows: Final = [json.loads(line) for line in capsys.readouterr().out.splitlines()] + assert rows[-1] == { + "summary": { + "cases": 12, + "findings": findings, + "phases": {"compilation": 24, "application": 24, **({"result": results} if results else {})}, + }, + } + + +@pytest.mark.oracle +@pytest.mark.parametrize( + "arguments", + [ + pytest.param(["--cases", "0"], id="empty"), + pytest.param(["--cases", "257"], id="too-many"), + pytest.param(["--broken-references"], id="missing-mutation"), + ], +) +def test_xslt_inline_cli_rejects_invalid_domain( + engines: tuple[ModuleType, ModuleType], + arguments: list[str], +) -> None: + with pytest.raises(SystemExit, match="2"): + engines[0].main(arguments) diff --git a/tests/validate/test_relaxng.py b/tests/validate/test_relaxng.py index 6169b6190..1aba4bc8b 100644 --- a/tests/validate/test_relaxng.py +++ b/tests/validate/test_relaxng.py @@ -232,9 +232,10 @@ def test_missing_start_raises() -> None: RelaxNG(grammar('')) -def test_ref_to_unknown_define_never_matches() -> None: +def test_ref_to_unknown_define_raises() -> None: schema = grammar('') - assert not check(schema, "").valid + with pytest.raises(ValueError, match="has no matching define"): + RelaxNG(schema) def test_malformed_schema_raises() -> None: @@ -346,7 +347,8 @@ def test_rng_many_defines_growth() -> None: def test_rng_many_defines_missing_ref() -> None: defines = "".join(f'' for number in (*range(10), 19, 20)) - assert not rng_ok(rgrammar(f'{defines}'), "") + with pytest.raises(ValueError, match="has no matching define"): + RelaxNG(rgrammar(f'{defines}')) def test_rng_start_requires_two_top_level() -> None: @@ -691,8 +693,8 @@ def test_rng_forbidden_grammar_rejected_at_compile(schema: str, message: str) -> pytest.param( rgrammar(''), "x", - False, - id="foreign-ref-is-unmatchable-content", + True, + id="foreign-ref-is-ignored-annotation", ), ], ) diff --git a/tests/validate/test_relaxng_references.py b/tests/validate/test_relaxng_references.py new file mode 100644 index 000000000..9c33a92ae --- /dev/null +++ b/tests/validate/test_relaxng_references.py @@ -0,0 +1,115 @@ +from __future__ import annotations + +from typing import Final + +import pytest + +from turbohtml import parse_xml +from turbohtml.validate import RelaxNG + +_NAMESPACE: Final = 'xmlns="http://relaxng.org/ns/structure/1.0"' + + +@pytest.mark.parametrize("count", [0, 1, 12], ids=["empty", "linear", "hashed"]) +@pytest.mark.parametrize( + "pattern", + [ + pytest.param('', id="start"), + pytest.param('', id="element"), + ], +) +def test_relaxng_undefined_reference_rejects_compilation(count: int, pattern: str) -> None: + definitions: Final = "".join(f'' for index in range(count)) + schema: Final = f"{pattern}{definitions}" + for _ in range(2): + with pytest.raises(ValueError, match="has no matching define"): + RelaxNG(schema) + + +def test_relaxng_undefined_short_reference_rejects_compilation() -> None: + for _ in range(2): + with pytest.raises(ValueError, match="has no matching define"): + RelaxNG(f'') + + +@pytest.mark.parametrize("count", [0, 12], ids=["linear", "hashed"]) +def test_relaxng_forward_reference_preserves_recursive_validation(count: int) -> None: + definitions: Final = "".join(f'' for index in range(count)) + schema: Final = ( + f'' + '' + f"{definitions}" + ) + validator: Final = RelaxNG(schema) + assert [ + [validator.is_valid(parse_xml(document)) for document in ("", "", "")] + for _ in range(2) + ] == [[True, True, False], [True, True, False]] + + +@pytest.mark.parametrize( + "pattern", + [ + pytest.param('', id="unknown"), + pytest.param("", id="absent-name"), + pytest.param('', id="nested"), + ], +) +def test_relaxng_undefined_unused_reference_rejects_compilation(pattern: str) -> None: + schema: Final = ( + f'' + f'{pattern}' + ) + with pytest.raises(ValueError, match="RELAX NG "): + RelaxNG(schema) + + +@pytest.mark.parametrize( + "definitions", + [ + pytest.param( + '', + id="unused-cycle", + ), + pytest.param( + '' + '', + id="combined-unused-cycle", + ), + ], +) +def test_relaxng_unused_reference_cycle_preserves_reachable_validation(definitions: str) -> None: + schema: Final = ( + f'{definitions}' + ) + validator: Final = RelaxNG(schema) + assert [validator.is_valid(parse_xml(document)) for document in ("", "")] == [True, False] + + +def test_relaxng_undefined_combined_unused_reference_rejects_compilation() -> None: + schema: Final = ( + f'' + '' + '' + ) + with pytest.raises(ValueError, match="has no matching define"): + RelaxNG(schema) + + +def test_relaxng_unused_foreign_annotation_is_ignored() -> None: + schema: Final = ( + f'' + '' + '' + ) + assert RelaxNG(schema).is_valid(parse_xml("")) + + +def test_relaxng_annotation_normalization_preserves_public_source() -> None: + source: Final = parse_xml( + f'' + '' + ) + serialized: Final = source.serialize() + validator: Final = RelaxNG(source) + assert (source.serialize(), validator.is_valid(parse_xml(""))) == (serialized, True) diff --git a/tools/fuzz/rng_inline.py b/tools/fuzz/rng_inline.py new file mode 100644 index 000000000..3e3cd12f5 --- /dev/null +++ b/tools/fuzz/rng_inline.py @@ -0,0 +1,184 @@ +"""Inline labels retain compiler errors alongside document verdicts.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import random +from dataclasses import asdict, dataclass, replace +from typing import TYPE_CHECKING, Final + +import lxml.etree + +from turbohtml import parse_xml +from turbohtml.validate import RelaxNG + +from .schema_mutation import mutate + +if TYPE_CHECKING: + from collections.abc import Iterator, Sequence + +__all__: Final = ["Case", "Verdict", "compare", "generate", "main"] + +_RNG: Final = 'xmlns:rng="http://relaxng.org/ns/structure/1.0"' + + +def main(argv: Sequence[str] | None = None) -> int: + """Keep compiler and validation findings in the exit status.""" + parser: Final = argparse.ArgumentParser(description="Compare bounded inline RELAX NG schemas with libxml2.") + parser.add_argument("--seed", type=int, default=0) + parser.add_argument("--cases", type=int, default=32) + parser.add_argument("--negative-control", action="store_true") + parser.add_argument("--mutate", action="store_true") + parser.add_argument("--broken-references", action="store_true") + arguments: Final = parser.parse_args(argv) + if not 1 <= arguments.cases <= 256: + parser.error("--cases must be between 1 and 256") + if arguments.broken_references and not arguments.mutate: + parser.error("--broken-references requires --mutate") + pool_seeds: Final = (0, 1, 2, 3, 4, 5) + pool: Final = tuple(generate(seed).schema for seed in pool_seeds) + findings = 0 + rows = 0 + for seed in range(arguments.seed, arguments.seed + arguments.cases): + case = generate(pool_seeds[seed % len(pool_seeds)] if arguments.mutate else seed) + if arguments.mutate: + case = replace( + case, + schema=mutate(case.schema, pool, seed, broken=arguments.broken_references), + compiles=not arguments.broken_references, + documents=() if arguments.broken_references else case.documents, + ) + for verdict in compare(case, negative_control=arguments.negative_control, reference_labels=arguments.mutate): + findings += verdict.actual != verdict.expected + rows += 1 + print(json.dumps(asdict(verdict), sort_keys=True)) + print(json.dumps({"summary": {"cases": arguments.cases, "rows": rows, "findings": findings}}, sort_keys=True)) + return int(findings != 0) + + +def compare(case: Case, *, negative_control: bool = False, reference_labels: bool = False) -> Iterator[Verdict]: + """Compile first so invalid schemas remain part of the comparison.""" + schema_hash: Final = hashlib.sha256(case.schema.encode()).hexdigest() + ours: Final = _compile_turbohtml(case.schema) + reference: Final = _compile_lxml(case.schema) + yield Verdict(case.seed, schema_hash, "turbohtml", "compilation", case.compiles, ours is not None) + yield Verdict(case.seed, schema_hash, "libxml2", "compilation", case.compiles, reference is not None) + for document, expected in case.documents: + document_hash: Final = hashlib.sha256((case.schema + "\\0" + document).encode()).hexdigest() + actual: Final = None if ours is None else ours.is_valid(parse_xml(document)) + reference_actual: Final = ( + None if reference is None else reference.validate(lxml.etree.fromstring(document.encode(), _parser())) + ) + expected_result: Final = reference_actual if reference_labels and reference_actual is not None else expected + yield Verdict( + case.seed, + document_hash, + "turbohtml", + "validation", + expected_result, + not actual if negative_control and actual is not None else actual, + ) + yield Verdict( + case.seed, + document_hash, + "libxml2", + "validation", + expected_result, + reference_actual, + ) + + +def generate(seed: int) -> Case: + """Explicit value types keep whitespace semantics independent of defaulting.""" + number: Final = random.Random(seed).randrange(1, 100) + word: Final = f"word{number}" + family: Final = seed % 8 + if family == 0: + pattern: Final = "" + documents: Final = ((f"{word}", True), ("", True), ("", False)) + elif family == 1: + pattern = '' + documents = ((f"{number}", True), ("wrong", False), ("", False)) + elif family == 2: + pattern = f'{word}' + documents = ((f"{word}", True), ("wrong", False), ("", False)) + elif family == 3: + pattern = ( + 'ab' + ) + documents = (("a", True), ("b", True), ("wrong", False)) + elif family == 4: + pattern = ( + '' + '' + ) + documents = ( + (f"{number}", True), + (f"{number}", False), + ("", False), + ) + elif family == 5: + pattern = '' + documents = ((f'', True), ('', False), ("", False)) + else: + schema: Final = ( + f'' + if family == 6 + else "" + ) + return Case(seed, schema, compiles=False, documents=()) + name: Final = f"entry{number}" + return Case( + seed, + f'' + f'' + f'{pattern}', + compiles=True, + documents=documents, + ) + + +def _compile_turbohtml(schema: str) -> RelaxNG | None: + try: + return RelaxNG(schema) + except ValueError: + return None + + +def _compile_lxml(schema: str) -> lxml.etree.RelaxNG | None: + try: + return lxml.etree.RelaxNG(lxml.etree.fromstring(schema.encode(), _parser())) + except (lxml.etree.RelaxNGParseError, lxml.etree.XMLSyntaxError): + return None + + +def _parser() -> lxml.etree.XMLParser: + return lxml.etree.XMLParser(resolve_entities=False, no_network=True) + + +@dataclass(frozen=True) +class Case: + """Expected labels retain shared compiler errors.""" + + seed: int + schema: str + compiles: bool + documents: tuple[tuple[str, bool], ...] + + +@dataclass(frozen=True) +class Verdict: + """Unavailable document results keep compilation failures visible.""" + + seed: int + sha256: str + engine: str + phase: str + expected: bool + actual: bool | None + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/fuzz/schema_mutation.py b/tools/fuzz/schema_mutation.py new file mode 100644 index 000000000..57c944deb --- /dev/null +++ b/tools/fuzz/schema_mutation.py @@ -0,0 +1,128 @@ +"""Seed subtrees retain recipient declarations after reference repair.""" + +from __future__ import annotations + +import copy +import random +import re +from typing import Final + +import lxml.etree + +__all__: Final = ["mutate"] + +_XSL: Final = "{http://www.w3.org/1999/XSL/Transform}" +_XS: Final = "{http://www.w3.org/2001/XMLSchema}" +_RNG: Final = "{http://relaxng.org/ns/structure/1.0}" +_PREFIXES: Final = {"x": _XSL[1:-1], "xs": _XS[1:-1], "rng": "http://relaxng.org/ns/structure/1.0"} + + +def mutate(source: str, pool: tuple[str, ...], seed: int, *, broken: bool = False) -> str: + """Repair donor references against declarations in the recipient tree.""" + root: Final = _parse(source) + donors: Final = tuple(_parse(source) for source in pool) + donor: Final = random.Random(seed).choice( + tuple(candidate for candidate in donors if candidate.tag == root.tag and _method(candidate) == _method(root)) + ) + if root.tag == _XSL + "stylesheet": + _stylesheet(root, donor, broken=broken) + elif root.tag == _XS + "schema": + _schema(root, donor, broken=broken) + elif root.tag == _RNG + "grammar": + _rng(root, donor, broken=broken) + else: + message = "only inline XSLT, XSD and RELAX NG seed trees are supported" + raise ValueError(message) + bound: Final = lxml.etree.Element(str(root.tag), dict(root.attrib), nsmap={**root.nsmap, **_PREFIXES}) + bound[:] = list(root) + return lxml.etree.tostring(bound, encoding="unicode") + + +def _parse(source: str) -> lxml.etree._Element: + wrapper: Final = lxml.etree.fromstring( + ( + '' + source + "" + ).encode(), + lxml.etree.XMLParser(resolve_entities=False, no_network=True), + ) + if len(wrapper) != 1: + message = "seed input requires one root element" + raise ValueError(message) + return wrapper[0] + + +def _method(root: lxml.etree._Element) -> str: + output: Final = root.find(_XSL + "output") + return "xml" if output is None else output.get("method", "xml") + + +def _stylesheet(root: lxml.etree._Element, donor: lxml.etree._Element, *, broken: bool) -> None: + target: Final = root.find(f"{_XSL}template[@match='/']") + replacement: Final = donor.find(f"{_XSL}template[@match='/']") + if target is None or replacement is None: + message = "stylesheet seeds require a root template" + raise ValueError(message) + target[:] = [copy.deepcopy(child) for child in replacement] + variables: Final = [ + name for node in root if node.tag in {_XSL + "variable", _XSL + "param"} and (name := node.get("name")) + ] + templates: Final = [name for node in root if node.tag == _XSL + "template" and (name := node.get("name"))] + keys: Final = [name for node in root if node.tag == _XSL + "key" and (name := node.get("name"))] + for node in target.iter(): + if node.tag == _XSL + "call-template" and templates: + node.set("name", templates[0]) + for attribute in ("select", "test"): + if (expression := node.get(attribute)) is not None: + expression = re.sub( + r"""(?P\bkey\s*\(\s*)(?P['"])[^'"]*(?P=quote)|(?P'[^']*'|"[^"]*")|(?P\$[A-Za-z_][\w.-]*)""", + lambda match: _reference(match, variables, keys), + expression, + ) + node.set(attribute, expression) + if broken: + lxml.etree.SubElement(target, _XSL + "call-template", {"name": "missing"}) + + +def _reference(match: re.Match[str], variables: list[str], keys: list[str]) -> str: + if match.group("variable") is not None and variables: + return "$" + variables[0] + if match.group("key") is not None and keys: + return match.group("key") + match.group("quote") + keys[0] + match.group("quote") + return match.group() + + +def _schema(root: lxml.etree._Element, donor: lxml.etree._Element, *, broken: bool) -> None: + target: Final = root.find(_XS + "element") + replacement: Final = donor.find(_XS + "element") + if target is None or replacement is None: + message = "schema seeds require a global element" + raise ValueError(message) + root.remove(target) + inserted: Final = copy.deepcopy(replacement) + inserted.set("name", target.get("name", "v")) + root.append(inserted) + types: Final = [ + name for node in root if node.tag in {_XS + "simpleType", _XS + "complexType"} and (name := node.get("name")) + ] + for node in inserted.iter(): + for attribute in ("type", "base"): + if (datatype := node.get(attribute)) is not None and not datatype.startswith("xs:"): + node.set(attribute, types[0] if types else "xs:string") + if broken: + inserted.set("type", "missing") + + +def _rng(root: lxml.etree._Element, donor: lxml.etree._Element, *, broken: bool) -> None: + target: Final = root.find(_RNG + "start") + replacement: Final = donor.find(_RNG + "start") + if target is None or replacement is None: + message = "RELAX NG seeds require a start pattern" + raise ValueError(message) + target[:] = [copy.deepcopy(child) for child in replacement] + definitions: Final = [name for node in root if node.tag == _RNG + "define" and (name := node.get("name"))] + for node in target.iter(_RNG + "ref"): + if definitions: + node.set("name", definitions[0]) + if broken: + target[:] = [lxml.etree.Element(_RNG + "ref", {"name": "missing"})] diff --git a/tools/fuzz/xsd_inline.py b/tools/fuzz/xsd_inline.py index f2f140572..bb927a277 100644 --- a/tools/fuzz/xsd_inline.py +++ b/tools/fuzz/xsd_inline.py @@ -6,7 +6,7 @@ import hashlib import json import random -from dataclasses import asdict, dataclass +from dataclasses import asdict, dataclass, replace from typing import TYPE_CHECKING, Final import lxml.etree @@ -14,6 +14,8 @@ from turbohtml import parse_xml from turbohtml.validate import XMLSchema +from .schema_mutation import mutate + if TYPE_CHECKING: from collections.abc import Iterator, Sequence @@ -28,13 +30,27 @@ def main(argv: Sequence[str] | None = None) -> int: parser.add_argument("--seed", type=int, default=0) parser.add_argument("--cases", type=int, default=32) parser.add_argument("--negative-control", action="store_true") + parser.add_argument("--mutate", action="store_true") + parser.add_argument("--broken-references", action="store_true") arguments: Final = parser.parse_args(argv) if not 1 <= arguments.cases <= 256: parser.error("--cases must be between 1 and 256") + if arguments.broken_references and not arguments.mutate: + parser.error("--broken-references requires --mutate") + pool_seeds: Final = (0, 1, 2, 3, 4, 5, 8) + pool: Final = tuple(generate(seed).schema for seed in pool_seeds) findings = 0 rows = 0 for seed in range(arguments.seed, arguments.seed + arguments.cases): - for verdict in compare(generate(seed), negative_control=arguments.negative_control): + case = generate(pool_seeds[seed % len(pool_seeds)] if arguments.mutate else seed) + if arguments.mutate: + case = replace( + case, + schema=mutate(case.schema, pool, seed, broken=arguments.broken_references), + compiles=not arguments.broken_references, + documents=() if arguments.broken_references else case.documents, + ) + for verdict in compare(case, negative_control=arguments.negative_control, reference_labels=arguments.mutate): findings += verdict.actual != verdict.expected rows += 1 print(json.dumps(asdict(verdict), sort_keys=True)) @@ -42,7 +58,7 @@ def main(argv: Sequence[str] | None = None) -> int: return int(findings != 0) -def compare(case: Case, *, negative_control: bool = False) -> Iterator[Verdict]: +def compare(case: Case, *, negative_control: bool = False, reference_labels: bool = False) -> Iterator[Verdict]: """Compile first so invalid schemas remain part of the comparison.""" schema_hash: Final = hashlib.sha256(case.schema.encode()).hexdigest() ours: Final = _compile_turbohtml(case.schema) @@ -52,12 +68,16 @@ def compare(case: Case, *, negative_control: bool = False) -> Iterator[Verdict]: for document, expected in case.documents: document_hash: Final = hashlib.sha256((case.schema + "\\0" + document).encode()).hexdigest() actual: Final = None if ours is None else ours.is_valid(parse_xml(document)) + reference_actual: Final = ( + None if reference is None else reference.validate(lxml.etree.fromstring(document.encode(), _parser())) + ) + expected_result: Final = reference_actual if reference_labels and reference_actual is not None else expected yield Verdict( case.seed, document_hash, "turbohtml", "validation", - expected, + expected_result, not actual if negative_control and actual is not None else actual, ) yield Verdict( @@ -65,8 +85,8 @@ def compare(case: Case, *, negative_control: bool = False) -> Iterator[Verdict]: document_hash, "libxml2", "validation", - expected, - None if reference is None else reference.validate(lxml.etree.fromstring(document.encode(), _parser())), + expected_result, + reference_actual, ) @@ -75,7 +95,7 @@ def generate(seed: int) -> Case: rng: Final = random.Random(seed) number: Final = rng.randrange(1, 100) word: Final = f"word{number}" - family: Final = seed % 8 + family: Final = seed % 9 if family == 0: declarations: Final = '' documents: Final = ((f"{word}", True), ("", True), ("", False)) @@ -107,10 +127,16 @@ def generate(seed: int) -> Case: "" ) documents = ((f"{number}", True), (f"{number + 3}", False), ("wrong", False)) + elif family == 8: + declarations = ( + f'' + f'' + ) + documents = ((f"{number}", True), ("wrong", False), ("", False)) else: declarations = '' if family == 6 else '' documents = () - return Case(seed, f"{declarations}", family < 6, documents) + return Case(seed, f"{declarations}", family < 6 or family == 8, documents) def _compile_turbohtml(schema: str) -> XMLSchema | None: diff --git a/tools/fuzz/xslt_inline.py b/tools/fuzz/xslt_inline.py new file mode 100644 index 000000000..27861b585 --- /dev/null +++ b/tools/fuzz/xslt_inline.py @@ -0,0 +1,217 @@ +"""Compiler and application outcomes keep malformed trees in the differential.""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import random +from dataclasses import asdict, dataclass, replace +from typing import TYPE_CHECKING, Final + +import lxml.etree + +from turbohtml import HTMLParseError, parse_xml +from turbohtml.transform import Transform + +from .schema_mutation import mutate + +if TYPE_CHECKING: + from collections.abc import Iterator, Sequence + +__all__: Final = ["Case", "ResultTree", "Verdict", "compare", "generate", "main"] + + +def main(argv: Sequence[str] | None = None) -> int: + """Keep compiler, application and result findings in the exit status.""" + parser: Final = argparse.ArgumentParser(description="Compare bounded inline XSLT trees with libxslt.") + parser.add_argument("--seed", type=int, default=0) + parser.add_argument("--cases", type=int, default=12) + parser.add_argument("--mutate", action="store_true") + parser.add_argument("--broken-references", action="store_true") + parser.add_argument("--negative-control", action="store_true") + arguments: Final = parser.parse_args(argv) + if not 1 <= arguments.cases <= 256: + parser.error("--cases must be between 1 and 256") + if arguments.broken_references and not arguments.mutate: + parser.error("--broken-references requires --mutate") + findings = 0 + phases: Final[dict[str, int]] = {} + pool: Final = tuple(generate(seed).stylesheet for seed in range(10)) + for seed in range(arguments.seed, arguments.seed + arguments.cases): + case = generate(seed) + if arguments.mutate: + case = replace( + case, + stylesheet=mutate(generate(seed % 10).stylesheet, pool, seed, broken=arguments.broken_references), + compiles=True, + applies=not arguments.broken_references, + output=None, + ) + for row in compare(case, negative_control=arguments.negative_control): + findings += row.expected != row.actual + phases[row.phase] = phases.get(row.phase, 0) + 1 + print(json.dumps(asdict(row), sort_keys=True)) + print(json.dumps({"summary": {"cases": arguments.cases, "findings": findings, "phases": phases}}, sort_keys=True)) + return int(findings != 0) + + +def compare(case: Case, *, negative_control: bool = False) -> Iterator[Verdict]: + """Known labels expose shared mistakes before engine agreement.""" + ours: Final = _turbohtml(case) + reference: Final = _libxslt(case) + digest: Final = hashlib.sha256((case.stylesheet + "\\0" + case.document).encode()).hexdigest() + for engine, outcome in (("turbohtml", ours), ("libxslt", reference)): + yield Verdict(case.seed, digest, engine, "compilation", case.compiles, outcome[0]) + yield Verdict(case.seed, digest, engine, "application", case.applies if case.compiles else None, outcome[1]) + if outcome[1]: + actual: Final = ( + _result("wrong" if case.method == "text" else "", case.method) + if negative_control and engine == "turbohtml" + else outcome[2] + ) + yield Verdict( + case.seed, + digest, + engine, + "result", + reference[2] if case.output is None else _result(case.output, case.method), + actual, + ) + + +def generate(seed: int) -> Case: + """Independent expected results distinguish a wrong reference verdict.""" + number: Final = random.Random(seed).randrange(1, 100) + word: Final = f"word{number}" + variable: Final = f"label{number}" + template: Final = f"render{number}" + key: Final = f"entry{number}" + family: Final = seed % 12 + method: Final = "xml" if family == 7 else "text" + bodies: Final = ( + (f" {word} & beta ", f" {word} & beta "), + ('', "BA"), + ('twoother', "two"), + (f'', word), + (f'', "BA"), + ('', "BA"), + ( + ( + '' + '' + ), + "AB", + ), + ( + f' alpha A beta ', + f' alpha A beta ', + ), + (f"", "A"), + ('', "parameter"), + ('', None), + ('', None), + ) + body, output = bodies[family] + stylesheet: Final = ( + '' + f'' + '' + f'' + f'' + '' + f'{body}' + ) + return Case( + seed, + stylesheet, + 'BA', + method, + family != 10, + family < 10, + output, + ) + + +def _turbohtml(case: Case) -> tuple[bool, bool | None, str | ResultTree | None]: + try: + compiled = Transform(parse_xml(case.stylesheet)) + except (HTMLParseError, ValueError): + return False, None, None + try: + return True, True, _result(compiled(parse_xml(case.document)), case.method) + except (HTMLParseError, ValueError, RuntimeError): + return True, False, None + + +def _libxslt(case: Case) -> tuple[bool, bool | None, str | ResultTree | None]: + parser: Final = lxml.etree.XMLParser(resolve_entities=False, no_network=True) + try: + compiled = lxml.etree.XSLT(lxml.etree.fromstring(case.stylesheet.encode(), parser)) + except (lxml.etree.XSLTParseError, lxml.etree.XMLSyntaxError): + return False, None, None + try: + return True, True, _result(str(compiled(lxml.etree.fromstring(case.document.encode(), parser))), case.method) + except (lxml.etree.XSLTApplyError, lxml.etree.XMLSyntaxError): + return True, False, None + + +def _result(output: str, method: str) -> str | ResultTree: + return ( + output + if method == "text" + else _tree( + lxml.etree.fromstring(output.encode(), lxml.etree.XMLParser(resolve_entities=False, no_network=True)) + ) + ) + + +def _tree(node: lxml.etree._Element) -> ResultTree: + return ResultTree( + node.tag if isinstance(node.tag, str) else "#" + lxml.etree.tostring(node, encoding="unicode", with_tail=False), + tuple(sorted(node.attrib.items())), + node.text or "", + node.tail or "", + tuple(_tree(child) for child in node), + ) + + +@dataclass(frozen=True) +class Case: + """Independent phase labels survive serialization and mutation.""" + + seed: int + stylesheet: str + document: str + method: str + compiles: bool + applies: bool + output: str | None + + +@dataclass(frozen=True) +class ResultTree: + """Expanded names preserve content while allowing serializer variation.""" + + name: str + attributes: tuple[tuple[str, str], ...] + text: str + tail: str + children: tuple[ResultTree, ...] + + +@dataclass(frozen=True) +class Verdict: + """Unavailable applications remain visible after compiler failures.""" + + seed: int + sha256: str + engine: str + phase: str + expected: bool | str | ResultTree | None + actual: bool | str | ResultTree | None + + +if __name__ == "__main__": + raise SystemExit(main()) From 2393a668ff38667226cd8a707a1b6a1957d4870f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Tue, 6 Oct 2026 21:40:49 -0700 Subject: [PATCH 2/3] =?UTF-8?q?=F0=9F=93=9D=20docs(schema):=20name=20refer?= =?UTF-8?q?ence=20fragments?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/changelog/{1016-reference.bugfix.rst => 2716.bugfix.rst} | 0 docs/changelog/{1016-reference.feature.rst => 2716.feature.rst} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename docs/changelog/{1016-reference.bugfix.rst => 2716.bugfix.rst} (100%) rename docs/changelog/{1016-reference.feature.rst => 2716.feature.rst} (100%) diff --git a/docs/changelog/1016-reference.bugfix.rst b/docs/changelog/2716.bugfix.rst similarity index 100% rename from docs/changelog/1016-reference.bugfix.rst rename to docs/changelog/2716.bugfix.rst diff --git a/docs/changelog/1016-reference.feature.rst b/docs/changelog/2716.feature.rst similarity index 100% rename from docs/changelog/1016-reference.feature.rst rename to docs/changelog/2716.feature.rst From 2ec126a5ccddcae3b52179ee9dd3ae867cf98902 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Bern=C3=A1t=20G=C3=A1bor?= Date: Tue, 6 Oct 2026 21:41:24 -0700 Subject: [PATCH 3/3] =?UTF-8?q?=F0=9F=93=9D=20docs(schema):=20reference=20?= =?UTF-8?q?PR=201198?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/changelog/{2716.bugfix.rst => 1198.bugfix.rst} | 0 docs/changelog/{2716.feature.rst => 1198.feature.rst} | 0 2 files changed, 0 insertions(+), 0 deletions(-) rename docs/changelog/{2716.bugfix.rst => 1198.bugfix.rst} (100%) rename docs/changelog/{2716.feature.rst => 1198.feature.rst} (100%) diff --git a/docs/changelog/2716.bugfix.rst b/docs/changelog/1198.bugfix.rst similarity index 100% rename from docs/changelog/2716.bugfix.rst rename to docs/changelog/1198.bugfix.rst diff --git a/docs/changelog/2716.feature.rst b/docs/changelog/1198.feature.rst similarity index 100% rename from docs/changelog/2716.feature.rst rename to docs/changelog/1198.feature.rst