diff --git a/.github/workflows/fuzz.yaml b/.github/workflows/fuzz.yaml index 1c14a06b6..2b4313b3c 100644 --- a/.github/workflows/fuzz.yaml +++ b/.github/workflows/fuzz.yaml @@ -32,6 +32,39 @@ jobs: run: uvx --with tox-uv tox run -e fuzz-smoke env: UV_PYTHON_PREFERENCE: only-managed + oracle: + name: ๐Ÿงช sanitizer wrong-output oracles + runs-on: ubuntu-24.04 + # a sanitize bug usually returns unsafe markup without crashing, so this lane compares the sanitizer with an + # independent re-parse (html5lib-python, parse5 to confirm), the random Policy it ran under, and the WHATWG URL + # scheme. Pull requests and pushes replay only the deterministic seed pass; scheduled and manual runs add + # MutaGen-style generation for a wall-clock budget. + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + fetch-depth: 0 + persist-credentials: false + - name: ๐ŸŸข Install Node + # zizmor: ignore[cache-poisoning] no cache: input is set, so setup-node writes no restorable cache + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version: "22" + - name: ๐Ÿ“ฆ Install parse5 for the confirming re-parse + run: npm ci + working-directory: tools/bench/node + - name: ๐Ÿ”„ Install uv + uses: astral-sh/setup-uv@c18668ad3cf93ea998bef934396af7bb5c839dc7 # v10.2.0 + with: + enable-cache: false + - name: โœ… Run fuzz-oracle + # the seed regenerates every finding from the public code, so it stays out of the tox command line it echoes + run: | + FUZZ_RNG_SEED="$(python3 -c 'import secrets; print(secrets.randbits(32))')" + export FUZZ_RNG_SEED + uvx --with tox-uv tox run -e fuzz-oracle -- --minutes "${MINUTES}" + env: + UV_PYTHON_PREFERENCE: only-managed + MINUTES: ${{ (github.event_name == 'pull_request' || github.event_name == 'push') && '0' || github.event.schedule == '0 4 * * 1' && '20' || '8' }} deep: name: ๐Ÿ•ณ๏ธ ASan/UBSan deep (mutation + structural) runs-on: ubuntu-24.04 diff --git a/.gitmodules b/.gitmodules index 55a805f24..6c2546d74 100644 --- a/.gitmodules +++ b/.gitmodules @@ -70,3 +70,16 @@ path = tests/conformance/linkify-it url = https://github.com/markdown-it/linkify-it shallow = true +[submodule "tools/fuzz-data/h5sc"] + path = tools/fuzz-data/h5sc + url = https://github.com/cure53/H5SC + shallow = true +[submodule "tools/fuzz-data/google-fuzzing"] + path = tools/fuzz-data/google-fuzzing + url = https://github.com/google/fuzzing + shallow = true +[submodule "tools/fuzz-data/wpt"] + path = tools/fuzz-data/wpt + url = https://github.com/web-platform-tests/wpt + shallow = true + update = none diff --git a/docs/development/index.rst b/docs/development/index.rst index 5c7128c19..102b56d5d 100644 --- a/docs/development/index.rst +++ b/docs/development/index.rst @@ -165,6 +165,17 @@ To read the crashers of a failed run, download its ``fuzz-crashes`` artifact and $ gh run download --name fuzz-crashes $ age --decrypt --identity fuzz-crashes.key --output crash- crash-.age +A sanitizer bug usually returns unsafe or altered markup without crashing, so ``fuzz-oracle`` (``fuzz.py --mode +oracle``) checks ``turbohtml.clean.sanitize`` against code it does not share. The output must re-parse in +html5lib-python into the tree the sanitizer judged, obey the random ``Policy`` it ran under, and keep or drop each URL +the way the WHATWG URL parser reads its scheme. ``--minutes 0`` replays only the seed corpora. The run writes findings +to a JSON report and logs only their hashes. + +.. code-block:: console + + $ tox r -e fuzz-oracle -- --minutes 0 # the per-PR seed pass + $ tox r -e fuzz-oracle -- --minutes 10 # adds generated markup and URL obfuscations + Add a target by registering a ``bytes``-taking callable in ``_TARGETS`` (in-process) and dropping a representative benign seed under ``tools/fuzz/corpus//``; add a standalone harness by mirroring ``idna_harness.c`` for any C unit that compiles free of the CPython boundary. macOS ships no ``libFuzzer`` runtime with Apple Clang, so the diff --git a/pyproject.toml b/pyproject.toml index 15cce1615..8e560bfa4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -264,6 +264,7 @@ extend-exclude = [ "tests/conformance/python-phonenumbers", "tests/conformance/unicodetools", "tests/html5lib-tests", + "tools/fuzz-data", "tools/html5lib-python" ] # vendored suites, not our code format.preview = true @@ -348,6 +349,7 @@ src.exclude = [ "tests/conformance/python-phonenumbers", "tests/conformance/unicodetools", "tests/html5lib-tests", + "tools/fuzz-data", "tools/html5lib-python" ] # vendored suites, not our code environment.extra-paths = [ diff --git a/src/turbohtml/_c/serialize/minify.c b/src/turbohtml/_c/serialize/minify.c index c2d4b7bee..e19191ab7 100644 --- a/src/turbohtml/_c/serialize/minify.c +++ b/src/turbohtml/_c/serialize/minify.c @@ -571,9 +571,7 @@ static int mini_emit_script_js(sbuf *out, th_tree *tree, th_node *node, const th } Py_ssize_t pos = 0; for (th_node *child = node->first_child; child != NULL; child = child->next_sibling) { - if (child->text_len > 0) { /* a parsed empty comment has a NULL text pointer */ - memcpy(src + pos, need_text(tree, child), (size_t)child->text_len * sizeof(Py_UCS4)); - } + memcpy(src + pos, need_text(tree, child), (size_t)child->text_len * sizeof(Py_UCS4)); pos += child->text_len; } /* errlen 0: the HTML path discards the message and falls back to verbatim instead */ @@ -614,9 +612,7 @@ static int mini_emit_style_css(sbuf *out, th_tree *tree, th_node *node, int base } Py_ssize_t pos = 0; for (th_node *child = node->first_child; child != NULL; child = child->next_sibling) { - if (child->text_len > 0) { /* a parsed empty comment has a NULL text pointer */ - memcpy(src + pos, need_text(tree, child), (size_t)child->text_len * sizeof(Py_UCS4)); - } + memcpy(src + pos, need_text(tree, child), (size_t)child->text_len * sizeof(Py_UCS4)); pos += child->text_len; } Py_ssize_t css_len; diff --git a/src/turbohtml/_c/tokenizer/xml.c b/src/turbohtml/_c/tokenizer/xml.c index 0b2ecacc3..8a3d1519f 100644 --- a/src/turbohtml/_c/tokenizer/xml.c +++ b/src/turbohtml/_c/tokenizer/xml.c @@ -396,15 +396,14 @@ static int push_open(xml_parser *parser, th_node *element) { return 0; } -/* Copy a code-point run (an entity-normalized attribute value) into the arena. */ +/* Copy a namespace URI into the arena. consume_namespace_decl rejects an empty one first, so src is never the + NULL scratch buffer. */ static Py_UCS4 *arena_copy(th_tree *tree, const Py_UCS4 *src, Py_ssize_t len) { Py_UCS4 *out = arena_alloc(tree, len * (Py_ssize_t)sizeof(Py_UCS4)); if (out == NULL) { /* GCOVR_EXCL_BR_LINE: allocation failure cannot be forced from a test */ return NULL; /* GCOVR_EXCL_LINE: allocation-failure path */ } - if (len > 0) { /* the scratch buffer stays NULL until a value pushes a code point */ - memcpy(out, src, (size_t)len * sizeof(Py_UCS4)); - } + memcpy(out, src, (size_t)len * sizeof(Py_UCS4)); return out; } diff --git a/tests/serialize/test_minify.py b/tests/serialize/test_minify.py index 064efc128..5573999b2 100644 --- a/tests/serialize/test_minify.py +++ b/tests/serialize/test_minify.py @@ -475,12 +475,8 @@ def test_body_start_omitted_with_empty_first_text() -> None: pytest.param("style", "a { color: red }", Minify(minify_css=CSSMinify()), "a{color:red}", id="style"), ], ) -def test_minify_raw_text_skips_an_empty_comment_child(tag: str, source: str, layout: Minify, expected: str) -> None: - # a parsed empty comment has a NULL text pointer, which memcpy may not receive even for length 0 - document: Final = parse(f"<{tag}>{source}") - element = document.find(tag) - assert isinstance(element, Element) - element.insert(0, document.children[0]) +def test_minify_raw_text_joins_an_empty_text_child(tag: str, source: str, layout: Minify, expected: str) -> None: + element: Final = Element(tag, children=[Text(""), Text(source)]) assert element.serialize(Html(layout=layout)) == f"<{tag}>{expected}" diff --git a/tests/tokenizer/test_xml.py b/tests/tokenizer/test_xml.py index d2136f6af..92931c82f 100644 --- a/tests/tokenizer/test_xml.py +++ b/tests/tokenizer/test_xml.py @@ -221,11 +221,6 @@ def test_attribute_value(markup: str, expected: str) -> None: assert dict(root_of(parse_xml(markup)).attrs) == {"a": expected} -def test_empty_prefixed_namespace_declaration_is_kept() -> None: - # the empty URI is copied from a NULL scratch buffer, which memcpy may not receive even for length 0 - assert dict(root_of(parse_xml('')).attrs) == {"xmlns:p": ""} - - @pytest.mark.parametrize( ("markup", "expected"), [ diff --git a/tools/bench/node/parse5_tree_runner.js b/tools/bench/node/parse5_tree_runner.js new file mode 100644 index 000000000..43c46d1a6 --- /dev/null +++ b/tools/bench/node/parse5_tree_runner.js @@ -0,0 +1,51 @@ +// Second independent re-parser for the sanitizer mutation-freedom oracle (tools/fuzz/sanitize_oracles.py). Reads a JSON +// array of HTML strings on stdin, parses each as a
fragment with scripting on (what innerHTML does), and writes a +// JSON array of html5lib-tests tree dumps. The dump mirrors parse5's own test serializer +// (tests/conformance/parse5/test/utils/serialize-to-dat-file-format.ts), which is TypeScript and not shipped in the npm +// package, so the Python side compares parse5, html5lib-python and turbohtml in one format. +const { defaultTreeAdapter: adapter, html, parseFragment } = require("parse5"); + +const PREFIXES = { [html.NS.SVG]: "svg ", [html.NS.MATHML]: "math " }; + +function dump(nodes, indent, out) { + const pad = "|".padEnd(indent + 2, " "); + for (let node of nodes) { + if (adapter.isCommentNode(node)) { + out.push(`${pad}`); + } else if (adapter.isTextNode(node)) { + out.push(`${pad}"${adapter.getTextNodeContent(node)}"`); + } else { + const tag = adapter.getTagName(node); + const namespace = adapter.getNamespaceURI(node); + out.push(`${pad}<${PREFIXES[namespace] ?? ""}${tag}>`); + let childIndent = indent + 2; + const attrPad = "|".padEnd(childIndent + 2, " "); + out.push( + ...adapter + .getAttrList(node) + .map((attr) => `${attrPad}${attr.prefix ? `${attr.prefix} ` : ""}${attr.name}="${attr.value}"`) + .sort(), + ); + if (tag === "template" && namespace === html.NS.HTML) { + out.push(`${attrPad}content`); + childIndent += 2; + node = adapter.getTemplateContent(node); + } + dump(adapter.getChildNodes(node), childIndent, out); + } + } + return out; +} + +let input = ""; +process.stdin.setEncoding("utf8"); +process.stdin.on("data", (chunk) => { + input += chunk; +}); +process.stdin.on("end", () => { + const dumps = JSON.parse(input).map((markup) => { + const context = adapter.createElement("div", html.NS.HTML, []); + return dump(adapter.getChildNodes(parseFragment(context, markup, { scriptingEnabled: true })), 0, []).join("\n"); + }); + process.stdout.write(JSON.stringify(dumps)); +}); diff --git a/tools/fuzz-data/google-fuzzing b/tools/fuzz-data/google-fuzzing new file mode 160000 index 000000000..734e55f3c --- /dev/null +++ b/tools/fuzz-data/google-fuzzing @@ -0,0 +1 @@ +Subproject commit 734e55f3cfed1adbb51bf6cb5c65b4c1197b7089 diff --git a/tools/fuzz-data/h5sc b/tools/fuzz-data/h5sc new file mode 160000 index 000000000..1b796a63b --- /dev/null +++ b/tools/fuzz-data/h5sc @@ -0,0 +1 @@ +Subproject commit 1b796a63b5ad8ca4734c52ed9cbe1035eb6f494b diff --git a/tools/fuzz-data/wpt b/tools/fuzz-data/wpt new file mode 160000 index 000000000..c5e80ef1d --- /dev/null +++ b/tools/fuzz-data/wpt @@ -0,0 +1 @@ +Subproject commit c5e80ef1dca982bfab50c54277fdf99bfb8c9a62 diff --git a/tools/fuzz/fuzz.py b/tools/fuzz/fuzz.py index ea8e8b905..d3171d288 100644 --- a/tools/fuzz/fuzz.py +++ b/tools/fuzz/fuzz.py @@ -13,9 +13,10 @@ ``smoke`` replays the past finds under ``tests/fuzz_regressions`` and a benign seed corpus once (fast, deterministic, gates every PR). ``deep`` adds a mutation loop and structural probes for a per-target budget (the scheduled/manual run -that hunts for crashes). A crashing input lands in ``--crash-dir`` as ``crash-``, and the log names it only by -hash, length, harness and seeds, because CI logs on a public repository are public. The in-process extension is -expected to be pre-built by the tox env; ``--build`` builds it here for a local run. +that hunts for crashes). ``oracle`` runs the sanitizer wrong-output oracles (``sanitize_oracles.py``) instead, +because a sanitizer bug usually returns unsafe markup without crashing. A crashing input lands in ``--crash-dir`` as +``crash-``, and the log names it only by hash, length and harness, because CI logs on a public repository are +public. The in-process extension is expected to be pre-built by the tox env; ``--build`` builds it here for a local run. """ from __future__ import annotations @@ -41,12 +42,13 @@ # pymalloc carves small objects out of pools, so an over-read that stays inside a pool never reaches ASan's redzones; # PYTHONMALLOC=malloc hands every PyMem call to the intercepted system allocator (v1 FUZZ-2 was silent under pymalloc) _ALLOCATORS: Final[tuple[str, ...]] = ("pymalloc", "malloc") +_WPT: Final = "tools/fuzz-data/wpt" def main() -> int: """Return 0 when every harness stays clean, nonzero on the first sanitizer abort or soft finding.""" parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--mode", choices=("smoke", "deep"), default="smoke") + parser.add_argument("--mode", choices=("smoke", "deep", "oracle"), default="smoke") parser.add_argument( "--minutes", type=float, default=1.0, help="deep-mode budget per in-process target, split across the allocators" ) @@ -65,6 +67,8 @@ def main() -> int: if args.build: _build_extension(Path(tempfile.mkdtemp(prefix="th-fuzz-build-"))) args.crash_dir.mkdir(parents=True, exist_ok=True) + if args.mode == "oracle": + return _run_oracles(args.minutes, args.rng_seed, args.crash_dir) if (code := _run_standalone(args.mode, args.extra_corpus, args.crash_dir)) != 0: return code return 0 if args.skip_inprocess else _run_inprocess(args.mode, args.minutes, args.rng_seed, args.crash_dir) @@ -89,6 +93,55 @@ def _build_extension(build_dir: Path) -> None: subprocess.run(cmd, check=True, env={**os.environ, "CC": _CC}) +def _run_oracles(minutes: float, rng_seed: int, crash_dir: Path) -> int: + """Run the sanitizer oracles under the ASan preload, so a C fault on the way still aborts with a stack trace.""" + _checkout_sparse_wpt() + env = _asan_preload() + if minutes > 0: + env = _private_reports(env, crash_dir) + result = subprocess.run( + [ + sys.executable, + str(_FUZZ / "sanitize_oracles.py"), + "--minutes", + str(minutes), + "--rng-seed", + str(rng_seed), + "--repro", + str(crash_dir / "current_input.html"), + "--report", + str(crash_dir / "crash-findings.json"), + ], + env=env, + check=False, + ) + if result.returncode not in {0, 1, 2}: + print(f"SANITIZER ABORT in the oracle run (exit {result.returncode})", file=sys.stderr) + return result.returncode + + +def _checkout_sparse_wpt() -> None: + """ + Check out only ``sanitizer-api/`` of the pinned WPT submodule. + + A depth-1 WPT clone is about 1 GB, so ``.gitmodules`` marks the submodule ``update = none`` (plain and recursive + ``git submodule update`` skip it) and this fetches the recorded commit blob-less with a sparse checkout instead. + """ + if (target := _ROOT / _WPT / "sanitizer-api").is_dir(): + return + commit = _git("rev-parse", f"HEAD:{_WPT}") + url = _git("config", "--file", ".gitmodules", f"submodule.{_WPT}.url") + _git("init", "--quiet", _WPT) + _git("-C", _WPT, "fetch", "--quiet", "--depth", "1", "--filter=blob:none", url, commit) + _git("-C", _WPT, "sparse-checkout", "set", "--no-cone", "/sanitizer-api/", "/LICENSE.md") + _git("-C", _WPT, "checkout", "--quiet", "FETCH_HEAD") + print(f"checked out {target.relative_to(_ROOT)} at {commit[:12]}") + + +def _git(*args: str) -> str: + return subprocess.run(["git", *args], cwd=_ROOT, capture_output=True, text=True, check=True).stdout.strip() + + def _run_standalone(mode: str, extra: Path | None, crash_dir: Path) -> int: work = Path(tempfile.mkdtemp(prefix="th-fuzz-")) idna = work / "idna_harness" diff --git a/tools/fuzz/sanitize_oracles.py b/tools/fuzz/sanitize_oracles.py new file mode 100644 index 000000000..c2718016e --- /dev/null +++ b/tools/fuzz/sanitize_oracles.py @@ -0,0 +1,1046 @@ +""" +Wrong-output oracles for ``turbohtml.clean.sanitize``: the ``--mode oracle`` lane of ``tools/fuzz/fuzz.py``. + +A sanitizer bug usually returns unsafe or altered markup without crashing, so each oracle compares the sanitizer with +code it does not share. ``mutation`` re-parses the output with the vendored html5lib-python and requires the tree the +sanitizer judged; ``policy`` requires that re-parsed tree to obey the random Policy that produced it; ``url`` compares +each keep or drop of a URL attribute with the scheme turbohtml's WHATWG ``normalize_url`` reads from the decoded value. +A negative control proves every oracle still fires before the run starts. The run minimizes each divergence and +re-checks the re-parse ones with parse5, which tells a stale html5lib rule apart from a turbohtml mutation. +""" + +from __future__ import annotations + +import argparse +import dataclasses +import hashlib +import html +import importlib.util +import json +import os +import random +import re +import shutil +import subprocess +import sys +import time +from collections import Counter +from collections.abc import Callable, Mapping +from dataclasses import dataclass +from functools import partial +from pathlib import Path +from typing import TYPE_CHECKING, Final, TypeVar +from urllib.parse import quote + +import html5lib +from html5lib.constants import namespaces + +from turbohtml import Comment, Element, Namespace, Node, Text, parse_fragment +from turbohtml.clean import DEFAULT_CSS_PROPERTIES, OnDisallowed, Policy, sanitize, sanitize_node +from turbohtml.extract import normalize_url + +if TYPE_CHECKING: + from collections.abc import Iterator, Sequence + from xml.dom.minidom import DocumentFragment as DomFragment + from xml.dom.minidom import Element as DomElement + +_T = TypeVar("_T") +_Check = Callable[[str], str | None] +_PolicyValue = frozenset[str] | Mapping[str, frozenset[str]] | bool | OnDisallowed + +_ROOT: Final[Path] = Path(__file__).resolve().parents[2] +_DATA: Final[Path] = _ROOT / "tools" / "fuzz-data" +_NODE_DIR: Final[Path] = _ROOT / "tools" / "bench" / "node" +# sanitize(str) parses its input as a
fragment with scripting off (clean/sanitize.c, th_tree_parse_fragment) +_CONTEXT: Final = "div" +# TH_MAX_TREE_DEPTH (dom/tree.h): past it turbohtml, Blink and Gecko flatten nesting while html5lib and parse5 do not +_DEPTH_CAP: Final = 512 +# meriyah's MINIMUM_COMPARED_RATIO: a run that skips more than this share compared too little to mean anything +_MIN_COMPARED_RATIO: Final = 0.9 +_MINIMIZE_BUDGET: Final = 1500 +_DOM_BUILDER: Final = html5lib.getTreeBuilder("dom") +_NAMESPACE_PREFIX: Final[dict[str, str]] = {namespaces["svg"]: "svg", namespaces["mathml"]: "math"} +_C0_OR_SPACE: Final = "".join(map(chr, range(0x21))) +_SCHEME: Final = re.compile(r"([a-z][a-z0-9+.-]*):") +_UNSAFE_HTML: Final = frozenset({ + "script", + "iframe", + "embed", + "object", + "noscript", + "noembed", + "noframes", + "base", + "basefont", + "title", + "plaintext", + "xmp", + "template", +}) +_SVG_ANIMATION: Final = frozenset({"animate", "animateColor", "animateMotion", "animateTransform", "set"}) +_URL_ATTRS: Final = frozenset({ + "href", + "src", + "action", + "formaction", + "xlink:href", + "poster", + "background", + "cite", + "data", + "ping", + "longdesc", +}) +_CSS_DANGER: Final = ("javascript:", "vbscript:", "expression(", "behavior:", "-moz-binding") +_FAILING_ORACLES: Final = frozenset({"mutation", "string-path", "policy", "url-unsafe", "decode"}) + + +def main() -> int: + """Return 0 when every oracle agrees, 1 on a failing finding, 2 when the run proved nothing.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--minutes", type=float, default=1.0, help="generation budget after the seed pass") + parser.add_argument("--rng-seed", type=int, default=0) + parser.add_argument("--repro", type=Path, required=True, help="each input is written here before it runs") + parser.add_argument("--report", type=Path, required=True, help="JSON file receiving every finding with its repro") + args = parser.parse_args() + + if blind := [name for name, fired in _negative_controls() if not fired]: + print(f"BLIND ORACLE: the negative control did not fire for {', '.join(blind)}", file=sys.stderr) + return 2 + rng = random.Random(args.rng_seed) + run = _Run(args.repro) + seeds = _seeds(rng) + for markup in seeds: + for policy in (Policy(), Policy.relaxed(), _draw_policy(rng)): + run.markup(markup, policy) + for wpt_markup, wpt_policy in _wpt_cases(): + run.markup(wpt_markup, wpt_policy) + tokens = _dictionary_tokens() + deadline = time.monotonic() + args.minutes * 60 + while time.monotonic() < deadline: + if (draw := rng.random()) < 0.5: + run.markup(_mutagen(rng, seeds), _draw_policy(rng)) + elif draw < 0.75: + run.markup(_splice(rng, rng.choice(seeds), tokens), _draw_policy(rng)) + else: + run.url(*_url_case(rng)) + print(f"oracle: {dict(run.stats)}") + if run.stats["compared"] < _MIN_COMPARED_RATIO * (run.stats["compared"] + run.stats["skipped"]): + print(f"VACUOUS RUN: compared below {_MIN_COMPARED_RATIO:.0%} of attempted cases", file=sys.stderr) + return 2 + findings = _confirm_with_parse5(run.minimized()) + args.report.write_text(json.dumps([dataclasses.asdict(finding) for finding, _ in findings], indent=2) + "\n") + for finding, _ in findings: + # only a hash reaches the log: the repro may be an unreported bypass (DESIGN-v2 decision D5) + digest = hashlib.sha256(finding.signature.encode()).hexdigest()[:12] + print(f"FINDING {finding.oracle} {digest} x{finding.hits} parse5={finding.parse5}", file=sys.stderr) + print(f"{len(findings)} finding(s) written to {args.report}") + return int(any(finding.oracle in _FAILING_ORACLES for finding, _ in findings)) + + +def _negative_controls() -> Iterator[tuple[str, bool]]: + """ + Feed each oracle a known-bad output and report whether it fired. + + Modeled on DOMPurify's ``ADD_ATTR: ['onerror']`` control (test/fuzz/sanitize.fast-check.js:357-376) and Fuzzilli's + startup tests: an oracle that stops detecting fails the run instead of passing silently. + """ + escaped = parse_fragment("

<img src=x onerror=alert(1)>

", _CONTEXT, positions=False) + # a serializer that forgets to escape text, the SI7 class of MutaGen's detectors + yield "mutation", _mutation_detail(escaped, html.unescape(escaped.inner_html)) is not None + select = parse_fragment("", _CONTEXT, positions=False) + yield "mutation in select", _mutation_detail(select, html.unescape(select.inner_html)) is not None + yield "policy on*", "event handler img@onerror" in _violations("", Policy()) + yield "policy url", "url a@href" in _violations('x', Policy()) + animation = Policy(tags=frozenset({"svg", "animate"}), attributes={"*": frozenset({"attributeName", "to"})}) + yield "policy baseline", "baseline element svg:animate" in _violations("", animation) + yield "policy css", "dangerous css in b@style" in _violations('', Policy()) + yield "url unsafe keep", _url_verdict("javascript:alert(1)", kept=True, policy=Policy()) == "url-unsafe" + yield "url strict drop", _url_verdict("https://example.com/", kept=False, policy=Policy()) == "url-strict" + + +def _mutation_detail(tree: Node, serialized: str) -> str | None: + """ + Describe where the independent re-parse of ``serialized`` first departs from ``tree``, or return None. + + The description names node shapes, not their text, so divergences that differ only in payload share a signature. + """ + expected = _normalized_dump(tree) + # html5lib-python and parse5 8.0 predate the customizable
), so only turbohtml can re-parse a select + if any(element.tag == "select" and element.namespace is Namespace.HTML for element in _elements(tree)): + reference = "turbohtml-reparse" + reparsed = _normalized_dump(parse_fragment(serialized, _CONTEXT, positions=False, scripting=True)) + else: + reference = "html5lib" + reparsed = _dump_html5lib(_reparse(serialized)) + if expected == reparsed: + return None + return next( + f"turbohtml {_shape(want)} != {reference} {_shape(got)}" + for want, got in zip([*expected.split("\n"), None], [*reparsed.split("\n"), None], strict=False) + if want != got + ) + + +def _normalized_dump(tree: Node) -> str: + # the serialization algorithm emits CR as is and input preprocessing turns it into LF, so no serializer keeps it + return "\n".join(_dump_turbohtml(tree.children)).replace("\r\n", "\n").replace("\r", "\n") + + +def _dump_turbohtml(nodes: Sequence[Node], depth: int = 0) -> list[str]: + """ + Render turbohtml nodes in the html5lib-tests tree format. + + Modeled on tests/dom/test_treebuilder_conformance.py ``_dump_node``. Adjacent text merges and empty text drops, + since no parser produces either and html5lib's ``testSerializer`` normalizes them away. + """ + pad = "| " + " " * depth + out: list[str] = [] + pending_text = "" + for node in nodes: + if isinstance(node, Text): + pending_text += node.data + continue + if pending_text: + out.append(f'{pad}"{pending_text}"') + pending_text = "" + if isinstance(node, Element): + foreign = node.namespace is not Namespace.HTML + out.append(f"{pad}<{node.namespace.value + ' ' if foreign else ''}{node.tag}>") + out.extend( + sorted( + f'{pad} {_attr_name(name, foreign=foreign)}="{_attr_text(value)}"' + for name, value in node.attrs.items() + ) + ) + if not foreign and node.tag == "template": + out.append(f"{pad} content") + out.extend(_dump_turbohtml(node.children[0].children, depth + 2)) + else: + out.extend(_dump_turbohtml(node.children, depth + 1)) + elif isinstance(node, Comment): + out.append(f"{pad}") + if pending_text: + out.append(f'{pad}"{pending_text}"') + return out + + +def _attr_name(name: str, *, foreign: bool) -> str: + """Spell a namespaced attribute on a foreign element as prefix and local name apart, the .dat way.""" + return name.replace(":", " ", 1) if foreign and name.startswith(("xlink:", "xml:", "xmlns:")) else name + + +def _attr_text(value: str | list[str] | None) -> str: + return " ".join(value) if isinstance(value, list) else value or "" + + +def _elements(root: Node) -> Iterator[Element]: + stack = [root] + while stack: + node = stack.pop() + if isinstance(node, Element): + yield node + stack.extend(node.children) + + +def _reparse(serialized: str) -> DomFragment: + """Parse ``serialized`` the way ``innerHTML`` assignment does: a
fragment with scripting on.""" + parser = html5lib.HTMLParser(tree=_DOM_BUILDER, namespaceHTMLElements=False) + return parser.parseFragment(serialized, container=_CONTEXT, scripting=True) + + +def _dump_html5lib(fragment: DomFragment) -> str: + """ + Render an html5lib tree in the html5lib-tests format with html5lib's own ``testSerializer``. + + It nests fragment children one column deeper than the .dat files, the offset ``convertTreeDump`` strips in + html5lib/tests/tree_construction.py; continuation lines of a multi-line text node carry no bar and stay as they are. + """ + lines = _DOM_BUILDER(namespaceHTMLElements=False).testSerializer(fragment).split("\n")[1:] + return "\n".join("| " + line[3:] if line.startswith("|") else line for line in lines) + + +def _shape(line: str | None) -> str: + if line is None: + return "#end" + body = line.lstrip("| ") + if not line.startswith("|") or body.startswith('"'): + return "#text" # a line without the bar continues a multi-line text node + if body.startswith("]]>]
`` shell vectors.txt puts around each.""" + text = _require(_DATA / "h5sc" / "vectors.txt").read_text(encoding="utf-8") + return re.findall(r'
(.*?)//\["\'`-->\]\]>\]
', text, flags=re.DOTALL) + + +def _dharma_payloads(rng: random.Random) -> list[str]: + """ + Generate payloads from Dharma's ``xss.dg`` grammar with the installed dharma package, unchanged. + + The child process takes a seed drawn from ``rng``, so the batch is reproducible from ``--rng-seed``. + """ + if (spec := importlib.util.find_spec("dharma")) is None or spec.origin is None: + msg = "the dharma package, which ships xss.dg, is missing; tox -e fuzz-oracle installs it" + raise ModuleNotFoundError(msg) + grammar = Path(spec.origin).parent / "grammars" / "xss.dg" + seed = str(rng.randrange(2**31)) + command = [ + sys.executable, + "-m", + "dharma", + "-grammars", + str(grammar), + "-count", + "200", + "-logging", + "40", + "-seed", + seed, + ] + output = subprocess.run(command, capture_output=True, text=True, check=True).stdout + return [line for line in output.splitlines() if line.strip()] + + +def _draw_policy(rng: random.Random) -> Policy: + """ + Draw a Policy from a base allowlist plus the hazard pool most sanitizer bypasses need. + + The pool is DESIGN-v2 ยง4d's: foreign content, raw-text and RCDATA elements, SVG animation, ``style``, URL-bearing + attributes and custom schemes, since 11 of MutaGen's 16 bypasses needed a relaxed configuration. + """ + tags = { + *rng.sample(sorted(Policy.relaxed().tags), rng.randint(0, 12)), + *rng.sample(_HAZARD_TAGS, rng.randint(1, 8)), + } + return Policy( + tags=frozenset(tags), + attributes={ + "*": frozenset(rng.sample(_HAZARD_ATTRS, rng.randint(0, 6))), + **{ + tag: frozenset(rng.sample(_HAZARD_ATTRS, rng.randint(1, 4))) + for tag in rng.sample(sorted(tags), min(3, len(tags))) + }, + }, + url_schemes=frozenset(rng.sample(_SCHEMES, rng.randint(0, 4))), + allow_relative_urls=rng.random() < 0.5, + allow_fragment_urls=rng.random() < 0.5, + on_disallowed_tag=rng.choice(list(OnDisallowed)), + strip_comments=rng.random() < 0.5, + remove_with_content=frozenset(rng.sample(_HAZARD_TAGS, rng.randint(0, 3))), + css_properties=DEFAULT_CSS_PROPERTIES | {"background-image", "list-style-image", "behavior"}, + attribute_prefixes=frozenset({"data-"}) if rng.random() < 0.3 else frozenset(), + allow_html=rng.random() < 0.9, + allow_svg=rng.random() < 0.8, + allow_mathml=rng.random() < 0.8, + ) + + +_HAZARD_TAGS: Final = ( + "svg", + "math", + "annotation-xml", + "foreignObject", + "desc", + "mtext", + "mglyph", + "mi", + "template", + "xmp", + "textarea", + "title", + "select", + "option", + "animate", + "set", + "style", + "noscript", + "iframe", + "a", + "form", + "button", + "img", + "image", + "use", + "table", + "noembed", + "noframes", + "listing", + "pre", + "plaintext", +) +_HAZARD_ATTRS: Final = ( + *sorted(_URL_ATTRS), + "style", + "onerror", + "onclick", + "attributeName", + "from", + "to", + "values", + "encoding", + "id", + "name", + "title", + "class", + "is", + "rel", + "type", +) +_SCHEMES: Final = ("http", "https", "mailto", "data", "javascript", "vbscript", "ftp", "tel", "x-custom") + + +def _wpt_cases() -> Iterator[tuple[str, Policy]]: + """ + Yield each WPT sanitizer-api ``#data`` input with its ``#config`` mapped onto a Policy. + + The block split follows tools/generate_wpt_tree_corpus.py ``_parse_file``. Only ``elements`` and ``attributes`` have + a Policy field to map to; a case naming neither runs under the default Policy. + """ + for path in sorted(_require(_DATA / "wpt" / "sanitizer-api").glob("*.sub.dat")): + for block in path.read_text(encoding="utf-8").split("#data\n")[1:]: + data, _, rest = block.partition("\n#") + config = json.loads(rest.partition("\n")[2].partition("\n#")[0]) if rest.startswith("config\n") else {} + tags = [entry if isinstance(entry, str) else entry["name"] for entry in config.get("elements", [])] + attributes = [entry if isinstance(entry, str) else entry["name"] for entry in config.get("attributes", [])] + yield ( + data.replace("{{host}}", "example.com"), + Policy( + tags=frozenset(tags) if tags else Policy().tags, + attributes={"*": frozenset(attributes)} if attributes else Policy().attributes, + ), + ) + + +def _dictionary_tokens() -> list[str]: + r""" + Read the google/fuzzing html, svg, mathml and css AFL dictionaries, the set jsoup concatenates into combo.dict. + + Entries are ``name="value"`` lines with ``\\``, ``\"`` and ``\xNN`` escapes, which AFL's ``load_extras_file`` + decodes the same way. + """ + return [ + re.sub( + r"\\(x[0-9a-fA-F]{2}|.)", + lambda escape: chr(int(escape[1][1:], 16)) if len(escape[1]) == 3 else escape[1], + match[1], + ) + for name in ("html", "svg", "mathml", "css") + for line in _require(_DATA / "google-fuzzing" / "dictionaries" / f"{name}.dict") + .read_text(encoding="utf-8") + .splitlines() + if not line.lstrip().startswith("#") and (match := re.search(r'"(.*)"\s*$', line)) is not None + ] + + +def _mutagen(rng: random.Random, seeds: Sequence[str]) -> str: + """ + Build an mXSS candidate inside-out from a trigger, the MutaGen generator (Klein and Johns, IEEE S&P 2024). + + ias-tubs/HTML_parsing_differentials carries no license, so this re-implements mutagen/lib/gen.ml from its structure: + up to 25 weighted actions wrap the trigger, an action family's odds decay with use (``decr``), and a result of 7 or + fewer steps, or mostly uninteresting ones, is drawn again. Vendored seed payloads join the three MutaGen triggers. + """ + while True: + counts: Counter[str] = Counter() + markup = rng.choice((*_TRIGGERS, rng.choice(seeds))) + steps = interesting = 0 + while ( + steps < 25 and (action := _weighted(rng, [(name, weight(counts)) for name, weight in _ACTIONS])) != "stop" + ): + counts[action] += 1 + markup = _apply(rng, action, markup) + interesting += action in {"open_tag", "enclose_tag", "enclose_tag_attribute"} + steps += 1 + if steps >= 7 and interesting * 3 >= steps: + return markup + + +_TRIGGERS: Final = ("", "", "") +# gen.ml Action.gen, with its comment, CDATA and URI-encoding variants folded into one family each and their weights +# summed; ``decr p c`` there is p ** (c + 1) +_ACTIONS: Final[tuple[tuple[str, Callable[[Counter[str]], float]], ...]] = ( + ("open_tag", lambda _counts: 1.0), + ("self_closing_tag", lambda _counts: 1.0), + ("enclose_tag", lambda _counts: 1.0), + ("enclose_tag_attribute", lambda _counts: 0.75), + ("close_tag", lambda counts: max(1.0, 0.1 * counts["open_tag"])), + ("xml_comment", lambda counts: 5 * 0.125 ** (counts["xml_comment"] + 1)), + ("js_comment", lambda counts: 0.01 ** (counts["js_comment"] + 1) + 2 * 0.005 ** (counts["js_comment"] + 1)), + ("uri_encode", lambda counts: 0.0005 ** (counts["uri_encode"] + 1) + 0.0001 ** (counts["uri_encode"] + 1)), + ("xml_encode", lambda counts: 0.025 ** (counts["xml_encode"] + 1)), + ("cdata", lambda counts: 3 * 0.05 ** (counts["cdata"] + 1)), + ("angle_bracket", lambda _counts: 0.2), + ("parsing_directive", lambda _counts: 0.05), + ("quote", lambda _counts: 0.25), + ("space", lambda _counts: 1.0), + ("stop", lambda _counts: 0.05), +) +_TAGS: Final = ( + ("div", 1.0), + ("span", 1.0), + ("title", 1.0), + ("form", 1.0), + ("dfn", 1.0), + ("header", 1.0), + ("p", 0.5), + ("br", 0.5), + ("a", 1.0), + ("style", 1.0), + ("noscript", 1.0), + ("table", 0.25), + ("td", 0.25), + ("tr", 0.25), + ("colgroup", 0.25), + ("svg", 1.0), + ("foreignobject", 1.0), + ("desc", 1.0), + ("path", 1.0), + ("math", 1.0), + ("mtext", 0.5), + ("mglyph", 0.5), + ("mi", 0.25), + ("mo", 0.25), + ("mn", 0.25), + ("ms", 0.25), + ("annotation-xml", 0.33), + ('annotation-xml encoding="text/html"', 0.33), + ('annotation-xml encoding="application/xhtml+xml"', 0.33), + ("select", 1.0), + ("input", 1.0), + ("option", 1.0), + ("textarea", 1.0), + ("keygen", 1.0), + ("xmp", 1.0), + ("noembed", 1.0), + ("listing", 1.0), + ("li", 0.5), + ("ul", 0.5), + ("pre", 1.0), + ("var", 1.0), + ("dl", 0.5), + ("dt", 0.5), + ("font", 1.0), + ("plaintext", 1.0), + ("noframes", 1.0), + ("iframe", 1.0), + ("object", 0.5), + ("embed", 0.5), + ("frameset", 0.5), +) +_ATTRIBUTE_QUOTES: Final = (("'", 0.4), ("'", 0.05), ('"', 0.4), (""", 0.05), ("`", 0.05), ("`", 0.05)) +_TOPLEVEL_QUOTES: Final = (("'", 0.4), ("'", 0.1), ('"', 0.4), (""", 0.1)) + + +def _weighted(rng: random.Random, choices: Sequence[tuple[_T, float]]) -> _T: + return rng.choices([choice for choice, _ in choices], weights=[weight for _, weight in choices])[0] + + +def _apply(rng: random.Random, action: str, inner: str) -> str: + """Wrap ``inner`` with one action as gen.ml's ``Operation.print`` renders it, placed before or after at random.""" + tag = _weighted(rng, _TAGS) + attrs = "".join( + _attribute(rng, rng.choice(("foo", "bar", "baz", "abc", "xyz"))) + for _ in range(_weighted(rng, ((0, 1.0), (1, 0.5), (2, 0.25)))) + ) + match action: + case "enclose_tag": + wrapped = f"<{tag}{attrs}>{inner}" + case "enclose_tag_attribute": + wrapped = f"<{tag}{attrs}{_attribute(rng, inner)}{'/' if rng.random() < 1 / 3 else ''}>" + case "xml_comment": + wrapped = rng.choice(( + f"", + f"", + f"{inner}", + f"{inner}--!>", + )) + case "js_comment": + wrapped = _weighted(rng, ((f"/*{inner}*/", 0.5), (f"/*{inner}", 0.25), (f"{inner}*/", 0.25))) + case "uri_encode": + # encodeURIComponent keeps only the unreserved marks, encodeURI also the reserved set + wrapped = quote(inner, safe=_weighted(rng, (("-_.!~*'()", 5.0), (";,/?:@&=+$#-_.!~*'()", 1.0)))) + case "xml_encode": + wrapped = html.escape(inner).replace("'", "'") + case "cdata": + wrapped = rng.choice((f"", f"")) + case _: + piece = { + "open_tag": f"<{tag}{attrs}>", + "self_closing_tag": f"<{tag}{attrs}/>", + "close_tag": f"", + "angle_bracket": rng.choice(("<", ">")), + "parsing_directive": " str: + """One attribute in gen.ml's ``Attribute`` forms: plain, space-after-equals or slash, then quoted per ``Quoted``.""" + name = rng.choice(("id", "title", "foo", "name", "data-foo")) + separator = _weighted(rng, ((f" {name}=", 0.9), (f" {name}= ", 0.05), (f"/{name}=", 0.05))) + left, right = _weighted(rng, _ATTRIBUTE_QUOTES), _weighted(rng, _ATTRIBUTE_QUOTES) + return separator + _weighted( + rng, + ( + (value, 0.5), + (f"{left}{value}{right}", 0.25), + (f"{left}{value}", 0.25), + (f"{value}{left}", 0.25), + (f"{left}{value}{left}", 1.0), + ), + ) + + +def _splice(rng: random.Random, seed: str, tokens: Sequence[str]) -> str: + """Insert dictionary tokens at random offsets, libFuzzer's dictionary mutation applied to a seed.""" + for _ in range(rng.randint(1, 6)): + at = rng.randint(0, len(seed)) + seed = seed[:at] + rng.choice(tokens) + seed[at:] + return seed + + +def _url_case(rng: random.Random) -> tuple[str, str, str, str, Policy]: + """ + Draw an element and URL attribute, a value with one obfuscation written as attribute source, and a URL policy. + + The obfuscations are scheme-check bypasses sanitizers shipped: character references without ``;`` and named + whitespace references (loofah GHSA-5qhf-9phg-95m2, GHSA-8whx-365g-h9vv), code points above U+00A0 (bleach + GHSA-8rfp-98v4-mmr6), NFKC look-alikes (html-sanitizer CVE-2024-34078), and ``javascript://://`` (Chromium's + ``ProtocolIsJavaScript`` disagreeing with KURL). + """ + base = rng.choice(_URL_BASES) + recipe, obfuscate = rng.choice(_OBFUSCATIONS) + at = rng.randint(0, colon if (colon := base.find(":")) > 0 else len(base)) + element, attribute = rng.choice(_URL_TARGETS) + policy = Policy( + tags=frozenset({"a", "img", "form", "button", "blockquote", "video", "svg"}), + attributes={"*": _URL_ATTRS}, + url_schemes=frozenset(rng.sample(_SCHEMES, rng.randint(0, 4))), + allow_relative_urls=rng.random() < 0.5, + allow_fragment_urls=rng.random() < 0.5, + ) + return element, attribute, recipe, obfuscate(base, at).replace('"', """), policy + + +_URL_BASES: Final = ( + "javascript:alert(1)", + "JaVaScRiPt:alert(1)", + "vbscript:msgbox(1)", + "data:text/html,", + "https://example.com/", + "http://example.com/x?y#z", + "mailto:a@example.com", + "x-custom:1", + "ftp://example.com/", + "tel:+1", + "//example.com/", + "/path", + "?query", + "#fragment", + "javascript://://alert(1)", + "javascript:/*\n*/alert(1)", + "c:\\x", + "\\\\example.com\\share", + "java", + "", +) +_URL_TARGETS: Final = ( + ("a", "href"), + ("img", "src"), + ("form", "action"), + ("button", "formaction"), + ("blockquote", "cite"), + ("video", "poster"), + ("svg", "xlink:href"), + ("svg", "href"), +) + + +def _insert(text: str, value: str, at: int) -> str: + return value[:at] + text + value[at:] + + +_FULLWIDTH: Final = {code: code + 0xFEE0 for code in range(ord("A"), ord("z") + 1) if chr(code).isalpha()} +_OBFUSCATIONS: Final[tuple[tuple[str, Callable[[str, int], str]], ...]] = ( + ("plain", lambda value, _at: value), + ("leading c0", lambda value, _at: f"\x01 \x1f{value}"), + ("leading space refs", lambda value, _at: f" {value}"), + (":", lambda value, _at: value.replace(":", ":", 1)), + ("uppercase", lambda value, _at: value.upper()), + ( + "letter as ref no ;", + lambda value, at: value[:at] + "".join(f"&#{ord(char)}" for char in value[at : at + 1]) + value[at + 1 :], + ), + ( + "letter as hex ref", + lambda value, at: value[:at] + "".join(f"&#x{ord(char):x};" for char in value[at : at + 1]) + value[at + 1 :], + ), + ("fullwidth", lambda value, at: value[:at] + value[at : at + 3].translate(_FULLWIDTH) + value[at + 3 :]), + *( + (recipe, partial(_insert, text)) + for recipe, text in ( + ("raw tab", "\t"), + ("raw newline", "\n"), + ("named ", " "), + ("named ", " "), + ("decimal ref no ;", " "), + ("hex ref no ;", " "), + ("zero-padded ref", " "), + ("nul ref", "�"), + ("c0 control", "\x01"), + ("del", "\x7f"), + ("nbsp", "\u00a0"), + ("zero width space", "\u200b"), + ("soft hyphen ref", "­"), + ("line separator", "\u2028"), + ) + ), +) + + +def _confirm_with_parse5(findings: list[tuple[_Finding, Policy]]) -> list[tuple[_Finding, Policy]]: + """ + Re-parse each minimized mutation finding's output with parse5 and record whether it confirms the mutation. + + ``confirmed`` means parse5 also departs from turbohtml's tree, so the mutation is turbohtml's; ``html5lib-only`` + means parse5 rebuilds turbohtml's tree, pointing at a stale html5lib rule. Node or the npm install may be absent on + a dev machine; the verdict then says so instead of failing the run. + """ + mutations = [(finding, policy) for finding, policy in findings if "!= html5lib " in finding.signature] + if (node := shutil.which("node")) is None or not (_NODE_DIR / "node_modules" / "parse5").is_dir(): + verdicts = dict.fromkeys((finding.signature for finding, _ in mutations), "unavailable") + else: + trees = [ + sanitize_node(parse_fragment(finding.markup, _CONTEXT, positions=False), policy) + for finding, policy in mutations + ] + dumps = subprocess.run( + [node, str(_NODE_DIR / "parse5_tree_runner.js")], + input=json.dumps([tree.inner_html for tree in trees]), + capture_output=True, + text=True, + cwd=_NODE_DIR, + # the in-process run preloads the ASan runtime, which node must not inherit + env={key: value for key, value in os.environ.items() if key not in {"LD_PRELOAD", "DYLD_INSERT_LIBRARIES"}}, + check=True, + ).stdout + verdicts = { + finding.signature: "html5lib-only" if dump == "\n".join(_dump_turbohtml(tree.children)) else "confirmed" + for (finding, _), tree, dump in zip(mutations, trees, json.loads(dumps), strict=True) + } + return [ + (dataclasses.replace(finding, parse5=verdicts.get(finding.signature, finding.parse5)), policy) + for finding, policy in findings + ] + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/prune_dist.py b/tools/prune_dist.py index a7676359b..6bbf27f37 100755 --- a/tools/prune_dist.py +++ b/tools/prune_dist.py @@ -15,5 +15,5 @@ if __name__ == "__main__": root = Path(os.environ["MESON_DIST_ROOT"]) - for relative in ("tests/html5lib-tests", "tools/bench-data", "tools/html5lib-python"): + for relative in ("tests/html5lib-tests", "tools/bench-data", "tools/fuzz-data", "tools/html5lib-python"): shutil.rmtree(root / relative, ignore_errors=True) diff --git a/tox.toml b/tox.toml index 10d657449..54ee4d80a 100644 --- a/tox.toml +++ b/tox.toml @@ -409,6 +409,59 @@ commands = [ ], ] +[env.fuzz-oracle] +description = """\ + run the sanitizer wrong-output oracles (independent re-parse, random Policy, URL scheme) under ASan/UBSan\ + """ +# dharma's xss.dg grammar generates seed payloads; six and webencodings are what the vendored html5lib-python imports +deps = [ "dharma>=1.3.2", "six>=1.17", "webencodings>=0.5.1" ] +dependency_groups = [ "test" ] +pass_env = [ "FUZZ_RNG_SEED" ] +# the independent re-parser is the vendored html5lib-python, imported from its checkout: its setup.py needs the +# pkg_resources module current setuptools no longer ships, so it cannot be installed +set_env = { CC = "clang", PYTHONHASHSEED = "0", PYTHONPATH = "{tox_root}{/}tools{/}html5lib-python" } +commands_pre = [ + # the WPT seeds are a sparse checkout fuzz.py makes itself; a plain update skips them (update = none in .gitmodules) + [ + "git", + "-C", + "{tox_root}", + "submodule", + "update", + "--init", + "--depth", + "1", + "tools/html5lib-python", + "tests/conformance/DOMPurify", + "tools/fuzz-data/h5sc", + "tools/fuzz-data/google-fuzzing", + ], + [ + "uv", + "pip", + "install", + "--reinstall", + "--no-deps", + "--no-build-isolation", + "--editable", + "{tox_root}", + "--config-settings=build-dir={env_dir}{/}cbuild", + "--config-settings=setup-args=-Dc_args=-fsanitize=address,undefined", + "--config-settings=setup-args=-Dc_link_args=-fsanitize=address,undefined", + "--config-settings=setup-args=-Dbuildtype=debugoptimized", + ], +] +commands = [ + [ + "python", + "{tox_root}{/}tools{/}fuzz{/}fuzz.py", + "--mode", + "oracle", + { replace = "posargs", default = [ "--minutes", "2" ], extend = true }, + ], +] +allowlist_externals = [ "git" ] + [env.fuzz-smoke] description = "replay past fuzz finds and the seed corpus through every harness under ASan/UBSan (the per-PR gate)" deps = [] # override the base env's gcovr; this env instruments for ASan, not coverage