diff --git a/.github/workflows/fuzz.yaml b/.github/workflows/fuzz.yaml index 635ba36e4..1c14a06b6 100644 --- a/.github/workflows/fuzz.yaml +++ b/.github/workflows/fuzz.yaml @@ -49,17 +49,40 @@ jobs: enable-cache: false - name: ✅ Run fuzz (deep) # the Monday 04:00 slot runs each target longer; the daily 06:00 slot and manual dispatch use the shorter budget - run: uvx --with tox-uv tox run -e fuzz -- --minutes "${MINUTES}" + # The seed is secret: with the public code it regenerates every crasher, so it stays out of the command line + # tox echoes and reaches fuzz.py only through the environment and the encrypted .replay files. + run: | + FUZZ_RNG_SEED="$(python3 -c 'import secrets; print(secrets.randbits(32))')" + export FUZZ_RNG_SEED + uvx --with tox-uv tox run -e fuzz -- --minutes "${MINUTES}" env: UV_PYTHON_PREFERENCE: only-managed MINUTES: ${{ github.event.schedule == '0 4 * * 1' && '20' || '8' }} - - name: 📤 Upload crash corpus - if: failure() + # A crasher of an unfixed bug is an undisclosed vulnerability, and artifacts on a public repository are readable + # by anyone, so crashers leave the runner only encrypted to the maintainers' age key; without the key nothing + # leaves. + - name: 🔐 Encrypt crashers + if: failure() && vars.FUZZ_AGE_RECIPIENT != '' + run: | + curl -sSfL -o "${RUNNER_TEMP}/age.tar.gz" \ + "https://github.com/FiloSottile/age/releases/download/${AGE_VERSION}/age-${AGE_VERSION}-linux-amd64.tar.gz" + echo "${AGE_SHA256} ${RUNNER_TEMP}/age.tar.gz" | sha256sum --check --strict + tar -xzf "${RUNNER_TEMP}/age.tar.gz" -C "${RUNNER_TEMP}" age/age + mkdir -p "${RUNNER_TEMP}/fuzz-crashes" + shopt -s nullglob + for crasher in .fuzz-crashes/crash-*; do + "${RUNNER_TEMP}/age/age" --recipient "${RECIPIENT}" \ + --output "${RUNNER_TEMP}/fuzz-crashes/$(basename "${crasher}").age" "${crasher}" + done + env: + AGE_VERSION: v1.3.2 + AGE_SHA256: cbe24006683f8eb669266162894b9a522a1af52f2665fbc63a4bb032ed26ac10 + RECIPIENT: ${{ vars.FUZZ_AGE_RECIPIENT }} + - name: 📤 Upload encrypted crashers + if: failure() && vars.FUZZ_AGE_RECIPIENT != '' uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: fuzz-crashes - path: | - **/crash-* - **/*.crash + path: ${{ runner.temp }}/fuzz-crashes/*.age if-no-files-found: ignore retention-days: 14 diff --git a/.gitignore b/.gitignore index 8f6f1a939..fb7e59599 100644 --- a/.gitignore +++ b/.gitignore @@ -32,3 +32,6 @@ tools/bench/node/node_modules/ # the conformance oracles build in place tests/conformance/oracles/*/target/ + +# tools/fuzz/fuzz.py stores crashing inputs here; they stay off the public repository +/.fuzz-crashes diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index d3772c3b4..b3b2284af 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -5,9 +5,9 @@ repos: rev: "v6.0.0" # v6.0.0 hooks: - id: end-of-file-fixer - exclude: ^tools/fuzz/corpus/.*$ # byte-exact crash inputs + exclude: ^(tools/fuzz/corpus|tests/fuzz_regressions)/.*$ # byte-exact crash inputs - id: trailing-whitespace - exclude: ^tools/fuzz/corpus/.*$ # byte-exact crash inputs + exclude: ^(tools/fuzz/corpus|tests/fuzz_regressions)/.*$ # byte-exact crash inputs - repo: https://github.com/python-jsonschema/check-jsonschema rev: "0.38.2" # 0.37.2 hooks: diff --git a/docs/development/index.rst b/docs/development/index.rst index 770793e1e..5c7128c19 100644 --- a/docs/development/index.rst +++ b/docs/development/index.rst @@ -128,20 +128,48 @@ Two mechanisms share one driver (``tools/fuzz/fuzz.py``): runs against an extension built with the sanitizers and calls the public API, so a C fault aborts the interpreter with a stack trace and the harness survives internal C refactors. -The ``fuzz-smoke`` environment replays a benign seed corpus (``tools/fuzz/corpus/``) once per target. It is fast and -deterministic and gates every pull request in the ``🔒 fuzz`` workflow, so it seeds with valid inputs, never known -crashers. The ``fuzz`` environment adds a mutation loop and escalating-depth structural probes for a per-target budget; -it is the continuous hunt, run weekly and on demand, not a merge gate. +The ``fuzz-smoke`` environment replays every past find in ``tests/fuzz_regressions/`` through every harness, then a +benign seed corpus (``tools/fuzz/corpus/``) once per target. It is deterministic and gates every pull request in the +``🔒 fuzz`` workflow. The ``fuzz`` environment adds a mutation loop and escalating-depth structural probes for a +per-target budget; it is the continuous hunt, run daily and on demand, not a merge gate. .. code-block:: console - $ tox r -e fuzz-smoke # the per-PR gate: benign corpus, no crash expected + $ tox r -e fuzz-smoke # the per-PR gate: past finds and benign corpus, no crash expected $ tox r -e fuzz -- --minutes 5 # the deep run: mutation + structural probes per target -Add a target by registering a ``bytes``-taking callable in ``TARGETS`` (in-process) and dropping a representative benign -seed under ``tools/fuzz/corpus//``; add a standalone harness by mirroring ``idna_harness.c`` for any C unit that -compiles free of the CPython boundary. macOS ships no ``libFuzzer`` runtime with Apple Clang, so the coverage-guided -mode needs an LLVM Clang (``brew install llvm``); the corpus-replay and mutation modes run under Apple Clang. +The in-process driver runs each input under pymalloc and again under ``PYTHONMALLOC=malloc``, because AddressSanitizer +cannot see an over-read that stays inside a pymalloc pool. The deep run splits ``--minutes`` between the two passes. +Both environments pin ``PYTHONHASHSEED=0``. ``--rng-seed`` (default ``$FUZZ_RNG_SEED``, else 0) fixes the mutation +sequence. The scheduled run draws a random seed and keeps it out of the log, since the seed regenerates every crasher. + +``fuzz.py`` stores a crashing input as ``.fuzz-crashes/crash-``, with its seed and mutation index in +``crash-.replay``, and logs only the SHA-256, length and harness, because anyone can read the CI logs of a +public repository. A deep run writes sanitizer reports to ``crash-sanitizer.`` beside them for the same reason. +Once the fix lands, copy the input into ``tests/fuzz_regressions/`` with the issue number in its name, and every later +run replays it first. + +The scheduled run uploads crashers only when the ``FUZZ_AGE_RECIPIENT`` repository variable holds an `age +`_ public key, and it encrypts each one to that key first. Without the variable it +uploads nothing. A maintainer sets the variable once and keeps the identity file private: + +.. code-block:: console + + $ age-keygen -o fuzz-crashes.key # prints "Public key: age1..." + $ gh variable set FUZZ_AGE_RECIPIENT --body age1... + +To read the crashers of a failed run, download its ``fuzz-crashes`` artifact and decrypt each file: + +.. code-block:: console + + $ gh run download --name fuzz-crashes + $ age --decrypt --identity fuzz-crashes.key --output crash- crash-.age + +Add a target by registering a ``bytes``-taking callable in ``_TARGETS`` (in-process) and dropping a representative +benign seed under ``tools/fuzz/corpus//``; add a standalone harness by mirroring ``idna_harness.c`` for any C +unit that compiles free of the CPython boundary. macOS ships no ``libFuzzer`` runtime with Apple Clang, so the +coverage-guided mode needs an LLVM Clang (``brew install llvm``); the corpus-replay and mutation modes run under Apple +Clang. **************** Project layout diff --git a/tools/fuzz/corpus/roundtrip/empty-run-null-src.html b/tests/fuzz_regressions/521-serialize-empty-run.html similarity index 100% rename from tools/fuzz/corpus/roundtrip/empty-run-null-src.html rename to tests/fuzz_regressions/521-serialize-empty-run.html diff --git a/tools/fuzz/corpus/parse/leading-cr-null-input.html b/tests/fuzz_regressions/752-tokenizer-leading-cr.html similarity index 100% rename from tools/fuzz/corpus/parse/leading-cr-null-input.html rename to tests/fuzz_regressions/752-tokenizer-leading-cr.html diff --git a/tests/fuzz_regressions/949-js-empty-member.js b/tests/fuzz_regressions/949-js-empty-member.js new file mode 100644 index 000000000..69ca5b257 --- /dev/null +++ b/tests/fuzz_regressions/949-js-empty-member.js @@ -0,0 +1 @@ +t.0. \ No newline at end of file diff --git a/tests/fuzz_regressions/952-css-selectorless-rule.css b/tests/fuzz_regressions/952-css-selectorless-rule.css new file mode 100644 index 000000000..0ce064102 --- /dev/null +++ b/tests/fuzz_regressions/952-css-selectorless-rule.css @@ -0,0 +1,2 @@ +{--x: +} \ No newline at end of file diff --git a/tools/fuzz/_targets.py b/tools/fuzz/_targets.py index 3d85ced62..0d2bc9865 100644 --- a/tools/fuzz/_targets.py +++ b/tools/fuzz/_targets.py @@ -9,10 +9,9 @@ operation aborts the interpreter with a sanitizer stack trace. Calling the public API keeps every target robust to the internal C refactors in flight (tox-dev/turbohtml#478). -``smoke`` replays a benign seed corpus once per target: fast, deterministic, and expected to stay green, so it gates -every PR. ``deep`` adds a mutation loop and structural probes (escalating nesting depth, which the research flags as the -most likely stack-overflow finding) for a wall-clock budget per target, for the scheduled/manual deep run. Before each -call the current input is written to the repro file, so an abort names its own crasher. +``smoke`` replays the past finds and a benign seed corpus once per target: deterministic and expected to stay green, so +it gates every PR. ``deep`` adds a mutation loop and structural probes for a wall-clock budget per target. Each input is +written to a file named after its target in the repro directory before the call, so an abort leaves its crasher behind. """ from __future__ import annotations @@ -20,7 +19,10 @@ import argparse import contextlib import dataclasses +import hashlib +import os import random +import re import sys import time from pathlib import Path @@ -45,14 +47,89 @@ # exception type is a finding. RecursionError from a deep Python walk is surfaced separately as a soft DoS signal. _EXPECTED: Final = (ValueError, UnicodeError, turbohtml.HTMLParseError) _SET_IDS: Final = (0, 1, 2) +# Script text where "" breaks the script content +# restrictions (https://html.spec.whatwg.org/multipage/scripting.html#restrictions-for-contents-of-script-elements): +# reparsing its serialization lands in the double-escaped state, which swallows the closing "". The standard +# warns that such parser-made trees need not survive serialize and reparse +# (https://html.spec.whatwg.org/multipage/parsing.html#serialising-html-fragments). +_SCRIPT_DOUBLE_ESCAPE: Final = re.compile(r").)*?]", re.IGNORECASE | re.DOTALL) +# TH_MAX_TREE_DEPTH (src/turbohtml/_c/dom/tree.h): past 512 open elements a start tag is inserted but not pushed, so +# further tags become siblings and reparsing the flattened output nests them differently +_MAX_TREE_DEPTH: Final = 512 -def _decode(data: bytes) -> str: - """Widen raw fuzz bytes to a str, keeping lone surrogates so the IDNA surrogate path is reachable.""" +def main() -> int: + """Drive one or all targets; return 1 on any soft finding, 0 if clean (an ASan abort exits nonzero on its own).""" + parser = argparse.ArgumentParser() + parser.add_argument("--mode", choices=("smoke", "deep"), required=True) + parser.add_argument("--target", default="all") + parser.add_argument("--corpus-dir", type=Path, required=True) + parser.add_argument("--regression-dir", type=Path, required=True) + parser.add_argument("--repro-dir", type=Path, required=True) + parser.add_argument("--crash-dir", type=Path, required=True) + parser.add_argument("--minutes", type=float, default=1.0) + parser.add_argument("--rng-seed", type=int, default=0) + parser.add_argument("--no-structural", action="store_true", help="skip the escalating-depth probes (a known DoS)") + args = parser.parse_args() + + regressions = _inputs(args.regression_dir) + findings: list[str] = [] + for name in _TARGETS if args.target == "all" else [args.target]: + repro = args.repro_dir / name + seeds = regressions + _inputs(args.corpus_dir / name) + findings += [ + _finding(name, f"seed {index}", found, data, args.crash_dir) + for index, data in enumerate(seeds) + if (found := _run_one(_TARGETS[name], data, repro)) is not None + ] + if args.mode == "deep": + findings += _hunt(name, seeds, repro, args) + else: + print(f"smoke {name}: {len(seeds)} seeds clean") + repro.unlink() + return 1 if findings else 0 + + +def _inputs(directory: Path) -> list[bytes]: + return [path.read_bytes() for path in sorted(directory.glob("*")) if path.is_file()] + + +def _run_one(func: Callable[[bytes], None], data: bytes, repro: Path) -> str | None: + repro.write_bytes(data) try: - return data.decode("utf-8", "surrogatepass") - except UnicodeError: - return data.decode("latin-1") + func(data) + except _EXPECTED: + return None + except RecursionError: + return "RecursionError (soft DoS)" + except AssertionError as exc: + return f"invariant broken: {exc}" + return None + + +_PHONES: Final = clean.PhoneNumbers(regions=("US", "GB", "DE", "IN")) +_PHONE_DETECTOR: Final = clean.LinkDetector(phones=_PHONES) +_PHONE_LINKER: Final = clean.Linker(clean.Linkify(phones=_PHONES, parse_email=True)) +_PHONE_POSSIBLE: Final = clean.LinkDetector(phones=clean.PhoneNumbers(regions=("US",), require_valid=False)) +_PHONE_EXACT: Final = clean.LinkDetector( + phones=clean.PhoneNumbers(regions=("US", "DE"), grouping=clean.PhoneGrouping.EXACT) +) + + +def _phone(data: bytes) -> None: + text = _decode(data) + for span in _PHONE_DETECTOR.find(text): + if span.phone is not None: + clean.PhoneNumber(*dataclasses.astuple(span.phone)) + for style in clean.PhoneFormat: + span.phone.format(style) + _PHONE_POSSIBLE.find(text) + _PHONE_EXACT.find(text) + _PHONE_LINKER.linkify(text) + for held in (text, "tel:" + text): + for require_valid in (True, False): + with contextlib.suppress(ValueError): + clean.PhoneNumber.parse(held, regions=("US", "DE"), require_valid=require_valid) def _parse(data: bytes) -> None: @@ -65,17 +142,16 @@ def _serialize(data: bytes) -> None: def _roundtrip(data: bytes) -> None: - once = turbohtml.parse(_decode(data)).serialize() - twice = turbohtml.parse(once).serialize() - if once != twice: - message = f"serialize not idempotent: {once!r} != {twice!r}" + document = turbohtml.parse(_decode(data)) + if (once := document.serialize()) != turbohtml.parse(once).serialize() and not _reparse_may_differ(document): + message = "serialize not idempotent" raise AssertionError(message) def _sanitize(data: bytes) -> None: - once = clean.sanitize(_decode(data)) - if clean.sanitize(once) != once: - message = f"sanitize not idempotent for {_decode(data)!r}" + text = _decode(data) + if clean.sanitize(once := clean.sanitize(text)) != once and not _past_depth_cap(turbohtml.parse_fragment(text)): + message = "sanitize not idempotent" raise AssertionError(message) @@ -94,31 +170,6 @@ def _idna(data: bytes) -> None: _url_to_ascii(_decode(data)) -_PHONES: Final = clean.PhoneNumbers(regions=("US", "GB", "DE", "IN")) -_PHONE_DETECTOR: Final = clean.LinkDetector(phones=_PHONES) -_PHONE_LINKER: Final = clean.Linker(clean.Linkify(phones=_PHONES, parse_email=True)) -_PHONE_POSSIBLE: Final = clean.LinkDetector(phones=clean.PhoneNumbers(regions=("US",), require_valid=False)) -_PHONE_EXACT: Final = clean.LinkDetector( - phones=clean.PhoneNumbers(regions=("US", "DE"), grouping=clean.PhoneGrouping.EXACT) -) - - -def _phone(data: bytes) -> None: - text = _decode(data) - for span in _PHONE_DETECTOR.find(text): - if span.phone is not None: - clean.PhoneNumber(*dataclasses.astuple(span.phone)) - for style in clean.PhoneFormat: - span.phone.format(style) - _PHONE_POSSIBLE.find(text) - _PHONE_EXACT.find(text) - _PHONE_LINKER.linkify(text) - for held in (text, "tel:" + text): - for require_valid in (True, False): - with contextlib.suppress(ValueError): - clean.PhoneNumber.parse(held, regions=("US", "DE"), require_valid=require_valid) - - def _minify_css(data: bytes) -> None: clean.minify_css(_decode(data)) @@ -127,7 +178,7 @@ def _minify_html(data: bytes) -> None: clean.minify(_decode(data)) -TARGETS: Final[dict[str, Callable[[bytes], None]]] = { +_TARGETS: Final[dict[str, Callable[[bytes], None]]] = { "phone": _phone, "parse": _parse, "serialize": _serialize, @@ -139,6 +190,73 @@ def _minify_html(data: bytes) -> None: "minify_html": _minify_html, } + +def _decode(data: bytes) -> str: + """Widen raw fuzz bytes to a str, keeping lone surrogates so the IDNA surrogate path is reachable.""" + try: + return data.decode("utf-8", "surrogatepass") + except UnicodeError: + return data.decode("latin-1") + + +def _reparse_may_differ(document: turbohtml.Document) -> bool: + return _past_depth_cap(document) or any( + _SCRIPT_DOUBLE_ESCAPE.search(script.text) for script in document.iter_elements("script") + ) + + +def _past_depth_cap(root: turbohtml.Node) -> bool: + # the root node plus 512 ancestors is the deepest placement the tree builder allows, reached only on a full stack + return any(sum(1 for _ in element.ancestors) > _MAX_TREE_DEPTH for element in root.iter_elements()) + + +def _finding(target: str, origin: str, kind: str, data: bytes, crash_dir: Path) -> str: + """ + Keep the input for the encrypted upload and log it by hash. + + A public CI log must carry neither the crasher bytes nor the origin (seed and mutation index) that regenerates + them, so the origin goes to the encrypted ``.replay`` file. It logs at once, not at exit, so a later sanitizer abort + in the same run cannot swallow the line. + """ + digest = hashlib.sha256(data).hexdigest() + (crash_dir / f"crash-{digest}").write_bytes(data) + (crash_dir / f"crash-{digest}.replay").write_text(f"{origin} PYTHONMALLOC={os.environ.get('PYTHONMALLOC')}\n") + finding = f"[{target}] {kind} sha256={digest} bytes={len(data)}" + print(f"FINDING {finding}", file=sys.stderr, flush=True) + return finding + + +def _hunt(name: str, seeds: list[bytes], repro: Path, args: argparse.Namespace) -> list[str]: + func = _TARGETS[name] + findings = [ + _finding(name, f"structural {index}", found, data, args.crash_dir) + for index, data in enumerate(() if args.no_structural else _structural(name)) + if (found := _run_one(func, data, repro)) is not None + ] + rng = random.Random(args.rng_seed) + deadline = time.monotonic() + args.minutes * 60 + runs = 0 + while time.monotonic() < deadline: + if (found := _run_one(func, data := _mutate(rng, rng.choice(seeds)), repro)) is not None: + findings.append(_finding(name, f"mutation {runs} of rng-seed {args.rng_seed}", found, data, args.crash_dir)) + runs += 1 + print(f"deep {name}: {runs} mutations + structural probes done") + return findings + + +def _structural(target: str) -> Iterator[bytes]: + if target in {"parse", "serialize", "roundtrip", "sanitize", "minify_html"}: + # deep enough that the recursive sanitizer walk overflows (around 8k frames under ASan), shallow enough that the + # quadratic parse stays bounded, so the deep run reaches the stack-overflow class without a multi-minute parse + for depth in (256, 2048, 16384, 32768): + yield b"
" * depth + yield b"" * depth + yield (b"" * depth) + (b"" * depth) + if target in {"url", "idna"}: + yield b"xn--" + b"a" * 20000 + yield b"a." * 20000 + + _INTERESTING: Final[tuple[bytes, ...]] = ( b"", @@ -165,23 +283,6 @@ def _minify_html(data: bytes) -> None: ) -def _structural(target: str) -> Iterator[bytes]: - """ - Yield escalating-depth probes for the recursive walks. - - Depths cap where the quadratic parse cost stays bounded yet the recursive sanitizer walk still overflows (it faults - around 8k frames under ASan), so the deep run surfaces the stack-overflow class without a multi-minute parse. - """ - if target in {"parse", "serialize", "roundtrip", "sanitize", "minify_html"}: - for depth in (256, 2048, 16384, 32768): - yield b"
" * depth - yield b"" * depth - yield (b"" * depth) + (b"" * depth) - if target in {"url", "idna"}: - yield b"xn--" + b"a" * 20000 - yield b"a." * 20000 - - def _mutate(rng: random.Random, seed: bytes) -> bytes: out = bytearray(seed) for _ in range(rng.randint(1, 8)): @@ -191,66 +292,9 @@ def _mutate(rng: random.Random, seed: bytes) -> bytes: index = rng.randrange(len(out)) out[index] ^= 1 << rng.randrange(8) else: - index = rng.randrange(len(out)) - del out[index] + del out[rng.randrange(len(out))] return bytes(out) -def _run_one(func: Callable[[bytes], None], data: bytes, repro: Path) -> str | None: - repro.write_bytes(data) - try: - func(data) - except _EXPECTED: - return None - except RecursionError: - return f"RecursionError (soft DoS) on {data[:80]!r}" - except AssertionError as exc: - return f"invariant broken: {exc}" - return None - - -def _corpus(corpus_dir: Path, target: str) -> list[bytes]: - return [path.read_bytes() for path in sorted((corpus_dir / target).glob("*")) if path.is_file()] - - -def main() -> int: - """Drive one or all targets; return 1 on any soft finding, 0 if clean (an ASan abort exits nonzero on its own).""" - parser = argparse.ArgumentParser() - parser.add_argument("--mode", choices=("smoke", "deep"), required=True) - parser.add_argument("--target", default="all") - parser.add_argument("--corpus-dir", type=Path, required=True) - parser.add_argument("--repro", type=Path, required=True) - parser.add_argument("--minutes", type=float, default=1.0) - parser.add_argument("--rng-seed", type=int, default=0) - parser.add_argument("--no-structural", action="store_true", help="skip the escalating-depth probes (a known DoS)") - args = parser.parse_args() - - names = list(TARGETS) if args.target == "all" else [args.target] - findings: list[str] = [] - for name in names: - func = TARGETS[name] - seeds = _corpus(args.corpus_dir, name) or [b""] - findings += [f"[{name}] {found}" for data in seeds if (found := _run_one(func, data, args.repro)) is not None] - if args.mode == "smoke": - print(f"smoke {name}: {len(seeds)} seeds clean") - continue - probes = () if args.no_structural else _structural(name) - findings += [ - f"[{name}] structural {f}" for data in probes if (f := _run_one(func, data, args.repro)) is not None - ] - rng = random.Random(args.rng_seed) - deadline = time.monotonic() + args.minutes * 60 - runs = 0 - while time.monotonic() < deadline: - if (found := _run_one(func, _mutate(rng, rng.choice(seeds)), args.repro)) is not None: - findings.append(f"[{name}] {found}") - runs += 1 - print(f"deep {name}: {runs} mutations + structural probes done") - - for finding in findings: - print(f"FINDING {finding}", file=sys.stderr) - return 1 if findings else 0 - - if __name__ == "__main__": raise SystemExit(main()) diff --git a/tools/fuzz/fuzz.py b/tools/fuzz/fuzz.py index c89bf524e..ea8e8b905 100644 --- a/tools/fuzz/fuzz.py +++ b/tools/fuzz/fuzz.py @@ -11,14 +11,18 @@ sanitize, the URL parser, and the HTML/CSS minifiers -- run against an extension compiled with the sanitizers so a C fault aborts the interpreter with a stack trace. It calls the public API, so it survives the in-flight C refactors. -``smoke`` replays a benign seed corpus once (fast, deterministic, gates every PR). ``deep`` adds a mutation loop and -structural probes for a per-target budget (the scheduled/manual run that hunts for crashes). The in-process extension is +``smoke`` replays the past finds under ``tests/fuzz_regressions`` and a benign seed corpus once (fast, deterministic, +gates every PR). ``deep`` adds a mutation loop and structural probes for a per-target budget (the scheduled/manual run +that hunts for crashes). A crashing input lands in ``--crash-dir`` as ``crash-``, and the log names it only by +hash, length, harness and seeds, because CI logs on a public repository are public. The in-process extension is expected to be pre-built by the tox env; ``--build`` builds it here for a local run. """ from __future__ import annotations import argparse +import hashlib +import json import os import platform import subprocess @@ -30,59 +34,62 @@ _ROOT: Final[Path] = Path(__file__).resolve().parent.parent.parent _FUZZ: Final[Path] = _ROOT / "tools" / "fuzz" _CORPUS: Final[Path] = _FUZZ / "corpus" -_JS_CORPUS: Final[Path] = _ROOT / "tests" / "serialize" / "js" / "_corpus" +_REGRESSIONS: Final[Path] = _ROOT / "tests" / "fuzz_regressions" +_JS_CORPUS: Final[Path] = _ROOT / "tests" / "js" / "_corpus" _CC: Final[str] = os.environ.get("CC", "clang") _JS_ENGINE: Final[tuple[str, ...]] = ("lexer", "ast", "parser", "printer", "fold", "mangle", "minify") +# pymalloc carves small objects out of pools, so an over-read that stays inside a pool never reaches ASan's redzones; +# PYTHONMALLOC=malloc hands every PyMem call to the intercepted system allocator (v1 FUZZ-2 was silent under pymalloc) +_ALLOCATORS: Final[tuple[str, ...]] = ("pymalloc", "malloc") -def _asan_preload() -> dict[str, str]: - """Build the env preloading the ASan runtime ahead of the interpreter so the instrumented .so's interceptors arm.""" - on_linux = platform.system() == "Linux" - flag = f"libclang_rt.asan-{platform.machine()}.so" if on_linux else "libclang_rt.asan_osx_dynamic.dylib" - runtime = subprocess.run( - [_CC, f"-print-file-name={flag}"], capture_output=True, text=True, check=True - ).stdout.strip() - env = dict(os.environ) - env["LD_PRELOAD" if on_linux else "DYLD_INSERT_LIBRARIES"] = runtime - env["ASAN_OPTIONS"] = "detect_leaks=0:halt_on_error=1:abort_on_error=1" - env["UBSAN_OPTIONS"] = "halt_on_error=1:print_stacktrace=1" - return env +def main() -> int: + """Return 0 when every harness stays clean, nonzero on the first sanitizer abort or soft finding.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--mode", choices=("smoke", "deep"), default="smoke") + parser.add_argument( + "--minutes", type=float, default=1.0, help="deep-mode budget per in-process target, split across the allocators" + ) + parser.add_argument( + "--rng-seed", + type=int, + default=int(os.environ.get("FUZZ_RNG_SEED", "0")), + help="seed of the deep-mode mutation sequence (default: $FUZZ_RNG_SEED, else 0)", + ) + parser.add_argument("--crash-dir", type=Path, default=_ROOT / ".fuzz-crashes", help="where crashing inputs land") + parser.add_argument("--build", action="store_true", help="build the ASan extension here (tox builds it otherwise)") + parser.add_argument("--extra-corpus", type=Path, default=None, help="a second seed directory (vendored test data)") + parser.add_argument("--skip-inprocess", action="store_true", help="only run the standalone C harnesses") + args = parser.parse_args() + if args.build: + _build_extension(Path(tempfile.mkdtemp(prefix="th-fuzz-build-"))) + args.crash_dir.mkdir(parents=True, exist_ok=True) + if (code := _run_standalone(args.mode, args.extra_corpus, args.crash_dir)) != 0: + return code + return 0 if args.skip_inprocess else _run_inprocess(args.mode, args.minutes, args.rng_seed, args.crash_dir) -def _compile(harness: Path, sources: list[Path], macro: str, binary: Path) -> None: + +def _build_extension(build_dir: Path) -> None: cmd = [ - _CC, - macro, - "-fsanitize=address,undefined", - "-fno-omit-frame-pointer", - "-g", - "-O1", - "-Wall", - "-Wextra", - "-Werror", - "-I", - str(_ROOT / "src" / "turbohtml" / "_c"), - str(harness), - *[str(path) for path in sources], - "-o", - str(binary), + "uv", + "pip", + "install", + "--reinstall", + "--no-deps", + "--no-build-isolation", + "--editable", + str(_ROOT), + f"--config-settings=build-dir={build_dir}", + "--config-settings=setup-args=-Dc_args=-fsanitize=address,undefined", + "--config-settings=setup-args=-Dc_link_args=-fsanitize=address,undefined", + "--config-settings=setup-args=-Dbuildtype=debugoptimized", ] print("$", " ".join(cmd)) - subprocess.run(cmd, check=True) - - -def _seed_files(target: str, extra: Path | None) -> list[str]: - files = sorted(str(path) for path in (_CORPUS / target).glob("*") if path.is_file()) - if extra is not None and (extra / target).is_dir(): - files += sorted(str(path) for path in (extra / target).glob("*") if path.is_file()) - return files + subprocess.run(cmd, check=True, env={**os.environ, "CC": _CC}) -def _run_standalone(mode: str, extra: Path | None) -> int: - """Compile and replay the malloc-backed C harnesses; a sanitizer abort returns its nonzero code.""" - env = dict(os.environ) - env["ASAN_OPTIONS"] = f"detect_leaks={1 if platform.system() == 'Linux' else 0}:halt_on_error=1" - env["UBSAN_OPTIONS"] = "halt_on_error=1:print_stacktrace=1" +def _run_standalone(mode: str, extra: Path | None, crash_dir: Path) -> int: work = Path(tempfile.mkdtemp(prefix="th-fuzz-")) idna = work / "idna_harness" js = work / "js_harness" @@ -95,7 +102,14 @@ def _run_standalone(mode: str, extra: Path | None) -> int: "-DJM_STANDALONE", js, ) - js_seeds = sorted(str(path) for path in _JS_CORPUS.glob("*")) if mode == "deep" else [] + js_seeds = _files(_REGRESSIONS) + (_js_corpus(work) if mode == "deep" else []) + env = { + **os.environ, + "ASAN_OPTIONS": f"detect_leaks={1 if platform.system() == 'Linux' else 0}:halt_on_error=1", + "UBSAN_OPTIONS": "halt_on_error=1:print_stacktrace=1", + } + if mode == "deep": + env = _private_reports(env, crash_dir) for binary, seeds in ((idna, _seed_files("idna", extra)), (phone, _seed_files("phone", extra)), (js, js_seeds)): if (result := subprocess.run([str(binary), *seeds], env=env, check=False)).returncode != 0: print(f"SANITIZER ABORT in {binary.name} (exit {result.returncode})", file=sys.stderr) @@ -103,65 +117,111 @@ def _run_standalone(mode: str, extra: Path | None) -> int: return 0 -def _build_extension(build_dir: Path) -> None: +def _compile(harness: Path, sources: list[Path], macro: str, binary: Path) -> None: cmd = [ - "uv", - "pip", - "install", - "--reinstall", - "--no-deps", - "--no-build-isolation", - "--editable", - str(_ROOT), - f"--config-settings=build-dir={build_dir}", - "--config-settings=setup-args=-Dc_args=-fsanitize=address,undefined", - "--config-settings=setup-args=-Dc_link_args=-fsanitize=address,undefined", - "--config-settings=setup-args=-Dbuildtype=debugoptimized", + _CC, + macro, + "-fsanitize=address,undefined", + "-fno-omit-frame-pointer", + "-g", + "-O1", + "-Wall", + "-Wextra", + "-Werror", + "-I", + str(_ROOT / "src" / "turbohtml" / "_c"), + str(harness), + *[str(path) for path in sources], + "-o", + str(binary), ] print("$", " ".join(cmd)) - subprocess.run(cmd, check=True, env={**os.environ, "CC": _CC}) + subprocess.run(cmd, check=True) -def _run_inprocess(mode: str, minutes: float) -> int: - """Re-exec the interpreter under the ASan preload to drive the coupled targets through the public API.""" - repro = Path(tempfile.mkdtemp(prefix="th-fuzz-repro-")) / "current_input.bin" - cmd = [ - sys.executable, - str(_FUZZ / "_targets.py"), - "--mode", - mode, - "--corpus-dir", - str(_CORPUS), - "--repro", - str(repro), - "--minutes", - str(minutes), - ] - result = subprocess.run(cmd, env=_asan_preload(), check=False) - if result.returncode not in {0, 1}: - crasher = repro.read_bytes() if repro.exists() else b"" - print(f"SANITIZER ABORT in-process (exit {result.returncode})", file=sys.stderr) - print(f"crashing input ({len(crasher)} bytes): {crasher[:400]!r}", file=sys.stderr) - return result.returncode +def _js_corpus(work: Path) -> list[str]: + # the fixtures are JSON rows, so each input becomes its own file for the harness, as tools/js_sanitize.py does + files = [] + for index, row in enumerate( + row for fixture in sorted(_JS_CORPUS.glob("*.json")) for row in json.loads(fixture.read_text(encoding="utf-8")) + ): + (path := work / f"corpus-{index}.js").write_text(row["input"], encoding="utf-8") + files.append(str(path)) + return files -def main() -> int: - """Return 0 when every harness stays clean, nonzero on the first sanitizer abort or soft finding.""" - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--mode", choices=("smoke", "deep"), default="smoke") - parser.add_argument("--minutes", type=float, default=1.0, help="deep-mode wall-clock budget per in-process target") - parser.add_argument("--build", action="store_true", help="build the ASan extension here (tox builds it otherwise)") - parser.add_argument("--extra-corpus", type=Path, default=None, help="a second seed directory (vendored test data)") - parser.add_argument("--skip-inprocess", action="store_true", help="only run the standalone C harnesses") - args = parser.parse_args() +def _seed_files(target: str, extra: Path | None) -> list[str]: + return _files(_REGRESSIONS) + _files(_CORPUS / target) + ([] if extra is None else _files(extra / target)) + + +def _files(directory: Path) -> list[str]: + return sorted(str(path) for path in directory.glob("*") if path.is_file()) + + +def _run_inprocess(mode: str, minutes: float, rng_seed: int, crash_dir: Path) -> int: + env = {"PYTHONHASHSEED": "0", **_asan_preload()} + if mode == "deep": + env = _private_reports(env, crash_dir) + status = 0 + for allocator in _ALLOCATORS: + repro_dir = Path(tempfile.mkdtemp(prefix="th-fuzz-repro-")) + context = f"PYTHONMALLOC={allocator} PYTHONHASHSEED={env['PYTHONHASHSEED']}" + print(f"in-process pass: {context}, crashers kept in {crash_dir}", flush=True) + result = subprocess.run( + [ + sys.executable, + str(_FUZZ / "_targets.py"), + "--mode", + mode, + "--corpus-dir", + str(_CORPUS), + "--regression-dir", + str(_REGRESSIONS), + "--repro-dir", + str(repro_dir), + "--crash-dir", + str(crash_dir), + "--minutes", + str(minutes / len(_ALLOCATORS)), + "--rng-seed", + str(rng_seed), + ], + env={**env, "PYTHONMALLOC": allocator}, + check=False, + ) + if result.returncode not in {0, 1}: + print(f"SANITIZER ABORT in-process (exit {result.returncode}) {context}", file=sys.stderr) + for repro in repro_dir.iterdir(): # the worker leaves only the input it died on, named after its target + data = repro.read_bytes() + digest = hashlib.sha256(data).hexdigest() + (crash_dir / f"crash-{digest}").write_bytes(data) + (crash_dir / f"crash-{digest}.replay").write_text(f"{context} rng-seed={rng_seed}\n") + print(f"crashing input: [{repro.name}] sha256={digest} bytes={len(data)}", file=sys.stderr) + return result.returncode + status |= result.returncode + return status - if args.build: - _build_extension(Path(tempfile.mkdtemp(prefix="th-fuzz-build-"))) - if (code := _run_standalone(args.mode, args.extra_corpus)) != 0: - return code - if args.skip_inprocess: - return 0 - return _run_inprocess(args.mode, args.minutes) + +def _asan_preload() -> dict[str, str]: + """Build the env preloading the ASan runtime ahead of the interpreter so the instrumented .so's interceptors arm.""" + on_linux = platform.system() == "Linux" + flag = f"libclang_rt.asan-{platform.machine()}.so" if on_linux else "libclang_rt.asan_osx_dynamic.dylib" + runtime = subprocess.run( + [_CC, f"-print-file-name={flag}"], capture_output=True, text=True, check=True + ).stdout.strip() + return { + **os.environ, + "LD_PRELOAD" if on_linux else "DYLD_INSERT_LIBRARIES": runtime, + "ASAN_OPTIONS": "detect_leaks=0:halt_on_error=1:abort_on_error=1", + "UBSAN_OPTIONS": "halt_on_error=1:print_stacktrace=1", + } + + +def _private_reports(env: dict[str, str], crash_dir: Path) -> dict[str, str]: + # a sanitizer report names the faulting function and line, which discloses an unfixed bug in a public CI log, so a + # deep run writes reports beside the crashers, where the workflow encrypts them + log = f":log_path={crash_dir / 'crash-sanitizer'}" + return {**env, "ASAN_OPTIONS": env["ASAN_OPTIONS"] + log, "UBSAN_OPTIONS": env["UBSAN_OPTIONS"] + log} if __name__ == "__main__": diff --git a/tox.toml b/tox.toml index cd8256cc2..10d657449 100644 --- a/tox.toml +++ b/tox.toml @@ -379,7 +379,10 @@ commands = [ description = "coverage-guided/mutation deep fuzz of every harness under ASan/UBSan (scheduled/manual, not a PR gate)" deps = [] # override the base env's gcovr; this env instruments for ASan, not coverage dependency_groups = [ "test" ] -set_env = { CC = "clang" } +pass_env = [ "FUZZ_RNG_SEED" ] +# tox draws a fresh hash seed per run unless set_env pins it, and the XML duplicate-attribute hash salts from it, so a +# find on a hash-dependent path would not replay; fuzz.py logs the seed at the start of each pass +set_env = { CC = "clang", PYTHONHASHSEED = "0" } commands_pre = [ [ "uv", @@ -407,10 +410,10 @@ commands = [ ] [env.fuzz-smoke] -description = "replay the benign fuzz seed corpus through every harness under ASan/UBSan (the fast per-PR gate)" +description = "replay past fuzz finds and the seed corpus through every harness under ASan/UBSan (the per-PR gate)" deps = [] # override the base env's gcovr; this env instruments for ASan, not coverage dependency_groups = [ "test" ] # carries the meson-python build backend the no-build-isolation install below needs -set_env = { CC = "clang" } +set_env = { CC = "clang", PYTHONHASHSEED = "0" } commands_pre = [ # build the extension with the sanitizers so the in-process harness faults on a C memory error. b_sanitize trips # meson's linker capability check for an extension module on macOS, so pass the flag through c_args/c_link_args, which