Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 33 additions & 0 deletions .github/workflows/fuzz.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,39 @@ jobs:
run: uvx --with tox-uv tox run -e fuzz-smoke
env:
UV_PYTHON_PREFERENCE: only-managed
oracle:
name: 🧪 sanitizer wrong-output oracles
runs-on: ubuntu-24.04
# a sanitize bug usually returns unsafe markup without crashing, so this lane compares the sanitizer with an
# independent re-parse (html5lib-python, parse5 to confirm), the random Policy it ran under, and the WHATWG URL
# scheme. Pull requests and pushes replay only the deterministic seed pass; scheduled and manual runs add
# MutaGen-style generation for a wall-clock budget.
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
with:
fetch-depth: 0
persist-credentials: false
- name: 🟢 Install Node
# zizmor: ignore[cache-poisoning] no cache: input is set, so setup-node writes no restorable cache
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
with:
node-version: "22"
- name: 📦 Install parse5 for the confirming re-parse
run: npm ci
working-directory: tools/bench/node
- name: 🔄 Install uv
uses: astral-sh/setup-uv@c18668ad3cf93ea998bef934396af7bb5c839dc7 # v10.2.0
with:
enable-cache: false
- name: ✅ Run fuzz-oracle
# the seed regenerates every finding from the public code, so it stays out of the tox command line it echoes
run: |
FUZZ_RNG_SEED="$(python3 -c 'import secrets; print(secrets.randbits(32))')"
export FUZZ_RNG_SEED
uvx --with tox-uv tox run -e fuzz-oracle -- --minutes "${MINUTES}"
env:
UV_PYTHON_PREFERENCE: only-managed
MINUTES: ${{ (github.event_name == 'pull_request' || github.event_name == 'push') && '0' || github.event.schedule == '0 4 * * 1' && '20' || '8' }}
deep:
name: 🕳️ ASan/UBSan deep (mutation + structural)
runs-on: ubuntu-24.04
Expand Down
13 changes: 13 additions & 0 deletions .gitmodules
Original file line number Diff line number Diff line change
Expand Up @@ -70,3 +70,16 @@
path = tests/conformance/linkify-it
url = https://github.com/markdown-it/linkify-it
shallow = true
[submodule "tools/fuzz-data/h5sc"]
path = tools/fuzz-data/h5sc
url = https://github.com/cure53/H5SC
shallow = true
[submodule "tools/fuzz-data/google-fuzzing"]
path = tools/fuzz-data/google-fuzzing
url = https://github.com/google/fuzzing
shallow = true
[submodule "tools/fuzz-data/wpt"]
path = tools/fuzz-data/wpt
url = https://github.com/web-platform-tests/wpt
shallow = true
update = none
11 changes: 11 additions & 0 deletions docs/development/index.rst
Original file line number Diff line number Diff line change
Expand Up @@ -165,6 +165,17 @@ To read the crashers of a failed run, download its ``fuzz-crashes`` artifact and
$ gh run download <run-id> --name fuzz-crashes
$ age --decrypt --identity fuzz-crashes.key --output crash-<sha256> crash-<sha256>.age

A sanitizer bug usually returns unsafe or altered markup without crashing, so ``fuzz-oracle`` (``fuzz.py --mode
oracle``) checks ``turbohtml.clean.sanitize`` against code it does not share. The output must re-parse in
html5lib-python into the tree the sanitizer judged, obey the random ``Policy`` it ran under, and keep or drop each URL
the way the WHATWG URL parser reads its scheme. ``--minutes 0`` replays only the seed corpora. The run writes findings
to a JSON report and logs only their hashes.

.. code-block:: console

$ tox r -e fuzz-oracle -- --minutes 0 # the per-PR seed pass
$ tox r -e fuzz-oracle -- --minutes 10 # adds generated markup and URL obfuscations

Add a target by registering a ``bytes``-taking callable in ``_TARGETS`` (in-process) and dropping a representative
benign seed under ``tools/fuzz/corpus/<target>/``; add a standalone harness by mirroring ``idna_harness.c`` for any C
unit that compiles free of the CPython boundary. macOS ships no ``libFuzzer`` runtime with Apple Clang, so the
Expand Down
2 changes: 2 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -264,6 +264,7 @@ extend-exclude = [
"tests/conformance/python-phonenumbers",
"tests/conformance/unicodetools",
"tests/html5lib-tests",
"tools/fuzz-data",
"tools/html5lib-python"
] # vendored suites, not our code
format.preview = true
Expand Down Expand Up @@ -348,6 +349,7 @@ src.exclude = [
"tests/conformance/python-phonenumbers",
"tests/conformance/unicodetools",
"tests/html5lib-tests",
"tools/fuzz-data",
"tools/html5lib-python"
] # vendored suites, not our code
environment.extra-paths = [
Expand Down
8 changes: 2 additions & 6 deletions src/turbohtml/_c/serialize/minify.c
Original file line number Diff line number Diff line change
Expand Up @@ -571,9 +571,7 @@ static int mini_emit_script_js(sbuf *out, th_tree *tree, th_node *node, const th
}
Py_ssize_t pos = 0;
for (th_node *child = node->first_child; child != NULL; child = child->next_sibling) {
if (child->text_len > 0) { /* a parsed empty comment has a NULL text pointer */
memcpy(src + pos, need_text(tree, child), (size_t)child->text_len * sizeof(Py_UCS4));
}
memcpy(src + pos, need_text(tree, child), (size_t)child->text_len * sizeof(Py_UCS4));
pos += child->text_len;
}
/* errlen 0: the HTML path discards the message and falls back to verbatim instead */
Expand Down Expand Up @@ -614,9 +612,7 @@ static int mini_emit_style_css(sbuf *out, th_tree *tree, th_node *node, int base
}
Py_ssize_t pos = 0;
for (th_node *child = node->first_child; child != NULL; child = child->next_sibling) {
if (child->text_len > 0) { /* a parsed empty comment has a NULL text pointer */
memcpy(src + pos, need_text(tree, child), (size_t)child->text_len * sizeof(Py_UCS4));
}
memcpy(src + pos, need_text(tree, child), (size_t)child->text_len * sizeof(Py_UCS4));
pos += child->text_len;
}
Py_ssize_t css_len;
Expand Down
7 changes: 3 additions & 4 deletions src/turbohtml/_c/tokenizer/xml.c
Original file line number Diff line number Diff line change
Expand Up @@ -396,15 +396,14 @@ static int push_open(xml_parser *parser, th_node *element) {
return 0;
}

/* Copy a code-point run (an entity-normalized attribute value) into the arena. */
/* Copy a namespace URI into the arena. consume_namespace_decl rejects an empty one first, so src is never the
NULL scratch buffer. */
static Py_UCS4 *arena_copy(th_tree *tree, const Py_UCS4 *src, Py_ssize_t len) {
Py_UCS4 *out = arena_alloc(tree, len * (Py_ssize_t)sizeof(Py_UCS4));
if (out == NULL) { /* GCOVR_EXCL_BR_LINE: allocation failure cannot be forced from a test */
return NULL; /* GCOVR_EXCL_LINE: allocation-failure path */
}
if (len > 0) { /* the scratch buffer stays NULL until a value pushes a code point */
memcpy(out, src, (size_t)len * sizeof(Py_UCS4));
}
memcpy(out, src, (size_t)len * sizeof(Py_UCS4));
return out;
}

Expand Down
8 changes: 2 additions & 6 deletions tests/serialize/test_minify.py
Original file line number Diff line number Diff line change
Expand Up @@ -475,12 +475,8 @@ def test_body_start_omitted_with_empty_first_text() -> None:
pytest.param("style", "a { color: red }", Minify(minify_css=CSSMinify()), "a{color:red}", id="style"),
],
)
def test_minify_raw_text_skips_an_empty_comment_child(tag: str, source: str, layout: Minify, expected: str) -> None:
# a parsed empty comment has a NULL text pointer, which memcpy may not receive even for length 0
document: Final = parse(f"<!----><{tag}>{source}</{tag}>")
element = document.find(tag)
assert isinstance(element, Element)
element.insert(0, document.children[0])
def test_minify_raw_text_joins_an_empty_text_child(tag: str, source: str, layout: Minify, expected: str) -> None:
element: Final = Element(tag, children=[Text(""), Text(source)])
assert element.serialize(Html(layout=layout)) == f"<{tag}>{expected}</{tag}>"


Expand Down
5 changes: 0 additions & 5 deletions tests/tokenizer/test_xml.py
Original file line number Diff line number Diff line change
Expand Up @@ -221,11 +221,6 @@ def test_attribute_value(markup: str, expected: str) -> None:
assert dict(root_of(parse_xml(markup)).attrs) == {"a": expected}


def test_empty_prefixed_namespace_declaration_is_kept() -> None:
# the empty URI is copied from a NULL scratch buffer, which memcpy may not receive even for length 0
assert dict(root_of(parse_xml('<r xmlns:p=""/>')).attrs) == {"xmlns:p": ""}


@pytest.mark.parametrize(
("markup", "expected"),
[
Expand Down
51 changes: 51 additions & 0 deletions tools/bench/node/parse5_tree_runner.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,51 @@
// Second independent re-parser for the sanitizer mutation-freedom oracle (tools/fuzz/sanitize_oracles.py). Reads a JSON
// array of HTML strings on stdin, parses each as a <div> fragment with scripting on (what innerHTML does), and writes a
// JSON array of html5lib-tests tree dumps. The dump mirrors parse5's own test serializer
// (tests/conformance/parse5/test/utils/serialize-to-dat-file-format.ts), which is TypeScript and not shipped in the npm
// package, so the Python side compares parse5, html5lib-python and turbohtml in one format.
const { defaultTreeAdapter: adapter, html, parseFragment } = require("parse5");

const PREFIXES = { [html.NS.SVG]: "svg ", [html.NS.MATHML]: "math " };

function dump(nodes, indent, out) {
const pad = "|".padEnd(indent + 2, " ");
for (let node of nodes) {
if (adapter.isCommentNode(node)) {
out.push(`${pad}<!-- ${adapter.getCommentNodeContent(node)} -->`);
} else if (adapter.isTextNode(node)) {
out.push(`${pad}"${adapter.getTextNodeContent(node)}"`);
} else {
const tag = adapter.getTagName(node);
const namespace = adapter.getNamespaceURI(node);
out.push(`${pad}<${PREFIXES[namespace] ?? ""}${tag}>`);
let childIndent = indent + 2;
const attrPad = "|".padEnd(childIndent + 2, " ");
out.push(
...adapter
.getAttrList(node)
.map((attr) => `${attrPad}${attr.prefix ? `${attr.prefix} ` : ""}${attr.name}="${attr.value}"`)
.sort(),
);
if (tag === "template" && namespace === html.NS.HTML) {
out.push(`${attrPad}content`);
childIndent += 2;
node = adapter.getTemplateContent(node);
}
dump(adapter.getChildNodes(node), childIndent, out);
}
}
return out;
}

let input = "";
process.stdin.setEncoding("utf8");
process.stdin.on("data", (chunk) => {
input += chunk;
});
process.stdin.on("end", () => {
const dumps = JSON.parse(input).map((markup) => {
const context = adapter.createElement("div", html.NS.HTML, []);
return dump(adapter.getChildNodes(parseFragment(context, markup, { scriptingEnabled: true })), 0, []).join("\n");
});
process.stdout.write(JSON.stringify(dumps));
});
1 change: 1 addition & 0 deletions tools/fuzz-data/google-fuzzing
Submodule google-fuzzing added at 734e55
1 change: 1 addition & 0 deletions tools/fuzz-data/h5sc
Submodule h5sc added at 1b796a
1 change: 1 addition & 0 deletions tools/fuzz-data/wpt
Submodule wpt added at c5e80e
61 changes: 57 additions & 4 deletions tools/fuzz/fuzz.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,9 +13,10 @@

``smoke`` replays the past finds under ``tests/fuzz_regressions`` and a benign seed corpus once (fast, deterministic,
gates every PR). ``deep`` adds a mutation loop and structural probes for a per-target budget (the scheduled/manual run
that hunts for crashes). A crashing input lands in ``--crash-dir`` as ``crash-<sha256>``, and the log names it only by
hash, length, harness and seeds, because CI logs on a public repository are public. The in-process extension is
expected to be pre-built by the tox env; ``--build`` builds it here for a local run.
that hunts for crashes). ``oracle`` runs the sanitizer wrong-output oracles (``sanitize_oracles.py``) instead,
because a sanitizer bug usually returns unsafe markup without crashing. A crashing input lands in ``--crash-dir`` as
``crash-<sha256>``, and the log names it only by hash, length and harness, because CI logs on a public repository are
public. The in-process extension is expected to be pre-built by the tox env; ``--build`` builds it here for a local run.
"""

from __future__ import annotations
Expand All @@ -41,12 +42,13 @@
# pymalloc carves small objects out of pools, so an over-read that stays inside a pool never reaches ASan's redzones;
# PYTHONMALLOC=malloc hands every PyMem call to the intercepted system allocator (v1 FUZZ-2 was silent under pymalloc)
_ALLOCATORS: Final[tuple[str, ...]] = ("pymalloc", "malloc")
_WPT: Final = "tools/fuzz-data/wpt"


def main() -> int:
"""Return 0 when every harness stays clean, nonzero on the first sanitizer abort or soft finding."""
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--mode", choices=("smoke", "deep"), default="smoke")
parser.add_argument("--mode", choices=("smoke", "deep", "oracle"), default="smoke")
parser.add_argument(
"--minutes", type=float, default=1.0, help="deep-mode budget per in-process target, split across the allocators"
)
Expand All @@ -65,6 +67,8 @@ def main() -> int:
if args.build:
_build_extension(Path(tempfile.mkdtemp(prefix="th-fuzz-build-")))
args.crash_dir.mkdir(parents=True, exist_ok=True)
if args.mode == "oracle":
return _run_oracles(args.minutes, args.rng_seed, args.crash_dir)
if (code := _run_standalone(args.mode, args.extra_corpus, args.crash_dir)) != 0:
return code
return 0 if args.skip_inprocess else _run_inprocess(args.mode, args.minutes, args.rng_seed, args.crash_dir)
Expand All @@ -89,6 +93,55 @@ def _build_extension(build_dir: Path) -> None:
subprocess.run(cmd, check=True, env={**os.environ, "CC": _CC})


def _run_oracles(minutes: float, rng_seed: int, crash_dir: Path) -> int:
"""Run the sanitizer oracles under the ASan preload, so a C fault on the way still aborts with a stack trace."""
_checkout_sparse_wpt()
env = _asan_preload()
if minutes > 0:
env = _private_reports(env, crash_dir)
result = subprocess.run(
[
sys.executable,
str(_FUZZ / "sanitize_oracles.py"),
"--minutes",
str(minutes),
"--rng-seed",
str(rng_seed),
"--repro",
str(crash_dir / "current_input.html"),
"--report",
str(crash_dir / "crash-findings.json"),
],
env=env,
check=False,
)
if result.returncode not in {0, 1, 2}:
print(f"SANITIZER ABORT in the oracle run (exit {result.returncode})", file=sys.stderr)
return result.returncode


def _checkout_sparse_wpt() -> None:
"""
Check out only ``sanitizer-api/`` of the pinned WPT submodule.

A depth-1 WPT clone is about 1 GB, so ``.gitmodules`` marks the submodule ``update = none`` (plain and recursive
``git submodule update`` skip it) and this fetches the recorded commit blob-less with a sparse checkout instead.
"""
if (target := _ROOT / _WPT / "sanitizer-api").is_dir():
return
commit = _git("rev-parse", f"HEAD:{_WPT}")
url = _git("config", "--file", ".gitmodules", f"submodule.{_WPT}.url")
_git("init", "--quiet", _WPT)
_git("-C", _WPT, "fetch", "--quiet", "--depth", "1", "--filter=blob:none", url, commit)
_git("-C", _WPT, "sparse-checkout", "set", "--no-cone", "/sanitizer-api/", "/LICENSE.md")
_git("-C", _WPT, "checkout", "--quiet", "FETCH_HEAD")
print(f"checked out {target.relative_to(_ROOT)} at {commit[:12]}")


def _git(*args: str) -> str:
return subprocess.run(["git", *args], cwd=_ROOT, capture_output=True, text=True, check=True).stdout.strip()


def _run_standalone(mode: str, extra: Path | None, crash_dir: Path) -> int:
work = Path(tempfile.mkdtemp(prefix="th-fuzz-"))
idna = work / "idna_harness"
Expand Down
Loading
Loading