Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs/changelog/1200.bugfix.rst
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Keep generated corpus bytes in separate executable target directories with their production metadata.
10 changes: 8 additions & 2 deletions tests/atheris/test_public_targets.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,13 +56,19 @@ def test_atheris_public_consumers_run_with_native_coverage(tmp_path: Path) -> No
check=True,
)
targets: Final = json.loads(inventory.stdout)
assert (targets["owners"], len(targets["targets"])) == (209, 28)
assert (targets["owners"], len(targets["targets"])) == (209, 29)
outcomes: Final[dict[str, dict[str, int | str]]] = {}
for target in targets["targets"]:
corpus: Final = tmp_path / target
corpus.mkdir()
seed: Final = (
b"<root>one</root>" if target == "xml-schema" else b"p.x" if target == "css-translate" else "水😀".encode()
b"<root>one</root>"
if target == "xml-schema"
else b"p.x"
if target == "css-translate"
else b"p { color: red; }"
if target == "css-stylesheet"
else "水😀".encode()
)
(corpus / "utf8").write_bytes(seed)
counters: Final = tmp_path / f"{target}-gcda"
Expand Down
2 changes: 1 addition & 1 deletion tests/test_fuzz_atheris_content_targets.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@ def test_content_target_behaviors(target: Target, data: bytes) -> None:

def test_content_targets_have_unique_owners() -> None:
exports: Final = [export for target in _TARGETS for export in target.exports]
assert len(exports) == len(set(exports)) == 92
assert len(exports) == len(set(exports)) == 91


def test_content_sanitizer_removes_attributes_and_comments() -> None:
Expand Down
112 changes: 112 additions & 0 deletions tests/test_fuzz_atheris_corpora.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,112 @@
from __future__ import annotations

import hashlib
import json
import random
from typing import TYPE_CHECKING, Final

import pytest
from fuzz.atheris_corpora import main, write_corpora
from fuzz.html_structure_generators import html_grammar_complete
from fuzz.structure_generators import Identifier, Production, compile_grammar, generate

if TYPE_CHECKING:
from pathlib import Path

from fuzz.atheris_corpora import CorpusProfile


@pytest.mark.parametrize(
("profile", "production", "source", "targets"),
[
pytest.param(
"html",
"html:document:fixture",
b"<html><body><p>x</p></body></html>",
("html-document", "html-incremental", "html-tokenizer"),
id="document",
),
pytest.param("html", "html:fragment:fixture", b"<p>x</p>", ("html-fragment", "html-tokenizer"), id="fragment"),
pytest.param("xml", "xml:fixture", b"<root>text</root>", ("xml-schema",), id="xml"),
pytest.param("css-stylesheet", "css:sheet", b"p { color: red; }", ("css-stylesheet",), id="stylesheet"),
pytest.param("css-declaration", "css:declaration", b"color: red;", ("css-object-model",), id="declaration"),
pytest.param("css-selector", "css:selector", b"p.x", ("css-translate",), id="selector"),
],
)
def test_corpus_routes_exact_bytes(
tmp_path: Path, profile: CorpusProfile, production: str, source: bytes, targets: tuple[str, ...]
) -> None:
grammar: Final = compile_grammar((Production(production, "root", (source,), 1, "fixture:1"),), "root")
case: Final = generate(grammar, random.SystemRandom())
manifest: Final = write_corpora((case,), grammar, tmp_path, profile)
assert (
tuple(entry.target for entry in manifest.entries),
tuple((tmp_path / entry.file).read_bytes() for entry in manifest.entries),
tuple(entry.productions for entry in manifest.entries),
manifest.rejected,
) == (targets, (source,) * len(targets), (((production, "fixture:1"),),) * len(targets), ())
assert (
json.loads((tmp_path / "manifest.json").read_text())["entries"][0]["sha256"]
== hashlib.sha256(source).hexdigest()
)


def test_corpus_deduplicates_bytes_preserves_traces(tmp_path: Path) -> None:
grammar: Final = compile_grammar(
(
Production("first", "root", (b"<p>x</p>",), 1, "fixture:1"),
Production("second", "root", (b"<p>x</p>",), 1, "fixture:2"),
),
"root",
)
cases: Final = tuple(generate(grammar, random.SystemRandom(), force=name) for name in ("first", "second"))
manifest: Final = write_corpora(cases, grammar, tmp_path, "html")
assert (
len(tuple((tmp_path / "html-fragment").iterdir())),
tuple(entry.productions for entry in manifest.entries if entry.target == "html-fragment"),
) == (1, ((("first", "fixture:1"),), (("second", "fixture:2"),)))


def test_corpus_rejects_invalid_input_before_writing(tmp_path: Path) -> None:
grammar: Final = compile_grammar((Production("invalid", "root", (b"\xff",), 1, "fixture:1"),), "root")
manifest: Final = write_corpora((generate(grammar, random.SystemRandom()),), grammar, tmp_path, "css-stylesheet")
assert (
manifest.entries,
tuple(entry.exception for entry in manifest.rejected),
(tmp_path / "css-stylesheet").exists(),
) == ((), ("UnicodeDecodeError",), False)


def test_corpus_cli_creates_target_files(tmp_path: Path) -> None:
assert main(("--output", str(tmp_path), "--count", "1", "--seed", "1")) == 0
manifest: Final = json.loads((tmp_path / "manifest.json").read_text())
assert manifest["entries"]
assert all((tmp_path / entry["file"]).is_file() for entry in manifest["entries"])


def test_corpus_cli_empty_count(tmp_path: Path) -> None:
assert main(("--output", str(tmp_path), "--count", "0")) == 0
assert json.loads((tmp_path / "manifest.json").read_text()) == {"entries": [], "rejected": []}


def test_corpus_cli_sweep_preserves_productions(tmp_path: Path) -> None:
assert main(("--output", str(tmp_path), "--count", "0", "--sweep")) == 0
manifest: Final = json.loads((tmp_path / "manifest.json").read_text())
assert {production.name for production in html_grammar_complete().productions} <= {
production[0] for entry in manifest["entries"] for production in entry["productions"]
}


def test_corpus_preserves_generated_bindings(tmp_path: Path) -> None:
grammar: Final = compile_grammar(
(
Production(
"bound", "root", (b'<p id="', Identifier("element", definition=True), b'">text</p>'), 1, "fixture:1"
),
),
"root",
)
case: Final = generate(grammar, random.SystemRandom())
manifest: Final = write_corpora((case,), grammar, tmp_path, "html")
assert tuple(entry.bindings for entry in manifest.entries) == (case.bindings, case.bindings)
assert json.loads((tmp_path / "manifest.json").read_text())["entries"][0]["bindings"] == [["element", "x"]]
42 changes: 42 additions & 0 deletions tests/test_fuzz_atheris_stylesheet_targets.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,42 @@
from __future__ import annotations

from typing import TYPE_CHECKING

import pytest
from fuzz.atheris_stylesheet_targets import stylesheet_observation, stylesheet_targets
from fuzz.atheris_targets import owner_inventory

if TYPE_CHECKING:
from fuzz.atheris_registry import Target


@pytest.mark.parametrize(
("source", "expected"),
[
pytest.param(b"", "", id="empty"),
pytest.param(b"p { color: #ff0000; margin: 0px; }", "p{color:red;margin:0}", id="sheet"),
pytest.param('p { content: "水😀"; }'.encode(), 'p{content:"水😀"}', id="unicode"),
pytest.param(b'p { content: "</style>"; color: red; }', 'p{content:"</style>";color:red}', id="html-boundary"),
],
)
def test_stylesheet_consumer_exact_source(source: bytes, expected: str) -> None:
assert stylesheet_observation(source) == expected


@pytest.mark.parametrize("target", stylesheet_targets(), ids=lambda target: target.name)
def test_stylesheet_callback_rejects_invalid_utf8(target: Target) -> None:
with pytest.raises(UnicodeDecodeError):
target.callback(b"\xff")


def test_stylesheet_owns_public_minifier() -> None:
assert owner_inventory()["turbohtml.clean.minify_css"] == "css-stylesheet"


@pytest.mark.parametrize("kind", ["fixpoint", "selector"])
def test_stylesheet_consumer_rejects_wrong_output(kind: str) -> None:
def wrong(source: str) -> str:
return source + "x" if kind == "fixpoint" else "q{color:red}"

with pytest.raises(AssertionError):
stylesheet_observation(b"p { color: red; }", wrong)
1 change: 0 additions & 1 deletion tools/fuzz/atheris_content_targets.py
Original file line number Diff line number Diff line change
Expand Up @@ -102,7 +102,6 @@ def content_targets() -> tuple[Target, ...]:
"JSMinify",
"Minify",
"minify",
"minify_css",
"minify_css_inline",
"minify_js",
)
Expand Down
136 changes: 136 additions & 0 deletions tools/fuzz/atheris_corpora.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,136 @@
"""Keep generation metadata outside byte files consumed by libFuzzer."""

from __future__ import annotations

import argparse
import hashlib
import json
import random
from dataclasses import asdict, dataclass
from pathlib import Path
from typing import TYPE_CHECKING, Final, Literal, cast

from .atheris_targets import public_targets
from .html_structure_generators import html_document, html_generate, html_grammar_complete
from .structure_generators import generation_sweep

if TYPE_CHECKING:
from collections.abc import Iterable, Sequence
from types import FunctionType

from .structure_generators import Generated, Grammar

__all__ = ["CorpusEntry", "CorpusManifest", "CorpusProfile", "RejectedEntry", "main", "write_corpora"]

CorpusProfile = Literal["html", "xml", "css-stylesheet", "css-declaration", "css-selector"]


def main(argv: Sequence[str] | None = None) -> int:
"""Export generated HTML into its executable parsing contexts."""
parser: Final = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--output", type=Path, required=True)
parser.add_argument("--count", type=int, default=1)
parser.add_argument("--seed", type=int, default=0)
parser.add_argument("--budget", type=int, default=30)
parser.add_argument("--sweep", action="store_true")
arguments: Final = parser.parse_args(argv)
generator: Final = random.Random(arguments.seed)
cases: Final = (
*(generation_sweep(html_grammar_complete(), budget=arguments.budget) if arguments.sweep else ()),
*(html_generate(generator, arguments.budget) for _ in range(arguments.count)),
)
write_corpora(cases, html_grammar_complete(), arguments.output, "html")
return 0


def write_corpora(
cases: Iterable[Generated], grammar: Grammar, directory: Path, profile: CorpusProfile
) -> CorpusManifest:
"""Preflight the executable consumer before admitting exact bytes to its corpus."""
targets: Final = {target.name: target for target in public_targets()}
origins: Final = {production.name: production.origin for production in grammar.productions}
entries: Final[list[CorpusEntry]] = []
rejected: Final[list[RejectedEntry]] = []
admitted: Final[set[tuple[str, str]]] = set()
directory.mkdir(parents=True, exist_ok=True)
for case in cases:
digest: Final = hashlib.sha256(case.data).hexdigest()
for name in _routes(case, profile):
target: Final = targets[name]
try:
target.callback(case.data)
except target.exceptions as error:
rejected.append(RejectedEntry(name, digest, type(error).__name__))
continue
destination: Final = directory / name / digest
if (name, digest) not in admitted:
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_bytes(case.data)
admitted.add((name, digest))
entries.append(
CorpusEntry(
name,
target.exports,
f"{target.callback.__module__}.{cast('FunctionType', target.callback).__qualname__}",
str(destination.relative_to(directory)),
digest,
case.nodes,
case.depth,
tuple((production, origins[production]) for production in case.productions),
case.bindings,
)
)
manifest: Final = CorpusManifest(tuple(entries), tuple(rejected))
(directory / "manifest.json").write_text(json.dumps(asdict(manifest), indent=2) + "\n", encoding="utf-8")
return manifest


def _routes(case: Generated, profile: CorpusProfile) -> tuple[str, ...]:
if profile == "html":
return (
("html-document", "html-incremental", "html-tokenizer")
if html_document(case)
else ("html-fragment", "html-tokenizer")
)
return {
"xml": ("xml-schema",),
"css-stylesheet": ("css-stylesheet",),
"css-declaration": ("css-object-model",),
"css-selector": ("css-translate",),
}[profile]


@dataclass(frozen=True)
class CorpusEntry:
"""Link exact corpus bytes to their generating trace and executable owner."""

target: str
exports: tuple[str, ...]
callback: str
file: str
sha256: str
generation_nodes: int
generation_depth: int
productions: tuple[tuple[str, str], ...]
bindings: tuple[tuple[str, str], ...]


@dataclass(frozen=True)
class RejectedEntry:
"""Retain documented input rejection outside accepted target directories."""

target: str
sha256: str
exception: str


@dataclass(frozen=True)
class CorpusManifest:
"""Keep accepted provenance and rejected-input evidence beside the corpus."""

entries: tuple[CorpusEntry, ...]
rejected: tuple[RejectedEntry, ...]


if __name__ == "__main__":
raise SystemExit(main())
40 changes: 40 additions & 0 deletions tools/fuzz/atheris_stylesheet_targets.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,40 @@
"""Whole stylesheet inputs must reach the public minifier without an HTML text boundary."""

from __future__ import annotations

from typing import TYPE_CHECKING, Final

from turbohtml.clean import minify_css

from .atheris_registry import Target
from .round_trip_oracles import OutOfScopeError, css_semantics_check, fixpoint_check

if TYPE_CHECKING:
from collections.abc import Callable

__all__ = ["stylesheet_observation", "stylesheet_targets"]


def stylesheet_targets() -> tuple[Target, ...]:
"""Own the public minifier through an exact stylesheet input."""
return (
Target("css-stylesheet", _stylesheet, ("turbohtml.clean.minify_css",), (UnicodeDecodeError, OutOfScopeError)),
)


def _stylesheet(data: bytes) -> None:
stylesheet_observation(data)


def stylesheet_observation(data: bytes, printer: Callable[[str], str] = minify_css) -> str:
"""Check supported values and selectors against a fixed DOM context."""
source: Final = data.decode("utf-8")
if failure := fixpoint_check(source, printer, numeric=True):
raise AssertionError(failure)
if failure := css_semantics_check(
'<style></style><p class="x">text<span id="target">child</span></p>',
printer,
stylesheet=source,
):
raise AssertionError(failure)
return printer(source)
3 changes: 2 additions & 1 deletion tools/fuzz/atheris_targets.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@
from .atheris_parser_targets import parser_targets
from .atheris_reference_targets import reference_targets
from .atheris_registry import validate_owners
from .atheris_stylesheet_targets import stylesheet_targets

if TYPE_CHECKING:
from collections.abc import Sequence
Expand Down Expand Up @@ -68,7 +69,7 @@ def main(argv: Sequence[str] | None = None) -> int:

def public_targets() -> tuple[Target, ...]:
"""Each group owns separate modules and qualified re-export aliases."""
targets: Final = parser_targets() + reference_targets() + content_targets() + dom_targets()
targets: Final = parser_targets() + reference_targets() + content_targets() + dom_targets() + stylesheet_targets()
validate_owners(targets, MODULES)
return targets

Expand Down
Loading
Loading