Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs/changelog/1197.bugfix.rst
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Exercise qualified public APIs through checked Atheris consumers.
111 changes: 111 additions & 0 deletions tests/atheris/test_public_targets.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,111 @@
from __future__ import annotations

import importlib
import importlib.util
import json
import os
import sys
from pathlib import Path
from subprocess import run # ruff: ignore[suspicious-subprocess-import] - libFuzzer owns a separate interpreter.
from typing import TYPE_CHECKING, Final, cast

import pytest
from fuzz.atheris_runtime import build_runtime

if TYPE_CHECKING:
from typing import Protocol

class _Runtime(Protocol):
def path(self) -> str: ...


_CONSUME: Final = """
import sys
from pathlib import Path
from typing import Final
from fuzz.atheris_targets import public_targets
_TARGET: Final = next(target for target in public_targets() if target.name == sys.argv[1])
_TARGET.callback(Path(sys.argv[2]).read_bytes())
"""


_LIST: Final = """
import json
from fuzz.atheris_targets import owner_inventory, public_targets
print(json.dumps({"owners": len(owner_inventory()), "targets": [target.name for target in public_targets()]}))
"""


@pytest.mark.oracle
def test_atheris_public_consumers_run_with_native_coverage(tmp_path: Path) -> None:
if sys.platform != "linux" or importlib.util.find_spec("atheris") is None:
pytest.skip("Atheris's native runtime requires its optional Linux wheel")
runtime: Final = cast("_Runtime", importlib.import_module("atheris"))
library: Final = build_runtime(Path(runtime.path()) / "libclang_rt.fuzzer_no_main.a", tmp_path / "runtime")
environment: Final = {
**os.environ,
"LD_PRELOAD": str(library),
"PYTHONPATH": str(Path(__file__).parents[2] / "tools"),
"GCOV_PREFIX": str(tmp_path / "inventory-gcda"),
}
inventory: Final = run( # ruff: ignore[subprocess-without-shell-equals-true] - fixed interpreter and owned modules.
[sys.executable, "-c", _LIST],
env=environment,
capture_output=True,
text=True,
check=True,
)
targets: Final = json.loads(inventory.stdout)
assert (targets["owners"], len(targets["targets"])) == (209, 28)
outcomes: Final[dict[str, dict[str, int | str]]] = {}
for target in targets["targets"]:
corpus: Final = tmp_path / target
corpus.mkdir()
seed: Final = (
b"<root>one</root>" if target == "xml-schema" else b"p.x" if target == "css-translate" else "水😀".encode()
)
(corpus / "utf8").write_bytes(seed)
counters: Final = tmp_path / f"{target}-gcda"
run( # ruff: ignore[subprocess-without-shell-equals-true] - public callback and private native counters.
[sys.executable, "-c", _CONSUME, target, str(corpus / "utf8")],
env={**environment, "GCOV_PREFIX": str(counters)},
capture_output=True,
check=True,
)
result = run( # ruff: ignore[subprocess-without-shell-equals-true] - fixed interpreter and finite public corpus.
[
sys.executable,
"-m",
"fuzz.atheris_targets",
"--target",
target,
"--corpus",
str(corpus),
"-atheris_runs=8",
"-seed=1",
"-max_len=64",
"-detect_leaks=0",
],
env={**environment, "GCOV_PREFIX": str(tmp_path / f"{target}-gcda")},
capture_output=True,
check=False,
)
(tmp_path / f"{target}.log").write_bytes(result.stdout + result.stderr)
assert result.returncode == 0, (target, result.stdout + result.stderr)
assert b"ATHERIS_REJECTION_BRIDGE=1" in result.stderr
assert b"Done 8" in result.stderr
assert b"inline 8-bit counters" in result.stderr
assert b"PC tables" in result.stderr
assert b"cov:" in result.stderr
assert b"ft:" in result.stderr
assert b"Coverage symbols are being provided by a library other than libFuzzer" not in result.stderr
manifest: Final = json.loads(corpus.with_suffix(".json").read_text())
assert (manifest["target"], bool(manifest["exports"])) == (target, True)
native_counters: Final = tuple(counters.rglob("*.gcda"))
assert native_counters, target
outcomes[target] = {
"exit": result.returncode,
"corpus": str(corpus),
"native_counter_files": len(native_counters),
}
(tmp_path / "public-targets.json").write_text(json.dumps(outcomes, indent=2) + "\n", encoding="utf-8")
74 changes: 74 additions & 0 deletions tests/test_fuzz_atheris_content_targets.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,74 @@
from __future__ import annotations

from typing import TYPE_CHECKING, Final

import pytest
from fuzz.atheris_content_targets import (
content_targets,
minifier_check,
minifier_observation,
sanitizer_check,
sanitizer_observation,
stdlib_observation,
url_check,
)

from turbohtml.clean import Removed

if TYPE_CHECKING:
from fuzz.atheris_registry import Target

_TARGETS: Final = content_targets()


@pytest.mark.parametrize("target", _TARGETS, ids=lambda target: target.name)
@pytest.mark.parametrize("data", [b"", b"hello", bytes(range(256))], ids=["empty", "ascii", "byte-range"])
def test_content_target_behaviors(target: Target, data: bytes) -> None:
target.callback(data)


def test_content_targets_have_unique_owners() -> None:
exports: Final = [export for target in _TARGETS for export in target.exports]
assert len(exports) == len(set(exports)) == 92


def test_content_sanitizer_removes_attributes_and_comments() -> None:
assert sanitizer_observation(b"x") == (
"<b>value78</b><b>value78</b>",
[Removed("b", "title"), Removed("b", "onclick")],
)


def test_content_minifiers_transform_source() -> None:
assert minifier_observation(b"x") == (
"<html><head></head><body><p>value78</p></body></html>",
".value78{color:red}",
"color:red",
'const value="value78"',
)


def test_content_stdlib_decodes_chunked_references() -> None:
assert stdlib_observation(b"x") == "value78&"


def test_content_sanitizer_rejects_wrong_renderer_result() -> None:
def render(data: bytes) -> tuple[str, list[Removed]]:
return data.decode(), []

with pytest.raises(AssertionError, match="Content API mismatch"):
sanitizer_check(b"wrong", render)


@pytest.mark.parametrize("kind", ["css-semantics", "css-fixpoint", "js-fixpoint"])
def test_content_minifier_rejects_wrong_output(kind: str) -> None:
def wrong(source: str) -> str:
return "q{color:blue}" if kind == "css-semantics" else source + "x"

with pytest.raises(AssertionError, match="Content API mismatch"):
minifier_check(b"x", js=wrong) if kind == "js-fixpoint" else minifier_check(b"x", css=wrong)


def test_content_url_rejects_stable_wrong_host() -> None:
with pytest.raises(AssertionError, match="Content API mismatch"):
url_check(b"x", lambda _source: "https://wrong.example/")
71 changes: 71 additions & 0 deletions tests/test_fuzz_atheris_dom_targets.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,71 @@
from __future__ import annotations

from typing import Final

import pytest
from fuzz.atheris_dom_targets import DomObservation, dom_observation, dom_targets

_DOMAINS: Final = (
"dom-construction",
"dom-traversal",
"dom-query",
"dom-mutation",
"dom-range",
"dom-shadow",
"dom-locations",
"dom-rewrite",
"dom-sax",
"dom-treebuild",
)


@pytest.mark.parametrize("domain", _DOMAINS)
@pytest.mark.parametrize(
"data",
[
pytest.param(b"", id="empty"),
pytest.param(b"text", id="plain"),
pytest.param("é水😀".encode(), id="unicode"),
pytest.param(b"<&>\r\n\x00", id="escaped"),
],
)
def test_dom_independent_contract(domain: str, data: bytes) -> None:
observation: Final = dom_observation(data, domain)
assert observation.actual == observation.expected


@pytest.mark.parametrize("domain", _DOMAINS)
def test_dom_registered_callback(domain: str) -> None:
target: Final = next(target for target in dom_targets() if target.name == domain)
assert target.exceptions == (UnicodeDecodeError,)
target.callback(b"text")
with pytest.raises(UnicodeDecodeError):
target.callback(b"\xff")


def test_dom_ownership() -> None:
targets: Final = dom_targets()
exports: Final = [export for target in targets for export in target.exports]
assert (tuple(target.name for target in targets), len(exports), len(set(exports))) == (_DOMAINS, 64, 64)


def test_dom_verification_rejects_changed_result() -> None:
with pytest.raises(AssertionError, match=r"actual.*expected"):
DomObservation("actual", "expected").verify()


def test_dom_verification_accepts_observed_result() -> None:
assert dom_observation(b"text", "dom-range").verify() is None


def test_dom_range_literal() -> None:
assert dom_observation(b"<&>", "dom-range").actual == repr(("<p>&lt;&amp;&gt;</p>", 0, 1, 0, 1, False, "<&>"))


def test_dom_input_bound() -> None:
assert dom_observation(b"x" * 65, "dom-range") == dom_observation(b"x" * 64, "dom-range")


def test_dom_unknown_domain() -> None:
with pytest.raises(KeyError, match="unknown"):
dom_observation(b"text", "unknown")
3 changes: 2 additions & 1 deletion tests/test_fuzz_atheris_driver.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@ def consume(data: bytes) -> None:
accepted.append(data)

target = Target("cli", consume, ("turbohtml.__main__.main",), (UnicodeError,))
runtime = mocker.MagicMock(spec=["Setup", "Fuzz", "instrument_func"])
runtime = mocker.MagicMock(spec=["Setup", "Fuzz", "instrument_func", "instrument_all"])
runtime.instrument_func.side_effect = lambda callback: callback
mocker.patch("fuzz.atheris_driver.import_module", autospec=True, return_value=runtime)
mocker.patch("fuzz.atheris_driver.rejection_hook", autospec=True, return_value=partial(rejected.append, 1))
Expand All @@ -36,6 +36,7 @@ def native_loop() -> None:

runtime.Fuzz.side_effect = native_loop
fuzz([target], ["turbohtml.__main__"], "cli", ["driver", "-runs=2"])
runtime.instrument_all.assert_called_once_with()
assert (accepted, rejected, runtime.Setup.call_args.args[0], runtime.Setup.call_args.kwargs) == (
[b"valid"],
[1],
Expand Down
65 changes: 65 additions & 0 deletions tests/test_fuzz_atheris_parser_targets.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,65 @@
from __future__ import annotations

from typing import TYPE_CHECKING

import pytest
from fuzz.atheris_parser_targets import (
document_observation,
fragment_observation,
incremental_observation,
parser_targets,
token_observation,
)

if TYPE_CHECKING:
from collections.abc import Callable

from fuzz.atheris_registry import Target


@pytest.mark.parametrize(
"observe",
[
pytest.param(document_observation, id="document"),
pytest.param(incremental_observation, id="incremental"),
],
)
def test_atheris_parser_document_output(observe: Callable[[bytes], str]) -> None:
assert observe("<p>水😀</p>".encode()) == "<html><head></head><body><p>水😀</p></body></html>"


def test_atheris_parser_fragment_output() -> None:
assert fragment_observation(b"<p>x&amp;y</p>") == "<div><p>x&amp;y</p></div>"


def test_atheris_parser_token_fields() -> None:
assert token_observation(b'<p id="a">x</p>') == (
("START_TAG", "p", None, (("id", "a"),), False, 1, 0),
("TEXT", None, "x", None, False, 1, 10),
("END_TAG", "p", None, (), False, 1, 11),
)


@pytest.mark.parametrize("target", parser_targets(), ids=lambda target: target.name)
def test_atheris_parser_callback_rejects_invalid_utf8(target: Target) -> None:
with pytest.raises(UnicodeDecodeError):
target.callback(b"\xff")


def test_atheris_parser_exports_have_unique_consumers() -> None:
assert tuple((target.name, target.exports) for target in parser_targets()) == (
("html-document", ("turbohtml.parse", "turbohtml.Document", "turbohtml.Node")),
(
"html-fragment",
(
"turbohtml.parse_fragment",
"turbohtml.Element",
"turbohtml.Html",
"turbohtml.Formatter",
"turbohtml.Indent",
"turbohtml.Minify",
),
),
("html-incremental", ("turbohtml.IncrementalParser",)),
("html-tokenizer", ("turbohtml.tokenize", "turbohtml.Tokenizer", "turbohtml.Token", "turbohtml.TokenType")),
)
Loading
Loading