diff --git a/docs/adr/0003-rfc9162-merkle-domain-separation.md b/docs/adr/0003-rfc9162-merkle-domain-separation.md index bc39f441..d3a4ab32 100644 --- a/docs/adr/0003-rfc9162-merkle-domain-separation.md +++ b/docs/adr/0003-rfc9162-merkle-domain-separation.md @@ -2,7 +2,7 @@ **Status**: Accepted **Date**: 2026-05-10 -**Spec section**: Section 4.1.1, Section 3.2.2 (composite policy bundle), Section 3.2.3 (tool catalog hash), Section 3.2.5 (RAG corpus) +**Spec section**: Section 4.1.1, Section 3.2.2 (composite policy bundle), Section 2.2.3 (composite policy bundle), Section 3.2.3 (tool catalog hash), Section 3.2.5 (RAG corpus) ## Context @@ -15,10 +15,14 @@ Use the RFC 9162 (Certificate Transparency v2) Merkle tree construction with exp - Leaf nodes: `SHA-256(0x00 || leaf_data)` - Internal nodes: `SHA-256(0x01 || left_hash || right_hash)` +This ADR defines three Merkle hash operations: tool catalog leaves, corpus document leaves, and composite policy sub-bundle leaves. + Leaf data for tool entries: RFC 8785 canonical JSON of the tool descriptor (schema + description, sorted by tool name). Leaf data for corpus documents: RFC 8785 canonical JSON of the document descriptor (hash + identifier + ingested_at). Leaf data for composite policy sub-bundles (Section 3.2.2): the **raw digest bytes** of each sub-bundle hash, not the `sha256:`-prefixed hex string and not a JSON descriptor. Unlike the two above, this leaf carries no structured descriptor, because the ordering rule already fixes which sub-bundle each leaf is. +Section 4.1.1 of the specification is the normative definition of the shared construction. Sections 3.2.2, 3.2.3, and 3.2.5.1 normatively define each artifact's leaf data and ordering; they take precedence over this ADR for those details. RFC 8785 applies only where those sections define JSON as an input to a hash. The composite policy sub-bundle leaf uses raw digest bytes and does not use JSON canonicalization. + ## Rationale - RFC 9162 is a published IETF standard for Merkle tree construction, used in Certificate Transparency - a deployed, audited system @@ -31,7 +35,7 @@ Leaf data for composite policy sub-bundles (Section 3.2.2): the **raw digest byt **Simple concatenation Merkle (no domain separation)**: Vulnerable to second-preimage attacks as described above. Rejected. **BLAKE3 Merkle**: BLAKE3 has built-in domain separation for its tree construction. Rejected because BLAKE3 is not yet in the standard library of all target languages, and SHA-256 is sufficient for this use case. - +id`. Composite policy sub-bundles sorted by policy language identifer (`cear`, `rego`, `yaml-agt), per Section 3.2.2 **Flat hash (hash of concatenated hashes)**: Not a Merkle tree - does not support efficient membership proofs. Rejected because the spec's design supports future membership proof extensions. ## Consequences diff --git a/python/src/agent_manifest/_canonicalize.py b/python/src/agent_manifest/_canonicalize.py deleted file mode 100644 index 44960864..00000000 --- a/python/src/agent_manifest/_canonicalize.py +++ /dev/null @@ -1,160 +0,0 @@ -"""RFC 8785 JSON Canonicalization Scheme (JCS). - -Reference: https://www.rfc-editor.org/rfc/rfc8785 - -Single canonicalization entry point for all signing, hashing, and Merkle -tree operations in the Agent Manifest SDK. Used for: - - - Manifest signature pre-image - - manifest_hash_in_report pre-image - - Memory snapshot hash input - - Evidence pack hash input - - Merkle tree leaf nodes containing JSON content - -Per spec Section 4.3: - - Null-valued optional fields are EXCLUDED from canonical form by default. - - @context and @type are treated as ordinary JSON fields (no JSON-LD normalization). - - Text artifact content (system_prompt, policy_bundle) is hashed as raw UTF-8 - NFC bytes, not as JSON — use hashlib directly for those, not this module. -""" -from __future__ import annotations - -import hashlib -import math -import unicodedata -from typing import Any - - -_MAX_DEPTH = 64 # DOS-006: prevent RecursionError from deeply nested JSON - - -def canonicalize(obj: Any, *, exclude_none: bool = True) -> bytes: - """Return RFC 8785 canonical JSON bytes for *obj*. - - Args: - obj: Any JSON-serializable Python value. - exclude_none: When True (default, per spec Section 4.3), mapping - entries whose value is None are omitted from the output. - Set to False only when verifying round-trips with external - producers that include explicit null fields. - - Returns: - UTF-8 encoded bytes with no trailing newline. - - Raises: - TypeError: If *obj* contains a type that cannot be serialized. - ValueError: If a float value is NaN or Infinity, or nesting exceeds - the maximum depth. - """ - return _serialize(obj, exclude_none=exclude_none, depth=0).encode("utf-8") - - -def canonical_hash(obj: Any, *, algorithm: str = "sha256", exclude_none: bool = True) -> str: - """Canonicalize *obj* and return a prefixed hex digest. - - Returns: - String in HashValue format: ``"sha256:<64-hex>"`` or - ``"shake256:<64-hex>"``. - """ - data = canonicalize(obj, exclude_none=exclude_none) - if algorithm == "sha256": - digest = hashlib.sha256(data).hexdigest() - elif algorithm == "shake256": - digest = hashlib.shake_256(data).hexdigest(32) # 256-bit = 32 bytes - else: - raise ValueError(f"Unsupported algorithm {algorithm!r}. Use 'sha256' or 'shake256'.") - return f"{algorithm}:{digest}" - - -# --------------------------------------------------------------------------- -# Internal helpers -# --------------------------------------------------------------------------- - - -def _serialize(obj: Any, *, exclude_none: bool, depth: int) -> str: - if depth > _MAX_DEPTH: - raise ValueError( - f"JSON nesting depth exceeds maximum of {_MAX_DEPTH}. " - "The manifest contains deeply nested structures." - ) - if obj is None: - return "null" - if isinstance(obj, bool): - # bool check must come before int — bool is a subclass of int in Python - return "true" if obj else "false" - if isinstance(obj, int): - return str(obj) - if isinstance(obj, float): - return _float_to_str(obj) - if isinstance(obj, str): - return _quote(obj) - if isinstance(obj, (list, tuple)): - return "[" + ",".join(_serialize(v, exclude_none=exclude_none, depth=depth + 1) for v in obj) + "]" - if isinstance(obj, dict): - return _serialize_dict(obj, exclude_none=exclude_none, depth=depth + 1) - raise TypeError( - f"Object of type {type(obj).__name__!r} is not JSON-serializable under RFC 8785" - ) - - -def _serialize_dict(d: dict[str, Any], *, exclude_none: bool, depth: int) -> str: - # RFC 8785 §3.2.3: sort keys by Unicode code point order. - # Python's str comparison uses Unicode code point order by default — no - # special locale or collation needed. - parts: list[str] = [] - for k in sorted(d.keys()): - v = d[k] - if exclude_none and v is None: - continue - parts.append(_quote(k) + ":" + _serialize(v, exclude_none=exclude_none, depth=depth)) - return "{" + ",".join(parts) + "}" - - -def _quote(s: str) -> str: - """Serialize a Python string as a JSON string per RFC 8785 §3.2.2.2. - - Applies NFC normalization (spec Section 4.3) before escaping. - """ - s = unicodedata.normalize("NFC", s) - buf: list[str] = ['"'] - for ch in s: - cp = ord(ch) - if ch == '"': - buf.append('\\"') - elif ch == "\\": - buf.append("\\\\") - elif ch == "\b": - buf.append("\\b") - elif ch == "\f": - buf.append("\\f") - elif ch == "\n": - buf.append("\\n") - elif ch == "\r": - buf.append("\\r") - elif ch == "\t": - buf.append("\\t") - elif cp <= 0x001F or 0x007F <= cp <= 0x009F or cp in (0x2028, 0x2029): - # Control characters and ECMAScript line terminators - buf.append(f"\\u{cp:04x}") - else: - buf.append(ch) - buf.append('"') - return "".join(buf) - - -def _float_to_str(f: float) -> str: - """Serialize a float per RFC 8785 §3.2.2.3 (ECMAScript number formatting). - - Raises: - ValueError: If *f* is NaN or Infinity (not permitted by RFC 8785). - """ - if math.isnan(f) or math.isinf(f): - raise ValueError(f"RFC 8785 does not permit NaN or Infinity ({f!r})") - # Integers stored as floats: no decimal point - if f == math.floor(f) and abs(f) < 1e15: - return str(int(f)) - # Use Python's shortest-round-trip repr, then normalize exponent notation - s = repr(f) - if "e" in s and "e+" not in s and "e-" not in s: - s = s.replace("e", "e+") - return s diff --git a/python/tests/interop/test_trace_canonicalization_boundary.py b/python/tests/interop/test_trace_canonicalization_boundary.py new file mode 100644 index 00000000..da8723f8 --- /dev/null +++ b/python/tests/interop/test_trace_canonicalization_boundary.py @@ -0,0 +1,60 @@ +"""Cross-repository RFC 8785 conformance guard (issue #322). + +trace-spec's canonicalization-boundary vectors are signed Trust Records whose +signature verifies only over the RFC 8785 canonical bytes of every field +except ``signature``. They are a black-box check on +``agent_manifest._canonicalize.canonicalize`` from an independent producer: +a non-conformant canonicalizer computes different signing bytes here and the +signature stops verifying, which is exactly the failure issue #322 reported. + +The vectors are not vendored in this repository yet -- see +``tests/interop/vectors/canonicalization-boundary/README.md`` for exact fetch +commands. This test skips cleanly with those instructions until the files +are present, and is not required for the rest of the suite to pass. +""" +from __future__ import annotations + +import base64 +import json +from pathlib import Path + +import pytest +from cryptography.exceptions import InvalidSignature +from cryptography.hazmat.primitives.asymmetric.ed25519 import Ed25519PublicKey + +from agent_manifest._canonicalize import canonicalize + +_VECTORS_DIR = Path(__file__).parent / "vectors" / "canonicalization-boundary" +_README = _VECTORS_DIR / "README.md" + + +def _b64url_decode(value: str) -> bytes: + return base64.urlsafe_b64decode(value + "=" * (-len(value) % 4)) + + +def _vector_files() -> list[Path]: + return sorted(_VECTORS_DIR.glob("*.json")) + + +def test_canonicalize_matches_trace_spec_signature(): + files = _vector_files() + if not files: + pytest.skip(f"vectors not vendored yet; see {_README}") + + for path in files: + vector = json.loads(path.read_text(encoding="utf-8")) + record = vector["record"] + jwk = vector["trusted_key"] + body = {k: v for k, v in record.items() if k != "signature"} + + pre_image = canonicalize(body) + public_key = Ed25519PublicKey.from_public_bytes(_b64url_decode(jwk["x"])) + signature = _b64url_decode(record["signature"]) + + try: + public_key.verify(signature, pre_image) + except InvalidSignature: + pytest.fail( + f"{path.name}: canonicalize() pre-image does not verify " + "against trace-spec's own signature over this record" + ) diff --git a/python/tests/interop/vectors/canonicalization-boundary/01-non-ascii-values.json b/python/tests/interop/vectors/canonicalization-boundary/01-non-ascii-values.json new file mode 100644 index 00000000..cd77d577 --- /dev/null +++ b/python/tests/interop/vectors/canonicalization-boundary/01-non-ascii-values.json @@ -0,0 +1,54 @@ +{ + "name": "non-ascii-values", + "description": "String values outside ASCII, all in the Basic Multilingual Plane. RFC 8785 emits them as literal UTF-8; a serializer that escapes to \\uXXXX signs different bytes and rejects this valid record.", + "spec": "trace-v0.2 section 3.2.2 — implementations MUST use an RFC 8785-conformant library", + "profile": "trace.canonicalization.boundary.v0", + "trusted_key": { + "kty": "OKP", + "crv": "Ed25519", + "x": "Be97jkxfFpVXzj9B-gwpMzv5t8PH30Edd-J7AIlrdoA" + }, + "record": { + "eat_profile": "tag:agentrust-io.com,2026:trace-v0.2", + "iat": 1785000000, + "subject": "spiffe://factory.example/agent/payments/prod", + "model": { + "provider": "anthropic", + "model_id": "claude-sonnet-4-6", + "version": "modèle-géant-4.6" + }, + "runtime": { + "platform": "software-only", + "measurement": "sha256:0000000000000000000000000000000000000000000000000000000000000000" + }, + "policy": { + "bundle_hash": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "enforcement_mode": "enforce" + }, + "data_class": "机密", + "build_provenance": { + "slsa_level": 0, + "digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" + }, + "appraisal": { + "status": "affirming", + "verifier": "https://verifier.example/v1" + }, + "transparency": "https://rekor.example/api/v1/log/entries/0", + "cnf": { + "jwk": { + "kty": "OKP", + "crv": "Ed25519", + "x": "Be97jkxfFpVXzj9B-gwpMzv5t8PH30Edd-J7AIlrdoA" + } + }, + "signature": "WehNEF0FqgUa_c85Hw7jbbz4_d_kg2GEyo4r4p242CNGjTkmmRNvVPuwTfjtKJwbOCuNspqEyrMNgZMOTh-OAA" + }, + "expected": { + "outcome": "verified" + }, + "diverges_under": [ + "sort_keys_default", + "sort_keys_compact" + ] +} diff --git a/python/tests/interop/vectors/canonicalization-boundary/02-non-bmp-values.json b/python/tests/interop/vectors/canonicalization-boundary/02-non-bmp-values.json new file mode 100644 index 00000000..46e9222b --- /dev/null +++ b/python/tests/interop/vectors/canonicalization-boundary/02-non-bmp-values.json @@ -0,0 +1,54 @@ +{ + "name": "non-bmp-values", + "description": "String values above U+FFFF, encoded as four UTF-8 bytes each. Under ASCII-escaping they become surrogate pairs; either way the bytes differ from RFC 8785's literal UTF-8.", + "spec": "trace-v0.2 section 3.2.2 — implementations MUST use an RFC 8785-conformant library", + "profile": "trace.canonicalization.boundary.v0", + "trusted_key": { + "kty": "OKP", + "crv": "Ed25519", + "x": "Be97jkxfFpVXzj9B-gwpMzv5t8PH30Edd-J7AIlrdoA" + }, + "record": { + "eat_profile": "tag:agentrust-io.com,2026:trace-v0.2", + "iat": 1785000000, + "subject": "spiffe://factory.example/agent/payments/prod", + "model": { + "provider": "anthropic", + "model_id": "claude-sonnet-4-6", + "version": "4.6-🤖" + }, + "runtime": { + "platform": "software-only", + "measurement": "sha256:0000000000000000000000000000000000000000000000000000000000000000" + }, + "policy": { + "bundle_hash": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "enforcement_mode": "enforce" + }, + "data_class": "confidential-🔒", + "build_provenance": { + "slsa_level": 0, + "digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" + }, + "appraisal": { + "status": "affirming", + "verifier": "https://verifier.example/v1" + }, + "transparency": "https://rekor.example/api/v1/log/entries/0", + "cnf": { + "jwk": { + "kty": "OKP", + "crv": "Ed25519", + "x": "Be97jkxfFpVXzj9B-gwpMzv5t8PH30Edd-J7AIlrdoA" + } + }, + "signature": "62CaOUWmDFPmgthTUkJ4cdwxmDQXzYg9hN6KaCB3EHjeDzeLiB_rdVFIRQrDVTzt-clmIoxNs7UxzJMFvWB_Bw" + }, + "expected": { + "outcome": "verified" + }, + "diverges_under": [ + "sort_keys_default", + "sort_keys_compact" + ] +} diff --git a/python/tests/interop/vectors/canonicalization-boundary/03-utf16-key-order.json b/python/tests/interop/vectors/canonicalization-boundary/03-utf16-key-order.json new file mode 100644 index 00000000..26309488 --- /dev/null +++ b/python/tests/interop/vectors/canonicalization-boundary/03-utf16-key-order.json @@ -0,0 +1,56 @@ +{ + "name": "utf16-key-order", + "description": "Two object keys whose order under RFC 8785's UTF-16 code-unit sort is the reverse of their code-point order. This is the record that distinguishes a true RFC 8785 serializer from json.dumps with every option set carefully: compact separators and ensure_ascii=False survive vectors 01 and 02, and fail here.", + "spec": "trace-v0.2 section 3.2.2 — implementations MUST use an RFC 8785-conformant library", + "profile": "trace.canonicalization.boundary.v0", + "trusted_key": { + "kty": "OKP", + "crv": "Ed25519", + "x": "Be97jkxfFpVXzj9B-gwpMzv5t8PH30Edd-J7AIlrdoA" + }, + "record": { + "eat_profile": "tag:agentrust-io.com,2026:trace-v0.2", + "iat": 1785000000, + "subject": "spiffe://factory.example/agent/payments/prod", + "model": { + "provider": "anthropic", + "model_id": "claude-sonnet-4-6" + }, + "runtime": { + "platform": "software-only", + "measurement": "sha256:0000000000000000000000000000000000000000000000000000000000000000" + }, + "policy": { + "bundle_hash": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "enforcement_mode": "enforce" + }, + "data_class": "confidential", + "build_provenance": { + "slsa_level": 0, + "digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" + }, + "appraisal": { + "status": "affirming", + "verifier": "https://verifier.example/v1" + }, + "transparency": "https://rekor.example/api/v1/log/entries/0", + "cnf": { + "jwk": { + "kty": "OKP", + "crv": "Ed25519", + "x": "Be97jkxfFpVXzj9B-gwpMzv5t8PH30Edd-J7AIlrdoA", + "zk😀": "sorts-first-under-rfc-8785", + "zk�": "sorts-second-under-rfc-8785" + } + }, + "signature": "CjOuPwCnxnwegFjguiSCi-_xPg3iOwnCgyKuKYnV0OorofjPJrkOLn3dUFa-6tVf0z8EDiHaczl6AN46MuBtCQ" + }, + "expected": { + "outcome": "verified" + }, + "diverges_under": [ + "sort_keys_default", + "sort_keys_compact", + "sort_keys_compact_utf8" + ] +} diff --git a/python/tests/interop/vectors/canonicalization-boundary/04-utf16-key-order-nested.json b/python/tests/interop/vectors/canonicalization-boundary/04-utf16-key-order-nested.json new file mode 100644 index 00000000..b7b9714a --- /dev/null +++ b/python/tests/interop/vectors/canonicalization-boundary/04-utf16-key-order-nested.json @@ -0,0 +1,58 @@ +{ + "name": "utf16-key-order-nested", + "description": "The divergence of vector 03 moved inside a nested object, so that a canonicalizer sorting by UTF-16 code units at the outer levels and by code points below them passes 03 and fails here. Without it the closest non-conformant form is caught by one vector, and the boundary disappears with that vector.", + "spec": "trace-v0.2 section 3.2.2 — implementations MUST use an RFC 8785-conformant library", + "profile": "trace.canonicalization.boundary.v0", + "trusted_key": { + "kty": "OKP", + "crv": "Ed25519", + "x": "Be97jkxfFpVXzj9B-gwpMzv5t8PH30Edd-J7AIlrdoA" + }, + "record": { + "eat_profile": "tag:agentrust-io.com,2026:trace-v0.2", + "iat": 1785000000, + "subject": "spiffe://factory.example/agent/payments/prod", + "model": { + "provider": "anthropic", + "model_id": "claude-sonnet-4-6" + }, + "runtime": { + "platform": "software-only", + "measurement": "sha256:0000000000000000000000000000000000000000000000000000000000000000" + }, + "policy": { + "bundle_hash": "sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "enforcement_mode": "enforce" + }, + "data_class": "confidential", + "build_provenance": { + "slsa_level": 0, + "digest": "sha256:bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" + }, + "appraisal": { + "status": "affirming", + "verifier": "https://verifier.example/v1" + }, + "transparency": "https://rekor.example/api/v1/log/entries/0", + "cnf": { + "jwk": { + "kty": "OKP", + "crv": "Ed25519", + "x": "Be97jkxfFpVXzj9B-gwpMzv5t8PH30Edd-J7AIlrdoA", + "zmeta": { + "zk😀": "sorts-first-under-rfc-8785", + "zk�": "sorts-second-under-rfc-8785" + } + } + }, + "signature": "yXsht9nU--Hvr8K7xHq72MOU6xyVhsCKw0_YcAdDff641JNlPG1d2qAZ_zwXaLe48agijvRk3MVZioG85aAiBg" + }, + "expected": { + "outcome": "verified" + }, + "diverges_under": [ + "sort_keys_default", + "sort_keys_compact", + "sort_keys_compact_utf8" + ] +} diff --git a/python/tests/interop/vectors/canonicalization-boundary/README.md b/python/tests/interop/vectors/canonicalization-boundary/README.md new file mode 100644 index 00000000..df70bd27 --- /dev/null +++ b/python/tests/interop/vectors/canonicalization-boundary/README.md @@ -0,0 +1,26 @@ +# Vendored trace-spec canonicalization-boundary vectors + +Empty until fetched. `test_trace_canonicalization_boundary.py` skips with +fetch instructions when no `*.json` files are present here. + +Source: https://github.com/agentrust-io/trace-spec/tree/main/examples/canonicalization-boundary + +These are signed fixtures — fetch the raw bytes directly rather than +retyping them, since the guard exists to catch canonicalization differences +that a retyped copy could silently hide. + +```powershell +git clone --depth 1 https://github.com/agentrust-io/trace-spec C:\Temp\trace-spec +Copy-Item C:\Temp\trace-spec\examples\canonicalization-boundary\*.json . +``` + +or per file: + +```powershell +curl.exe -o 01-non-ascii-values.json https://raw.githubusercontent.com/agentrust-io/trace-spec/main/examples/canonicalization-boundary/01-non-ascii-values.json +curl.exe -o 02-non-bmp-values.json https://raw.githubusercontent.com/agentrust-io/trace-spec/main/examples/canonicalization-boundary/02-non-bmp-values.json +curl.exe -o 03-utf16-key-order.json https://raw.githubusercontent.com/agentrust-io/trace-spec/main/examples/canonicalization-boundary/03-utf16-key-order.json +curl.exe -o 04-utf16-key-order-nested.json https://raw.githubusercontent.com/agentrust-io/trace-spec/main/examples/canonicalization-boundary/04-utf16-key-order-nested.json +``` + +Run `pytest tests/interop/test_trace_canonicalization_boundary.py -v` afterward. diff --git a/python/tests/test_canonicalize.py b/python/tests/test_canonicalize.py deleted file mode 100644 index c2e499d7..00000000 --- a/python/tests/test_canonicalize.py +++ /dev/null @@ -1,235 +0,0 @@ -"""RFC 8785 canonical JSON test suite. - -Covers: - - Appendix D test vector (verified via sha256sum) - - Key sort ordering, whitespace, null exclusion - - NFC normalization, string escaping, boolean/float handling - - @context / @type as ordinary fields -""" -import hashlib -import math - -import pytest - -from agent_manifest._canonicalize import canonical_hash, canonicalize - - -# --------------------------------------------------------------------------- -# Spec Appendix D test vector (SHA-256 verified via bash sha256sum) -# --------------------------------------------------------------------------- - -APPENDIX_D_INPUT = { - "version": "0.1", - "issued_at": "2026-06-23T09:00:00Z", - "agent_id": "spiffe://trust.example/agent/kyc/prod-001", -} -APPENDIX_D_CANONICAL = ( - b'{"agent_id":"spiffe://trust.example/agent/kyc/prod-001"' - b',"issued_at":"2026-06-23T09:00:00Z","version":"0.1"}' -) -APPENDIX_D_SHA256 = "b83293348255f4427dc030478f354b83f4f82662223be0926ad9f2db946b5319" - - -def test_appendix_d_canonical_form(): - assert canonicalize(APPENDIX_D_INPUT) == APPENDIX_D_CANONICAL - - -def test_appendix_d_sha256(): - assert hashlib.sha256(APPENDIX_D_CANONICAL).hexdigest() == APPENDIX_D_SHA256 - - -def test_appendix_d_canonical_hash(): - assert canonical_hash(APPENDIX_D_INPUT) == f"sha256:{APPENDIX_D_SHA256}" - - -# --------------------------------------------------------------------------- -# Key ordering -# --------------------------------------------------------------------------- - - -def test_keys_sorted_lexicographic(): - assert canonicalize({"z": 1, "a": 2, "m": 3}) == b'{"a":2,"m":3,"z":1}' - - -def test_nested_keys_sorted(): - assert canonicalize({"b": {"y": 1, "x": 2}, "a": 0}) == b'{"a":0,"b":{"x":2,"y":1}}' - - -def test_unicode_key_ordering(): - # chr(233) = U+00E9 (é) > chr(101) = 'e' - obj = {chr(233): 1, "e": 2} - result = canonicalize(obj) - assert result == ('{"e":2,"' + chr(233) + '":1}').encode("utf-8") - - -# --------------------------------------------------------------------------- -# Whitespace -# --------------------------------------------------------------------------- - - -def test_no_whitespace(): - result = canonicalize({"a": 1, "b": [1, 2, 3]}) - assert b" " not in result and b"\n" not in result and b"\t" not in result - - -# --------------------------------------------------------------------------- -# Null handling (spec Section 4.3) -# --------------------------------------------------------------------------- - - -def test_null_excluded_by_default(): - assert canonicalize({"a": 1, "b": None, "c": 3}) == b'{"a":1,"c":3}' - - -def test_null_included_when_opted_in(): - assert canonicalize({"a": 1, "b": None}, exclude_none=False) == b'{"a":1,"b":null}' - - -def test_nested_null_excluded(): - assert canonicalize({"outer": {"present": 1, "absent": None}}) == b'{"outer":{"present":1}}' - - -# --------------------------------------------------------------------------- -# Boolean serialization -# --------------------------------------------------------------------------- - - -def test_boolean_true(): - assert canonicalize({"v": True}) == b'{"v":true}' - - -def test_boolean_false(): - assert canonicalize({"v": False}) == b'{"v":false}' - - -def test_bool_not_confused_with_int(): - # bool is a subclass of int — must not serialize True as 1 - assert canonicalize({"a": True, "b": 1}) == b'{"a":true,"b":1}' - - -# --------------------------------------------------------------------------- -# String escaping — using chr() to avoid embedding control chars in source -# --------------------------------------------------------------------------- - - -def test_null_byte_escaped(): - assert canonicalize({"v": chr(0)}) == b'{"v":"\\u0000"}' - - -def test_unit_separator_escaped(): - assert canonicalize({"v": chr(31)}) == b'{"v":"\\u001f"}' - - -def test_backslash_escaped(): - assert canonicalize({"v": "\\"}) == b'{"v":"\\\\"}' - - -def test_double_quote_escaped(): - assert canonicalize({"v": '"'}) == b'{"v":"\\""}' - - -def test_tab_newline_escaped(): - assert canonicalize({"v": "\t\n"}) == b'{"v":"\\t\\n"}' - - -def test_line_separator_escaped(): - # U+2028 LINE SEPARATOR must be - assert b"\\u2028" in canonicalize({"v": chr(0x2028)}) - - -def test_regular_unicode_verbatim(): - # Non-control chars pass through after NFC normalization - assert canonicalize({"v": "é"}) == '{"v":"é"}'.encode("utf-8") - - -# --------------------------------------------------------------------------- -# NFC normalization -# --------------------------------------------------------------------------- - - -def test_nfc_normalization(): - precomposed = "é" # é as single code point - decomposed = "é" # e + combining accent - assert canonicalize({"v": precomposed}) == canonicalize({"v": decomposed}) - - -# --------------------------------------------------------------------------- -# Arrays -# --------------------------------------------------------------------------- - - -def test_array_order_preserved(): - assert canonicalize([3, 1, 2]) == b"[3,1,2]" - - -def test_nested_array(): - assert canonicalize([[1, 2], [3, 4]]) == b"[[1,2],[3,4]]" - - -def test_empty_array(): - assert canonicalize([]) == b"[]" - - -def test_empty_object(): - assert canonicalize({}) == b"{}" - - -# --------------------------------------------------------------------------- -# Numbers -# --------------------------------------------------------------------------- - - -def test_integer(): - assert canonicalize({"v": 42}) == b'{"v":42}' - - -def test_negative_integer(): - assert canonicalize({"v": -7}) == b'{"v":-7}' - - -def test_float_integer_value_no_decimal(): - assert canonicalize({"v": 1.0}) == b'{"v":1}' - - -def test_float_nan_raises(): - with pytest.raises(ValueError, match="NaN"): - canonicalize({"v": math.nan}) - - -def test_float_infinity_raises(): - with pytest.raises(ValueError, match="Infinity"): - canonicalize({"v": math.inf}) - - -# --------------------------------------------------------------------------- -# @context / @type as ordinary fields -# --------------------------------------------------------------------------- - - -def test_context_type_ordinary(): - obj = { - "@context": "https://manifest.agentrust-io.com/v0.2/context.json", - "@type": "AgentManifest", - "manifest_id": "test", - } - result = canonicalize(obj) - # '@' (U+0040) sorts before all letters, so @context comes first - assert result.startswith(b'{"@context"') - assert b'"@type"' in result - assert b'"manifest_id"' in result - - -# --------------------------------------------------------------------------- -# shake256 -# --------------------------------------------------------------------------- - - -def test_shake256_length(): - result = canonical_hash({"v": 1}, algorithm="shake256") - assert result.startswith("shake256:") - assert len(result) == len("shake256:") + 64 - - -def test_unsupported_algorithm_raises(): - with pytest.raises(ValueError, match="Unsupported"): - canonical_hash({"v": 1}, algorithm="md5")