import hashlib from datetime import UTC, datetime import pytest from tht.evidence.contracts import AcquiredDocument, SourceObject from tht.evidence.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize def acquired(content: bytes, *, media_type: str = "text/markdown") -> AcquiredDocument: return AcquiredDocument( source=SourceObject( source_id="source:guide", uri="https://host/guide.md?signature=transport#part", fingerprint="etag:abc", modified_at=datetime(2026, 7, 12, 12, 0, tzinfo=UTC), metadata={"owner": "docs"}, ), content=content, media_type=media_type, metadata={"transport": "http"}, ) def test_normalize_utf8_bom_newlines_unicode_and_frontmatter(): raw = ( "\ufeff---\r\ntitle: Café\r\ntags: [uno, due]\r\n---\r\n" "Cafe\u0301\rBody\r\n" ).encode() document = normalize(acquired(raw, media_type="text/markdown; charset=UTF-8"), "pipe:v1") assert document.content == "Café\nBody\n" assert document.title == "Café" assert document.metadata["frontmatter"] == {"tags": ["uno", "due"], "title": "Café"} assert document.metadata["source"] == {"owner": "docs"} assert document.metadata["acquisition"] == {"transport": "http"} assert document.source_uri == "https://host/guide.md" assert document.modified_at == datetime(2026, 7, 12, 12, 0, tzinfo=UTC) assert document.content_hash == "sha256:" + hashlib.sha256(document.content.encode()).hexdigest() def test_plain_text_that_only_resembles_frontmatter_is_not_dropped(): document = normalize(acquired(b"---\nnot: closed\nbody"), "pipe:v1") assert document.content == "---\nnot: closed\nbody" assert "frontmatter" not in document.metadata def test_frontmatter_can_end_at_eof_without_inventing_content(): document = normalize(acquired(b"---\ntitle: Empty\n---"), "pipe:v1") assert document.title == "Empty" assert document.content == "" @pytest.mark.parametrize( "frontmatter", [ "title: first\ntitle: second", "title: &shared value\ncopy: *shared", "nested: " + "[" * 25 + "x" + "]" * 25, "items: [" + ",".join("x" for _ in range(1100)) + "]", "api_key: secret", ], ) def test_rejects_unsafe_frontmatter_as_typed_permanent_error(frontmatter): raw = f"---\n{frontmatter}\n---\nbody".encode() with pytest.raises(PermanentNormalizationError) as caught: normalize(acquired(raw), "pipe:v1") assert caught.value.reason == "invalid_frontmatter" def test_pipeline_policy_errors_are_not_misclassified_as_bad_frontmatter(): with pytest.raises(ValueError, match="pipeline_version") as caught: normalize(acquired(b"---\ntitle: valid\n---\nbody"), "") assert not isinstance(caught.value, PermanentNormalizationError) @pytest.mark.parametrize( ("content", "media_type", "reason"), [ (b"bad: \xff", "text/plain", "undecodable"), (b"hello", "text/plain; charset=iso-8859-1", "unsupported_charset"), (b"x" * (MAX_DOCUMENT_BYTES + 1), "text/plain", "oversized"), ], ) def test_rejects_invalid_input_as_typed_permanent_error(content, media_type, reason): with pytest.raises(PermanentNormalizationError) as caught: normalize(acquired(content, media_type=media_type), "pipe:v1") assert caught.value.permanent is True assert caught.value.reason == reason