import hashlib from datetime import UTC, datetime import pytest from tht.evidence.contracts import AcquiredDocument, SourceObject from tht.evidence.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize def acquired(content: bytes, *, media_type: str = "text/markdown") -> AcquiredDocument: return AcquiredDocument( source=SourceObject( source_id="source:guide", uri="https://host/guide.md?signature=transport#part", fingerprint="etag:abc", modified_at=datetime(2026, 7, 12, 12, 0, tzinfo=UTC), metadata={"owner": "docs"}, ), content=content, media_type=media_type, metadata={"transport": "http"}, ) def test_normalize_utf8_bom_newlines_unicode_and_frontmatter(): raw = ( "\ufeff---\r\ntitle: Café\r\ntags: [uno, due]\r\n---\r\n" "Cafe\u0301\rBody\r\n" ).encode() document = normalize(acquired(raw, media_type="text/markdown; charset=UTF-8"), "pipe:v1") assert document.content == "Café\nBody\n" assert document.title == "Café" assert document.metadata["frontmatter"] == {"tags": ["uno", "due"], "title": "Café"} assert document.metadata["source"] == {"owner": "docs"} assert document.metadata["acquisition"] == {"transport": "http"} assert document.source_uri == "https://host/guide.md" assert document.modified_at == datetime(2026, 7, 12, 12, 0, tzinfo=UTC) assert document.content_hash == "sha256:" + hashlib.sha256(document.content.encode()).hexdigest() def test_plain_text_that_only_resembles_frontmatter_is_not_dropped(): document = normalize(acquired(b"---\nnot: closed\nbody"), "pipe:v1") assert document.content == "---\nnot: closed\nbody" assert "frontmatter" not in document.metadata def test_frontmatter_can_end_at_eof_without_inventing_content(): document = normalize(acquired(b"---\ntitle: Empty\n---"), "pipe:v1") assert document.title == "Empty" assert document.content == "" def test_normalize_preserves_validated_curated_evidence_for_semantic_projection(): source = SourceObject( source_id="filesystem:fascia-pediatrica", uri="file:///safe/evidence/curated/formula/fascia-pediatrica.md", fingerprint="sha256:" + "a" * 64, metadata={"relative_path": "curated/formula/fascia-pediatrica.md"}, ) raw = ( "---\n" "schema_version: 1\n" "id: evidence:fascia-pediatrica\n" "title: Fascia pediatrica\n" "kind: formula\n" "purposes: [sql_generation]\n" "applies_to:\n" " columns: [clinical.patient.birth_date]\n" "language: it\n" "provenance:\n" " source_file: source/paziente.md\n" " source_sha256: sha256:" + "b" * 64 + "\n" " supporting_excerpts: [Pazienti con età inferiore a 18 anni.]\n" "review_items: []\n" "formula:\n" " concept: fascia pediatrica\n" " columns: [clinical.patient.birth_date]\n" " sql: CASE WHEN age < 18 THEN 'pediatric' END\n" "---\n" ).encode() document = normalize(AcquiredDocument(source=source, content=raw, media_type="text/markdown"), "pipe:v1") assert document.content == raw.decode() assert document.title == "Fascia pediatrica" assert document.metadata["curated_evidence"]["id"] == "evidence:fascia-pediatrica" assert document.metadata["curated_evidence"]["payload"]["concept"] == "fascia pediatrica" def test_only_curated_markdown_is_treated_as_canonical_evidence_without_relative_metadata(): source = SourceObject( source_id="filesystem:notes", uri="file:///safe/evidence/curated/formula/notes.txt", fingerprint="sha256:" + "a" * 64, ) document = normalize( AcquiredDocument(source=source, content=b"ordinary note", media_type="text/plain"), "pipe:v1", ) assert document.content == "ordinary note" assert "curated_evidence" not in document.metadata @pytest.mark.parametrize( "frontmatter", [ "title: first\ntitle: second", "title: &shared value\ncopy: *shared", "nested: " + "[" * 25 + "x" + "]" * 25, "items: [" + ",".join("x" for _ in range(1100)) + "]", "api_key: secret", ], ) def test_rejects_unsafe_frontmatter_as_typed_permanent_error(frontmatter): raw = f"---\n{frontmatter}\n---\nbody".encode() with pytest.raises(PermanentNormalizationError) as caught: normalize(acquired(raw), "pipe:v1") assert caught.value.reason == "invalid_frontmatter" def test_pipeline_policy_errors_are_not_misclassified_as_bad_frontmatter(): with pytest.raises(ValueError, match="pipeline_version") as caught: normalize(acquired(b"---\ntitle: valid\n---\nbody"), "") assert not isinstance(caught.value, PermanentNormalizationError) @pytest.mark.parametrize( ("content", "media_type", "reason"), [ (b"bad: \xff", "text/plain", "undecodable"), (b"hello", "text/plain; charset=iso-8859-1", "unsupported_charset"), (b"x" * (MAX_DOCUMENT_BYTES + 1), "text/plain", "oversized"), ], ) def test_rejects_invalid_input_as_typed_permanent_error(content, media_type, reason): with pytest.raises(PermanentNormalizationError) as caught: normalize(acquired(content, media_type=media_type), "pipe:v1") assert caught.value.permanent is True assert caught.value.reason == reason