144 lines
5.3 KiB
Python
144 lines
5.3 KiB
Python
import hashlib
|
|
from datetime import UTC, datetime
|
|
|
|
import pytest
|
|
|
|
from tht.evidence.contracts import AcquiredDocument, SourceObject
|
|
from tht.evidence.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize
|
|
|
|
|
|
def acquired(content: bytes, *, media_type: str = "text/markdown") -> AcquiredDocument:
|
|
return AcquiredDocument(
|
|
source=SourceObject(
|
|
source_id="source:guide",
|
|
uri="https://host/guide.md?signature=transport#part",
|
|
fingerprint="etag:abc",
|
|
modified_at=datetime(2026, 7, 12, 12, 0, tzinfo=UTC),
|
|
metadata={"owner": "docs"},
|
|
),
|
|
content=content,
|
|
media_type=media_type,
|
|
metadata={"transport": "http"},
|
|
)
|
|
|
|
|
|
def test_normalize_utf8_bom_newlines_unicode_and_frontmatter():
|
|
raw = (
|
|
"\ufeff---\r\ntitle: Café\r\ntags: [uno, due]\r\n---\r\n"
|
|
"Cafe\u0301\rBody\r\n"
|
|
).encode()
|
|
|
|
document = normalize(acquired(raw, media_type="text/markdown; charset=UTF-8"), "pipe:v1")
|
|
|
|
assert document.content == "Café\nBody\n"
|
|
assert document.title == "Café"
|
|
assert document.metadata["frontmatter"] == {"tags": ["uno", "due"], "title": "Café"}
|
|
assert document.metadata["source"] == {"owner": "docs"}
|
|
assert document.metadata["acquisition"] == {"transport": "http"}
|
|
assert document.source_uri == "https://host/guide.md"
|
|
assert document.modified_at == datetime(2026, 7, 12, 12, 0, tzinfo=UTC)
|
|
assert document.content_hash == "sha256:" + hashlib.sha256(document.content.encode()).hexdigest()
|
|
|
|
|
|
def test_plain_text_that_only_resembles_frontmatter_is_not_dropped():
|
|
document = normalize(acquired(b"---\nnot: closed\nbody"), "pipe:v1")
|
|
assert document.content == "---\nnot: closed\nbody"
|
|
assert "frontmatter" not in document.metadata
|
|
|
|
|
|
def test_frontmatter_can_end_at_eof_without_inventing_content():
|
|
document = normalize(acquired(b"---\ntitle: Empty\n---"), "pipe:v1")
|
|
assert document.title == "Empty"
|
|
assert document.content == ""
|
|
|
|
|
|
def test_normalize_preserves_validated_curated_evidence_for_semantic_projection():
|
|
source = SourceObject(
|
|
source_id="filesystem:fascia-pediatrica",
|
|
uri="file:///safe/evidence/curated/formula/fascia-pediatrica.md",
|
|
fingerprint="sha256:" + "a" * 64,
|
|
metadata={"relative_path": "curated/formula/fascia-pediatrica.md"},
|
|
)
|
|
raw = (
|
|
"---\n"
|
|
"schema_version: 1\n"
|
|
"id: evidence:fascia-pediatrica\n"
|
|
"title: Fascia pediatrica\n"
|
|
"kind: formula\n"
|
|
"purposes: [sql_generation]\n"
|
|
"applies_to:\n"
|
|
" columns: [clinical.patient.birth_date]\n"
|
|
"language: it\n"
|
|
"provenance:\n"
|
|
" source_file: source/paziente.md\n"
|
|
" source_sha256: sha256:" + "b" * 64 + "\n"
|
|
" supporting_excerpts: [Pazienti con età inferiore a 18 anni.]\n"
|
|
"review_items: []\n"
|
|
"formula:\n"
|
|
" concept: fascia pediatrica\n"
|
|
" columns: [clinical.patient.birth_date]\n"
|
|
" sql: CASE WHEN age < 18 THEN 'pediatric' END\n"
|
|
"---\n"
|
|
).encode()
|
|
|
|
document = normalize(AcquiredDocument(source=source, content=raw, media_type="text/markdown"), "pipe:v1")
|
|
|
|
assert document.content == raw.decode()
|
|
assert document.title == "Fascia pediatrica"
|
|
assert document.metadata["curated_evidence"]["id"] == "evidence:fascia-pediatrica"
|
|
assert document.metadata["curated_evidence"]["payload"]["concept"] == "fascia pediatrica"
|
|
|
|
|
|
def test_only_curated_markdown_is_treated_as_canonical_evidence_without_relative_metadata():
|
|
source = SourceObject(
|
|
source_id="filesystem:notes",
|
|
uri="file:///safe/evidence/curated/formula/notes.txt",
|
|
fingerprint="sha256:" + "a" * 64,
|
|
)
|
|
|
|
document = normalize(
|
|
AcquiredDocument(source=source, content=b"ordinary note", media_type="text/plain"),
|
|
"pipe:v1",
|
|
)
|
|
|
|
assert document.content == "ordinary note"
|
|
assert "curated_evidence" not in document.metadata
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"frontmatter",
|
|
[
|
|
"title: first\ntitle: second",
|
|
"title: &shared value\ncopy: *shared",
|
|
"nested: " + "[" * 25 + "x" + "]" * 25,
|
|
"items: [" + ",".join("x" for _ in range(1100)) + "]",
|
|
"api_key: secret",
|
|
],
|
|
)
|
|
def test_rejects_unsafe_frontmatter_as_typed_permanent_error(frontmatter):
|
|
raw = f"---\n{frontmatter}\n---\nbody".encode()
|
|
with pytest.raises(PermanentNormalizationError) as caught:
|
|
normalize(acquired(raw), "pipe:v1")
|
|
assert caught.value.reason == "invalid_frontmatter"
|
|
|
|
|
|
def test_pipeline_policy_errors_are_not_misclassified_as_bad_frontmatter():
|
|
with pytest.raises(ValueError, match="pipeline_version") as caught:
|
|
normalize(acquired(b"---\ntitle: valid\n---\nbody"), "")
|
|
assert not isinstance(caught.value, PermanentNormalizationError)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("content", "media_type", "reason"),
|
|
[
|
|
(b"bad: \xff", "text/plain", "undecodable"),
|
|
(b"hello", "text/plain; charset=iso-8859-1", "unsupported_charset"),
|
|
(b"x" * (MAX_DOCUMENT_BYTES + 1), "text/plain", "oversized"),
|
|
],
|
|
)
|
|
def test_rejects_invalid_input_as_typed_permanent_error(content, media_type, reason):
|
|
with pytest.raises(PermanentNormalizationError) as caught:
|
|
normalize(acquired(content, media_type=media_type), "pipe:v1")
|
|
assert caught.value.permanent is True
|
|
assert caught.value.reason == reason
|