91 lines
3.4 KiB
Python
91 lines
3.4 KiB
Python
import hashlib
|
|
from datetime import UTC, datetime
|
|
|
|
import pytest
|
|
|
|
from tht.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize
|
|
from tht.ports.evidence import AcquiredDocument, SourceObject
|
|
|
|
|
|
def acquired(content: bytes, *, media_type: str = "text/markdown") -> AcquiredDocument:
|
|
return AcquiredDocument(
|
|
source=SourceObject(
|
|
source_id="source:guide",
|
|
uri="https://host/guide.md?signature=transport#part",
|
|
fingerprint="etag:abc",
|
|
modified_at=datetime(2026, 7, 12, 12, 0, tzinfo=UTC),
|
|
metadata={"owner": "docs"},
|
|
),
|
|
content=content,
|
|
media_type=media_type,
|
|
metadata={"transport": "http"},
|
|
)
|
|
|
|
|
|
def test_normalize_utf8_bom_newlines_unicode_and_frontmatter():
|
|
raw = (
|
|
"\ufeff---\r\ntitle: Café\r\ntags: [uno, due]\r\n---\r\n"
|
|
"Cafe\u0301\rBody\r\n"
|
|
).encode()
|
|
|
|
document = normalize(acquired(raw, media_type="text/markdown; charset=UTF-8"), "pipe:v1")
|
|
|
|
assert document.content == "Café\nBody\n"
|
|
assert document.title == "Café"
|
|
assert document.metadata["frontmatter"] == {"tags": ["uno", "due"], "title": "Café"}
|
|
assert document.metadata["source"] == {"owner": "docs"}
|
|
assert document.metadata["acquisition"] == {"transport": "http"}
|
|
assert document.source_uri == "https://host/guide.md"
|
|
assert document.modified_at == datetime(2026, 7, 12, 12, 0, tzinfo=UTC)
|
|
assert document.content_hash == "sha256:" + hashlib.sha256(document.content.encode()).hexdigest()
|
|
|
|
|
|
def test_plain_text_that_only_resembles_frontmatter_is_not_dropped():
|
|
document = normalize(acquired(b"---\nnot: closed\nbody"), "pipe:v1")
|
|
assert document.content == "---\nnot: closed\nbody"
|
|
assert "frontmatter" not in document.metadata
|
|
|
|
|
|
def test_frontmatter_can_end_at_eof_without_inventing_content():
|
|
document = normalize(acquired(b"---\ntitle: Empty\n---"), "pipe:v1")
|
|
assert document.title == "Empty"
|
|
assert document.content == ""
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"frontmatter",
|
|
[
|
|
"title: first\ntitle: second",
|
|
"title: &shared value\ncopy: *shared",
|
|
"nested: " + "[" * 25 + "x" + "]" * 25,
|
|
"items: [" + ",".join("x" for _ in range(1100)) + "]",
|
|
"api_key: secret",
|
|
],
|
|
)
|
|
def test_rejects_unsafe_frontmatter_as_typed_permanent_error(frontmatter):
|
|
raw = f"---\n{frontmatter}\n---\nbody".encode()
|
|
with pytest.raises(PermanentNormalizationError) as caught:
|
|
normalize(acquired(raw), "pipe:v1")
|
|
assert caught.value.reason == "invalid_frontmatter"
|
|
|
|
|
|
def test_pipeline_policy_errors_are_not_misclassified_as_bad_frontmatter():
|
|
with pytest.raises(ValueError, match="pipeline_version") as caught:
|
|
normalize(acquired(b"---\ntitle: valid\n---\nbody"), "")
|
|
assert not isinstance(caught.value, PermanentNormalizationError)
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("content", "media_type", "reason"),
|
|
[
|
|
(b"bad: \xff", "text/plain", "undecodable"),
|
|
(b"hello", "text/plain; charset=iso-8859-1", "unsupported_charset"),
|
|
(b"x" * (MAX_DOCUMENT_BYTES + 1), "text/plain", "oversized"),
|
|
],
|
|
)
|
|
def test_rejects_invalid_input_as_typed_permanent_error(content, media_type, reason):
|
|
with pytest.raises(PermanentNormalizationError) as caught:
|
|
normalize(acquired(content, media_type=media_type), "pipe:v1")
|
|
assert caught.value.permanent is True
|
|
assert caught.value.reason == reason
|