100 lines
3.6 KiB
Python
100 lines
3.6 KiB
Python
"""Pure, deterministic conversion of acquired bytes into canonical text."""
|
|
|
|
import hashlib
|
|
import re
|
|
import unicodedata
|
|
from collections.abc import Mapping
|
|
|
|
import yaml
|
|
from pydantic import JsonValue, TypeAdapter, ValidationError
|
|
|
|
from tht.corpus.models import CanonicalDocument
|
|
from tht.ports.evidence import AcquiredDocument, canonical_provenance_uri
|
|
|
|
|
|
MAX_DOCUMENT_BYTES = 10 * 1024 * 1024
|
|
_CHARSET = re.compile(r"(?:^|;)\s*charset\s*=\s*[\"']?([^;\s\"']+)", re.IGNORECASE)
|
|
_FRONTMATTER = re.compile(r"\A---\n(.*?)\n---(?:\n|\Z)", re.DOTALL)
|
|
_JSON_OBJECT = TypeAdapter(dict[str, JsonValue])
|
|
|
|
|
|
class PermanentNormalizationError(ValueError):
|
|
"""A deterministic input failure which retrying cannot repair."""
|
|
|
|
def __init__(self, reason: str) -> None:
|
|
super().__init__(f"document normalization failed: {reason}")
|
|
self.reason = reason
|
|
self.permanent = True
|
|
|
|
|
|
def _sha256(value: str) -> str:
|
|
return hashlib.sha256(value.encode("utf-8")).hexdigest()
|
|
|
|
|
|
def _decode(acquired: AcquiredDocument) -> str:
|
|
if len(acquired.content) > MAX_DOCUMENT_BYTES:
|
|
raise PermanentNormalizationError("oversized")
|
|
|
|
media_type = acquired.media_type or "text/plain"
|
|
charset = _CHARSET.search(media_type)
|
|
if charset and charset.group(1).lower().replace("_", "-") not in {
|
|
"utf-8",
|
|
"utf8",
|
|
"us-ascii",
|
|
"ascii",
|
|
}:
|
|
raise PermanentNormalizationError("unsupported_charset")
|
|
try:
|
|
return acquired.content.decode("utf-8-sig", errors="strict")
|
|
except UnicodeDecodeError as error:
|
|
raise PermanentNormalizationError("undecodable") from error
|
|
|
|
|
|
def _frontmatter(text: str) -> tuple[dict[str, JsonValue], str]:
|
|
match = _FRONTMATTER.match(text)
|
|
if match is None:
|
|
return {}, text
|
|
try:
|
|
loaded = yaml.safe_load(match.group(1))
|
|
if loaded is None:
|
|
loaded = {}
|
|
if not isinstance(loaded, Mapping):
|
|
raise TypeError("frontmatter is not a mapping")
|
|
metadata = _JSON_OBJECT.validate_python(dict(loaded))
|
|
except (TypeError, UnicodeError, ValidationError, yaml.YAMLError) as error:
|
|
raise PermanentNormalizationError("invalid_frontmatter") from error
|
|
return metadata, text[match.end() :]
|
|
|
|
|
|
def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDocument:
|
|
"""Normalize one transport result without I/O or implicit data loss."""
|
|
if not pipeline_version:
|
|
raise ValueError("pipeline_version must not be empty")
|
|
|
|
decoded = _decode(acquired)
|
|
canonical = unicodedata.normalize("NFC", decoded.replace("\r\n", "\n").replace("\r", "\n"))
|
|
frontmatter, content = _frontmatter(canonical)
|
|
source_uri = canonical_provenance_uri(acquired.source.uri)
|
|
identity = f"{acquired.source.source_id}\n{source_uri}"
|
|
media_type = (acquired.media_type or "text/plain").split(";", 1)[0].strip().lower()
|
|
metadata: dict[str, JsonValue] = {
|
|
"source": acquired.source.model_dump(mode="json")["metadata"],
|
|
"acquisition": acquired.model_dump(mode="json")["metadata"],
|
|
}
|
|
if frontmatter:
|
|
metadata["frontmatter"] = frontmatter
|
|
|
|
return CanonicalDocument(
|
|
document_id=f"doc:{_sha256(identity)}",
|
|
source_id=acquired.source.source_id,
|
|
source_uri=source_uri,
|
|
source_fingerprint=acquired.source.fingerprint,
|
|
content_hash=f"sha256:{_sha256(content)}",
|
|
title=str(frontmatter.get("title", "")),
|
|
content=content,
|
|
media_type=media_type,
|
|
modified_at=acquired.source.modified_at,
|
|
pipeline_version=pipeline_version,
|
|
metadata=metadata,
|
|
)
|