feat(evidence): build semantic fragments from typed units
This commit is contained in:
@@ -4,12 +4,15 @@ import hashlib
|
||||
import re
|
||||
import unicodedata
|
||||
from collections.abc import Mapping
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
import yaml
|
||||
from pydantic import JsonValue, TypeAdapter, ValidationError
|
||||
from yaml.events import AliasEvent
|
||||
from yaml.nodes import MappingNode
|
||||
|
||||
from tht.evidence.canonical import EVIDENCE_KINDS, parse_curated_markdown
|
||||
from tht.evidence.contracts import AcquiredDocument, canonical_provenance_uri
|
||||
from tht.evidence.corpus.models import CanonicalDocument
|
||||
|
||||
@@ -108,6 +111,20 @@ def _frontmatter(text: str) -> tuple[dict[str, JsonValue], str]:
|
||||
return metadata, text[match.end() :]
|
||||
|
||||
|
||||
def _is_curated_filesystem_document(acquired: AcquiredDocument) -> bool:
|
||||
if urlsplit(acquired.source.uri).scheme != "file":
|
||||
return False
|
||||
path = Path(urlsplit(acquired.source.uri).path)
|
||||
if path.suffix != ".md":
|
||||
return False
|
||||
parts = path.parts
|
||||
try:
|
||||
curated_index = parts.index("curated")
|
||||
except ValueError:
|
||||
return False
|
||||
return len(parts) >= curated_index + 3 and parts[curated_index + 1] in EVIDENCE_KINDS
|
||||
|
||||
|
||||
def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDocument:
|
||||
"""Normalize one transport result without I/O or implicit data loss."""
|
||||
if not pipeline_version:
|
||||
@@ -115,7 +132,6 @@ def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDoc
|
||||
|
||||
decoded = _decode(acquired)
|
||||
canonical = unicodedata.normalize("NFC", decoded.replace("\r\n", "\n").replace("\r", "\n"))
|
||||
frontmatter, content = _frontmatter(canonical)
|
||||
source_uri = canonical_provenance_uri(acquired.source.uri)
|
||||
identity = f"{acquired.source.source_id}\n{source_uri}"
|
||||
media_type = (acquired.media_type or "text/plain").split(";", 1)[0].strip().lower()
|
||||
@@ -123,8 +139,21 @@ def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDoc
|
||||
"source": acquired.source.model_dump(mode="json")["metadata"],
|
||||
"acquisition": acquired.model_dump(mode="json")["metadata"],
|
||||
}
|
||||
if frontmatter:
|
||||
metadata["frontmatter"] = frontmatter
|
||||
if _is_curated_filesystem_document(acquired):
|
||||
try:
|
||||
evidence = parse_curated_markdown(canonical, path=Path(urlsplit(acquired.source.uri).path))
|
||||
except ValueError as error:
|
||||
raise PermanentNormalizationError("invalid_curated_evidence") from error
|
||||
if evidence.review_items:
|
||||
raise PermanentNormalizationError("curated_evidence_requires_review")
|
||||
content = canonical
|
||||
title = evidence.title
|
||||
metadata["curated_evidence"] = evidence.model_dump(mode="json")
|
||||
else:
|
||||
frontmatter, content = _frontmatter(canonical)
|
||||
title = str(frontmatter.get("title", ""))
|
||||
if frontmatter:
|
||||
metadata["frontmatter"] = frontmatter
|
||||
|
||||
try:
|
||||
return CanonicalDocument(
|
||||
@@ -133,7 +162,7 @@ def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDoc
|
||||
source_uri=source_uri,
|
||||
source_fingerprint=acquired.source.fingerprint,
|
||||
content_hash=f"sha256:{_sha256(content)}",
|
||||
title=str(frontmatter.get("title", "")),
|
||||
title=title,
|
||||
content=content,
|
||||
media_type=media_type,
|
||||
modified_at=acquired.source.modified_at,
|
||||
@@ -141,6 +170,6 @@ def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDoc
|
||||
metadata=metadata,
|
||||
)
|
||||
except ValidationError as error:
|
||||
if frontmatter:
|
||||
if "frontmatter" in metadata:
|
||||
raise PermanentNormalizationError("invalid_frontmatter") from error
|
||||
raise
|
||||
|
||||
Reference in New Issue
Block a user