feat(corpus): add deterministic normalization and chunking
This commit is contained in:
@@ -0,0 +1,89 @@
|
||||
"""Versioned deterministic chunking for canonical corpus documents."""
|
||||
|
||||
import hashlib
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
from tht.corpus.models import CanonicalChunk, CanonicalDocument
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ChunkPolicy:
|
||||
version: str
|
||||
max_chars: int
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not self.version:
|
||||
raise ValueError("chunk policy version must not be empty")
|
||||
if self.max_chars <= 0:
|
||||
raise ValueError("max_chars must be greater than zero")
|
||||
|
||||
|
||||
def _hash(text: str) -> str:
|
||||
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _hard_split(text: str, maximum: int) -> list[str]:
|
||||
return [text[start : start + maximum] for start in range(0, len(text), maximum)]
|
||||
|
||||
|
||||
def _split_block(block: str, maximum: int) -> list[str]:
|
||||
if len(block) <= maximum:
|
||||
return [block]
|
||||
tokens = re.findall(r"\S+", block)
|
||||
chunks: list[str] = []
|
||||
current = ""
|
||||
for token in tokens:
|
||||
if len(token) > maximum:
|
||||
if current:
|
||||
chunks.append(current)
|
||||
current = ""
|
||||
chunks.extend(_hard_split(token, maximum))
|
||||
continue
|
||||
candidate = f"{current} {token}" if current else token
|
||||
if len(candidate) <= maximum:
|
||||
current = candidate
|
||||
else:
|
||||
chunks.append(current)
|
||||
current = token
|
||||
if current:
|
||||
chunks.append(current)
|
||||
return chunks
|
||||
|
||||
|
||||
def _contents(content: str, maximum: int) -> list[str]:
|
||||
if not content:
|
||||
return []
|
||||
result: list[str] = []
|
||||
for block in re.split(r"\n{2,}", content):
|
||||
if block:
|
||||
result.extend(_split_block(block, maximum))
|
||||
return result
|
||||
|
||||
|
||||
def chunk(document: CanonicalDocument, policy: ChunkPolicy) -> list[CanonicalChunk]:
|
||||
"""Split canonical text with stable character-count boundaries and identifiers."""
|
||||
chunks: list[CanonicalChunk] = []
|
||||
for ordinal, content in enumerate(_contents(document.content, policy.max_chars)):
|
||||
identifier = _hash(f"{document.content_hash}:{ordinal}:{policy.version}")
|
||||
chunks.append(
|
||||
CanonicalChunk(
|
||||
chunk_id=f"chunk:{identifier}",
|
||||
document_id=document.document_id,
|
||||
ordinal=ordinal,
|
||||
content=content,
|
||||
content_hash=f"sha256:{_hash(content)}",
|
||||
source_uri=document.source_uri,
|
||||
pipeline_version=document.pipeline_version,
|
||||
metadata={
|
||||
"chunk_policy": {
|
||||
"version": policy.version,
|
||||
"max_chars": policy.max_chars,
|
||||
},
|
||||
"document": document.model_dump(mode="json")["metadata"],
|
||||
"source_fingerprint": document.source_fingerprint,
|
||||
"title": document.title,
|
||||
},
|
||||
)
|
||||
)
|
||||
return chunks
|
||||
@@ -0,0 +1,99 @@
|
||||
"""Pure, deterministic conversion of acquired bytes into canonical text."""
|
||||
|
||||
import hashlib
|
||||
import re
|
||||
import unicodedata
|
||||
from collections.abc import Mapping
|
||||
|
||||
import yaml
|
||||
from pydantic import JsonValue, TypeAdapter, ValidationError
|
||||
|
||||
from tht.corpus.models import CanonicalDocument
|
||||
from tht.ports.evidence import AcquiredDocument, canonical_provenance_uri
|
||||
|
||||
|
||||
MAX_DOCUMENT_BYTES = 10 * 1024 * 1024
|
||||
_CHARSET = re.compile(r"(?:^|;)\s*charset\s*=\s*[\"']?([^;\s\"']+)", re.IGNORECASE)
|
||||
_FRONTMATTER = re.compile(r"\A---\n(.*?)\n---(?:\n|\Z)", re.DOTALL)
|
||||
_JSON_OBJECT = TypeAdapter(dict[str, JsonValue])
|
||||
|
||||
|
||||
class PermanentNormalizationError(ValueError):
|
||||
"""A deterministic input failure which retrying cannot repair."""
|
||||
|
||||
def __init__(self, reason: str) -> None:
|
||||
super().__init__(f"document normalization failed: {reason}")
|
||||
self.reason = reason
|
||||
self.permanent = True
|
||||
|
||||
|
||||
def _sha256(value: str) -> str:
|
||||
return hashlib.sha256(value.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _decode(acquired: AcquiredDocument) -> str:
|
||||
if len(acquired.content) > MAX_DOCUMENT_BYTES:
|
||||
raise PermanentNormalizationError("oversized")
|
||||
|
||||
media_type = acquired.media_type or "text/plain"
|
||||
charset = _CHARSET.search(media_type)
|
||||
if charset and charset.group(1).lower().replace("_", "-") not in {
|
||||
"utf-8",
|
||||
"utf8",
|
||||
"us-ascii",
|
||||
"ascii",
|
||||
}:
|
||||
raise PermanentNormalizationError("unsupported_charset")
|
||||
try:
|
||||
return acquired.content.decode("utf-8-sig", errors="strict")
|
||||
except UnicodeDecodeError as error:
|
||||
raise PermanentNormalizationError("undecodable") from error
|
||||
|
||||
|
||||
def _frontmatter(text: str) -> tuple[dict[str, JsonValue], str]:
|
||||
match = _FRONTMATTER.match(text)
|
||||
if match is None:
|
||||
return {}, text
|
||||
try:
|
||||
loaded = yaml.safe_load(match.group(1))
|
||||
if loaded is None:
|
||||
loaded = {}
|
||||
if not isinstance(loaded, Mapping):
|
||||
raise TypeError("frontmatter is not a mapping")
|
||||
metadata = _JSON_OBJECT.validate_python(dict(loaded))
|
||||
except (TypeError, UnicodeError, ValidationError, yaml.YAMLError) as error:
|
||||
raise PermanentNormalizationError("invalid_frontmatter") from error
|
||||
return metadata, text[match.end() :]
|
||||
|
||||
|
||||
def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDocument:
|
||||
"""Normalize one transport result without I/O or implicit data loss."""
|
||||
if not pipeline_version:
|
||||
raise ValueError("pipeline_version must not be empty")
|
||||
|
||||
decoded = _decode(acquired)
|
||||
canonical = unicodedata.normalize("NFC", decoded.replace("\r\n", "\n").replace("\r", "\n"))
|
||||
frontmatter, content = _frontmatter(canonical)
|
||||
source_uri = canonical_provenance_uri(acquired.source.uri)
|
||||
identity = f"{acquired.source.source_id}\n{source_uri}"
|
||||
media_type = (acquired.media_type or "text/plain").split(";", 1)[0].strip().lower()
|
||||
metadata: dict[str, JsonValue] = {
|
||||
"source": acquired.source.model_dump(mode="json")["metadata"],
|
||||
"acquisition": acquired.model_dump(mode="json")["metadata"],
|
||||
}
|
||||
if frontmatter:
|
||||
metadata["frontmatter"] = frontmatter
|
||||
|
||||
return CanonicalDocument(
|
||||
document_id=f"doc:{_sha256(identity)}",
|
||||
source_id=acquired.source.source_id,
|
||||
source_uri=source_uri,
|
||||
source_fingerprint=acquired.source.fingerprint,
|
||||
content_hash=f"sha256:{_sha256(content)}",
|
||||
title=str(frontmatter.get("title", "")),
|
||||
content=content,
|
||||
media_type=media_type,
|
||||
modified_at=acquired.source.modified_at,
|
||||
pipeline_version=pipeline_version,
|
||||
metadata=metadata,
|
||||
)
|
||||
Reference in New Issue
Block a user