feat(corpus): add deterministic normalization and chunking

This commit is contained in:
2026-07-12 03:44:54 +02:00
parent d1fdf7d9f5
commit 015715d092
5 changed files with 365 additions and 0 deletions
+89
View File
@@ -0,0 +1,89 @@
"""Versioned deterministic chunking for canonical corpus documents."""
import hashlib
import re
from dataclasses import dataclass
from tht.corpus.models import CanonicalChunk, CanonicalDocument
@dataclass(frozen=True, slots=True)
class ChunkPolicy:
version: str
max_chars: int
def __post_init__(self) -> None:
if not self.version:
raise ValueError("chunk policy version must not be empty")
if self.max_chars <= 0:
raise ValueError("max_chars must be greater than zero")
def _hash(text: str) -> str:
return hashlib.sha256(text.encode("utf-8")).hexdigest()
def _hard_split(text: str, maximum: int) -> list[str]:
return [text[start : start + maximum] for start in range(0, len(text), maximum)]
def _split_block(block: str, maximum: int) -> list[str]:
if len(block) <= maximum:
return [block]
tokens = re.findall(r"\S+", block)
chunks: list[str] = []
current = ""
for token in tokens:
if len(token) > maximum:
if current:
chunks.append(current)
current = ""
chunks.extend(_hard_split(token, maximum))
continue
candidate = f"{current} {token}" if current else token
if len(candidate) <= maximum:
current = candidate
else:
chunks.append(current)
current = token
if current:
chunks.append(current)
return chunks
def _contents(content: str, maximum: int) -> list[str]:
if not content:
return []
result: list[str] = []
for block in re.split(r"\n{2,}", content):
if block:
result.extend(_split_block(block, maximum))
return result
def chunk(document: CanonicalDocument, policy: ChunkPolicy) -> list[CanonicalChunk]:
"""Split canonical text with stable character-count boundaries and identifiers."""
chunks: list[CanonicalChunk] = []
for ordinal, content in enumerate(_contents(document.content, policy.max_chars)):
identifier = _hash(f"{document.content_hash}:{ordinal}:{policy.version}")
chunks.append(
CanonicalChunk(
chunk_id=f"chunk:{identifier}",
document_id=document.document_id,
ordinal=ordinal,
content=content,
content_hash=f"sha256:{_hash(content)}",
source_uri=document.source_uri,
pipeline_version=document.pipeline_version,
metadata={
"chunk_policy": {
"version": policy.version,
"max_chars": policy.max_chars,
},
"document": document.model_dump(mode="json")["metadata"],
"source_fingerprint": document.source_fingerprint,
"title": document.title,
},
)
)
return chunks
+99
View File
@@ -0,0 +1,99 @@
"""Pure, deterministic conversion of acquired bytes into canonical text."""
import hashlib
import re
import unicodedata
from collections.abc import Mapping
import yaml
from pydantic import JsonValue, TypeAdapter, ValidationError
from tht.corpus.models import CanonicalDocument
from tht.ports.evidence import AcquiredDocument, canonical_provenance_uri
MAX_DOCUMENT_BYTES = 10 * 1024 * 1024
_CHARSET = re.compile(r"(?:^|;)\s*charset\s*=\s*[\"']?([^;\s\"']+)", re.IGNORECASE)
_FRONTMATTER = re.compile(r"\A---\n(.*?)\n---(?:\n|\Z)", re.DOTALL)
_JSON_OBJECT = TypeAdapter(dict[str, JsonValue])
class PermanentNormalizationError(ValueError):
"""A deterministic input failure which retrying cannot repair."""
def __init__(self, reason: str) -> None:
super().__init__(f"document normalization failed: {reason}")
self.reason = reason
self.permanent = True
def _sha256(value: str) -> str:
return hashlib.sha256(value.encode("utf-8")).hexdigest()
def _decode(acquired: AcquiredDocument) -> str:
if len(acquired.content) > MAX_DOCUMENT_BYTES:
raise PermanentNormalizationError("oversized")
media_type = acquired.media_type or "text/plain"
charset = _CHARSET.search(media_type)
if charset and charset.group(1).lower().replace("_", "-") not in {
"utf-8",
"utf8",
"us-ascii",
"ascii",
}:
raise PermanentNormalizationError("unsupported_charset")
try:
return acquired.content.decode("utf-8-sig", errors="strict")
except UnicodeDecodeError as error:
raise PermanentNormalizationError("undecodable") from error
def _frontmatter(text: str) -> tuple[dict[str, JsonValue], str]:
match = _FRONTMATTER.match(text)
if match is None:
return {}, text
try:
loaded = yaml.safe_load(match.group(1))
if loaded is None:
loaded = {}
if not isinstance(loaded, Mapping):
raise TypeError("frontmatter is not a mapping")
metadata = _JSON_OBJECT.validate_python(dict(loaded))
except (TypeError, UnicodeError, ValidationError, yaml.YAMLError) as error:
raise PermanentNormalizationError("invalid_frontmatter") from error
return metadata, text[match.end() :]
def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDocument:
"""Normalize one transport result without I/O or implicit data loss."""
if not pipeline_version:
raise ValueError("pipeline_version must not be empty")
decoded = _decode(acquired)
canonical = unicodedata.normalize("NFC", decoded.replace("\r\n", "\n").replace("\r", "\n"))
frontmatter, content = _frontmatter(canonical)
source_uri = canonical_provenance_uri(acquired.source.uri)
identity = f"{acquired.source.source_id}\n{source_uri}"
media_type = (acquired.media_type or "text/plain").split(";", 1)[0].strip().lower()
metadata: dict[str, JsonValue] = {
"source": acquired.source.model_dump(mode="json")["metadata"],
"acquisition": acquired.model_dump(mode="json")["metadata"],
}
if frontmatter:
metadata["frontmatter"] = frontmatter
return CanonicalDocument(
document_id=f"doc:{_sha256(identity)}",
source_id=acquired.source.source_id,
source_uri=source_uri,
source_fingerprint=acquired.source.fingerprint,
content_hash=f"sha256:{_sha256(content)}",
title=str(frontmatter.get("title", "")),
content=content,
media_type=media_type,
modified_at=acquired.source.modified_at,
pipeline_version=pipeline_version,
metadata=metadata,
)