feat(corpus): add deterministic normalization and chunking
This commit is contained in:
@@ -0,0 +1,89 @@
|
||||
"""Versioned deterministic chunking for canonical corpus documents."""
|
||||
|
||||
import hashlib
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
|
||||
from tht.corpus.models import CanonicalChunk, CanonicalDocument
|
||||
|
||||
|
||||
@dataclass(frozen=True, slots=True)
|
||||
class ChunkPolicy:
|
||||
version: str
|
||||
max_chars: int
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
if not self.version:
|
||||
raise ValueError("chunk policy version must not be empty")
|
||||
if self.max_chars <= 0:
|
||||
raise ValueError("max_chars must be greater than zero")
|
||||
|
||||
|
||||
def _hash(text: str) -> str:
|
||||
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _hard_split(text: str, maximum: int) -> list[str]:
|
||||
return [text[start : start + maximum] for start in range(0, len(text), maximum)]
|
||||
|
||||
|
||||
def _split_block(block: str, maximum: int) -> list[str]:
|
||||
if len(block) <= maximum:
|
||||
return [block]
|
||||
tokens = re.findall(r"\S+", block)
|
||||
chunks: list[str] = []
|
||||
current = ""
|
||||
for token in tokens:
|
||||
if len(token) > maximum:
|
||||
if current:
|
||||
chunks.append(current)
|
||||
current = ""
|
||||
chunks.extend(_hard_split(token, maximum))
|
||||
continue
|
||||
candidate = f"{current} {token}" if current else token
|
||||
if len(candidate) <= maximum:
|
||||
current = candidate
|
||||
else:
|
||||
chunks.append(current)
|
||||
current = token
|
||||
if current:
|
||||
chunks.append(current)
|
||||
return chunks
|
||||
|
||||
|
||||
def _contents(content: str, maximum: int) -> list[str]:
|
||||
if not content:
|
||||
return []
|
||||
result: list[str] = []
|
||||
for block in re.split(r"\n{2,}", content):
|
||||
if block:
|
||||
result.extend(_split_block(block, maximum))
|
||||
return result
|
||||
|
||||
|
||||
def chunk(document: CanonicalDocument, policy: ChunkPolicy) -> list[CanonicalChunk]:
|
||||
"""Split canonical text with stable character-count boundaries and identifiers."""
|
||||
chunks: list[CanonicalChunk] = []
|
||||
for ordinal, content in enumerate(_contents(document.content, policy.max_chars)):
|
||||
identifier = _hash(f"{document.content_hash}:{ordinal}:{policy.version}")
|
||||
chunks.append(
|
||||
CanonicalChunk(
|
||||
chunk_id=f"chunk:{identifier}",
|
||||
document_id=document.document_id,
|
||||
ordinal=ordinal,
|
||||
content=content,
|
||||
content_hash=f"sha256:{_hash(content)}",
|
||||
source_uri=document.source_uri,
|
||||
pipeline_version=document.pipeline_version,
|
||||
metadata={
|
||||
"chunk_policy": {
|
||||
"version": policy.version,
|
||||
"max_chars": policy.max_chars,
|
||||
},
|
||||
"document": document.model_dump(mode="json")["metadata"],
|
||||
"source_fingerprint": document.source_fingerprint,
|
||||
"title": document.title,
|
||||
},
|
||||
)
|
||||
)
|
||||
return chunks
|
||||
Reference in New Issue
Block a user