90 lines
2.9 KiB
Python
90 lines
2.9 KiB
Python
"""Versioned deterministic chunking for canonical corpus documents."""
|
|
|
|
import hashlib
|
|
import re
|
|
from dataclasses import dataclass
|
|
|
|
from tht.corpus.models import CanonicalChunk, CanonicalDocument
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class ChunkPolicy:
|
|
version: str
|
|
max_chars: int
|
|
|
|
def __post_init__(self) -> None:
|
|
if not self.version:
|
|
raise ValueError("chunk policy version must not be empty")
|
|
if self.max_chars <= 0:
|
|
raise ValueError("max_chars must be greater than zero")
|
|
|
|
|
|
def _hash(text: str) -> str:
|
|
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
|
|
|
|
def _hard_split(text: str, maximum: int) -> list[str]:
|
|
return [text[start : start + maximum] for start in range(0, len(text), maximum)]
|
|
|
|
|
|
def _split_block(block: str, maximum: int) -> list[str]:
|
|
if len(block) <= maximum:
|
|
return [block]
|
|
tokens = re.findall(r"\S+", block)
|
|
chunks: list[str] = []
|
|
current = ""
|
|
for token in tokens:
|
|
if len(token) > maximum:
|
|
if current:
|
|
chunks.append(current)
|
|
current = ""
|
|
chunks.extend(_hard_split(token, maximum))
|
|
continue
|
|
candidate = f"{current} {token}" if current else token
|
|
if len(candidate) <= maximum:
|
|
current = candidate
|
|
else:
|
|
chunks.append(current)
|
|
current = token
|
|
if current:
|
|
chunks.append(current)
|
|
return chunks
|
|
|
|
|
|
def _contents(content: str, maximum: int) -> list[str]:
|
|
if not content:
|
|
return []
|
|
result: list[str] = []
|
|
for block in re.split(r"\n{2,}", content):
|
|
if block:
|
|
result.extend(_split_block(block, maximum))
|
|
return result
|
|
|
|
|
|
def chunk(document: CanonicalDocument, policy: ChunkPolicy) -> list[CanonicalChunk]:
|
|
"""Split canonical text with stable character-count boundaries and identifiers."""
|
|
chunks: list[CanonicalChunk] = []
|
|
for ordinal, content in enumerate(_contents(document.content, policy.max_chars)):
|
|
identifier = _hash(f"{document.content_hash}:{ordinal}:{policy.version}")
|
|
chunks.append(
|
|
CanonicalChunk(
|
|
chunk_id=f"chunk:{identifier}",
|
|
document_id=document.document_id,
|
|
ordinal=ordinal,
|
|
content=content,
|
|
content_hash=f"sha256:{_hash(content)}",
|
|
source_uri=document.source_uri,
|
|
pipeline_version=document.pipeline_version,
|
|
metadata={
|
|
"chunk_policy": {
|
|
"version": policy.version,
|
|
"max_chars": policy.max_chars,
|
|
},
|
|
"document": document.model_dump(mode="json")["metadata"],
|
|
"source_fingerprint": document.source_fingerprint,
|
|
"title": document.title,
|
|
},
|
|
)
|
|
)
|
|
return chunks
|