246 lines
9.5 KiB
Python
246 lines
9.5 KiB
Python
"""Evidence-owned deterministic chunking for canonical corpus documents."""
|
|
|
|
import hashlib
|
|
import json
|
|
import re
|
|
from collections.abc import Mapping
|
|
from dataclasses import asdict, dataclass
|
|
|
|
from tht.evidence.canonical import CuratedEvidence, ReviewItem
|
|
from tht.evidence.corpus.models import CanonicalChunk, CanonicalDocument
|
|
|
|
|
|
@dataclass(frozen=True, slots=True)
|
|
class ChunkPolicy:
|
|
version: str
|
|
max_chars: int
|
|
|
|
def __post_init__(self) -> None:
|
|
if not self.version:
|
|
raise ValueError("chunk policy version must not be empty")
|
|
if self.max_chars <= 0:
|
|
raise ValueError("max_chars must be greater than zero")
|
|
|
|
|
|
class AtomicContentTooLargeError(ValueError):
|
|
"""A semantic Evidence element exceeds the configured embedding boundary."""
|
|
|
|
code = "atomic_content_too_large"
|
|
|
|
def __init__(self, evidence: CuratedEvidence) -> None:
|
|
super().__init__(self.code)
|
|
field = {
|
|
"formula": "formula.sql",
|
|
"enum": "enum.values",
|
|
"mapping": "mapping",
|
|
"normalization": "normalization.rule",
|
|
"glossary": "glossary.definition",
|
|
"domain": "domain.rule",
|
|
"example": "example",
|
|
"reference": "reference.url",
|
|
}[evidence.kind]
|
|
self.review_item = ReviewItem(
|
|
code=self.code,
|
|
message=(
|
|
f"Il contenuto atomico di {evidence.id} supera max_chunk_chars e non può essere diviso."
|
|
),
|
|
field=field,
|
|
)
|
|
|
|
|
|
def _hash(text: str) -> str:
|
|
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
|
|
|
|
def _contents(content: str, maximum: int) -> list[str]:
|
|
result: list[str] = []
|
|
start = 0
|
|
while start < len(content):
|
|
end = min(start + maximum, len(content))
|
|
if end < len(content):
|
|
boundaries = list(re.finditer(r"\s+", content[start:end]))
|
|
if boundaries:
|
|
end = start + boundaries[-1].end()
|
|
result.append(content[start:end])
|
|
start = end
|
|
return result
|
|
|
|
|
|
def _policy_fingerprint(policy: ChunkPolicy) -> str:
|
|
serialized = json.dumps(asdict(policy), ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
|
return f"sha256:{_hash(serialized)}"
|
|
|
|
|
|
def _curated_evidence(document: CanonicalDocument) -> CuratedEvidence | None:
|
|
raw = document.metadata.get("curated_evidence")
|
|
if raw is None:
|
|
return None
|
|
if not isinstance(raw, Mapping):
|
|
raise TypeError("invalid curated evidence projection")
|
|
try:
|
|
return CuratedEvidence.model_validate(raw)
|
|
except ValueError as error:
|
|
raise ValueError("invalid curated evidence projection") from error
|
|
|
|
|
|
_ENGLISH_LABELS = {
|
|
"purpose": "Purpose", "concept": "Concept", "tables": "Tables", "columns": "Columns",
|
|
"column": "Column", "value": "Value", "meaning": "Meaning", "input": "Input",
|
|
"output": "Output", "rule": "Rule", "definition": "Definition", "synonyms": "Synonyms",
|
|
"variants": "Variants", "question": "Question", "interpretation": "Interpretation",
|
|
"label": "Label", "url": "URL", "description": "Description", "provenance": "Provenance",
|
|
}
|
|
_ITALIAN_LABELS = {
|
|
"purpose": "Scopi", "concept": "Concetto", "tables": "Tabelle", "columns": "Colonne",
|
|
"column": "Colonna", "value": "Valore", "meaning": "Significato", "input": "Input",
|
|
"output": "Output", "rule": "Regola", "definition": "Definizione", "synonyms": "Sinonimi",
|
|
"variants": "Varianti", "question": "Domanda", "interpretation": "Interpretazione",
|
|
"label": "Etichetta", "url": "URL", "description": "Descrizione", "provenance": "Provenienza",
|
|
}
|
|
_ITALIAN_KIND_LABELS = {
|
|
"glossary": "Glossario", "domain": "Dominio", "enum": "Enum", "example": "Esempio",
|
|
"mapping": "Mappatura", "normalization": "Normalizzazione", "formula": "Formula",
|
|
"reference": "Riferimento",
|
|
}
|
|
|
|
|
|
def _labels(evidence: CuratedEvidence) -> Mapping[str, str]:
|
|
return _ITALIAN_LABELS if evidence.language.lower().startswith("it") else _ENGLISH_LABELS
|
|
|
|
|
|
def _kind_label(evidence: CuratedEvidence) -> str:
|
|
if evidence.language.lower().startswith("it"):
|
|
return _ITALIAN_KIND_LABELS[evidence.kind]
|
|
return evidence.kind.title()
|
|
|
|
|
|
def _scope_lines(evidence: CuratedEvidence, labels: Mapping[str, str]) -> list[str]:
|
|
lines: list[str] = []
|
|
if evidence.applies_to.concepts:
|
|
lines.append(f"{labels['concept']}: " + ", ".join(evidence.applies_to.concepts))
|
|
if evidence.applies_to.tables:
|
|
lines.append(f"{labels['tables']}: " + ", ".join(evidence.applies_to.tables))
|
|
if evidence.applies_to.columns:
|
|
lines.append(f"{labels['columns']}: " + ", ".join(evidence.applies_to.columns))
|
|
return lines
|
|
|
|
|
|
def _curated_fragment_content(evidence: CuratedEvidence) -> list[str]:
|
|
"""Render semantic atoms without treating structured values as arbitrary text."""
|
|
label = _kind_label(evidence)
|
|
labels = _labels(evidence)
|
|
common = [
|
|
f"{label}: {evidence.title}",
|
|
f"{labels['purpose']}: " + ", ".join(evidence.purposes),
|
|
*_scope_lines(evidence, labels),
|
|
]
|
|
payload = evidence.payload
|
|
if evidence.kind == "formula":
|
|
body = [
|
|
f"{labels['concept']}: {payload.concept}",
|
|
f"{labels['columns']}: " + ", ".join(payload.columns),
|
|
f"SQL: {payload.sql}",
|
|
]
|
|
return ["\n".join([*common, *body, f"{labels['provenance']}: {evidence.provenance.source_file}"])]
|
|
if evidence.kind == "enum":
|
|
return [
|
|
"\n".join([
|
|
*common,
|
|
f"{labels['column']}: {payload.column}",
|
|
f"{labels['value']}: {value}",
|
|
f"{labels['meaning']}: {meaning}",
|
|
f"{labels['provenance']}: {evidence.provenance.source_file}",
|
|
])
|
|
for value, meaning in sorted(payload.values.items())
|
|
]
|
|
if evidence.kind == "mapping":
|
|
body = [
|
|
f"{labels['concept']}: {payload.concept}",
|
|
f"{labels['tables']}: " + ", ".join(payload.tables),
|
|
f"{labels['columns']}: " + ", ".join(payload.columns),
|
|
]
|
|
elif evidence.kind == "normalization":
|
|
body = [
|
|
f"{labels['input']}: {payload.input}",
|
|
f"{labels['output']}: {payload.output}",
|
|
f"{labels['rule']}: {payload.rule}",
|
|
]
|
|
elif evidence.kind == "glossary":
|
|
body = [
|
|
f"{labels['definition']}: {payload.definition}",
|
|
*( [f"{labels['synonyms']}: " + ", ".join(payload.synonyms)] if payload.synonyms else []),
|
|
*( [f"{labels['variants']}: " + ", ".join(payload.variants)] if payload.variants else []),
|
|
]
|
|
elif evidence.kind == "domain":
|
|
body = [f"{labels['rule']}: {payload.rule}"]
|
|
elif evidence.kind == "example":
|
|
body = [f"{labels['question']}: {payload.question}", f"{labels['interpretation']}: {payload.interpretation}"]
|
|
elif evidence.kind == "reference":
|
|
body = [f"{labels['label']}: {payload.label}", f"{labels['url']}: {payload.url}", f"{labels['description']}: {payload.description}"]
|
|
else: # pragma: no cover - CuratedEvidence validates the finite kind set.
|
|
raise ValueError("unsupported curated evidence kind")
|
|
return ["\n".join([*common, *body, f"{labels['provenance']}: {evidence.provenance.source_file}"])]
|
|
|
|
|
|
def _curated_metadata(evidence: CuratedEvidence) -> dict:
|
|
return {
|
|
"evidence_id": evidence.id,
|
|
"evidence_kind": evidence.kind,
|
|
"purposes": list(evidence.purposes),
|
|
"scope": evidence.applies_to.model_dump(mode="json"),
|
|
"language": evidence.language,
|
|
"provenance": evidence.provenance.model_dump(mode="json"),
|
|
}
|
|
|
|
|
|
def chunk(document: CanonicalDocument, policy: ChunkPolicy) -> list[CanonicalChunk]:
|
|
"""Split canonical text with stable character-count boundaries and identifiers."""
|
|
chunks: list[CanonicalChunk] = []
|
|
policy_fingerprint = _policy_fingerprint(policy)
|
|
curated = _curated_evidence(document)
|
|
contents = (
|
|
_curated_fragment_content(curated)
|
|
if curated is not None
|
|
else _contents(document.content, policy.max_chars)
|
|
)
|
|
if any(len(content) > policy.max_chars for content in contents):
|
|
if curated is not None:
|
|
raise AtomicContentTooLargeError(curated)
|
|
raise ValueError("chunk content exceeds max_chars")
|
|
for ordinal, content in enumerate(contents):
|
|
chunk_hash = f"sha256:{_hash(content)}"
|
|
identifier = _hash(
|
|
":".join(
|
|
(
|
|
document.document_id,
|
|
document.content_hash,
|
|
policy_fingerprint,
|
|
str(ordinal),
|
|
chunk_hash,
|
|
)
|
|
)
|
|
)
|
|
chunks.append(
|
|
CanonicalChunk(
|
|
chunk_id=f"chunk:{identifier}",
|
|
document_id=document.document_id,
|
|
ordinal=ordinal,
|
|
content=content,
|
|
content_hash=chunk_hash,
|
|
source_uri=document.source_uri,
|
|
pipeline_version=document.pipeline_version,
|
|
metadata={
|
|
"chunk_policy": {
|
|
"version": policy.version,
|
|
"max_chars": policy.max_chars,
|
|
"fingerprint": policy_fingerprint,
|
|
},
|
|
"document": document.model_dump(mode="json")["metadata"],
|
|
"source_fingerprint": document.source_fingerprint,
|
|
"title": document.title,
|
|
**(_curated_metadata(curated) if curated is not None else {}),
|
|
},
|
|
)
|
|
)
|
|
return chunks
|