feat(evidence): validate canonical curated corpus

This commit is contained in:
2026-08-24 17:38:22 +02:00
parent 5c6228f8c2
commit ae0976a4aa
5 changed files with 1217 additions and 0 deletions
+24
View File
@@ -1,6 +1,20 @@
"""Cohesive public entrypoint for Evidence domain capabilities."""
from tht.evidence.acquisition import acquire, discover
from tht.evidence.authoring import (
EvidenceManifest,
ValidationFinding,
ValidationReport,
dump_manifest,
load_manifest,
validate_workspace_evidence,
)
from tht.evidence.canonical import (
CuratedEvidence,
dump_curated_markdown,
load_curated_tree,
parse_curated_markdown,
)
from tht.evidence.contracts import (
AcquiredDocument,
EvidenceSource,
@@ -28,11 +42,15 @@ __all__ = [
"AcquiredDocument",
"ActiveEvidenceSearcher",
"CorpusWorkspaceMismatchError",
"CuratedEvidence",
"EvidenceEmbedder",
"EvidenceManifest",
"EvidenceSource",
"EvidenceSourceError",
"EvidenceSourceErrorCategory",
"SourceObject",
"ValidationFinding",
"ValidationReport",
"acquire",
"active_searcher",
"build_preprocessing_pipeline",
@@ -40,10 +58,16 @@ __all__ = [
"build_sources",
"canonical_provenance_uri",
"discover",
"dump_curated_markdown",
"dump_manifest",
"load_curated_tree",
"load_manifest",
"normalize_aware_datetime",
"parse_curated_markdown",
"project_session",
"resolve_citation",
"validate_corpus_workspace",
"validate_namespaced_value",
"validate_safe_metadata",
"validate_workspace_evidence",
]
+256
View File
@@ -0,0 +1,256 @@
"""Validation for the Git-reviewed Evidence authoring workspace."""
from __future__ import annotations
import hashlib
import unicodedata
from dataclasses import dataclass
from pathlib import Path
from typing import Literal
import yaml
from pydantic import ValidationError, field_validator, model_validator
from tht.evidence.canonical import (
CuratedEvidence,
StrictModel,
is_evidence_id,
load_curated_tree,
validate_source_file,
)
MAX_AUTHORING_FILE_BYTES = 10 * 1024 * 1024
@dataclass(frozen=True)
class ValidationFinding:
severity: Literal["error", "warning"]
code: str
path: str
message: str
@dataclass(frozen=True)
class ValidationReport:
findings: tuple[ValidationFinding, ...]
@property
def publishable(self) -> bool:
return not any(finding.severity == "error" for finding in self.findings)
class ManifestSource(StrictModel):
sha256: str
units: tuple[str, ...]
@field_validator("sha256")
@classmethod
def _validate_sha256(cls, value: str) -> str:
if not value.startswith("sha256:") or len(value) != 71:
raise ValueError("sha256 must be a sha256 digest")
try:
int(value.removeprefix("sha256:"), 16)
except ValueError as error:
raise ValueError("sha256 must be a sha256 digest") from error
return value
@field_validator("units")
@classmethod
def _validate_units(cls, value: tuple[str, ...]) -> tuple[str, ...]:
if any(not is_evidence_id(unit) for unit in value):
raise ValueError("units must use stable evidence identifiers")
return value
class EvidenceManifest(StrictModel):
schema_version: Literal[1]
pipeline_version: Literal["evidence-authoring-v1"]
sources: dict[str, ManifestSource]
orphans: tuple[str, ...]
@field_validator("sources")
@classmethod
def _validate_sources(cls, value: dict[str, ManifestSource]) -> dict[str, ManifestSource]:
for source_path in value:
validate_source_file(source_path)
return value
@field_validator("orphans")
@classmethod
def _validate_orphans(cls, value: tuple[str, ...]) -> tuple[str, ...]:
if any(not is_evidence_id(unit) for unit in value):
raise ValueError("orphans must use stable evidence identifiers")
return value
@model_validator(mode="after")
def _validate_unit_membership(self) -> EvidenceManifest:
seen: set[str] = set()
for source in self.sources.values():
duplicate = seen.intersection(source.units)
if duplicate:
raise ValueError("a unit may belong to only one source")
seen.update(source.units)
if len(set(self.orphans)) != len(self.orphans):
raise ValueError("orphans must be unique")
return self
def load_manifest(path: Path) -> EvidenceManifest:
"""Load the managed, versioned authoring manifest."""
try:
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError, yaml.YAMLError) as error:
raise ValueError("manifest cannot be read") from error
try:
return EvidenceManifest.model_validate(raw)
except ValidationError as error:
raise ValueError("manifest is invalid") from error
def dump_manifest(manifest: EvidenceManifest) -> str:
"""Serialize the manifest deterministically for Git review."""
return yaml.safe_dump(manifest.model_dump(mode="json"), allow_unicode=True, sort_keys=True)
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
"""Validate the curated corpus without writing the workspace."""
evidence_root = workspace_root / "evidence"
findings: list[ValidationFinding] = []
manifest_path = evidence_root / "manifest.yaml"
if not manifest_path.is_file():
return ValidationReport((ValidationFinding(
severity="error",
code="manifest_missing",
path="manifest.yaml",
message="The managed Evidence manifest is missing.",
),))
try:
manifest = load_manifest(manifest_path)
except ValueError:
return ValidationReport((ValidationFinding(
severity="error",
code="manifest_invalid",
path="manifest.yaml",
message="The managed Evidence manifest is invalid.",
),))
for orphan in manifest.orphans:
findings.append(ValidationFinding(
severity="error",
code="orphaned_unit",
path="manifest.yaml",
message=f"Orphaned Evidence {orphan} must be resolved before publication.",
))
try:
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError):
return ValidationReport(tuple(findings + [ValidationFinding(
severity="error",
code="curated_invalid",
path="curated",
message="A Curated Evidence document is invalid or cannot be read.",
)]))
source_texts = {
source_path: _validate_manifest_source(evidence_root, source_path, source, findings)
for source_path, source in manifest.sources.items()
}
seen_ids: set[str] = set()
for evidence in documents:
if evidence.id in seen_ids:
findings.append(ValidationFinding(
severity="error",
code="duplicate_evidence_id",
path="curated",
message=f"Evidence id {evidence.id} appears more than once.",
))
seen_ids.add(evidence.id)
findings.extend(_validate_unit(manifest, evidence, source_texts.get(evidence.provenance.source_file)))
for source in manifest.sources.values():
for unit in source.units:
if unit not in seen_ids:
findings.append(ValidationFinding(
severity="error",
code="manifest_unit_without_curated",
path="manifest.yaml",
message=f"Manifest Evidence {unit} has no Curated document.",
))
return ValidationReport(tuple(findings))
def _validate_manifest_source(
evidence_root: Path,
source_path: str,
manifest_source: ManifestSource,
findings: list[ValidationFinding],
) -> str | None:
path = evidence_root / source_path
if not path.is_file():
findings.append(ValidationFinding("error", "source_missing", source_path,
"The manifest source does not exist."))
return None
if path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
findings.append(ValidationFinding("error", "source_oversized", source_path,
"The manifest source exceeds the authoring size limit."))
return None
try:
source = _normalize(path.read_text(encoding="utf-8"))
except UnicodeDecodeError:
findings.append(ValidationFinding("error", "source_unreadable", source_path,
"The manifest source is not valid UTF-8."))
return None
digest = "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
if digest != manifest_source.sha256:
findings.append(ValidationFinding("error", "source_hash_mismatch", source_path,
"The manifest hash does not match the normalized source."))
return source
def _validate_unit(
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
) -> list[ValidationFinding]:
path = evidence.provenance.source_file
findings: list[ValidationFinding] = []
manifest_source = manifest.sources.get(path)
if manifest_source is None:
findings.append(ValidationFinding(
severity="error",
code="manifest_source_missing",
path="manifest.yaml",
message="The manifest does not contain the provenance source.",
))
else:
if manifest_source.sha256 != evidence.provenance.source_sha256:
findings.append(ValidationFinding(
severity="error",
code="manifest_source_hash_mismatch",
path="manifest.yaml",
message="The manifest hash does not match the Evidence provenance.",
))
if evidence.id not in manifest_source.units:
findings.append(ValidationFinding(
severity="error",
code="manifest_unit_missing",
path="manifest.yaml",
message="The manifest does not link the Evidence unit to its source.",
))
if source is None:
return findings
for excerpt in evidence.provenance.supporting_excerpts:
if _normalize(excerpt) not in source:
findings.append(ValidationFinding(
severity="error",
code="supporting_excerpt_missing",
path=path,
message="A supporting excerpt is absent from the normalized source.",
))
for item in evidence.review_items:
findings.append(ValidationFinding(
severity="error",
code="unresolved_review_item",
path=path,
message=f"Review item {item.code} must be resolved before publication.",
))
return findings
def _normalize(text: str) -> str:
return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n"))
+332
View File
@@ -0,0 +1,332 @@
"""Typed, reviewable Evidence units stored in the workspace repository."""
from __future__ import annotations
import re
from pathlib import Path, PurePosixPath
from typing import Literal
import sqlglot
import yaml
from pydantic import AnyHttpUrl, BaseModel, ConfigDict, Field, field_validator, model_validator
from sqlglot import exp
from tht.evidence.contracts import validate_canonical_uri
class StrictModel(BaseModel):
"""Reject undeclared fields in the repository's canonical format."""
model_config = ConfigDict(extra="forbid")
EvidenceKind = Literal[
"glossary",
"domain",
"enum",
"example",
"mapping",
"normalization",
"formula",
"reference",
]
EvidencePurpose = Literal[
"disambiguation",
"rewriting",
"schema_linking",
"sql_generation",
]
_IDENTIFIER = r"[A-Za-z_][A-Za-z0-9_$]*"
_TABLE_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}$")
_COLUMN_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}\.{_IDENTIFIER}$")
MAX_CURATED_FILE_BYTES = 10 * 1024 * 1024
class EvidenceScope(StrictModel):
concepts: tuple[str, ...] = ()
tables: tuple[str, ...] = ()
columns: tuple[str, ...] = ()
@field_validator("tables")
@classmethod
def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables")
@field_validator("columns")
@classmethod
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
class EvidenceProvenance(StrictModel):
model_config = ConfigDict(extra="forbid", frozen=True)
source_file: str
source_sha256: str
supporting_excerpts: tuple[str, ...]
@field_validator("source_file")
@classmethod
def _validate_source_file(cls, value: str) -> str:
return validate_source_file(value)
@field_validator("source_sha256")
@classmethod
def _validate_sha256(cls, value: str) -> str:
if not re.fullmatch(r"sha256:[0-9a-f]{64}", value):
raise ValueError("source_sha256 must be a sha256 digest")
return value
@field_validator("supporting_excerpts")
@classmethod
def _validate_excerpts(cls, value: tuple[str, ...]) -> tuple[str, ...]:
if not 1 <= len(value) <= 5:
raise ValueError("supporting_excerpts must contain one to five items")
if any(not excerpt.strip() or len(excerpt) > 1000 for excerpt in value):
raise ValueError("supporting excerpts must be nonempty and at most 1000 characters")
return value
class ReviewItem(StrictModel):
code: str
message: str
field: str | None = None
class FormulaPayload(StrictModel):
concept: str
columns: tuple[str, ...]
sql: str
@field_validator("columns")
@classmethod
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
@field_validator("sql")
@classmethod
def _validate_expression(cls, value: str) -> str:
try:
statements = [statement for statement in sqlglot.parse(value, read="postgres") if statement]
except sqlglot.errors.ParseError as error:
raise ValueError("formula.sql must be valid PostgreSQL") from error
if len(statements) != 1:
raise ValueError("formula.sql must contain exactly one expression")
expression = statements[0]
if expression.find(exp.Select) is not None or expression.find(exp.With) is not None:
raise ValueError("formula.sql must not contain a query")
if any(
expression.find(statement_type) is not None
for statement_type in (
exp.Insert,
exp.Update,
exp.Delete,
exp.Create,
exp.Drop,
exp.Alter,
exp.Merge,
exp.TruncateTable,
exp.Grant,
exp.Revoke,
exp.Command,
exp.Values,
exp.Set,
exp.Table,
)
):
raise ValueError("formula.sql must not contain DDL or DML")
return value
class ReferencePayload(StrictModel):
url: AnyHttpUrl
label: str
description: str
@field_validator("url")
@classmethod
def _reject_credentials(cls, value: AnyHttpUrl) -> AnyHttpUrl:
validate_canonical_uri(str(value))
return value
class GlossaryPayload(StrictModel):
definition: str
synonyms: tuple[str, ...] = ()
variants: tuple[str, ...] = ()
class DomainPayload(StrictModel):
rule: str
class EnumPayload(StrictModel):
column: str
values: dict[str, str]
@field_validator("column")
@classmethod
def _validate_column(cls, value: str) -> str:
_validate_identifiers((value,), _COLUMN_IDENTIFIER, "column")
return value
class ExamplePayload(StrictModel):
question: str
interpretation: str
class MappingPayload(StrictModel):
concept: str
tables: tuple[str, ...]
columns: tuple[str, ...]
@field_validator("tables")
@classmethod
def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables")
@field_validator("columns")
@classmethod
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
class NormalizationPayload(StrictModel):
input: str
output: str
rule: str
EvidencePayload = (
GlossaryPayload
| DomainPayload
| EnumPayload
| ExamplePayload
| MappingPayload
| NormalizationPayload
| FormulaPayload
| ReferencePayload
)
_PAYLOAD_TYPE_BY_KIND = {
"glossary": GlossaryPayload,
"domain": DomainPayload,
"enum": EnumPayload,
"example": ExamplePayload,
"mapping": MappingPayload,
"normalization": NormalizationPayload,
"formula": FormulaPayload,
"reference": ReferencePayload,
}
_EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
class CuratedEvidence(StrictModel):
schema_version: Literal[1]
id: str
title: str
kind: EvidenceKind
purposes: tuple[EvidencePurpose, ...]
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
language: str
provenance: EvidenceProvenance
review_items: tuple[ReviewItem, ...] = ()
payload: EvidencePayload
@model_validator(mode="after")
def _validate_kind_payload(self) -> CuratedEvidence:
if not is_evidence_id(self.id):
raise ValueError("id must use the evidence:<slug> form")
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
if expected is not None and not isinstance(self.payload, expected):
raise ValueError(f"{self.kind} requires its typed payload")
return self
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
"""Parse the canonical frontmatter representation of one Curated Evidence unit."""
if not text.startswith("---\n"):
raise ValueError("curated evidence requires YAML frontmatter")
try:
_, frontmatter, body = text.split("---\n", 2)
except ValueError as error:
raise ValueError("curated evidence frontmatter is malformed") from error
raw = yaml.safe_load(frontmatter)
try:
data = dict(raw)
except (TypeError, ValueError) as error:
raise ValueError("curated evidence frontmatter must be a mapping") from error
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
kind = data.get("kind")
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
data["payload"] = data.pop(kind, None)
evidence = CuratedEvidence.model_validate(data)
if path is not None:
_validate_kind_directory(path, evidence.kind)
return evidence
def dump_curated_markdown(value: CuratedEvidence) -> str:
"""Render canonical frontmatter with a human-readable kind-specific payload key."""
data = value.model_dump(mode="json", exclude={"payload"})
data[value.kind] = value.payload.model_dump(mode="json")
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
return f"---\n{frontmatter}---\n"
def load_curated_tree(root: Path) -> list[CuratedEvidence]:
"""Load canonical Evidence units in stable path order from a curated root."""
if not root.is_dir():
return []
documents: list[CuratedEvidence] = []
for path in sorted(root.rglob("*.md")):
if path.name.upper().startswith("README"):
continue
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError("curated evidence exceeds the size limit")
try:
text = path.read_text(encoding="utf-8")
except UnicodeDecodeError as error:
raise ValueError("curated evidence must be UTF-8") from error
documents.append(parse_curated_markdown(text, path=path))
return documents
def _validate_kind_directory(path: Path, kind: EvidenceKind) -> None:
parts = path.parts
try:
curated_index = parts.index("curated")
except ValueError:
return
if len(parts) <= curated_index + 1 or parts[curated_index + 1] != kind:
raise ValueError("curated evidence kind must match its directory")
def _validate_identifiers(
values: tuple[str, ...], pattern: re.Pattern[str], field: str,
) -> tuple[str, ...]:
if any(pattern.fullmatch(value) is None for value in values):
raise ValueError(f"{field} must use canonical schema identifiers")
return values
def validate_source_file(value: str) -> str:
"""Validate a repository-relative, credential-free Source Evidence path."""
path = PurePosixPath(value)
if (
path.is_absolute()
or ".." in path.parts
or not path.parts
or path.parts[0] != "source"
or not value.endswith((".md", ".txt", ".sql.md"))
):
raise ValueError("source_file must be a supported path below source/")
return value
def is_evidence_id(value: str) -> bool:
"""Whether a value uses the stable public Evidence identifier format."""
return _EVIDENCE_ID.fullmatch(value) is not None