From ae0976a4aa08e824f64a231fac1178c05a6daa3a Mon Sep 17 00:00:00 2001 From: mptyl Date: Mon, 24 Aug 2026 17:38:22 +0200 Subject: [PATCH] feat(evidence): validate canonical curated corpus --- harness/tests/test_evidence_authoring.py | 281 +++++++++++++++++++ harness/tests/test_evidence_canonical.py | 324 ++++++++++++++++++++++ harness/tht/evidence/__init__.py | 24 ++ harness/tht/evidence/authoring.py | 256 +++++++++++++++++ harness/tht/evidence/canonical.py | 332 +++++++++++++++++++++++ 5 files changed, 1217 insertions(+) create mode 100644 harness/tests/test_evidence_authoring.py create mode 100644 harness/tests/test_evidence_canonical.py create mode 100644 harness/tht/evidence/authoring.py create mode 100644 harness/tht/evidence/canonical.py diff --git a/harness/tests/test_evidence_authoring.py b/harness/tests/test_evidence_authoring.py new file mode 100644 index 00000000..cb6be003 --- /dev/null +++ b/harness/tests/test_evidence_authoring.py @@ -0,0 +1,281 @@ +import hashlib + +import pytest +from pydantic import ValidationError + +from tht.evidence import ( + CuratedEvidence, + EvidenceManifest, + dump_curated_markdown, + dump_manifest, + load_manifest, + validate_workspace_evidence, +) + + +def _evidence(source_text: str, *, review_items=()): + return CuratedEvidence.model_validate({ + "schema_version": 1, + "id": "evidence:fascia-pediatrica", + "title": "Fascia pediatrica", + "kind": "domain", + "purposes": ["disambiguation"], + "applies_to": {"concepts": ["fascia pediatrica"]}, + "language": "it", + "provenance": { + "source_file": "source/domain/patient.md", + "source_sha256": "sha256:" + hashlib.sha256(source_text.encode()).hexdigest(), + "supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."], + }, + "review_items": list(review_items), + "payload": {"rule": "La fascia pediatrica comprende i minori."}, + }) + + +def _write_workspace(root, evidence, source_text): + evidence_root = root / "evidence" + source_path = evidence_root / "source" / "domain" / "patient.md" + curated_path = evidence_root / "curated" / "domain" / "fascia-pediatrica.md" + source_path.parent.mkdir(parents=True) + curated_path.parent.mkdir(parents=True) + source_path.write_text(source_text, encoding="utf-8") + curated_path.write_text(dump_curated_markdown(evidence), encoding="utf-8") + (evidence_root / "manifest.yaml").write_text(dump_manifest(EvidenceManifest.model_validate({ + "schema_version": 1, + "pipeline_version": "evidence-authoring-v1", + "sources": { + evidence.provenance.source_file: { + "sha256": evidence.provenance.source_sha256, + "units": [evidence.id], + }, + }, + "orphans": [], + })), encoding="utf-8") + + +def test_workspace_validation_allows_a_coherent_curated_unit(tmp_path): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + _write_workspace(tmp_path, _evidence(source_text), source_text) + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is True + assert report.findings == () + + +def test_workspace_validation_keeps_review_items_visible_and_blocks_publication(tmp_path): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + _write_workspace(tmp_path, _evidence(source_text, review_items=[{ + "code": "ambiguous_source_statement", + "message": "Il sorgente non chiarisce la data di riferimento.", + "field": "payload.rule", + }]), source_text) + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is False + assert [finding.code for finding in report.findings] == ["unresolved_review_item"] + + +def test_manifest_round_trip_is_versioned_and_deterministic(tmp_path): + manifest = EvidenceManifest.model_validate({ + "schema_version": 1, + "pipeline_version": "evidence-authoring-v1", + "sources": { + "source/domain/patient.md": { + "sha256": "sha256:" + "a" * 64, + "units": ["evidence:fascia-pediatrica"], + }, + }, + "orphans": [], + }) + path = tmp_path / "evidence" / "manifest.yaml" + path.parent.mkdir(parents=True) + path.write_text(dump_manifest(manifest), encoding="utf-8") + + loaded = load_manifest(path) + + assert loaded == manifest + assert path.read_text(encoding="utf-8") == dump_manifest(manifest) + + +def test_manifest_rejects_unknown_fields_and_incompatible_pipeline_version(): + with pytest.raises(ValidationError): + EvidenceManifest.model_validate({ + "schema_version": 1, + "pipeline_version": "evidence-authoring-v2", + "sources": {}, + "orphans": [], + "approved": True, + }) + + +def test_workspace_validation_requires_a_manifest(tmp_path): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + _write_workspace(tmp_path, _evidence(source_text), source_text) + (tmp_path / "evidence" / "manifest.yaml").unlink() + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is False + assert [finding.code for finding in report.findings] == ["manifest_missing"] + + +def test_workspace_validation_reports_a_non_utf8_manifest_without_raising(tmp_path): + evidence_root = tmp_path / "evidence" + evidence_root.mkdir() + (evidence_root / "manifest.yaml").write_bytes(b"\xff") + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is False + assert [finding.code for finding in report.findings] == ["manifest_invalid"] + + +def test_workspace_validation_reports_invalid_curated_documents_without_raising(tmp_path): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + _write_workspace(tmp_path, _evidence(source_text), source_text) + (tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md").write_text( + "not canonical frontmatter", encoding="utf-8", + ) + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is False + assert [finding.code for finding in report.findings] == ["curated_invalid"] + + +def test_workspace_validation_preserves_manifest_orphans_but_blocks_publication(tmp_path): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + _write_workspace(tmp_path, _evidence(source_text), source_text) + manifest_path = tmp_path / "evidence" / "manifest.yaml" + manifest = load_manifest(manifest_path).model_copy(update={"orphans": ("evidence:retired",)}) + manifest_path.write_text(dump_manifest(manifest), encoding="utf-8") + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is False + assert [finding.code for finding in report.findings] == ["orphaned_unit"] + + +def test_workspace_validation_requires_manifest_and_provenance_to_agree(tmp_path): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + evidence = _evidence(source_text) + _write_workspace(tmp_path, evidence, source_text) + manifest_path = tmp_path / "evidence" / "manifest.yaml" + manifest = EvidenceManifest.model_validate({ + "schema_version": 1, + "pipeline_version": "evidence-authoring-v1", + "sources": { + evidence.provenance.source_file: { + "sha256": "sha256:" + "b" * 64, + "units": ["evidence:another-unit"], + }, + }, + "orphans": [], + }) + manifest_path.write_text(dump_manifest(manifest), encoding="utf-8") + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is False + assert {finding.code for finding in report.findings} == { + "manifest_source_hash_mismatch", + "manifest_unit_missing", + "source_hash_mismatch", + "manifest_unit_without_curated", + } + + +def test_workspace_validation_checks_unreferenced_manifest_sources_and_units(tmp_path): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + _write_workspace(tmp_path, _evidence(source_text), source_text) + manifest_path = tmp_path / "evidence" / "manifest.yaml" + manifest = EvidenceManifest.model_validate({ + "schema_version": 1, + "pipeline_version": "evidence-authoring-v1", + "sources": { + "source/domain/missing.md": { + "sha256": "sha256:" + "a" * 64, + "units": ["evidence:removed"], + }, + }, + "orphans": [], + }) + manifest_path.write_text(dump_manifest(manifest), encoding="utf-8") + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is False + assert {finding.code for finding in report.findings} == { + "manifest_source_missing", + "source_missing", + "manifest_unit_without_curated", + } + + +def test_manifest_rejects_unsafe_source_paths_and_legacy_unit_identifiers(): + with pytest.raises(ValidationError): + EvidenceManifest.model_validate({ + "schema_version": 1, + "pipeline_version": "evidence-authoring-v1", + "sources": {"../patient.pdf": {"sha256": "sha256:" + "a" * 64, "units": ["domain:patient"]}}, + "orphans": [], + }) + + +def test_manifest_rejects_a_unit_declared_by_two_sources(): + with pytest.raises(ValidationError, match="only one source"): + EvidenceManifest.model_validate({ + "schema_version": 1, + "pipeline_version": "evidence-authoring-v1", + "sources": { + "source/domain/one.md": { + "sha256": "sha256:" + "a" * 64, + "units": ["evidence:shared"], + }, + "source/domain/two.md": { + "sha256": "sha256:" + "b" * 64, + "units": ["evidence:shared"], + }, + }, + "orphans": [], + }) + + +def test_workspace_validation_rejects_duplicate_curated_ids_and_changed_sources(tmp_path): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + evidence = _evidence(source_text) + _write_workspace(tmp_path, evidence, source_text) + duplicate_path = tmp_path / "evidence" / "curated" / "domain" / "duplicate.md" + duplicate_path.write_text(dump_curated_markdown(evidence), encoding="utf-8") + (tmp_path / "evidence" / "source" / "domain" / "patient.md").write_text( + "Il testo sorgente è cambiato.", encoding="utf-8", + ) + + report = validate_workspace_evidence(tmp_path) + + assert report.publishable is False + assert {finding.code for finding in report.findings} == { + "duplicate_evidence_id", + "source_hash_mismatch", + "supporting_excerpt_missing", + } + + +def test_workspace_validation_reports_unreadable_or_oversized_sources(tmp_path, monkeypatch): + source_text = "I pazienti sotto i 18 anni sono pediatrici." + _write_workspace(tmp_path, _evidence(source_text), source_text) + source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md" + source_path.write_bytes(b"\xff") + + unreadable = validate_workspace_evidence(tmp_path) + + assert [finding.code for finding in unreadable.findings] == ["source_unreadable"] + + source_path.write_text(source_text, encoding="utf-8") + monkeypatch.setattr("tht.evidence.authoring.MAX_AUTHORING_FILE_BYTES", 1) + + oversized = validate_workspace_evidence(tmp_path) + + assert [finding.code for finding in oversized.findings] == ["source_oversized"] diff --git a/harness/tests/test_evidence_canonical.py b/harness/tests/test_evidence_canonical.py new file mode 100644 index 00000000..f7946e99 --- /dev/null +++ b/harness/tests/test_evidence_canonical.py @@ -0,0 +1,324 @@ +import pytest +from pydantic import ValidationError + +from tht.evidence import ( + CuratedEvidence, + dump_curated_markdown, + load_curated_tree, + parse_curated_markdown, +) + +COMMON = { + "schema_version": 1, + "id": "evidence:fascia-pediatrica", + "title": "Fascia pediatrica", + "purposes": ["sql_generation"], + "applies_to": { + "concepts": ["fascia pediatrica"], + "tables": ["clinical.patient"], + "columns": ["clinical.patient.birth_date"], + }, + "language": "it", + "provenance": { + "source_file": "source/domain/patient.md", + "source_sha256": "sha256:" + "a" * 64, + "supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."], + }, + "review_items": [], +} + + +def test_formula_requires_its_typed_payload(): + with pytest.raises(ValidationError): + CuratedEvidence.model_validate({ + **COMMON, + "kind": "formula", + "payload": {"concept": "fascia pediatrica"}, + }) + + +def test_reference_rejects_formula_payload(): + with pytest.raises(ValidationError): + CuratedEvidence.model_validate({ + **COMMON, + "kind": "reference", + "payload": { + "concept": "fascia pediatrica", + "columns": ["clinical.patient.birth_date"], + "sql": "CASE WHEN true THEN 1 END", + }, + }) + + +@pytest.mark.parametrize(("kind", "payload"), [ + ("glossary", { + "definition": "Un paziente con età inferiore a 18 anni.", + "synonyms": ["minore"], + "variants": ["pediatrico"], + }), + ("domain", {"rule": "L'età è calcolata alla data di ricovero."}), + ("enum", { + "column": "clinical.episode.discharge_status", + "values": {"D": "dimesso"}, + }), + ("example", { + "question": "Come riconosco un paziente pediatrico?", + "interpretation": "Applicare la formula della fascia pediatrica.", + }), + ("mapping", { + "concept": "fascia pediatrica", + "tables": ["clinical.patient"], + "columns": ["clinical.patient.birth_date"], + }), + ("normalization", { + "input": "PEDS", + "output": "pediatrico", + "rule": "Converte il codice abbreviato nella forma canonica.", + }), + ("formula", { + "concept": "fascia pediatrica", + "columns": ["clinical.patient.birth_date"], + "sql": "CASE WHEN age < 18 THEN 'pediatrica' ELSE 'adulta' END", + }), + ("reference", { + "url": "https://example.test/linea-guida", + "label": "Linea guida", + "description": "Criteri clinici di riferimento.", + }), +]) +def test_every_kind_requires_a_typed_payload(kind, payload): + evidence = CuratedEvidence.model_validate({ + **COMMON, + "kind": kind, + "payload": payload, + }) + + assert evidence.kind == kind + + +def test_canonical_evidence_rejects_unknown_envelope_and_payload_fields(): + with pytest.raises(ValidationError): + CuratedEvidence.model_validate({ + **COMMON, + "kind": "domain", + "payload": {"rule": "Una regola", "unknown": "no"}, + "unknown": "no", + }) + + +@pytest.mark.parametrize("purpose", ["disambiguation", "rewriting", "schema_linking", "sql_generation"]) +def test_all_public_evidence_purposes_are_accepted(purpose): + evidence = CuratedEvidence.model_validate({ + **COMMON, + "kind": "domain", + "purposes": [purpose], + "payload": {"rule": "Una regola di dominio."}, + }) + + assert evidence.purposes == (purpose,) + + +@pytest.mark.parametrize("field, value", [ + ("id", "formula:fascia-pediatrica"), + ("purposes", ["memory"]), + ("provenance", { + **COMMON["provenance"], + "source_sha256": "sha256:not-a-digest", + }), + ("provenance", { + **COMMON["provenance"], + "supporting_excerpts": [], + }), + ("provenance", { + **COMMON["provenance"], + "supporting_excerpts": ["x" * 1001], + }), +]) +def test_common_evidence_constraints_reject_invalid_values(field, value): + with pytest.raises(ValidationError): + CuratedEvidence.model_validate({ + **COMMON, + field: value, + "kind": "domain", + "payload": {"rule": "Una regola di dominio."}, + }) + + +def test_provenance_is_immutable_after_validation(): + evidence = CuratedEvidence.model_validate({ + **COMMON, + "kind": "domain", + "payload": {"rule": "Una regola di dominio."}, + }) + + with pytest.raises(ValidationError): + evidence.provenance.source_file = "source/other.md" + + +@pytest.mark.parametrize("sql", [ + "SELECT clinical.patient.birth_date FROM clinical.patient", + "WITH patients AS (SELECT 1) SELECT * FROM patients", + "DELETE FROM clinical.patient", + "CREATE TABLE scratch (id integer)", + "TRUNCATE TABLE clinical.patient", + "GRANT SELECT ON clinical.patient TO reader", + "VALUES (1)", + "TABLE clinical.patient", + "SET search_path = clinical", + "CASE WHEN age < 18 THEN 'pediatrica' ELSE 'adulta' END; SELECT 1", +]) +def test_formula_rejects_complete_sql_statements(sql): + with pytest.raises(ValidationError): + CuratedEvidence.model_validate({ + **COMMON, + "kind": "formula", + "payload": { + "concept": "fascia pediatrica", + "columns": ["clinical.patient.birth_date"], + "sql": sql, + }, + }) + + +def test_formula_accepts_one_composable_postgresql_expression(): + evidence = CuratedEvidence.model_validate({ + **COMMON, + "kind": "formula", + "payload": { + "concept": "fascia pediatrica", + "columns": ["clinical.patient.birth_date"], + "sql": "CASE WHEN age < 18 THEN 'pediatrica' ELSE 'adulta' END", + }, + }) + + assert evidence.payload.sql.startswith("CASE WHEN") + + +def _formula_evidence() -> CuratedEvidence: + return CuratedEvidence.model_validate({ + **COMMON, + "kind": "formula", + "payload": { + "concept": "fascia pediatrica", + "columns": ["clinical.patient.birth_date"], + "sql": "CASE WHEN age < 18 THEN 'pediatrica' ELSE 'adulta' END", + }, + }) + + +def test_curated_markdown_round_trip_uses_the_kind_specific_key(tmp_path): + evidence = _formula_evidence() + path = tmp_path / "curated" / "formula" / "fascia-pediatrica.md" + + text = dump_curated_markdown(evidence) + parsed = parse_curated_markdown(text, path=path) + + assert "formula:" in text + assert "payload:" not in text + assert parsed == evidence + + +def test_curated_markdown_rejects_a_kind_that_disagrees_with_its_directory(tmp_path): + text = dump_curated_markdown(_formula_evidence()) + + with pytest.raises(ValueError, match="directory"): + parse_curated_markdown(text, path=tmp_path / "curated" / "reference" / "guide.md") + + +def test_load_curated_tree_returns_documents_in_path_order_and_ignores_readmes(tmp_path): + root = tmp_path / "curated" + formula_path = root / "formula" / "fascia-pediatrica.md" + domain_path = root / "domain" / "eta.md" + formula_path.parent.mkdir(parents=True) + domain_path.parent.mkdir(parents=True) + formula_path.write_text(dump_curated_markdown(_formula_evidence()), encoding="utf-8") + domain_path.write_text(dump_curated_markdown(CuratedEvidence.model_validate({ + **COMMON, + "id": "evidence:eta-ricovero", + "title": "Età al ricovero", + "kind": "domain", + "payload": {"rule": "L'età è calcolata al ricovero."}, + })), encoding="utf-8") + (root / "README.md").write_text("solo navigazione", encoding="utf-8") + + loaded = load_curated_tree(root) + + assert [item.id for item in loaded] == [ + "evidence:eta-ricovero", + "evidence:fascia-pediatrica", + ] + + +@pytest.mark.parametrize(("source_file", "reference_url"), [ + ("source/domain/patient.pdf", "https://example.test/linea-guida"), + ("../patient.md", "https://example.test/linea-guida"), + ("source/domain/patient.md", "https://user:secret@example.test/linea-guida"), + ("source/domain/patient.md", "https://example.test/linea-guida?access_token=secret"), +]) +def test_canonical_evidence_rejects_unsafe_source_or_url(source_file, reference_url): + with pytest.raises(ValidationError): + CuratedEvidence.model_validate({ + **COMMON, + "kind": "reference", + "provenance": {**COMMON["provenance"], "source_file": source_file}, + "payload": { + "url": reference_url, + "label": "Linea guida", + "description": "Criteri clinici.", + }, + }) + + +def test_curated_markdown_rejects_nonempty_body_instead_of_ignoring_it(): + text = dump_curated_markdown(_formula_evidence()) + "password: should-not-be-ignored\n" + + with pytest.raises(ValueError, match="body"): + parse_curated_markdown(text) + + +@pytest.mark.parametrize(("scope", "payload"), [ + ( + {"tables": ["patient"], "columns": ["clinical.patient.birth_date"]}, + {"rule": "Una regola."}, + ), + ( + {"tables": ["clinical.patient"], "columns": ["clinical.patient.date.of.birth"]}, + {"rule": "Una regola."}, + ), + ( + {"tables": ["clinical.patient"], "columns": ["clinical.patient.birth_date"]}, + { + "concept": "fascia pediatrica", + "columns": ["clinical.patient"], + "sql": "CASE WHEN age < 18 THEN 1 ELSE 0 END", + }, + ), +]) +def test_schema_identifiers_require_table_or_column_shape(scope, payload): + with pytest.raises(ValidationError): + CuratedEvidence.model_validate({ + **COMMON, + "kind": "formula" if "sql" in payload else "domain", + "applies_to": scope, + "payload": payload, + }) + + +def test_load_curated_tree_rejects_non_utf8_and_oversized_files(tmp_path, monkeypatch): + root = tmp_path / "curated" + path = root / "domain" / "eta.md" + path.parent.mkdir(parents=True) + path.write_bytes(b"\xff") + + with pytest.raises(ValueError, match="UTF-8"): + load_curated_tree(root) + + path.write_text(dump_curated_markdown(CuratedEvidence.model_validate({ + **COMMON, + "kind": "domain", + "payload": {"rule": "Una regola."}, + })), encoding="utf-8") + monkeypatch.setattr("tht.evidence.canonical.MAX_CURATED_FILE_BYTES", 1) + + with pytest.raises(ValueError, match="size limit"): + load_curated_tree(root) diff --git a/harness/tht/evidence/__init__.py b/harness/tht/evidence/__init__.py index 3648e8af..287d5263 100644 --- a/harness/tht/evidence/__init__.py +++ b/harness/tht/evidence/__init__.py @@ -1,6 +1,20 @@ """Cohesive public entrypoint for Evidence domain capabilities.""" from tht.evidence.acquisition import acquire, discover +from tht.evidence.authoring import ( + EvidenceManifest, + ValidationFinding, + ValidationReport, + dump_manifest, + load_manifest, + validate_workspace_evidence, +) +from tht.evidence.canonical import ( + CuratedEvidence, + dump_curated_markdown, + load_curated_tree, + parse_curated_markdown, +) from tht.evidence.contracts import ( AcquiredDocument, EvidenceSource, @@ -28,11 +42,15 @@ __all__ = [ "AcquiredDocument", "ActiveEvidenceSearcher", "CorpusWorkspaceMismatchError", + "CuratedEvidence", "EvidenceEmbedder", + "EvidenceManifest", "EvidenceSource", "EvidenceSourceError", "EvidenceSourceErrorCategory", "SourceObject", + "ValidationFinding", + "ValidationReport", "acquire", "active_searcher", "build_preprocessing_pipeline", @@ -40,10 +58,16 @@ __all__ = [ "build_sources", "canonical_provenance_uri", "discover", + "dump_curated_markdown", + "dump_manifest", + "load_curated_tree", + "load_manifest", "normalize_aware_datetime", + "parse_curated_markdown", "project_session", "resolve_citation", "validate_corpus_workspace", "validate_namespaced_value", "validate_safe_metadata", + "validate_workspace_evidence", ] diff --git a/harness/tht/evidence/authoring.py b/harness/tht/evidence/authoring.py new file mode 100644 index 00000000..d3e4a39d --- /dev/null +++ b/harness/tht/evidence/authoring.py @@ -0,0 +1,256 @@ +"""Validation for the Git-reviewed Evidence authoring workspace.""" + +from __future__ import annotations + +import hashlib +import unicodedata +from dataclasses import dataclass +from pathlib import Path +from typing import Literal + +import yaml +from pydantic import ValidationError, field_validator, model_validator + +from tht.evidence.canonical import ( + CuratedEvidence, + StrictModel, + is_evidence_id, + load_curated_tree, + validate_source_file, +) + +MAX_AUTHORING_FILE_BYTES = 10 * 1024 * 1024 + + +@dataclass(frozen=True) +class ValidationFinding: + severity: Literal["error", "warning"] + code: str + path: str + message: str + + +@dataclass(frozen=True) +class ValidationReport: + findings: tuple[ValidationFinding, ...] + + @property + def publishable(self) -> bool: + return not any(finding.severity == "error" for finding in self.findings) + + +class ManifestSource(StrictModel): + sha256: str + units: tuple[str, ...] + + @field_validator("sha256") + @classmethod + def _validate_sha256(cls, value: str) -> str: + if not value.startswith("sha256:") or len(value) != 71: + raise ValueError("sha256 must be a sha256 digest") + try: + int(value.removeprefix("sha256:"), 16) + except ValueError as error: + raise ValueError("sha256 must be a sha256 digest") from error + return value + + @field_validator("units") + @classmethod + def _validate_units(cls, value: tuple[str, ...]) -> tuple[str, ...]: + if any(not is_evidence_id(unit) for unit in value): + raise ValueError("units must use stable evidence identifiers") + return value + + +class EvidenceManifest(StrictModel): + schema_version: Literal[1] + pipeline_version: Literal["evidence-authoring-v1"] + sources: dict[str, ManifestSource] + orphans: tuple[str, ...] + + @field_validator("sources") + @classmethod + def _validate_sources(cls, value: dict[str, ManifestSource]) -> dict[str, ManifestSource]: + for source_path in value: + validate_source_file(source_path) + return value + + @field_validator("orphans") + @classmethod + def _validate_orphans(cls, value: tuple[str, ...]) -> tuple[str, ...]: + if any(not is_evidence_id(unit) for unit in value): + raise ValueError("orphans must use stable evidence identifiers") + return value + + @model_validator(mode="after") + def _validate_unit_membership(self) -> EvidenceManifest: + seen: set[str] = set() + for source in self.sources.values(): + duplicate = seen.intersection(source.units) + if duplicate: + raise ValueError("a unit may belong to only one source") + seen.update(source.units) + if len(set(self.orphans)) != len(self.orphans): + raise ValueError("orphans must be unique") + return self + + +def load_manifest(path: Path) -> EvidenceManifest: + """Load the managed, versioned authoring manifest.""" + try: + raw = yaml.safe_load(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, yaml.YAMLError) as error: + raise ValueError("manifest cannot be read") from error + try: + return EvidenceManifest.model_validate(raw) + except ValidationError as error: + raise ValueError("manifest is invalid") from error + + +def dump_manifest(manifest: EvidenceManifest) -> str: + """Serialize the manifest deterministically for Git review.""" + return yaml.safe_dump(manifest.model_dump(mode="json"), allow_unicode=True, sort_keys=True) + + +def validate_workspace_evidence(workspace_root: Path) -> ValidationReport: + """Validate the curated corpus without writing the workspace.""" + evidence_root = workspace_root / "evidence" + findings: list[ValidationFinding] = [] + manifest_path = evidence_root / "manifest.yaml" + if not manifest_path.is_file(): + return ValidationReport((ValidationFinding( + severity="error", + code="manifest_missing", + path="manifest.yaml", + message="The managed Evidence manifest is missing.", + ),)) + try: + manifest = load_manifest(manifest_path) + except ValueError: + return ValidationReport((ValidationFinding( + severity="error", + code="manifest_invalid", + path="manifest.yaml", + message="The managed Evidence manifest is invalid.", + ),)) + for orphan in manifest.orphans: + findings.append(ValidationFinding( + severity="error", + code="orphaned_unit", + path="manifest.yaml", + message=f"Orphaned Evidence {orphan} must be resolved before publication.", + )) + try: + documents = load_curated_tree(evidence_root / "curated") + except (OSError, ValidationError, ValueError): + return ValidationReport(tuple(findings + [ValidationFinding( + severity="error", + code="curated_invalid", + path="curated", + message="A Curated Evidence document is invalid or cannot be read.", + )])) + source_texts = { + source_path: _validate_manifest_source(evidence_root, source_path, source, findings) + for source_path, source in manifest.sources.items() + } + seen_ids: set[str] = set() + for evidence in documents: + if evidence.id in seen_ids: + findings.append(ValidationFinding( + severity="error", + code="duplicate_evidence_id", + path="curated", + message=f"Evidence id {evidence.id} appears more than once.", + )) + seen_ids.add(evidence.id) + findings.extend(_validate_unit(manifest, evidence, source_texts.get(evidence.provenance.source_file))) + for source in manifest.sources.values(): + for unit in source.units: + if unit not in seen_ids: + findings.append(ValidationFinding( + severity="error", + code="manifest_unit_without_curated", + path="manifest.yaml", + message=f"Manifest Evidence {unit} has no Curated document.", + )) + return ValidationReport(tuple(findings)) + + +def _validate_manifest_source( + evidence_root: Path, + source_path: str, + manifest_source: ManifestSource, + findings: list[ValidationFinding], +) -> str | None: + path = evidence_root / source_path + if not path.is_file(): + findings.append(ValidationFinding("error", "source_missing", source_path, + "The manifest source does not exist.")) + return None + if path.stat().st_size > MAX_AUTHORING_FILE_BYTES: + findings.append(ValidationFinding("error", "source_oversized", source_path, + "The manifest source exceeds the authoring size limit.")) + return None + try: + source = _normalize(path.read_text(encoding="utf-8")) + except UnicodeDecodeError: + findings.append(ValidationFinding("error", "source_unreadable", source_path, + "The manifest source is not valid UTF-8.")) + return None + digest = "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest() + if digest != manifest_source.sha256: + findings.append(ValidationFinding("error", "source_hash_mismatch", source_path, + "The manifest hash does not match the normalized source.")) + return source + + +def _validate_unit( + manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None, +) -> list[ValidationFinding]: + path = evidence.provenance.source_file + findings: list[ValidationFinding] = [] + manifest_source = manifest.sources.get(path) + if manifest_source is None: + findings.append(ValidationFinding( + severity="error", + code="manifest_source_missing", + path="manifest.yaml", + message="The manifest does not contain the provenance source.", + )) + else: + if manifest_source.sha256 != evidence.provenance.source_sha256: + findings.append(ValidationFinding( + severity="error", + code="manifest_source_hash_mismatch", + path="manifest.yaml", + message="The manifest hash does not match the Evidence provenance.", + )) + if evidence.id not in manifest_source.units: + findings.append(ValidationFinding( + severity="error", + code="manifest_unit_missing", + path="manifest.yaml", + message="The manifest does not link the Evidence unit to its source.", + )) + if source is None: + return findings + for excerpt in evidence.provenance.supporting_excerpts: + if _normalize(excerpt) not in source: + findings.append(ValidationFinding( + severity="error", + code="supporting_excerpt_missing", + path=path, + message="A supporting excerpt is absent from the normalized source.", + )) + for item in evidence.review_items: + findings.append(ValidationFinding( + severity="error", + code="unresolved_review_item", + path=path, + message=f"Review item {item.code} must be resolved before publication.", + )) + return findings + + +def _normalize(text: str) -> str: + return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n")) diff --git a/harness/tht/evidence/canonical.py b/harness/tht/evidence/canonical.py new file mode 100644 index 00000000..289f1e36 --- /dev/null +++ b/harness/tht/evidence/canonical.py @@ -0,0 +1,332 @@ +"""Typed, reviewable Evidence units stored in the workspace repository.""" + +from __future__ import annotations + +import re +from pathlib import Path, PurePosixPath +from typing import Literal + +import sqlglot +import yaml +from pydantic import AnyHttpUrl, BaseModel, ConfigDict, Field, field_validator, model_validator +from sqlglot import exp + +from tht.evidence.contracts import validate_canonical_uri + + +class StrictModel(BaseModel): + """Reject undeclared fields in the repository's canonical format.""" + + model_config = ConfigDict(extra="forbid") + + +EvidenceKind = Literal[ + "glossary", + "domain", + "enum", + "example", + "mapping", + "normalization", + "formula", + "reference", +] +EvidencePurpose = Literal[ + "disambiguation", + "rewriting", + "schema_linking", + "sql_generation", +] +_IDENTIFIER = r"[A-Za-z_][A-Za-z0-9_$]*" +_TABLE_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}$") +_COLUMN_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}\.{_IDENTIFIER}$") +MAX_CURATED_FILE_BYTES = 10 * 1024 * 1024 + + +class EvidenceScope(StrictModel): + concepts: tuple[str, ...] = () + tables: tuple[str, ...] = () + columns: tuple[str, ...] = () + + @field_validator("tables") + @classmethod + def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]: + return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables") + + @field_validator("columns") + @classmethod + def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]: + return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns") + + +class EvidenceProvenance(StrictModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + source_file: str + source_sha256: str + supporting_excerpts: tuple[str, ...] + + @field_validator("source_file") + @classmethod + def _validate_source_file(cls, value: str) -> str: + return validate_source_file(value) + + @field_validator("source_sha256") + @classmethod + def _validate_sha256(cls, value: str) -> str: + if not re.fullmatch(r"sha256:[0-9a-f]{64}", value): + raise ValueError("source_sha256 must be a sha256 digest") + return value + + @field_validator("supporting_excerpts") + @classmethod + def _validate_excerpts(cls, value: tuple[str, ...]) -> tuple[str, ...]: + if not 1 <= len(value) <= 5: + raise ValueError("supporting_excerpts must contain one to five items") + if any(not excerpt.strip() or len(excerpt) > 1000 for excerpt in value): + raise ValueError("supporting excerpts must be nonempty and at most 1000 characters") + return value + + +class ReviewItem(StrictModel): + code: str + message: str + field: str | None = None + + +class FormulaPayload(StrictModel): + concept: str + columns: tuple[str, ...] + sql: str + + @field_validator("columns") + @classmethod + def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]: + return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns") + + @field_validator("sql") + @classmethod + def _validate_expression(cls, value: str) -> str: + try: + statements = [statement for statement in sqlglot.parse(value, read="postgres") if statement] + except sqlglot.errors.ParseError as error: + raise ValueError("formula.sql must be valid PostgreSQL") from error + if len(statements) != 1: + raise ValueError("formula.sql must contain exactly one expression") + expression = statements[0] + if expression.find(exp.Select) is not None or expression.find(exp.With) is not None: + raise ValueError("formula.sql must not contain a query") + if any( + expression.find(statement_type) is not None + for statement_type in ( + exp.Insert, + exp.Update, + exp.Delete, + exp.Create, + exp.Drop, + exp.Alter, + exp.Merge, + exp.TruncateTable, + exp.Grant, + exp.Revoke, + exp.Command, + exp.Values, + exp.Set, + exp.Table, + ) + ): + raise ValueError("formula.sql must not contain DDL or DML") + return value + + +class ReferencePayload(StrictModel): + url: AnyHttpUrl + label: str + description: str + + @field_validator("url") + @classmethod + def _reject_credentials(cls, value: AnyHttpUrl) -> AnyHttpUrl: + validate_canonical_uri(str(value)) + return value + + +class GlossaryPayload(StrictModel): + definition: str + synonyms: tuple[str, ...] = () + variants: tuple[str, ...] = () + + +class DomainPayload(StrictModel): + rule: str + + +class EnumPayload(StrictModel): + column: str + values: dict[str, str] + + @field_validator("column") + @classmethod + def _validate_column(cls, value: str) -> str: + _validate_identifiers((value,), _COLUMN_IDENTIFIER, "column") + return value + + +class ExamplePayload(StrictModel): + question: str + interpretation: str + + +class MappingPayload(StrictModel): + concept: str + tables: tuple[str, ...] + columns: tuple[str, ...] + + @field_validator("tables") + @classmethod + def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]: + return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables") + + @field_validator("columns") + @classmethod + def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]: + return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns") + + +class NormalizationPayload(StrictModel): + input: str + output: str + rule: str + + +EvidencePayload = ( + GlossaryPayload + | DomainPayload + | EnumPayload + | ExamplePayload + | MappingPayload + | NormalizationPayload + | FormulaPayload + | ReferencePayload +) + + +_PAYLOAD_TYPE_BY_KIND = { + "glossary": GlossaryPayload, + "domain": DomainPayload, + "enum": EnumPayload, + "example": ExamplePayload, + "mapping": MappingPayload, + "normalization": NormalizationPayload, + "formula": FormulaPayload, + "reference": ReferencePayload, +} +_EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$") + + +class CuratedEvidence(StrictModel): + schema_version: Literal[1] + id: str + title: str + kind: EvidenceKind + purposes: tuple[EvidencePurpose, ...] + applies_to: EvidenceScope = Field(default_factory=EvidenceScope) + language: str + provenance: EvidenceProvenance + review_items: tuple[ReviewItem, ...] = () + payload: EvidencePayload + + @model_validator(mode="after") + def _validate_kind_payload(self) -> CuratedEvidence: + if not is_evidence_id(self.id): + raise ValueError("id must use the evidence: form") + expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind) + if expected is not None and not isinstance(self.payload, expected): + raise ValueError(f"{self.kind} requires its typed payload") + return self + + +def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence: + """Parse the canonical frontmatter representation of one Curated Evidence unit.""" + if not text.startswith("---\n"): + raise ValueError("curated evidence requires YAML frontmatter") + try: + _, frontmatter, body = text.split("---\n", 2) + except ValueError as error: + raise ValueError("curated evidence frontmatter is malformed") from error + raw = yaml.safe_load(frontmatter) + try: + data = dict(raw) + except (TypeError, ValueError) as error: + raise ValueError("curated evidence frontmatter must be a mapping") from error + if body.strip(): + raise ValueError("curated evidence must not contain an ignored body") + kind = data.get("kind") + if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND: + data["payload"] = data.pop(kind, None) + evidence = CuratedEvidence.model_validate(data) + if path is not None: + _validate_kind_directory(path, evidence.kind) + return evidence + + +def dump_curated_markdown(value: CuratedEvidence) -> str: + """Render canonical frontmatter with a human-readable kind-specific payload key.""" + data = value.model_dump(mode="json", exclude={"payload"}) + data[value.kind] = value.payload.model_dump(mode="json") + frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False) + return f"---\n{frontmatter}---\n" + + +def load_curated_tree(root: Path) -> list[CuratedEvidence]: + """Load canonical Evidence units in stable path order from a curated root.""" + if not root.is_dir(): + return [] + documents: list[CuratedEvidence] = [] + for path in sorted(root.rglob("*.md")): + if path.name.upper().startswith("README"): + continue + if path.stat().st_size > MAX_CURATED_FILE_BYTES: + raise ValueError("curated evidence exceeds the size limit") + try: + text = path.read_text(encoding="utf-8") + except UnicodeDecodeError as error: + raise ValueError("curated evidence must be UTF-8") from error + documents.append(parse_curated_markdown(text, path=path)) + return documents + + +def _validate_kind_directory(path: Path, kind: EvidenceKind) -> None: + parts = path.parts + try: + curated_index = parts.index("curated") + except ValueError: + return + if len(parts) <= curated_index + 1 or parts[curated_index + 1] != kind: + raise ValueError("curated evidence kind must match its directory") + + +def _validate_identifiers( + values: tuple[str, ...], pattern: re.Pattern[str], field: str, +) -> tuple[str, ...]: + if any(pattern.fullmatch(value) is None for value in values): + raise ValueError(f"{field} must use canonical schema identifiers") + return values + + +def validate_source_file(value: str) -> str: + """Validate a repository-relative, credential-free Source Evidence path.""" + path = PurePosixPath(value) + if ( + path.is_absolute() + or ".." in path.parts + or not path.parts + or path.parts[0] != "source" + or not value.endswith((".md", ".txt", ".sql.md")) + ): + raise ValueError("source_file must be a supported path below source/") + return value + + +def is_evidence_id(value: str) -> bool: + """Whether a value uses the stable public Evidence identifier format.""" + return _EVIDENCE_ID.fullmatch(value) is not None