feat(evidence): validate canonical curated corpus
This commit is contained in:
@@ -0,0 +1,281 @@
|
|||||||
|
import hashlib
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from pydantic import ValidationError
|
||||||
|
|
||||||
|
from tht.evidence import (
|
||||||
|
CuratedEvidence,
|
||||||
|
EvidenceManifest,
|
||||||
|
dump_curated_markdown,
|
||||||
|
dump_manifest,
|
||||||
|
load_manifest,
|
||||||
|
validate_workspace_evidence,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _evidence(source_text: str, *, review_items=()):
|
||||||
|
return CuratedEvidence.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"id": "evidence:fascia-pediatrica",
|
||||||
|
"title": "Fascia pediatrica",
|
||||||
|
"kind": "domain",
|
||||||
|
"purposes": ["disambiguation"],
|
||||||
|
"applies_to": {"concepts": ["fascia pediatrica"]},
|
||||||
|
"language": "it",
|
||||||
|
"provenance": {
|
||||||
|
"source_file": "source/domain/patient.md",
|
||||||
|
"source_sha256": "sha256:" + hashlib.sha256(source_text.encode()).hexdigest(),
|
||||||
|
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
|
||||||
|
},
|
||||||
|
"review_items": list(review_items),
|
||||||
|
"payload": {"rule": "La fascia pediatrica comprende i minori."},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def _write_workspace(root, evidence, source_text):
|
||||||
|
evidence_root = root / "evidence"
|
||||||
|
source_path = evidence_root / "source" / "domain" / "patient.md"
|
||||||
|
curated_path = evidence_root / "curated" / "domain" / "fascia-pediatrica.md"
|
||||||
|
source_path.parent.mkdir(parents=True)
|
||||||
|
curated_path.parent.mkdir(parents=True)
|
||||||
|
source_path.write_text(source_text, encoding="utf-8")
|
||||||
|
curated_path.write_text(dump_curated_markdown(evidence), encoding="utf-8")
|
||||||
|
(evidence_root / "manifest.yaml").write_text(dump_manifest(EvidenceManifest.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"pipeline_version": "evidence-authoring-v1",
|
||||||
|
"sources": {
|
||||||
|
evidence.provenance.source_file: {
|
||||||
|
"sha256": evidence.provenance.source_sha256,
|
||||||
|
"units": [evidence.id],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"orphans": [],
|
||||||
|
})), encoding="utf-8")
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_allows_a_coherent_curated_unit(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is True
|
||||||
|
assert report.findings == ()
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_keeps_review_items_visible_and_blocks_publication(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text, review_items=[{
|
||||||
|
"code": "ambiguous_source_statement",
|
||||||
|
"message": "Il sorgente non chiarisce la data di riferimento.",
|
||||||
|
"field": "payload.rule",
|
||||||
|
}]), source_text)
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is False
|
||||||
|
assert [finding.code for finding in report.findings] == ["unresolved_review_item"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_manifest_round_trip_is_versioned_and_deterministic(tmp_path):
|
||||||
|
manifest = EvidenceManifest.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"pipeline_version": "evidence-authoring-v1",
|
||||||
|
"sources": {
|
||||||
|
"source/domain/patient.md": {
|
||||||
|
"sha256": "sha256:" + "a" * 64,
|
||||||
|
"units": ["evidence:fascia-pediatrica"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"orphans": [],
|
||||||
|
})
|
||||||
|
path = tmp_path / "evidence" / "manifest.yaml"
|
||||||
|
path.parent.mkdir(parents=True)
|
||||||
|
path.write_text(dump_manifest(manifest), encoding="utf-8")
|
||||||
|
|
||||||
|
loaded = load_manifest(path)
|
||||||
|
|
||||||
|
assert loaded == manifest
|
||||||
|
assert path.read_text(encoding="utf-8") == dump_manifest(manifest)
|
||||||
|
|
||||||
|
|
||||||
|
def test_manifest_rejects_unknown_fields_and_incompatible_pipeline_version():
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
EvidenceManifest.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"pipeline_version": "evidence-authoring-v2",
|
||||||
|
"sources": {},
|
||||||
|
"orphans": [],
|
||||||
|
"approved": True,
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_requires_a_manifest(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
(tmp_path / "evidence" / "manifest.yaml").unlink()
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is False
|
||||||
|
assert [finding.code for finding in report.findings] == ["manifest_missing"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_reports_a_non_utf8_manifest_without_raising(tmp_path):
|
||||||
|
evidence_root = tmp_path / "evidence"
|
||||||
|
evidence_root.mkdir()
|
||||||
|
(evidence_root / "manifest.yaml").write_bytes(b"\xff")
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is False
|
||||||
|
assert [finding.code for finding in report.findings] == ["manifest_invalid"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_reports_invalid_curated_documents_without_raising(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
(tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md").write_text(
|
||||||
|
"not canonical frontmatter", encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is False
|
||||||
|
assert [finding.code for finding in report.findings] == ["curated_invalid"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_preserves_manifest_orphans_but_blocks_publication(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
manifest_path = tmp_path / "evidence" / "manifest.yaml"
|
||||||
|
manifest = load_manifest(manifest_path).model_copy(update={"orphans": ("evidence:retired",)})
|
||||||
|
manifest_path.write_text(dump_manifest(manifest), encoding="utf-8")
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is False
|
||||||
|
assert [finding.code for finding in report.findings] == ["orphaned_unit"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_requires_manifest_and_provenance_to_agree(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
evidence = _evidence(source_text)
|
||||||
|
_write_workspace(tmp_path, evidence, source_text)
|
||||||
|
manifest_path = tmp_path / "evidence" / "manifest.yaml"
|
||||||
|
manifest = EvidenceManifest.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"pipeline_version": "evidence-authoring-v1",
|
||||||
|
"sources": {
|
||||||
|
evidence.provenance.source_file: {
|
||||||
|
"sha256": "sha256:" + "b" * 64,
|
||||||
|
"units": ["evidence:another-unit"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"orphans": [],
|
||||||
|
})
|
||||||
|
manifest_path.write_text(dump_manifest(manifest), encoding="utf-8")
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is False
|
||||||
|
assert {finding.code for finding in report.findings} == {
|
||||||
|
"manifest_source_hash_mismatch",
|
||||||
|
"manifest_unit_missing",
|
||||||
|
"source_hash_mismatch",
|
||||||
|
"manifest_unit_without_curated",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_checks_unreferenced_manifest_sources_and_units(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
manifest_path = tmp_path / "evidence" / "manifest.yaml"
|
||||||
|
manifest = EvidenceManifest.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"pipeline_version": "evidence-authoring-v1",
|
||||||
|
"sources": {
|
||||||
|
"source/domain/missing.md": {
|
||||||
|
"sha256": "sha256:" + "a" * 64,
|
||||||
|
"units": ["evidence:removed"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"orphans": [],
|
||||||
|
})
|
||||||
|
manifest_path.write_text(dump_manifest(manifest), encoding="utf-8")
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is False
|
||||||
|
assert {finding.code for finding in report.findings} == {
|
||||||
|
"manifest_source_missing",
|
||||||
|
"source_missing",
|
||||||
|
"manifest_unit_without_curated",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_manifest_rejects_unsafe_source_paths_and_legacy_unit_identifiers():
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
EvidenceManifest.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"pipeline_version": "evidence-authoring-v1",
|
||||||
|
"sources": {"../patient.pdf": {"sha256": "sha256:" + "a" * 64, "units": ["domain:patient"]}},
|
||||||
|
"orphans": [],
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_manifest_rejects_a_unit_declared_by_two_sources():
|
||||||
|
with pytest.raises(ValidationError, match="only one source"):
|
||||||
|
EvidenceManifest.model_validate({
|
||||||
|
"schema_version": 1,
|
||||||
|
"pipeline_version": "evidence-authoring-v1",
|
||||||
|
"sources": {
|
||||||
|
"source/domain/one.md": {
|
||||||
|
"sha256": "sha256:" + "a" * 64,
|
||||||
|
"units": ["evidence:shared"],
|
||||||
|
},
|
||||||
|
"source/domain/two.md": {
|
||||||
|
"sha256": "sha256:" + "b" * 64,
|
||||||
|
"units": ["evidence:shared"],
|
||||||
|
},
|
||||||
|
},
|
||||||
|
"orphans": [],
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_rejects_duplicate_curated_ids_and_changed_sources(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
evidence = _evidence(source_text)
|
||||||
|
_write_workspace(tmp_path, evidence, source_text)
|
||||||
|
duplicate_path = tmp_path / "evidence" / "curated" / "domain" / "duplicate.md"
|
||||||
|
duplicate_path.write_text(dump_curated_markdown(evidence), encoding="utf-8")
|
||||||
|
(tmp_path / "evidence" / "source" / "domain" / "patient.md").write_text(
|
||||||
|
"Il testo sorgente è cambiato.", encoding="utf-8",
|
||||||
|
)
|
||||||
|
|
||||||
|
report = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert report.publishable is False
|
||||||
|
assert {finding.code for finding in report.findings} == {
|
||||||
|
"duplicate_evidence_id",
|
||||||
|
"source_hash_mismatch",
|
||||||
|
"supporting_excerpt_missing",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_workspace_validation_reports_unreadable_or_oversized_sources(tmp_path, monkeypatch):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
source_path = tmp_path / "evidence" / "source" / "domain" / "patient.md"
|
||||||
|
source_path.write_bytes(b"\xff")
|
||||||
|
|
||||||
|
unreadable = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert [finding.code for finding in unreadable.findings] == ["source_unreadable"]
|
||||||
|
|
||||||
|
source_path.write_text(source_text, encoding="utf-8")
|
||||||
|
monkeypatch.setattr("tht.evidence.authoring.MAX_AUTHORING_FILE_BYTES", 1)
|
||||||
|
|
||||||
|
oversized = validate_workspace_evidence(tmp_path)
|
||||||
|
|
||||||
|
assert [finding.code for finding in oversized.findings] == ["source_oversized"]
|
||||||
@@ -0,0 +1,324 @@
|
|||||||
|
import pytest
|
||||||
|
from pydantic import ValidationError
|
||||||
|
|
||||||
|
from tht.evidence import (
|
||||||
|
CuratedEvidence,
|
||||||
|
dump_curated_markdown,
|
||||||
|
load_curated_tree,
|
||||||
|
parse_curated_markdown,
|
||||||
|
)
|
||||||
|
|
||||||
|
COMMON = {
|
||||||
|
"schema_version": 1,
|
||||||
|
"id": "evidence:fascia-pediatrica",
|
||||||
|
"title": "Fascia pediatrica",
|
||||||
|
"purposes": ["sql_generation"],
|
||||||
|
"applies_to": {
|
||||||
|
"concepts": ["fascia pediatrica"],
|
||||||
|
"tables": ["clinical.patient"],
|
||||||
|
"columns": ["clinical.patient.birth_date"],
|
||||||
|
},
|
||||||
|
"language": "it",
|
||||||
|
"provenance": {
|
||||||
|
"source_file": "source/domain/patient.md",
|
||||||
|
"source_sha256": "sha256:" + "a" * 64,
|
||||||
|
"supporting_excerpts": ["I pazienti sotto i 18 anni sono pediatrici."],
|
||||||
|
},
|
||||||
|
"review_items": [],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_formula_requires_its_typed_payload():
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "formula",
|
||||||
|
"payload": {"concept": "fascia pediatrica"},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_reference_rejects_formula_payload():
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "reference",
|
||||||
|
"payload": {
|
||||||
|
"concept": "fascia pediatrica",
|
||||||
|
"columns": ["clinical.patient.birth_date"],
|
||||||
|
"sql": "CASE WHEN true THEN 1 END",
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(("kind", "payload"), [
|
||||||
|
("glossary", {
|
||||||
|
"definition": "Un paziente con età inferiore a 18 anni.",
|
||||||
|
"synonyms": ["minore"],
|
||||||
|
"variants": ["pediatrico"],
|
||||||
|
}),
|
||||||
|
("domain", {"rule": "L'età è calcolata alla data di ricovero."}),
|
||||||
|
("enum", {
|
||||||
|
"column": "clinical.episode.discharge_status",
|
||||||
|
"values": {"D": "dimesso"},
|
||||||
|
}),
|
||||||
|
("example", {
|
||||||
|
"question": "Come riconosco un paziente pediatrico?",
|
||||||
|
"interpretation": "Applicare la formula della fascia pediatrica.",
|
||||||
|
}),
|
||||||
|
("mapping", {
|
||||||
|
"concept": "fascia pediatrica",
|
||||||
|
"tables": ["clinical.patient"],
|
||||||
|
"columns": ["clinical.patient.birth_date"],
|
||||||
|
}),
|
||||||
|
("normalization", {
|
||||||
|
"input": "PEDS",
|
||||||
|
"output": "pediatrico",
|
||||||
|
"rule": "Converte il codice abbreviato nella forma canonica.",
|
||||||
|
}),
|
||||||
|
("formula", {
|
||||||
|
"concept": "fascia pediatrica",
|
||||||
|
"columns": ["clinical.patient.birth_date"],
|
||||||
|
"sql": "CASE WHEN age < 18 THEN 'pediatrica' ELSE 'adulta' END",
|
||||||
|
}),
|
||||||
|
("reference", {
|
||||||
|
"url": "https://example.test/linea-guida",
|
||||||
|
"label": "Linea guida",
|
||||||
|
"description": "Criteri clinici di riferimento.",
|
||||||
|
}),
|
||||||
|
])
|
||||||
|
def test_every_kind_requires_a_typed_payload(kind, payload):
|
||||||
|
evidence = CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": kind,
|
||||||
|
"payload": payload,
|
||||||
|
})
|
||||||
|
|
||||||
|
assert evidence.kind == kind
|
||||||
|
|
||||||
|
|
||||||
|
def test_canonical_evidence_rejects_unknown_envelope_and_payload_fields():
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "domain",
|
||||||
|
"payload": {"rule": "Una regola", "unknown": "no"},
|
||||||
|
"unknown": "no",
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("purpose", ["disambiguation", "rewriting", "schema_linking", "sql_generation"])
|
||||||
|
def test_all_public_evidence_purposes_are_accepted(purpose):
|
||||||
|
evidence = CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "domain",
|
||||||
|
"purposes": [purpose],
|
||||||
|
"payload": {"rule": "Una regola di dominio."},
|
||||||
|
})
|
||||||
|
|
||||||
|
assert evidence.purposes == (purpose,)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("field, value", [
|
||||||
|
("id", "formula:fascia-pediatrica"),
|
||||||
|
("purposes", ["memory"]),
|
||||||
|
("provenance", {
|
||||||
|
**COMMON["provenance"],
|
||||||
|
"source_sha256": "sha256:not-a-digest",
|
||||||
|
}),
|
||||||
|
("provenance", {
|
||||||
|
**COMMON["provenance"],
|
||||||
|
"supporting_excerpts": [],
|
||||||
|
}),
|
||||||
|
("provenance", {
|
||||||
|
**COMMON["provenance"],
|
||||||
|
"supporting_excerpts": ["x" * 1001],
|
||||||
|
}),
|
||||||
|
])
|
||||||
|
def test_common_evidence_constraints_reject_invalid_values(field, value):
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
field: value,
|
||||||
|
"kind": "domain",
|
||||||
|
"payload": {"rule": "Una regola di dominio."},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_provenance_is_immutable_after_validation():
|
||||||
|
evidence = CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "domain",
|
||||||
|
"payload": {"rule": "Una regola di dominio."},
|
||||||
|
})
|
||||||
|
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
evidence.provenance.source_file = "source/other.md"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("sql", [
|
||||||
|
"SELECT clinical.patient.birth_date FROM clinical.patient",
|
||||||
|
"WITH patients AS (SELECT 1) SELECT * FROM patients",
|
||||||
|
"DELETE FROM clinical.patient",
|
||||||
|
"CREATE TABLE scratch (id integer)",
|
||||||
|
"TRUNCATE TABLE clinical.patient",
|
||||||
|
"GRANT SELECT ON clinical.patient TO reader",
|
||||||
|
"VALUES (1)",
|
||||||
|
"TABLE clinical.patient",
|
||||||
|
"SET search_path = clinical",
|
||||||
|
"CASE WHEN age < 18 THEN 'pediatrica' ELSE 'adulta' END; SELECT 1",
|
||||||
|
])
|
||||||
|
def test_formula_rejects_complete_sql_statements(sql):
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "formula",
|
||||||
|
"payload": {
|
||||||
|
"concept": "fascia pediatrica",
|
||||||
|
"columns": ["clinical.patient.birth_date"],
|
||||||
|
"sql": sql,
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_formula_accepts_one_composable_postgresql_expression():
|
||||||
|
evidence = CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "formula",
|
||||||
|
"payload": {
|
||||||
|
"concept": "fascia pediatrica",
|
||||||
|
"columns": ["clinical.patient.birth_date"],
|
||||||
|
"sql": "CASE WHEN age < 18 THEN 'pediatrica' ELSE 'adulta' END",
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
assert evidence.payload.sql.startswith("CASE WHEN")
|
||||||
|
|
||||||
|
|
||||||
|
def _formula_evidence() -> CuratedEvidence:
|
||||||
|
return CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "formula",
|
||||||
|
"payload": {
|
||||||
|
"concept": "fascia pediatrica",
|
||||||
|
"columns": ["clinical.patient.birth_date"],
|
||||||
|
"sql": "CASE WHEN age < 18 THEN 'pediatrica' ELSE 'adulta' END",
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_curated_markdown_round_trip_uses_the_kind_specific_key(tmp_path):
|
||||||
|
evidence = _formula_evidence()
|
||||||
|
path = tmp_path / "curated" / "formula" / "fascia-pediatrica.md"
|
||||||
|
|
||||||
|
text = dump_curated_markdown(evidence)
|
||||||
|
parsed = parse_curated_markdown(text, path=path)
|
||||||
|
|
||||||
|
assert "formula:" in text
|
||||||
|
assert "payload:" not in text
|
||||||
|
assert parsed == evidence
|
||||||
|
|
||||||
|
|
||||||
|
def test_curated_markdown_rejects_a_kind_that_disagrees_with_its_directory(tmp_path):
|
||||||
|
text = dump_curated_markdown(_formula_evidence())
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="directory"):
|
||||||
|
parse_curated_markdown(text, path=tmp_path / "curated" / "reference" / "guide.md")
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_curated_tree_returns_documents_in_path_order_and_ignores_readmes(tmp_path):
|
||||||
|
root = tmp_path / "curated"
|
||||||
|
formula_path = root / "formula" / "fascia-pediatrica.md"
|
||||||
|
domain_path = root / "domain" / "eta.md"
|
||||||
|
formula_path.parent.mkdir(parents=True)
|
||||||
|
domain_path.parent.mkdir(parents=True)
|
||||||
|
formula_path.write_text(dump_curated_markdown(_formula_evidence()), encoding="utf-8")
|
||||||
|
domain_path.write_text(dump_curated_markdown(CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"id": "evidence:eta-ricovero",
|
||||||
|
"title": "Età al ricovero",
|
||||||
|
"kind": "domain",
|
||||||
|
"payload": {"rule": "L'età è calcolata al ricovero."},
|
||||||
|
})), encoding="utf-8")
|
||||||
|
(root / "README.md").write_text("solo navigazione", encoding="utf-8")
|
||||||
|
|
||||||
|
loaded = load_curated_tree(root)
|
||||||
|
|
||||||
|
assert [item.id for item in loaded] == [
|
||||||
|
"evidence:eta-ricovero",
|
||||||
|
"evidence:fascia-pediatrica",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(("source_file", "reference_url"), [
|
||||||
|
("source/domain/patient.pdf", "https://example.test/linea-guida"),
|
||||||
|
("../patient.md", "https://example.test/linea-guida"),
|
||||||
|
("source/domain/patient.md", "https://user:secret@example.test/linea-guida"),
|
||||||
|
("source/domain/patient.md", "https://example.test/linea-guida?access_token=secret"),
|
||||||
|
])
|
||||||
|
def test_canonical_evidence_rejects_unsafe_source_or_url(source_file, reference_url):
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "reference",
|
||||||
|
"provenance": {**COMMON["provenance"], "source_file": source_file},
|
||||||
|
"payload": {
|
||||||
|
"url": reference_url,
|
||||||
|
"label": "Linea guida",
|
||||||
|
"description": "Criteri clinici.",
|
||||||
|
},
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_curated_markdown_rejects_nonempty_body_instead_of_ignoring_it():
|
||||||
|
text = dump_curated_markdown(_formula_evidence()) + "password: should-not-be-ignored\n"
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="body"):
|
||||||
|
parse_curated_markdown(text)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(("scope", "payload"), [
|
||||||
|
(
|
||||||
|
{"tables": ["patient"], "columns": ["clinical.patient.birth_date"]},
|
||||||
|
{"rule": "Una regola."},
|
||||||
|
),
|
||||||
|
(
|
||||||
|
{"tables": ["clinical.patient"], "columns": ["clinical.patient.date.of.birth"]},
|
||||||
|
{"rule": "Una regola."},
|
||||||
|
),
|
||||||
|
(
|
||||||
|
{"tables": ["clinical.patient"], "columns": ["clinical.patient.birth_date"]},
|
||||||
|
{
|
||||||
|
"concept": "fascia pediatrica",
|
||||||
|
"columns": ["clinical.patient"],
|
||||||
|
"sql": "CASE WHEN age < 18 THEN 1 ELSE 0 END",
|
||||||
|
},
|
||||||
|
),
|
||||||
|
])
|
||||||
|
def test_schema_identifiers_require_table_or_column_shape(scope, payload):
|
||||||
|
with pytest.raises(ValidationError):
|
||||||
|
CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "formula" if "sql" in payload else "domain",
|
||||||
|
"applies_to": scope,
|
||||||
|
"payload": payload,
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_curated_tree_rejects_non_utf8_and_oversized_files(tmp_path, monkeypatch):
|
||||||
|
root = tmp_path / "curated"
|
||||||
|
path = root / "domain" / "eta.md"
|
||||||
|
path.parent.mkdir(parents=True)
|
||||||
|
path.write_bytes(b"\xff")
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="UTF-8"):
|
||||||
|
load_curated_tree(root)
|
||||||
|
|
||||||
|
path.write_text(dump_curated_markdown(CuratedEvidence.model_validate({
|
||||||
|
**COMMON,
|
||||||
|
"kind": "domain",
|
||||||
|
"payload": {"rule": "Una regola."},
|
||||||
|
})), encoding="utf-8")
|
||||||
|
monkeypatch.setattr("tht.evidence.canonical.MAX_CURATED_FILE_BYTES", 1)
|
||||||
|
|
||||||
|
with pytest.raises(ValueError, match="size limit"):
|
||||||
|
load_curated_tree(root)
|
||||||
@@ -1,6 +1,20 @@
|
|||||||
"""Cohesive public entrypoint for Evidence domain capabilities."""
|
"""Cohesive public entrypoint for Evidence domain capabilities."""
|
||||||
|
|
||||||
from tht.evidence.acquisition import acquire, discover
|
from tht.evidence.acquisition import acquire, discover
|
||||||
|
from tht.evidence.authoring import (
|
||||||
|
EvidenceManifest,
|
||||||
|
ValidationFinding,
|
||||||
|
ValidationReport,
|
||||||
|
dump_manifest,
|
||||||
|
load_manifest,
|
||||||
|
validate_workspace_evidence,
|
||||||
|
)
|
||||||
|
from tht.evidence.canonical import (
|
||||||
|
CuratedEvidence,
|
||||||
|
dump_curated_markdown,
|
||||||
|
load_curated_tree,
|
||||||
|
parse_curated_markdown,
|
||||||
|
)
|
||||||
from tht.evidence.contracts import (
|
from tht.evidence.contracts import (
|
||||||
AcquiredDocument,
|
AcquiredDocument,
|
||||||
EvidenceSource,
|
EvidenceSource,
|
||||||
@@ -28,11 +42,15 @@ __all__ = [
|
|||||||
"AcquiredDocument",
|
"AcquiredDocument",
|
||||||
"ActiveEvidenceSearcher",
|
"ActiveEvidenceSearcher",
|
||||||
"CorpusWorkspaceMismatchError",
|
"CorpusWorkspaceMismatchError",
|
||||||
|
"CuratedEvidence",
|
||||||
"EvidenceEmbedder",
|
"EvidenceEmbedder",
|
||||||
|
"EvidenceManifest",
|
||||||
"EvidenceSource",
|
"EvidenceSource",
|
||||||
"EvidenceSourceError",
|
"EvidenceSourceError",
|
||||||
"EvidenceSourceErrorCategory",
|
"EvidenceSourceErrorCategory",
|
||||||
"SourceObject",
|
"SourceObject",
|
||||||
|
"ValidationFinding",
|
||||||
|
"ValidationReport",
|
||||||
"acquire",
|
"acquire",
|
||||||
"active_searcher",
|
"active_searcher",
|
||||||
"build_preprocessing_pipeline",
|
"build_preprocessing_pipeline",
|
||||||
@@ -40,10 +58,16 @@ __all__ = [
|
|||||||
"build_sources",
|
"build_sources",
|
||||||
"canonical_provenance_uri",
|
"canonical_provenance_uri",
|
||||||
"discover",
|
"discover",
|
||||||
|
"dump_curated_markdown",
|
||||||
|
"dump_manifest",
|
||||||
|
"load_curated_tree",
|
||||||
|
"load_manifest",
|
||||||
"normalize_aware_datetime",
|
"normalize_aware_datetime",
|
||||||
|
"parse_curated_markdown",
|
||||||
"project_session",
|
"project_session",
|
||||||
"resolve_citation",
|
"resolve_citation",
|
||||||
"validate_corpus_workspace",
|
"validate_corpus_workspace",
|
||||||
"validate_namespaced_value",
|
"validate_namespaced_value",
|
||||||
"validate_safe_metadata",
|
"validate_safe_metadata",
|
||||||
|
"validate_workspace_evidence",
|
||||||
]
|
]
|
||||||
|
|||||||
@@ -0,0 +1,256 @@
|
|||||||
|
"""Validation for the Git-reviewed Evidence authoring workspace."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import unicodedata
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Literal
|
||||||
|
|
||||||
|
import yaml
|
||||||
|
from pydantic import ValidationError, field_validator, model_validator
|
||||||
|
|
||||||
|
from tht.evidence.canonical import (
|
||||||
|
CuratedEvidence,
|
||||||
|
StrictModel,
|
||||||
|
is_evidence_id,
|
||||||
|
load_curated_tree,
|
||||||
|
validate_source_file,
|
||||||
|
)
|
||||||
|
|
||||||
|
MAX_AUTHORING_FILE_BYTES = 10 * 1024 * 1024
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ValidationFinding:
|
||||||
|
severity: Literal["error", "warning"]
|
||||||
|
code: str
|
||||||
|
path: str
|
||||||
|
message: str
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ValidationReport:
|
||||||
|
findings: tuple[ValidationFinding, ...]
|
||||||
|
|
||||||
|
@property
|
||||||
|
def publishable(self) -> bool:
|
||||||
|
return not any(finding.severity == "error" for finding in self.findings)
|
||||||
|
|
||||||
|
|
||||||
|
class ManifestSource(StrictModel):
|
||||||
|
sha256: str
|
||||||
|
units: tuple[str, ...]
|
||||||
|
|
||||||
|
@field_validator("sha256")
|
||||||
|
@classmethod
|
||||||
|
def _validate_sha256(cls, value: str) -> str:
|
||||||
|
if not value.startswith("sha256:") or len(value) != 71:
|
||||||
|
raise ValueError("sha256 must be a sha256 digest")
|
||||||
|
try:
|
||||||
|
int(value.removeprefix("sha256:"), 16)
|
||||||
|
except ValueError as error:
|
||||||
|
raise ValueError("sha256 must be a sha256 digest") from error
|
||||||
|
return value
|
||||||
|
|
||||||
|
@field_validator("units")
|
||||||
|
@classmethod
|
||||||
|
def _validate_units(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||||
|
if any(not is_evidence_id(unit) for unit in value):
|
||||||
|
raise ValueError("units must use stable evidence identifiers")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceManifest(StrictModel):
|
||||||
|
schema_version: Literal[1]
|
||||||
|
pipeline_version: Literal["evidence-authoring-v1"]
|
||||||
|
sources: dict[str, ManifestSource]
|
||||||
|
orphans: tuple[str, ...]
|
||||||
|
|
||||||
|
@field_validator("sources")
|
||||||
|
@classmethod
|
||||||
|
def _validate_sources(cls, value: dict[str, ManifestSource]) -> dict[str, ManifestSource]:
|
||||||
|
for source_path in value:
|
||||||
|
validate_source_file(source_path)
|
||||||
|
return value
|
||||||
|
|
||||||
|
@field_validator("orphans")
|
||||||
|
@classmethod
|
||||||
|
def _validate_orphans(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||||
|
if any(not is_evidence_id(unit) for unit in value):
|
||||||
|
raise ValueError("orphans must use stable evidence identifiers")
|
||||||
|
return value
|
||||||
|
|
||||||
|
@model_validator(mode="after")
|
||||||
|
def _validate_unit_membership(self) -> EvidenceManifest:
|
||||||
|
seen: set[str] = set()
|
||||||
|
for source in self.sources.values():
|
||||||
|
duplicate = seen.intersection(source.units)
|
||||||
|
if duplicate:
|
||||||
|
raise ValueError("a unit may belong to only one source")
|
||||||
|
seen.update(source.units)
|
||||||
|
if len(set(self.orphans)) != len(self.orphans):
|
||||||
|
raise ValueError("orphans must be unique")
|
||||||
|
return self
|
||||||
|
|
||||||
|
|
||||||
|
def load_manifest(path: Path) -> EvidenceManifest:
|
||||||
|
"""Load the managed, versioned authoring manifest."""
|
||||||
|
try:
|
||||||
|
raw = yaml.safe_load(path.read_text(encoding="utf-8"))
|
||||||
|
except (OSError, UnicodeDecodeError, yaml.YAMLError) as error:
|
||||||
|
raise ValueError("manifest cannot be read") from error
|
||||||
|
try:
|
||||||
|
return EvidenceManifest.model_validate(raw)
|
||||||
|
except ValidationError as error:
|
||||||
|
raise ValueError("manifest is invalid") from error
|
||||||
|
|
||||||
|
|
||||||
|
def dump_manifest(manifest: EvidenceManifest) -> str:
|
||||||
|
"""Serialize the manifest deterministically for Git review."""
|
||||||
|
return yaml.safe_dump(manifest.model_dump(mode="json"), allow_unicode=True, sort_keys=True)
|
||||||
|
|
||||||
|
|
||||||
|
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
|
||||||
|
"""Validate the curated corpus without writing the workspace."""
|
||||||
|
evidence_root = workspace_root / "evidence"
|
||||||
|
findings: list[ValidationFinding] = []
|
||||||
|
manifest_path = evidence_root / "manifest.yaml"
|
||||||
|
if not manifest_path.is_file():
|
||||||
|
return ValidationReport((ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="manifest_missing",
|
||||||
|
path="manifest.yaml",
|
||||||
|
message="The managed Evidence manifest is missing.",
|
||||||
|
),))
|
||||||
|
try:
|
||||||
|
manifest = load_manifest(manifest_path)
|
||||||
|
except ValueError:
|
||||||
|
return ValidationReport((ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="manifest_invalid",
|
||||||
|
path="manifest.yaml",
|
||||||
|
message="The managed Evidence manifest is invalid.",
|
||||||
|
),))
|
||||||
|
for orphan in manifest.orphans:
|
||||||
|
findings.append(ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="orphaned_unit",
|
||||||
|
path="manifest.yaml",
|
||||||
|
message=f"Orphaned Evidence {orphan} must be resolved before publication.",
|
||||||
|
))
|
||||||
|
try:
|
||||||
|
documents = load_curated_tree(evidence_root / "curated")
|
||||||
|
except (OSError, ValidationError, ValueError):
|
||||||
|
return ValidationReport(tuple(findings + [ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="curated_invalid",
|
||||||
|
path="curated",
|
||||||
|
message="A Curated Evidence document is invalid or cannot be read.",
|
||||||
|
)]))
|
||||||
|
source_texts = {
|
||||||
|
source_path: _validate_manifest_source(evidence_root, source_path, source, findings)
|
||||||
|
for source_path, source in manifest.sources.items()
|
||||||
|
}
|
||||||
|
seen_ids: set[str] = set()
|
||||||
|
for evidence in documents:
|
||||||
|
if evidence.id in seen_ids:
|
||||||
|
findings.append(ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="duplicate_evidence_id",
|
||||||
|
path="curated",
|
||||||
|
message=f"Evidence id {evidence.id} appears more than once.",
|
||||||
|
))
|
||||||
|
seen_ids.add(evidence.id)
|
||||||
|
findings.extend(_validate_unit(manifest, evidence, source_texts.get(evidence.provenance.source_file)))
|
||||||
|
for source in manifest.sources.values():
|
||||||
|
for unit in source.units:
|
||||||
|
if unit not in seen_ids:
|
||||||
|
findings.append(ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="manifest_unit_without_curated",
|
||||||
|
path="manifest.yaml",
|
||||||
|
message=f"Manifest Evidence {unit} has no Curated document.",
|
||||||
|
))
|
||||||
|
return ValidationReport(tuple(findings))
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_manifest_source(
|
||||||
|
evidence_root: Path,
|
||||||
|
source_path: str,
|
||||||
|
manifest_source: ManifestSource,
|
||||||
|
findings: list[ValidationFinding],
|
||||||
|
) -> str | None:
|
||||||
|
path = evidence_root / source_path
|
||||||
|
if not path.is_file():
|
||||||
|
findings.append(ValidationFinding("error", "source_missing", source_path,
|
||||||
|
"The manifest source does not exist."))
|
||||||
|
return None
|
||||||
|
if path.stat().st_size > MAX_AUTHORING_FILE_BYTES:
|
||||||
|
findings.append(ValidationFinding("error", "source_oversized", source_path,
|
||||||
|
"The manifest source exceeds the authoring size limit."))
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
source = _normalize(path.read_text(encoding="utf-8"))
|
||||||
|
except UnicodeDecodeError:
|
||||||
|
findings.append(ValidationFinding("error", "source_unreadable", source_path,
|
||||||
|
"The manifest source is not valid UTF-8."))
|
||||||
|
return None
|
||||||
|
digest = "sha256:" + hashlib.sha256(source.encode("utf-8")).hexdigest()
|
||||||
|
if digest != manifest_source.sha256:
|
||||||
|
findings.append(ValidationFinding("error", "source_hash_mismatch", source_path,
|
||||||
|
"The manifest hash does not match the normalized source."))
|
||||||
|
return source
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_unit(
|
||||||
|
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
|
||||||
|
) -> list[ValidationFinding]:
|
||||||
|
path = evidence.provenance.source_file
|
||||||
|
findings: list[ValidationFinding] = []
|
||||||
|
manifest_source = manifest.sources.get(path)
|
||||||
|
if manifest_source is None:
|
||||||
|
findings.append(ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="manifest_source_missing",
|
||||||
|
path="manifest.yaml",
|
||||||
|
message="The manifest does not contain the provenance source.",
|
||||||
|
))
|
||||||
|
else:
|
||||||
|
if manifest_source.sha256 != evidence.provenance.source_sha256:
|
||||||
|
findings.append(ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="manifest_source_hash_mismatch",
|
||||||
|
path="manifest.yaml",
|
||||||
|
message="The manifest hash does not match the Evidence provenance.",
|
||||||
|
))
|
||||||
|
if evidence.id not in manifest_source.units:
|
||||||
|
findings.append(ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="manifest_unit_missing",
|
||||||
|
path="manifest.yaml",
|
||||||
|
message="The manifest does not link the Evidence unit to its source.",
|
||||||
|
))
|
||||||
|
if source is None:
|
||||||
|
return findings
|
||||||
|
for excerpt in evidence.provenance.supporting_excerpts:
|
||||||
|
if _normalize(excerpt) not in source:
|
||||||
|
findings.append(ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="supporting_excerpt_missing",
|
||||||
|
path=path,
|
||||||
|
message="A supporting excerpt is absent from the normalized source.",
|
||||||
|
))
|
||||||
|
for item in evidence.review_items:
|
||||||
|
findings.append(ValidationFinding(
|
||||||
|
severity="error",
|
||||||
|
code="unresolved_review_item",
|
||||||
|
path=path,
|
||||||
|
message=f"Review item {item.code} must be resolved before publication.",
|
||||||
|
))
|
||||||
|
return findings
|
||||||
|
|
||||||
|
|
||||||
|
def _normalize(text: str) -> str:
|
||||||
|
return unicodedata.normalize("NFC", text.replace("\r\n", "\n").replace("\r", "\n"))
|
||||||
@@ -0,0 +1,332 @@
|
|||||||
|
"""Typed, reviewable Evidence units stored in the workspace repository."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
from pathlib import Path, PurePosixPath
|
||||||
|
from typing import Literal
|
||||||
|
|
||||||
|
import sqlglot
|
||||||
|
import yaml
|
||||||
|
from pydantic import AnyHttpUrl, BaseModel, ConfigDict, Field, field_validator, model_validator
|
||||||
|
from sqlglot import exp
|
||||||
|
|
||||||
|
from tht.evidence.contracts import validate_canonical_uri
|
||||||
|
|
||||||
|
|
||||||
|
class StrictModel(BaseModel):
|
||||||
|
"""Reject undeclared fields in the repository's canonical format."""
|
||||||
|
|
||||||
|
model_config = ConfigDict(extra="forbid")
|
||||||
|
|
||||||
|
|
||||||
|
EvidenceKind = Literal[
|
||||||
|
"glossary",
|
||||||
|
"domain",
|
||||||
|
"enum",
|
||||||
|
"example",
|
||||||
|
"mapping",
|
||||||
|
"normalization",
|
||||||
|
"formula",
|
||||||
|
"reference",
|
||||||
|
]
|
||||||
|
EvidencePurpose = Literal[
|
||||||
|
"disambiguation",
|
||||||
|
"rewriting",
|
||||||
|
"schema_linking",
|
||||||
|
"sql_generation",
|
||||||
|
]
|
||||||
|
_IDENTIFIER = r"[A-Za-z_][A-Za-z0-9_$]*"
|
||||||
|
_TABLE_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}$")
|
||||||
|
_COLUMN_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}\.{_IDENTIFIER}$")
|
||||||
|
MAX_CURATED_FILE_BYTES = 10 * 1024 * 1024
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceScope(StrictModel):
|
||||||
|
concepts: tuple[str, ...] = ()
|
||||||
|
tables: tuple[str, ...] = ()
|
||||||
|
columns: tuple[str, ...] = ()
|
||||||
|
|
||||||
|
@field_validator("tables")
|
||||||
|
@classmethod
|
||||||
|
def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||||
|
return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables")
|
||||||
|
|
||||||
|
@field_validator("columns")
|
||||||
|
@classmethod
|
||||||
|
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||||
|
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceProvenance(StrictModel):
|
||||||
|
model_config = ConfigDict(extra="forbid", frozen=True)
|
||||||
|
|
||||||
|
source_file: str
|
||||||
|
source_sha256: str
|
||||||
|
supporting_excerpts: tuple[str, ...]
|
||||||
|
|
||||||
|
@field_validator("source_file")
|
||||||
|
@classmethod
|
||||||
|
def _validate_source_file(cls, value: str) -> str:
|
||||||
|
return validate_source_file(value)
|
||||||
|
|
||||||
|
@field_validator("source_sha256")
|
||||||
|
@classmethod
|
||||||
|
def _validate_sha256(cls, value: str) -> str:
|
||||||
|
if not re.fullmatch(r"sha256:[0-9a-f]{64}", value):
|
||||||
|
raise ValueError("source_sha256 must be a sha256 digest")
|
||||||
|
return value
|
||||||
|
|
||||||
|
@field_validator("supporting_excerpts")
|
||||||
|
@classmethod
|
||||||
|
def _validate_excerpts(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||||
|
if not 1 <= len(value) <= 5:
|
||||||
|
raise ValueError("supporting_excerpts must contain one to five items")
|
||||||
|
if any(not excerpt.strip() or len(excerpt) > 1000 for excerpt in value):
|
||||||
|
raise ValueError("supporting excerpts must be nonempty and at most 1000 characters")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
class ReviewItem(StrictModel):
|
||||||
|
code: str
|
||||||
|
message: str
|
||||||
|
field: str | None = None
|
||||||
|
|
||||||
|
|
||||||
|
class FormulaPayload(StrictModel):
|
||||||
|
concept: str
|
||||||
|
columns: tuple[str, ...]
|
||||||
|
sql: str
|
||||||
|
|
||||||
|
@field_validator("columns")
|
||||||
|
@classmethod
|
||||||
|
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||||
|
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
|
||||||
|
|
||||||
|
@field_validator("sql")
|
||||||
|
@classmethod
|
||||||
|
def _validate_expression(cls, value: str) -> str:
|
||||||
|
try:
|
||||||
|
statements = [statement for statement in sqlglot.parse(value, read="postgres") if statement]
|
||||||
|
except sqlglot.errors.ParseError as error:
|
||||||
|
raise ValueError("formula.sql must be valid PostgreSQL") from error
|
||||||
|
if len(statements) != 1:
|
||||||
|
raise ValueError("formula.sql must contain exactly one expression")
|
||||||
|
expression = statements[0]
|
||||||
|
if expression.find(exp.Select) is not None or expression.find(exp.With) is not None:
|
||||||
|
raise ValueError("formula.sql must not contain a query")
|
||||||
|
if any(
|
||||||
|
expression.find(statement_type) is not None
|
||||||
|
for statement_type in (
|
||||||
|
exp.Insert,
|
||||||
|
exp.Update,
|
||||||
|
exp.Delete,
|
||||||
|
exp.Create,
|
||||||
|
exp.Drop,
|
||||||
|
exp.Alter,
|
||||||
|
exp.Merge,
|
||||||
|
exp.TruncateTable,
|
||||||
|
exp.Grant,
|
||||||
|
exp.Revoke,
|
||||||
|
exp.Command,
|
||||||
|
exp.Values,
|
||||||
|
exp.Set,
|
||||||
|
exp.Table,
|
||||||
|
)
|
||||||
|
):
|
||||||
|
raise ValueError("formula.sql must not contain DDL or DML")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
class ReferencePayload(StrictModel):
|
||||||
|
url: AnyHttpUrl
|
||||||
|
label: str
|
||||||
|
description: str
|
||||||
|
|
||||||
|
@field_validator("url")
|
||||||
|
@classmethod
|
||||||
|
def _reject_credentials(cls, value: AnyHttpUrl) -> AnyHttpUrl:
|
||||||
|
validate_canonical_uri(str(value))
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
class GlossaryPayload(StrictModel):
|
||||||
|
definition: str
|
||||||
|
synonyms: tuple[str, ...] = ()
|
||||||
|
variants: tuple[str, ...] = ()
|
||||||
|
|
||||||
|
|
||||||
|
class DomainPayload(StrictModel):
|
||||||
|
rule: str
|
||||||
|
|
||||||
|
|
||||||
|
class EnumPayload(StrictModel):
|
||||||
|
column: str
|
||||||
|
values: dict[str, str]
|
||||||
|
|
||||||
|
@field_validator("column")
|
||||||
|
@classmethod
|
||||||
|
def _validate_column(cls, value: str) -> str:
|
||||||
|
_validate_identifiers((value,), _COLUMN_IDENTIFIER, "column")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
class ExamplePayload(StrictModel):
|
||||||
|
question: str
|
||||||
|
interpretation: str
|
||||||
|
|
||||||
|
|
||||||
|
class MappingPayload(StrictModel):
|
||||||
|
concept: str
|
||||||
|
tables: tuple[str, ...]
|
||||||
|
columns: tuple[str, ...]
|
||||||
|
|
||||||
|
@field_validator("tables")
|
||||||
|
@classmethod
|
||||||
|
def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||||
|
return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables")
|
||||||
|
|
||||||
|
@field_validator("columns")
|
||||||
|
@classmethod
|
||||||
|
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
||||||
|
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
|
||||||
|
|
||||||
|
|
||||||
|
class NormalizationPayload(StrictModel):
|
||||||
|
input: str
|
||||||
|
output: str
|
||||||
|
rule: str
|
||||||
|
|
||||||
|
|
||||||
|
EvidencePayload = (
|
||||||
|
GlossaryPayload
|
||||||
|
| DomainPayload
|
||||||
|
| EnumPayload
|
||||||
|
| ExamplePayload
|
||||||
|
| MappingPayload
|
||||||
|
| NormalizationPayload
|
||||||
|
| FormulaPayload
|
||||||
|
| ReferencePayload
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
_PAYLOAD_TYPE_BY_KIND = {
|
||||||
|
"glossary": GlossaryPayload,
|
||||||
|
"domain": DomainPayload,
|
||||||
|
"enum": EnumPayload,
|
||||||
|
"example": ExamplePayload,
|
||||||
|
"mapping": MappingPayload,
|
||||||
|
"normalization": NormalizationPayload,
|
||||||
|
"formula": FormulaPayload,
|
||||||
|
"reference": ReferencePayload,
|
||||||
|
}
|
||||||
|
_EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
|
||||||
|
|
||||||
|
|
||||||
|
class CuratedEvidence(StrictModel):
|
||||||
|
schema_version: Literal[1]
|
||||||
|
id: str
|
||||||
|
title: str
|
||||||
|
kind: EvidenceKind
|
||||||
|
purposes: tuple[EvidencePurpose, ...]
|
||||||
|
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
|
||||||
|
language: str
|
||||||
|
provenance: EvidenceProvenance
|
||||||
|
review_items: tuple[ReviewItem, ...] = ()
|
||||||
|
payload: EvidencePayload
|
||||||
|
|
||||||
|
@model_validator(mode="after")
|
||||||
|
def _validate_kind_payload(self) -> CuratedEvidence:
|
||||||
|
if not is_evidence_id(self.id):
|
||||||
|
raise ValueError("id must use the evidence:<slug> form")
|
||||||
|
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
|
||||||
|
if expected is not None and not isinstance(self.payload, expected):
|
||||||
|
raise ValueError(f"{self.kind} requires its typed payload")
|
||||||
|
return self
|
||||||
|
|
||||||
|
|
||||||
|
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
|
||||||
|
"""Parse the canonical frontmatter representation of one Curated Evidence unit."""
|
||||||
|
if not text.startswith("---\n"):
|
||||||
|
raise ValueError("curated evidence requires YAML frontmatter")
|
||||||
|
try:
|
||||||
|
_, frontmatter, body = text.split("---\n", 2)
|
||||||
|
except ValueError as error:
|
||||||
|
raise ValueError("curated evidence frontmatter is malformed") from error
|
||||||
|
raw = yaml.safe_load(frontmatter)
|
||||||
|
try:
|
||||||
|
data = dict(raw)
|
||||||
|
except (TypeError, ValueError) as error:
|
||||||
|
raise ValueError("curated evidence frontmatter must be a mapping") from error
|
||||||
|
if body.strip():
|
||||||
|
raise ValueError("curated evidence must not contain an ignored body")
|
||||||
|
kind = data.get("kind")
|
||||||
|
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
|
||||||
|
data["payload"] = data.pop(kind, None)
|
||||||
|
evidence = CuratedEvidence.model_validate(data)
|
||||||
|
if path is not None:
|
||||||
|
_validate_kind_directory(path, evidence.kind)
|
||||||
|
return evidence
|
||||||
|
|
||||||
|
|
||||||
|
def dump_curated_markdown(value: CuratedEvidence) -> str:
|
||||||
|
"""Render canonical frontmatter with a human-readable kind-specific payload key."""
|
||||||
|
data = value.model_dump(mode="json", exclude={"payload"})
|
||||||
|
data[value.kind] = value.payload.model_dump(mode="json")
|
||||||
|
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
|
||||||
|
return f"---\n{frontmatter}---\n"
|
||||||
|
|
||||||
|
|
||||||
|
def load_curated_tree(root: Path) -> list[CuratedEvidence]:
|
||||||
|
"""Load canonical Evidence units in stable path order from a curated root."""
|
||||||
|
if not root.is_dir():
|
||||||
|
return []
|
||||||
|
documents: list[CuratedEvidence] = []
|
||||||
|
for path in sorted(root.rglob("*.md")):
|
||||||
|
if path.name.upper().startswith("README"):
|
||||||
|
continue
|
||||||
|
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||||
|
raise ValueError("curated evidence exceeds the size limit")
|
||||||
|
try:
|
||||||
|
text = path.read_text(encoding="utf-8")
|
||||||
|
except UnicodeDecodeError as error:
|
||||||
|
raise ValueError("curated evidence must be UTF-8") from error
|
||||||
|
documents.append(parse_curated_markdown(text, path=path))
|
||||||
|
return documents
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_kind_directory(path: Path, kind: EvidenceKind) -> None:
|
||||||
|
parts = path.parts
|
||||||
|
try:
|
||||||
|
curated_index = parts.index("curated")
|
||||||
|
except ValueError:
|
||||||
|
return
|
||||||
|
if len(parts) <= curated_index + 1 or parts[curated_index + 1] != kind:
|
||||||
|
raise ValueError("curated evidence kind must match its directory")
|
||||||
|
|
||||||
|
|
||||||
|
def _validate_identifiers(
|
||||||
|
values: tuple[str, ...], pattern: re.Pattern[str], field: str,
|
||||||
|
) -> tuple[str, ...]:
|
||||||
|
if any(pattern.fullmatch(value) is None for value in values):
|
||||||
|
raise ValueError(f"{field} must use canonical schema identifiers")
|
||||||
|
return values
|
||||||
|
|
||||||
|
|
||||||
|
def validate_source_file(value: str) -> str:
|
||||||
|
"""Validate a repository-relative, credential-free Source Evidence path."""
|
||||||
|
path = PurePosixPath(value)
|
||||||
|
if (
|
||||||
|
path.is_absolute()
|
||||||
|
or ".." in path.parts
|
||||||
|
or not path.parts
|
||||||
|
or path.parts[0] != "source"
|
||||||
|
or not value.endswith((".md", ".txt", ".sql.md"))
|
||||||
|
):
|
||||||
|
raise ValueError("source_file must be a supported path below source/")
|
||||||
|
return value
|
||||||
|
|
||||||
|
|
||||||
|
def is_evidence_id(value: str) -> bool:
|
||||||
|
"""Whether a value uses the stable public Evidence identifier format."""
|
||||||
|
return _EVIDENCE_ID.fullmatch(value) is not None
|
||||||
Reference in New Issue
Block a user