feat(evidence): add table-free v3 and design guidance

This commit is contained in:
Codex
2026-08-26 12:15:40 +02:00
parent 38f02cfd08
commit 9d4f994d3e
11 changed files with 983 additions and 67 deletions
+1 -1
View File
@@ -121,7 +121,7 @@ def pipeline(tmp_path, source, *, embedder=None, vectors=None, model="model-a",
def test_pipeline_embeds_validated_curated_evidence_as_semantic_fragments(tmp_path):
evidence = CuratedEvidence.model_validate(
{
"schema_version": 2,
"schema_version": 3,
"id": "evidence:fascia-pediatrica",
"title": "Fascia pediatrica",
"kind": "formula",
+24 -4
View File
@@ -350,12 +350,12 @@ def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tm
assert restructurer.requests[0].previous_units[0].id == "evidence:fascia-pediatrica"
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
curated = load_curated_tree(tmp_path / "evidence" / "curated")[0]
assert curated.schema_version == 2
assert curated.schema_version == 3
assert "# Fascia pediatrica\n" in curated_path.read_text(encoding="utf-8")
assert validate_workspace_evidence(tmp_path).publishable is True
def test_migrate_workspace_evidence_rewrites_v1_units_without_a_model_call(tmp_path):
def test_migrate_workspace_evidence_rewrites_v1_units_as_v3_without_a_model_call(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
@@ -365,7 +365,7 @@ def test_migrate_workspace_evidence_rewrites_v1_units_without_a_model_call(tmp_p
migrated = load_curated_tree(tmp_path / "evidence" / "curated")[0]
assert report.migrated == ("evidence:fascia-pediatrica",)
assert report.unchanged == ()
assert migrated.schema_version == 2
assert migrated.schema_version == 3
assert migrated.payload.rule == "La fascia pediatrica comprende i minori."
assert "## Regola\n\nLa fascia pediatrica comprende i minori." in curated_path.read_text(
encoding="utf-8",
@@ -373,6 +373,26 @@ def test_migrate_workspace_evidence_rewrites_v1_units_without_a_model_call(tmp_p
assert report.findings == ()
def test_migrate_workspace_evidence_rewrites_v2_units_as_table_free_v3(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
evidence = _evidence(source_text).model_copy(update={"schema_version": 2})
_write_workspace(tmp_path, evidence, source_text)
first = migrate_workspace_evidence(tmp_path, git_status=lambda _: ())
second = migrate_workspace_evidence(tmp_path, git_status=lambda _: ())
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
text = curated_path.read_text(encoding="utf-8")
migrated = load_curated_tree(tmp_path / "evidence" / "curated")[0]
assert first.migrated == ("evidence:fascia-pediatrica",)
assert first.unchanged == ()
assert second.migrated == ()
assert second.unchanged == ("evidence:fascia-pediatrica",)
assert migrated.schema_version == 3
assert text.startswith("<!-- tht:metadata:")
assert not any(line.startswith("|") for line in text.splitlines())
def test_migrate_workspace_evidence_rejects_dirty_curated_files_in_a_nested_workspace(tmp_path):
subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True)
workspace_root = tmp_path / "psd-clinical"
@@ -546,7 +566,7 @@ def test_prepare_marks_an_omitted_prior_unit_for_human_review(tmp_path):
"supporting_excerpt_missing", "unresolved_review_item",
]
retained = load_curated_tree(tmp_path / "evidence" / "curated")[0]
assert retained.schema_version == 2
assert retained.schema_version == 3
assert retained.review_items[0].code == "source_no_longer_supports_unit"
+73
View File
@@ -221,6 +221,79 @@ def _domain_evidence_v2() -> CuratedEvidence:
})
def _domain_evidence_v3() -> CuratedEvidence:
return CuratedEvidence.model_validate({
**COMMON,
"schema_version": 3,
"title": "Dominio Ablazione",
"kind": "domain",
"purposes": ["disambiguation", "schema_linking", "rewriting"],
"applies_to": {
"concepts": ["ablazione", "studio elettrofisiologico", "SEE"],
"tables": ["clinical.fact_ablazione"],
"columns": ["clinical.fact_ablazione.patient_id"],
},
"payload": {
"rule": (
"Il dominio Ablazione rappresenta la procedura transcatetere.\n\n"
"La fact centrale è `clinical.fact_ablazione`."
),
},
})
def test_v3_curated_markdown_replaces_frontmatter_tables_with_readable_sections(tmp_path):
evidence = _domain_evidence_v3()
path = tmp_path / "curated" / "domain" / "dominio-ablazione.md"
text = dump_curated_markdown(evidence)
parsed = parse_curated_markdown(text, path=path)
assert text.startswith("<!-- tht:metadata:")
assert not text.startswith("---\n")
assert not any(line.startswith("|") for line in text.splitlines())
assert "> **Dominio** · Italiano" in text
assert "**Scopi:** Disambiguazione · Collegamento allo schema · Riscrittura" in text
assert "## Ambito di applicazione" in text
assert "### Concetti\n\n- ablazione\n- studio elettrofisiologico\n- SEE" in text
assert "### Tabelle\n\n- `clinical.fact_ablazione`" in text
assert "### Colonne\n\n- `clinical.fact_ablazione.patient_id`" in text
assert "<summary>Dettagli tecnici e provenienza</summary>" in text
assert parsed == evidence
def test_v3_curated_markdown_renders_enum_values_as_a_list_instead_of_a_table(tmp_path):
evidence = CuratedEvidence.model_validate({
**COMMON,
"schema_version": 3,
"kind": "enum",
"payload": {
"column": "clinical.episode.discharge_status",
"values": {"D": "dimesso", "T": "trasferito | altra struttura"},
},
})
text = dump_curated_markdown(evidence)
assert "## Valori\n\n- `D`: dimesso\n- `T`: trasferito | altra struttura" in text
assert not any(line.startswith("|") for line in text.splitlines())
assert parse_curated_markdown(
text,
path=tmp_path / "curated" / "enum" / "discharge-status.md",
) == evidence
def test_v3_curated_markdown_rejects_visible_metadata_that_drifted_from_canonical_data():
text = dump_curated_markdown(_domain_evidence_v3()).replace(
"**Scopi:** Disambiguazione",
"**Scopi:** Testo alterato",
1,
)
with pytest.raises(ValueError, match="not canonical"):
parse_curated_markdown(text)
def test_v2_curated_markdown_renders_domain_content_in_the_markdown_body(tmp_path):
evidence = _domain_evidence_v2()
path = tmp_path / "curated" / "domain" / "dominio-ablazione.md"
+1 -1
View File
@@ -180,7 +180,7 @@ def migrate_cmd(
workspace_root: Path,
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
) -> None:
"""Rewrite legacy Curated units as readable Markdown without model calls."""
"""Rewrite legacy Curated units as table-free v3 Markdown without model calls."""
root = _canonical_worktree(workspace_root)
try:
report = migrate_workspace_evidence(root)
+6 -6
View File
@@ -694,7 +694,7 @@ def migrate_workspace_evidence(
*,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
) -> EvidenceMigrationReport:
"""Rewrite v1 Curated units as readable v2 Markdown without changing semantics."""
"""Rewrite legacy Curated units as table-free v3 Markdown without changing semantics."""
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
@@ -708,10 +708,10 @@ def migrate_workspace_evidence(
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
migrated = tuple(sorted(
document.id for document in documents if document.schema_version == 1
document.id for document in documents if document.schema_version in {1, 2}
))
unchanged = tuple(sorted(
document.id for document in documents if document.schema_version == 2
document.id for document in documents if document.schema_version == 3
))
if not migrated:
return EvidenceMigrationReport(
@@ -720,7 +720,7 @@ def migrate_workspace_evidence(
findings=validate_workspace_evidence(workspace_root).findings,
)
upgraded = {
evidence_id: document.model_copy(update={"schema_version": 2})
evidence_id: document.model_copy(update={"schema_version": 3})
for evidence_id, document in documents_by_id.items()
}
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
@@ -1007,7 +1007,7 @@ def _candidate_to_evidence(
mode="json",
exclude={"schema_version", "existing_id", "supporting_excerpts"},
)
data["schema_version"] = 2
data["schema_version"] = 3
data["id"] = evidence_id
data["provenance"] = {
"source_file": source_file,
@@ -1028,7 +1028,7 @@ def _unsupported_unit(
message="The current source no longer supports this Evidence unit.",
),)
return evidence.model_copy(update={
"schema_version": 2,
"schema_version": 3,
"provenance": evidence.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": source_hash,
+306 -20
View File
@@ -2,6 +2,9 @@
from __future__ import annotations
import base64
import binascii
import json
import re
from pathlib import Path, PurePosixPath
from typing import Literal
@@ -226,7 +229,7 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
class CuratedEvidence(StrictModel):
schema_version: Literal[1, 2]
schema_version: Literal[1, 2, 3]
id: str
title: str
kind: EvidenceKind
@@ -272,6 +275,10 @@ _V2_LABELS = {
"values": "Values",
"meaning": "Meaning",
"variants": "Variants",
"applies_to": "Applies to",
"concepts": "Concepts",
"technical_details": "Technical details and provenance",
"purposes": "Purposes",
},
"it": {
"column": "Colonna",
@@ -297,6 +304,10 @@ _V2_LABELS = {
"values": "Valori",
"meaning": "Significato",
"variants": "Varianti",
"applies_to": "Ambito di applicazione",
"concepts": "Concetti",
"technical_details": "Dettagli tecnici e provenienza",
"purposes": "Scopi",
},
}
_V2_FIELD = re.compile(
@@ -307,6 +318,43 @@ _V2_EXCERPT_SEPARATOR = "<!-- tht:excerpt-separator -->"
_V2_EMPTY_LIST = "<!-- tht:empty-list -->"
_V2_REVIEW_SEPARATOR = "<!-- tht:review-separator -->"
_V2_REVIEW_FIELD = "<!-- tht:review-field -->"
_V3_METADATA = re.compile(r"\A<!-- tht:metadata:([A-Za-z0-9+/=]+) -->\n")
_V3_KIND_LABELS = {
"en": {
"glossary": "Glossary",
"domain": "Domain",
"enum": "Enumeration",
"example": "Example",
"mapping": "Mapping",
"normalization": "Normalization",
"formula": "Formula",
"reference": "Reference",
},
"it": {
"glossary": "Glossario",
"domain": "Dominio",
"enum": "Enumerazione",
"example": "Esempio",
"mapping": "Mappatura",
"normalization": "Normalizzazione",
"formula": "Formula",
"reference": "Riferimento",
},
}
_V3_PURPOSE_LABELS = {
"en": {
"disambiguation": "Disambiguation",
"rewriting": "Rewriting",
"schema_linking": "Schema linking",
"sql_generation": "SQL generation",
},
"it": {
"disambiguation": "Disambiguazione",
"rewriting": "Riscrittura",
"schema_linking": "Collegamento allo schema",
"sql_generation": "Generazione SQL",
},
}
def _v2_labels(language: str) -> dict[str, str]:
@@ -425,6 +473,40 @@ def _parse_v2_values(value: str, labels: dict[str, str]) -> dict[str, str]:
return parsed
def _render_v3_values(values: dict[str, str], labels: dict[str, str]) -> str:
if not values:
return f"{_V2_EMPTY_LIST}\n_{labels['empty']}._"
if any("`" in value for value in values):
raise ValueError("curated evidence enum values must not contain backticks")
rendered: list[str] = []
for value, meaning in sorted(values.items()):
lines = meaning.split("\n")
rendered.append(f"- `{value}`: {lines[0]}")
rendered.extend(f" {line}" for line in lines[1:])
return "\n".join(rendered)
def _parse_v3_values(value: str, labels: dict[str, str]) -> dict[str, str]:
if value == f"{_V2_EMPTY_LIST}\n_{labels['empty']}._":
return {}
parsed: dict[str, list[str]] = {}
current: str | None = None
for line in value.split("\n"):
match = re.fullmatch(r"- `([^`]+)`: ?(.*)", line)
if match is not None:
current = match.group(1)
if current in parsed:
raise ValueError("curated evidence enum value appears more than once")
parsed[current] = [match.group(2)]
continue
if current is None or not line.startswith(" "):
raise ValueError("curated evidence values list is malformed")
parsed[current].append(line[2:])
if not parsed:
raise ValueError("curated evidence values list is malformed")
return {key: "\n".join(lines) for key, lines in parsed.items()}
def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
payload = value.payload
if value.kind == "glossary":
@@ -485,6 +567,18 @@ def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[s
raise ValueError(f"unsupported curated evidence kind {value.kind}")
def _render_v3_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
if value.kind != "enum":
return _render_v2_payload(value, labels)
payload = value.payload
return [
_render_v2_field("column", labels["column"], f"`{payload.column}`"),
_render_v2_field("values", labels["values"], _render_v3_values(
payload.values, labels,
)),
]
def _render_v2_review_items(value: CuratedEvidence, labels: dict[str, str]) -> str:
rendered: list[str] = []
for item in value.review_items:
@@ -523,6 +617,121 @@ def _render_v2_body(value: CuratedEvidence) -> str:
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
def _render_v3_block(name: str, content: str) -> str:
if "<!-- tht:field:" in content or "<!-- /tht:field:" in content:
raise ValueError(f"curated evidence {name} contains a reserved marker")
return (
f"<!-- tht:field:{name} -->\n"
f"{content}\n"
f"<!-- /tht:field:{name} -->"
)
def _v3_locale(value: CuratedEvidence) -> str:
return "it" if value.language.lower().startswith("it") else "en"
def _render_v3_overview(value: CuratedEvidence, labels: dict[str, str]) -> str:
locale = _v3_locale(value)
language = "Italiano" if locale == "it" else "English"
kind = _V3_KIND_LABELS[locale][value.kind]
purposes = " · ".join(_V3_PURPOSE_LABELS[locale][purpose] for purpose in value.purposes)
if not purposes:
purposes = labels["empty"]
return _render_v3_block(
"overview",
f"> **{kind}** · {language}\n>\n> **{labels['purposes']}:** {purposes}",
)
def _render_v3_scope(value: CuratedEvidence, labels: dict[str, str]) -> str:
sections: list[str] = []
for label, values, code in (
(labels["concepts"], value.applies_to.concepts, False),
(labels["tables"], value.applies_to.tables, True),
(labels["columns"], value.applies_to.columns, True),
):
if values:
sections.append(
f"### {label}\n\n"
f"{_render_v2_list(values, code=code, empty_label=labels['empty'])}"
)
content = "\n\n".join(sections) if sections else f"_{labels['empty']}._"
return _render_v3_block(
"applies_to",
f"## {labels['applies_to']}\n\n{content}",
)
def _render_v3_provenance(value: CuratedEvidence, labels: dict[str, str]) -> str:
locale = _v3_locale(value)
technical_labels = {
"en": {
"schema": "Schema version",
"kind": "Kind",
"language": "Language",
"source": "Source file",
},
"it": {
"schema": "Versione schema",
"kind": "Tipo",
"language": "Lingua",
"source": "File sorgente",
},
}[locale]
content = (
"<details>\n"
f"<summary>{labels['technical_details']}</summary>\n\n"
f"- **ID:** `{value.id}`\n"
f"- **{technical_labels['schema']}:** `{value.schema_version}`\n"
f"- **{technical_labels['kind']}:** `{value.kind}`\n"
f"- **{technical_labels['language']}:** `{value.language}`\n"
f"- **{technical_labels['source']}:** `{value.provenance.source_file}`\n"
f"- **SHA-256:** `{value.provenance.source_sha256}`\n\n"
"</details>"
)
return _render_v3_block("provenance", content)
def _render_v3_metadata(value: CuratedEvidence) -> str:
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
data["provenance"].pop("supporting_excerpts")
encoded = base64.b64encode(json.dumps(
data,
ensure_ascii=False,
separators=(",", ":"),
sort_keys=True,
).encode("utf-8")).decode("ascii")
return f"<!-- tht:metadata:{encoded} -->"
def _render_v3_body(value: CuratedEvidence) -> str:
if "\n" in value.title:
raise ValueError("curated evidence title must be single-line in v3")
labels = _v2_labels(value.language)
fields = [
_render_v3_overview(value, labels),
_render_v3_scope(value, labels),
*_render_v3_payload(value, labels),
_render_v2_field(
"supporting_excerpts",
labels["supporting_excerpts"],
f"\n{_V2_EXCERPT_SEPARATOR}\n".join(
_render_v2_excerpt(excerpt)
for excerpt in value.provenance.supporting_excerpts
),
),
]
if value.review_items:
fields.append(_render_v2_field(
"review_items",
labels["review_items"],
_render_v2_review_items(value, labels),
))
fields.append(_render_v3_provenance(value, labels))
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
def _parse_v2_field_content(name: str, block: str) -> str:
try:
heading, content = block.split("\n\n", 1)
@@ -615,6 +824,17 @@ def _parse_v2_payload(
raise ValueError("curated evidence body kind is unsupported")
def _parse_v3_payload(
kind: str, fields: dict[str, str], labels: dict[str, str],
) -> tuple[dict, set[str]]:
if kind != "enum":
return _parse_v2_payload(kind, fields, labels)
return {
"column": _parse_inline_code(fields.get("column", ""), "column"),
"values": _parse_v3_values(fields.get("values", ""), labels),
}, {"column", "values"}
def _parse_v2_review_items(value: str) -> tuple[ReviewItem, ...]:
items: list[ReviewItem] = []
for raw_item in value.split(f"\n{_V2_REVIEW_SEPARATOR}\n"):
@@ -680,35 +900,101 @@ def _parse_v2_body(data: dict, body: str) -> dict:
return data
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
"""Parse the canonical frontmatter representation of one Curated Evidence unit."""
if not text.startswith("---\n"):
raise ValueError("curated evidence requires YAML frontmatter")
try:
_, frontmatter, body = text.split("---\n", 2)
except ValueError as error:
raise ValueError("curated evidence frontmatter is malformed") from error
raw = yaml.safe_load(frontmatter)
def _parse_v3_body(data: dict, body: str) -> dict:
kind = data.get("kind")
body_owned = {"payload", "review_items"}
if isinstance(kind, str):
body_owned.add(kind)
if body_owned.intersection(data):
raise ValueError("curated evidence v3 metadata contains body-owned fields")
title = data.get("title")
if not isinstance(title, str) or not body.startswith(f"# {title}\n"):
raise ValueError("curated evidence body title must match its metadata")
fields: dict[str, str] = {}
for match in _V2_FIELD.finditer(body):
name = match.group(1)
if name in fields:
raise ValueError(f"curated evidence field {name} appears more than once")
raw_content = match.group(2)
fields[name] = (
raw_content
if name in {"overview", "applies_to", "provenance"}
else _parse_v2_field_content(name, raw_content)
)
skeleton = _V2_FIELD.sub("", body).strip()
if skeleton != f"# {title}":
raise ValueError("curated evidence body contains unstructured content")
labels = _v2_labels(str(data.get("language", "")))
payload, payload_fields = _parse_v3_payload(kind, fields, labels)
common_fields = {"overview", "applies_to", "supporting_excerpts", "provenance"}
if "review_items" in fields:
common_fields.add("review_items")
if set(fields) != payload_fields | common_fields:
raise ValueError("curated evidence body fields do not match its kind")
provenance = data.get("provenance")
if not isinstance(provenance, dict) or "supporting_excerpts" in provenance:
raise ValueError("curated evidence v3 provenance is malformed")
provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"])
data["review_items"] = (
_parse_v2_review_items(fields["review_items"])
if "review_items" in fields
else []
)
data["payload"] = payload
return data
def _parse_v3_document(text: str) -> dict:
match = _V3_METADATA.match(text)
if match is None:
raise ValueError("curated evidence v3 metadata is malformed")
try:
decoded = base64.b64decode(match.group(1), validate=True).decode("utf-8")
raw = json.loads(decoded)
data = dict(raw)
except (TypeError, ValueError) as error:
raise ValueError("curated evidence frontmatter must be a mapping") from error
if data.get("schema_version") == 2:
data = _parse_v2_body(data, body)
except (binascii.Error, UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as error:
raise ValueError("curated evidence v3 metadata is malformed") from error
if data.get("schema_version") != 3:
raise ValueError("curated evidence v3 metadata has the wrong schema version")
return _parse_v3_body(data, text[match.end():])
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
"""Parse one canonical Curated Evidence Markdown document."""
if text.startswith("<!-- tht:metadata:"):
data = _parse_v3_document(text)
else:
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
kind = data.get("kind")
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
data["payload"] = data.pop(kind, None)
if not text.startswith("---\n"):
raise ValueError("curated evidence requires canonical metadata")
try:
_, frontmatter, body = text.split("---\n", 2)
except ValueError as error:
raise ValueError("curated evidence frontmatter is malformed") from error
raw = yaml.safe_load(frontmatter)
try:
data = dict(raw)
except (TypeError, ValueError) as error:
raise ValueError("curated evidence frontmatter must be a mapping") from error
if data.get("schema_version") == 2:
data = _parse_v2_body(data, body)
else:
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
kind = data.get("kind")
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
data["payload"] = data.pop(kind, None)
evidence = CuratedEvidence.model_validate(data)
if evidence.schema_version == 3 and dump_curated_markdown(evidence) != text:
raise ValueError("curated evidence v3 presentation is not canonical")
if path is not None:
_validate_kind_directory(path, evidence.kind)
return evidence
def dump_curated_markdown(value: CuratedEvidence) -> str:
"""Render canonical frontmatter with a human-readable kind-specific payload key."""
"""Render one canonical Curated Evidence Markdown document."""
if value.schema_version == 3:
return f"{_render_v3_metadata(value)}\n{_render_v3_body(value)}"
if value.schema_version == 2:
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
data["provenance"].pop("supporting_excerpts")