Files
ThothII/harness/tht/evidence/canonical.py
T

1062 lines
38 KiB
Python

"""Typed, reviewable Evidence units stored in the workspace repository."""
from __future__ import annotations
import base64
import binascii
import json
import re
from pathlib import Path, PurePosixPath
from typing import Literal
import sqlglot
import yaml
from pydantic import AnyHttpUrl, BaseModel, ConfigDict, Field, field_validator, model_validator
from sqlglot import exp
from tht.evidence.contracts import validate_canonical_uri
class StrictModel(BaseModel):
"""Reject undeclared fields in the repository's canonical format."""
model_config = ConfigDict(extra="forbid")
EVIDENCE_KINDS = (
"glossary",
"domain",
"enum",
"example",
"mapping",
"normalization",
"formula",
"reference",
)
EVIDENCE_PURPOSES = (
"disambiguation",
"rewriting",
"schema_linking",
"sql_generation",
)
EvidenceKind = Literal[*EVIDENCE_KINDS]
EvidencePurpose = Literal[*EVIDENCE_PURPOSES]
_IDENTIFIER = r"[A-Za-z_][A-Za-z0-9_$]*"
_TABLE_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}$")
_COLUMN_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}\.{_IDENTIFIER}$")
MAX_CURATED_FILE_BYTES = 10 * 1024 * 1024
class EvidenceScope(StrictModel):
concepts: tuple[str, ...] = ()
tables: tuple[str, ...] = ()
columns: tuple[str, ...] = ()
@field_validator("tables")
@classmethod
def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables")
@field_validator("columns")
@classmethod
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
class EvidenceProvenance(StrictModel):
model_config = ConfigDict(extra="forbid", frozen=True)
source_file: str
source_sha256: str
supporting_excerpts: tuple[str, ...]
@field_validator("source_file")
@classmethod
def _validate_source_file(cls, value: str) -> str:
return validate_source_file(value)
@field_validator("source_sha256")
@classmethod
def _validate_sha256(cls, value: str) -> str:
if not re.fullmatch(r"sha256:[0-9a-f]{64}", value):
raise ValueError("source_sha256 must be a sha256 digest")
return value
@field_validator("supporting_excerpts")
@classmethod
def _validate_excerpts(cls, value: tuple[str, ...]) -> tuple[str, ...]:
if not 1 <= len(value) <= 5:
raise ValueError("supporting_excerpts must contain one to five items")
if any(not excerpt.strip() or len(excerpt) > 1000 for excerpt in value):
raise ValueError("supporting excerpts must be nonempty and at most 1000 characters")
return value
class ReviewItem(StrictModel):
code: str
message: str
field: str | None = None
class FormulaPayload(StrictModel):
concept: str
columns: tuple[str, ...]
sql: str
@field_validator("columns")
@classmethod
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
@field_validator("sql")
@classmethod
def _validate_expression(cls, value: str) -> str:
try:
statements = [statement for statement in sqlglot.parse(value, read="postgres") if statement]
except sqlglot.errors.ParseError as error:
raise ValueError("formula.sql must be valid PostgreSQL") from error
if len(statements) != 1:
raise ValueError("formula.sql must contain exactly one expression")
expression = statements[0]
if expression.find(exp.Select) is not None or expression.find(exp.With) is not None:
raise ValueError("formula.sql must not contain a query")
if any(
expression.find(statement_type) is not None
for statement_type in (
exp.Insert,
exp.Update,
exp.Delete,
exp.Create,
exp.Drop,
exp.Alter,
exp.Merge,
exp.TruncateTable,
exp.Grant,
exp.Revoke,
exp.Command,
exp.Values,
exp.Set,
exp.Table,
)
):
raise ValueError("formula.sql must not contain DDL or DML")
return value
class ReferencePayload(StrictModel):
url: AnyHttpUrl
label: str
description: str
@field_validator("url")
@classmethod
def _reject_credentials(cls, value: AnyHttpUrl) -> AnyHttpUrl:
validate_canonical_uri(str(value))
return value
class GlossaryPayload(StrictModel):
definition: str
synonyms: tuple[str, ...] = ()
variants: tuple[str, ...] = ()
class DomainPayload(StrictModel):
rule: str
class EnumPayload(StrictModel):
column: str
values: dict[str, str]
@field_validator("column")
@classmethod
def _validate_column(cls, value: str) -> str:
_validate_identifiers((value,), _COLUMN_IDENTIFIER, "column")
return value
class ExamplePayload(StrictModel):
question: str
interpretation: str
class MappingPayload(StrictModel):
concept: str
tables: tuple[str, ...]
columns: tuple[str, ...]
@field_validator("tables")
@classmethod
def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables")
@field_validator("columns")
@classmethod
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
class NormalizationPayload(StrictModel):
input: str
output: str
rule: str
EvidencePayload = (
GlossaryPayload
| DomainPayload
| EnumPayload
| ExamplePayload
| MappingPayload
| NormalizationPayload
| FormulaPayload
| ReferencePayload
)
_PAYLOAD_TYPE_BY_KIND = {
"glossary": GlossaryPayload,
"domain": DomainPayload,
"enum": EnumPayload,
"example": ExamplePayload,
"mapping": MappingPayload,
"normalization": NormalizationPayload,
"formula": FormulaPayload,
"reference": ReferencePayload,
}
_EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
class CuratedEvidence(StrictModel):
schema_version: Literal[1, 2, 3]
id: str
title: str
kind: EvidenceKind
purposes: tuple[EvidencePurpose, ...]
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
language: str
provenance: EvidenceProvenance
review_items: tuple[ReviewItem, ...] = ()
payload: EvidencePayload
@model_validator(mode="after")
def _validate_kind_payload(self) -> CuratedEvidence:
if not is_evidence_id(self.id):
raise ValueError("id must use the evidence:<slug> form")
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
if expected is not None and not isinstance(self.payload, expected):
raise ValueError(f"{self.kind} requires its typed payload")
return self
_V2_LABELS = {
"en": {
"column": "Column",
"columns": "Columns",
"concept": "Concept",
"definition": "Definition",
"description": "Description",
"empty": "No items",
"input": "Input",
"interpretation": "Interpretation",
"label": "Label",
"output": "Output",
"question": "Question",
"review_items": "Review items",
"field": "Field",
"rule": "Rule",
"sql": "SQL",
"supporting_excerpts": "Supporting excerpts",
"synonyms": "Synonyms",
"tables": "Tables",
"url": "URL",
"value": "Value",
"values": "Values",
"meaning": "Meaning",
"variants": "Variants",
"applies_to": "Applies to",
"concepts": "Concepts",
"technical_details": "Technical details and provenance",
"purposes": "Purposes",
},
"it": {
"column": "Colonna",
"columns": "Colonne",
"concept": "Concetto",
"definition": "Definizione",
"description": "Descrizione",
"empty": "Nessun elemento",
"input": "Input",
"interpretation": "Interpretazione",
"label": "Etichetta",
"output": "Output",
"question": "Domanda",
"review_items": "Elementi da rivedere",
"field": "Campo",
"rule": "Regola",
"sql": "SQL",
"supporting_excerpts": "Estratti di supporto",
"synonyms": "Sinonimi",
"tables": "Tabelle",
"url": "URL",
"value": "Valore",
"values": "Valori",
"meaning": "Significato",
"variants": "Varianti",
"applies_to": "Ambito di applicazione",
"concepts": "Concetti",
"technical_details": "Dettagli tecnici e provenienza",
"purposes": "Scopi",
},
}
_V2_FIELD = re.compile(
r"<!-- tht:field:([a-z_]+) -->\n(.*?)\n<!-- /tht:field:\1 -->",
re.DOTALL,
)
_V2_EXCERPT_SEPARATOR = "<!-- tht:excerpt-separator -->"
_V2_EMPTY_LIST = "<!-- tht:empty-list -->"
_V2_REVIEW_SEPARATOR = "<!-- tht:review-separator -->"
_V2_REVIEW_FIELD = "<!-- tht:review-field -->"
_V3_METADATA = re.compile(r"\A<!-- tht:metadata:([A-Za-z0-9+/=]+) -->\n")
_V3_KIND_LABELS = {
"en": {
"glossary": "Glossary",
"domain": "Domain",
"enum": "Enumeration",
"example": "Example",
"mapping": "Mapping",
"normalization": "Normalization",
"formula": "Formula",
"reference": "Reference",
},
"it": {
"glossary": "Glossario",
"domain": "Dominio",
"enum": "Enumerazione",
"example": "Esempio",
"mapping": "Mappatura",
"normalization": "Normalizzazione",
"formula": "Formula",
"reference": "Riferimento",
},
}
_V3_PURPOSE_LABELS = {
"en": {
"disambiguation": "Disambiguation",
"rewriting": "Rewriting",
"schema_linking": "Schema linking",
"sql_generation": "SQL generation",
},
"it": {
"disambiguation": "Disambiguazione",
"rewriting": "Riscrittura",
"schema_linking": "Collegamento allo schema",
"sql_generation": "Generazione SQL",
},
}
def _v2_labels(language: str) -> dict[str, str]:
return _V2_LABELS["it" if language.lower().startswith("it") else "en"]
def _render_v2_field(name: str, label: str, content: str) -> str:
closing_marker = f"<!-- /tht:field:{name} -->"
if "<!-- tht:field:" in content or "<!-- /tht:field:" in content:
raise ValueError(f"curated evidence {name} contains a reserved marker")
return (
f"<!-- tht:field:{name} -->\n"
f"## {label}\n\n"
f"{content}\n"
f"{closing_marker}"
)
def _render_v2_excerpt(value: str) -> str:
if _V2_EXCERPT_SEPARATOR in value:
raise ValueError("supporting excerpt contains a reserved marker")
quoted = "\n".join(">" if not line else f"> {line}" for line in value.split("\n"))
return quoted
def _render_v2_list(
values: tuple[str, ...], *, code: bool = False, empty_label: str = "No items",
) -> str:
if not values:
return f"{_V2_EMPTY_LIST}\n_{empty_label}._"
if any("\n" in value for value in values):
raise ValueError("curated evidence list values must be single-line")
if code and any("`" in value for value in values):
raise ValueError("curated evidence code values must not contain backticks")
return "\n".join(f"- `{value}`" if code else f"- {value}" for value in values)
def _parse_v2_list(
value: str, *, code: bool = False, empty_label: str = "No items",
) -> tuple[str, ...]:
if value == f"{_V2_EMPTY_LIST}\n_{empty_label}._":
return ()
parsed: list[str] = []
for line in value.split("\n"):
if not line.startswith("- "):
raise ValueError("curated evidence list is malformed")
item = line[2:]
if code:
if len(item) < 2 or not item.startswith("`") or not item.endswith("`"):
raise ValueError("curated evidence code list is malformed")
item = item[1:-1]
parsed.append(item)
return tuple(parsed)
def _escape_v2_table_value(value: str) -> str:
return value.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "\\n")
def _unescape_v2_table_value(value: str) -> str:
output: list[str] = []
index = 0
while index < len(value):
if value[index] != "\\":
output.append(value[index])
index += 1
continue
if index + 1 >= len(value):
raise ValueError("curated evidence table escape is malformed")
escaped = value[index + 1]
if escaped not in {"\\", "|", "n"}:
raise ValueError("curated evidence table escape is malformed")
output.append("\n" if escaped == "n" else escaped)
index += 2
return "".join(output)
def _render_v2_values(values: dict[str, str], labels: dict[str, str]) -> str:
if any("`" in value for value in values):
raise ValueError("curated evidence enum values must not contain backticks")
rows = [
f"| {labels['value']} | {labels['meaning']} |",
"| --- | --- |",
]
if not values:
rows.append(f"| _{labels['empty']}._ | |")
return "\n".join(rows)
rows.extend(
f"| `{_escape_v2_table_value(value)}` | {_escape_v2_table_value(meaning)} |"
for value, meaning in sorted(values.items())
)
return "\n".join(rows)
def _parse_v2_values(value: str, labels: dict[str, str]) -> dict[str, str]:
lines = value.split("\n")
if (
len(lines) < 3
or lines[0] != f"| {labels['value']} | {labels['meaning']} |"
or lines[1] != "| --- | --- |"
):
raise ValueError("curated evidence values table is malformed")
if lines[2:] == [f"| _{labels['empty']}._ | |"]:
return {}
parsed: dict[str, str] = {}
for line in lines[2:]:
if not line.startswith("| ") or not line.endswith(" |"):
raise ValueError("curated evidence values table is malformed")
cells = re.split(r"(?<!\\)\s\|\s", line[2:-2], maxsplit=1)
if len(cells) != 2 or not cells[0].startswith("`") or not cells[0].endswith("`"):
raise ValueError("curated evidence values table is malformed")
key = _unescape_v2_table_value(cells[0][1:-1])
if key in parsed:
raise ValueError("curated evidence enum value appears more than once")
parsed[key] = _unescape_v2_table_value(cells[1])
return parsed
def _render_v3_values(values: dict[str, str], labels: dict[str, str]) -> str:
if not values:
return f"{_V2_EMPTY_LIST}\n_{labels['empty']}._"
if any("`" in value for value in values):
raise ValueError("curated evidence enum values must not contain backticks")
rendered: list[str] = []
for value, meaning in sorted(values.items()):
lines = meaning.split("\n")
rendered.append(f"- `{value}`: {lines[0]}")
rendered.extend(f" {line}" for line in lines[1:])
return "\n".join(rendered)
def _parse_v3_values(value: str, labels: dict[str, str]) -> dict[str, str]:
if value == f"{_V2_EMPTY_LIST}\n_{labels['empty']}._":
return {}
parsed: dict[str, list[str]] = {}
current: str | None = None
for line in value.split("\n"):
match = re.fullmatch(r"- `([^`]+)`: ?(.*)", line)
if match is not None:
current = match.group(1)
if current in parsed:
raise ValueError("curated evidence enum value appears more than once")
parsed[current] = [match.group(2)]
continue
if current is None or not line.startswith(" "):
raise ValueError("curated evidence values list is malformed")
parsed[current].append(line[2:])
if not parsed:
raise ValueError("curated evidence values list is malformed")
return {key: "\n".join(lines) for key, lines in parsed.items()}
def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
payload = value.payload
if value.kind == "glossary":
fields = [_render_v2_field("definition", labels["definition"], payload.definition)]
if payload.synonyms:
fields.append(_render_v2_field(
"synonyms", labels["synonyms"], _render_v2_list(payload.synonyms),
))
if payload.variants:
fields.append(_render_v2_field(
"variants", labels["variants"], _render_v2_list(payload.variants),
))
return fields
if value.kind == "domain":
return [_render_v2_field("rule", labels["rule"], payload.rule)]
if value.kind == "enum":
return [
_render_v2_field("column", labels["column"], f"`{payload.column}`"),
_render_v2_field("values", labels["values"], _render_v2_values(payload.values, labels)),
]
if value.kind == "example":
return [
_render_v2_field("question", labels["question"], payload.question),
_render_v2_field("interpretation", labels["interpretation"], payload.interpretation),
]
if value.kind == "mapping":
return [
_render_v2_field("concept", labels["concept"], payload.concept),
_render_v2_field("tables", labels["tables"], _render_v2_list(
payload.tables, code=True, empty_label=labels["empty"],
)),
_render_v2_field("columns", labels["columns"], _render_v2_list(
payload.columns, code=True, empty_label=labels["empty"],
)),
]
if value.kind == "normalization":
return [
_render_v2_field("input", labels["input"], payload.input),
_render_v2_field("output", labels["output"], payload.output),
_render_v2_field("rule", labels["rule"], payload.rule),
]
if value.kind == "formula":
if "```" in payload.sql:
raise ValueError("curated evidence SQL contains a reserved Markdown fence")
return [
_render_v2_field("concept", labels["concept"], payload.concept),
_render_v2_field("columns", labels["columns"], _render_v2_list(
payload.columns, code=True, empty_label=labels["empty"],
)),
_render_v2_field("sql", labels["sql"], f"```sql\n{payload.sql}\n```"),
]
if value.kind == "reference":
return [
_render_v2_field("label", labels["label"], payload.label),
_render_v2_field("url", labels["url"], f"<{payload.url}>"),
_render_v2_field("description", labels["description"], payload.description),
]
raise ValueError(f"unsupported curated evidence kind {value.kind}")
def _render_v3_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
if value.kind != "enum":
return _render_v2_payload(value, labels)
payload = value.payload
return [
_render_v2_field("column", labels["column"], f"`{payload.column}`"),
_render_v2_field("values", labels["values"], _render_v3_values(
payload.values, labels,
)),
]
def _render_v2_review_items(value: CuratedEvidence, labels: dict[str, str]) -> str:
rendered: list[str] = []
for item in value.review_items:
if "`" in item.code or (item.field is not None and "`" in item.field):
raise ValueError("curated evidence review identifiers must not contain backticks")
if _V2_REVIEW_SEPARATOR in item.message or _V2_REVIEW_FIELD in item.message:
raise ValueError("curated evidence review message contains a reserved marker")
block = f"### `{item.code}`\n\n{item.message}"
if item.field is not None:
block += f"\n\n{_V2_REVIEW_FIELD}\n**{labels['field']}:** `{item.field}`"
rendered.append(block)
return f"\n{_V2_REVIEW_SEPARATOR}\n".join(rendered)
def _render_v2_body(value: CuratedEvidence) -> str:
if "\n" in value.title:
raise ValueError("curated evidence title must be single-line in v2")
labels = _v2_labels(value.language)
fields = [
*_render_v2_payload(value, labels),
_render_v2_field(
"supporting_excerpts",
labels["supporting_excerpts"],
f"\n{_V2_EXCERPT_SEPARATOR}\n".join(
_render_v2_excerpt(excerpt)
for excerpt in value.provenance.supporting_excerpts
),
),
]
if value.review_items:
fields.append(_render_v2_field(
"review_items",
labels["review_items"],
_render_v2_review_items(value, labels),
))
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
def _render_v3_block(name: str, content: str) -> str:
if "<!-- tht:field:" in content or "<!-- /tht:field:" in content:
raise ValueError(f"curated evidence {name} contains a reserved marker")
return (
f"<!-- tht:field:{name} -->\n"
f"{content}\n"
f"<!-- /tht:field:{name} -->"
)
def _v3_locale(value: CuratedEvidence) -> str:
return "it" if value.language.lower().startswith("it") else "en"
def _render_v3_overview(value: CuratedEvidence, labels: dict[str, str]) -> str:
locale = _v3_locale(value)
language = "Italiano" if locale == "it" else "English"
kind = _V3_KIND_LABELS[locale][value.kind]
purposes = " · ".join(_V3_PURPOSE_LABELS[locale][purpose] for purpose in value.purposes)
if not purposes:
purposes = labels["empty"]
return _render_v3_block(
"overview",
f"> **{kind}** · {language}\n>\n> **{labels['purposes']}:** {purposes}",
)
def _render_v3_scope(value: CuratedEvidence, labels: dict[str, str]) -> str:
sections: list[str] = []
for label, values, code in (
(labels["concepts"], value.applies_to.concepts, False),
(labels["tables"], value.applies_to.tables, True),
(labels["columns"], value.applies_to.columns, True),
):
if values:
sections.append(
f"### {label}\n\n"
f"{_render_v2_list(values, code=code, empty_label=labels['empty'])}"
)
content = "\n\n".join(sections) if sections else f"_{labels['empty']}._"
return _render_v3_block(
"applies_to",
f"## {labels['applies_to']}\n\n{content}",
)
def _render_v3_provenance(value: CuratedEvidence, labels: dict[str, str]) -> str:
locale = _v3_locale(value)
technical_labels = {
"en": {
"schema": "Schema version",
"kind": "Kind",
"language": "Language",
"source": "Source file",
},
"it": {
"schema": "Versione schema",
"kind": "Tipo",
"language": "Lingua",
"source": "File sorgente",
},
}[locale]
content = (
"<details>\n"
f"<summary>{labels['technical_details']}</summary>\n\n"
f"- **ID:** `{value.id}`\n"
f"- **{technical_labels['schema']}:** `{value.schema_version}`\n"
f"- **{technical_labels['kind']}:** `{value.kind}`\n"
f"- **{technical_labels['language']}:** `{value.language}`\n"
f"- **{technical_labels['source']}:** `{value.provenance.source_file}`\n"
f"- **SHA-256:** `{value.provenance.source_sha256}`\n\n"
"</details>"
)
return _render_v3_block("provenance", content)
def _render_v3_metadata(value: CuratedEvidence) -> str:
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
data["provenance"].pop("supporting_excerpts")
encoded = base64.b64encode(json.dumps(
data,
ensure_ascii=False,
separators=(",", ":"),
sort_keys=True,
).encode("utf-8")).decode("ascii")
return f"<!-- tht:metadata:{encoded} -->"
def _render_v3_body(value: CuratedEvidence) -> str:
if "\n" in value.title:
raise ValueError("curated evidence title must be single-line in v3")
labels = _v2_labels(value.language)
fields = [
_render_v3_overview(value, labels),
_render_v3_scope(value, labels),
*_render_v3_payload(value, labels),
_render_v2_field(
"supporting_excerpts",
labels["supporting_excerpts"],
f"\n{_V2_EXCERPT_SEPARATOR}\n".join(
_render_v2_excerpt(excerpt)
for excerpt in value.provenance.supporting_excerpts
),
),
]
if value.review_items:
fields.append(_render_v2_field(
"review_items",
labels["review_items"],
_render_v2_review_items(value, labels),
))
fields.append(_render_v3_provenance(value, labels))
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
def _parse_v2_field_content(name: str, block: str) -> str:
try:
heading, content = block.split("\n\n", 1)
except ValueError as error:
raise ValueError(f"curated evidence field {name} is malformed") from error
if not heading.startswith("## ") or not content:
raise ValueError(f"curated evidence field {name} is malformed")
return content
def _parse_v2_excerpts(block: str) -> tuple[str, ...]:
excerpts: list[str] = []
for raw_excerpt in block.split(f"\n{_V2_EXCERPT_SEPARATOR}\n"):
lines = raw_excerpt.split("\n")
if any(line != ">" and not line.startswith("> ") for line in lines):
raise ValueError("curated evidence supporting excerpt is malformed")
excerpts.append("\n".join(line[2:] if line.startswith("> ") else "" for line in lines))
if not excerpts:
raise ValueError("curated evidence supporting excerpts are malformed")
return tuple(excerpts)
def _parse_inline_code(value: str, name: str) -> str:
if len(value) < 2 or not value.startswith("`") or not value.endswith("`"):
raise ValueError(f"curated evidence field {name} must be inline code")
return value[1:-1]
def _parse_v2_payload(
kind: str, fields: dict[str, str], labels: dict[str, str],
) -> tuple[dict, set[str]]:
if kind == "glossary":
expected = {"definition"}
payload: dict = {"definition": fields.get("definition")}
for name in ("synonyms", "variants"):
if name in fields:
expected.add(name)
payload[name] = _parse_v2_list(fields[name], empty_label=labels["empty"])
else:
payload[name] = ()
return payload, expected
if kind == "domain":
return {"rule": fields.get("rule")}, {"rule"}
if kind == "enum":
return {
"column": _parse_inline_code(fields.get("column", ""), "column"),
"values": _parse_v2_values(fields.get("values", ""), labels),
}, {"column", "values"}
if kind == "example":
return {
"question": fields.get("question"),
"interpretation": fields.get("interpretation"),
}, {"question", "interpretation"}
if kind == "mapping":
return {
"concept": fields.get("concept"),
"tables": _parse_v2_list(
fields.get("tables", ""), code=True, empty_label=labels["empty"],
),
"columns": _parse_v2_list(
fields.get("columns", ""), code=True, empty_label=labels["empty"],
),
}, {"concept", "tables", "columns"}
if kind == "normalization":
return {
"input": fields.get("input"),
"output": fields.get("output"),
"rule": fields.get("rule"),
}, {"input", "output", "rule"}
if kind == "formula":
sql = fields.get("sql", "")
if not sql.startswith("```sql\n") or not sql.endswith("\n```"):
raise ValueError("curated evidence SQL block is malformed")
return {
"concept": fields.get("concept"),
"columns": _parse_v2_list(
fields.get("columns", ""), code=True, empty_label=labels["empty"],
),
"sql": sql.removeprefix("```sql\n").removesuffix("\n```"),
}, {"concept", "columns", "sql"}
if kind == "reference":
url = fields.get("url", "")
if not url.startswith("<") or not url.endswith(">"):
raise ValueError("curated evidence reference URL is malformed")
return {
"label": fields.get("label"),
"url": url[1:-1],
"description": fields.get("description"),
}, {"label", "url", "description"}
raise ValueError("curated evidence body kind is unsupported")
def _parse_v3_payload(
kind: str, fields: dict[str, str], labels: dict[str, str],
) -> tuple[dict, set[str]]:
if kind != "enum":
return _parse_v2_payload(kind, fields, labels)
return {
"column": _parse_inline_code(fields.get("column", ""), "column"),
"values": _parse_v3_values(fields.get("values", ""), labels),
}, {"column", "values"}
def _parse_v2_review_items(value: str) -> tuple[ReviewItem, ...]:
items: list[ReviewItem] = []
for raw_item in value.split(f"\n{_V2_REVIEW_SEPARATOR}\n"):
try:
heading, detail = raw_item.split("\n\n", 1)
except ValueError as error:
raise ValueError("curated evidence review item is malformed") from error
if not heading.startswith("### `") or not heading.endswith("`"):
raise ValueError("curated evidence review item code is malformed")
code = heading.removeprefix("### `").removesuffix("`")
field = None
marker = f"\n\n{_V2_REVIEW_FIELD}\n"
if marker in detail:
message, rendered_field = detail.split(marker, 1)
match = re.fullmatch(r"\*\*[^*]+:\*\* `([^`]+)`", rendered_field)
if match is None:
raise ValueError("curated evidence review item field is malformed")
field = match.group(1)
else:
message = detail
if not code or not message:
raise ValueError("curated evidence review item is malformed")
items.append(ReviewItem(code=code, message=message, field=field))
return tuple(items)
def _parse_v2_body(data: dict, body: str) -> dict:
kind = data.get("kind")
body_owned = {"payload", "review_items"}
if isinstance(kind, str):
body_owned.add(kind)
if body_owned.intersection(data):
raise ValueError("curated evidence v2 frontmatter contains body-owned fields")
title = data.get("title")
if not isinstance(title, str) or not body.startswith(f"# {title}\n"):
raise ValueError("curated evidence body title must match its metadata")
fields: dict[str, str] = {}
for match in _V2_FIELD.finditer(body):
name = match.group(1)
if name in fields:
raise ValueError(f"curated evidence field {name} appears more than once")
fields[name] = _parse_v2_field_content(name, match.group(2))
skeleton = _V2_FIELD.sub("", body).strip()
if skeleton != f"# {title}":
raise ValueError("curated evidence body contains unstructured content")
labels = _v2_labels(str(data.get("language", "")))
payload, payload_fields = _parse_v2_payload(kind, fields, labels)
common_fields = {"supporting_excerpts"}
if "review_items" in fields:
common_fields.add("review_items")
if set(fields) != payload_fields | common_fields:
raise ValueError("curated evidence body fields do not match its kind")
provenance = data.get("provenance")
if not isinstance(provenance, dict) or "supporting_excerpts" in provenance:
raise ValueError("curated evidence v2 provenance is malformed")
provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"])
data["review_items"] = (
_parse_v2_review_items(fields["review_items"])
if "review_items" in fields
else []
)
data["payload"] = payload
return data
def _parse_v3_body(data: dict, body: str) -> dict:
kind = data.get("kind")
body_owned = {"payload", "review_items"}
if isinstance(kind, str):
body_owned.add(kind)
if body_owned.intersection(data):
raise ValueError("curated evidence v3 metadata contains body-owned fields")
title = data.get("title")
if not isinstance(title, str) or not body.startswith(f"# {title}\n"):
raise ValueError("curated evidence body title must match its metadata")
fields: dict[str, str] = {}
for match in _V2_FIELD.finditer(body):
name = match.group(1)
if name in fields:
raise ValueError(f"curated evidence field {name} appears more than once")
raw_content = match.group(2)
fields[name] = (
raw_content
if name in {"overview", "applies_to", "provenance"}
else _parse_v2_field_content(name, raw_content)
)
skeleton = _V2_FIELD.sub("", body).strip()
if skeleton != f"# {title}":
raise ValueError("curated evidence body contains unstructured content")
labels = _v2_labels(str(data.get("language", "")))
payload, payload_fields = _parse_v3_payload(kind, fields, labels)
common_fields = {"overview", "applies_to", "supporting_excerpts", "provenance"}
if "review_items" in fields:
common_fields.add("review_items")
if set(fields) != payload_fields | common_fields:
raise ValueError("curated evidence body fields do not match its kind")
provenance = data.get("provenance")
if not isinstance(provenance, dict) or "supporting_excerpts" in provenance:
raise ValueError("curated evidence v3 provenance is malformed")
provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"])
data["review_items"] = (
_parse_v2_review_items(fields["review_items"])
if "review_items" in fields
else []
)
data["payload"] = payload
return data
def _parse_v3_document(text: str) -> dict:
match = _V3_METADATA.match(text)
if match is None:
raise ValueError("curated evidence v3 metadata is malformed")
try:
decoded = base64.b64decode(match.group(1), validate=True).decode("utf-8")
raw = json.loads(decoded)
data = dict(raw)
except (binascii.Error, UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as error:
raise ValueError("curated evidence v3 metadata is malformed") from error
if data.get("schema_version") != 3:
raise ValueError("curated evidence v3 metadata has the wrong schema version")
return _parse_v3_body(data, text[match.end():])
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
"""Parse one canonical Curated Evidence Markdown document."""
if text.startswith("<!-- tht:metadata:"):
data = _parse_v3_document(text)
else:
if not text.startswith("---\n"):
raise ValueError("curated evidence requires canonical metadata")
try:
_, frontmatter, body = text.split("---\n", 2)
except ValueError as error:
raise ValueError("curated evidence frontmatter is malformed") from error
raw = yaml.safe_load(frontmatter)
try:
data = dict(raw)
except (TypeError, ValueError) as error:
raise ValueError("curated evidence frontmatter must be a mapping") from error
if data.get("schema_version") == 2:
data = _parse_v2_body(data, body)
else:
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
kind = data.get("kind")
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
data["payload"] = data.pop(kind, None)
evidence = CuratedEvidence.model_validate(data)
if evidence.schema_version == 3 and dump_curated_markdown(evidence) != text:
raise ValueError("curated evidence v3 presentation is not canonical")
if path is not None:
_validate_kind_directory(path, evidence.kind)
return evidence
def dump_curated_markdown(value: CuratedEvidence) -> str:
"""Render one canonical Curated Evidence Markdown document."""
if value.schema_version == 3:
return f"{_render_v3_metadata(value)}\n{_render_v3_body(value)}"
if value.schema_version == 2:
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
data["provenance"].pop("supporting_excerpts")
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
return f"---\n{frontmatter}---\n{_render_v2_body(value)}"
data = value.model_dump(mode="json", exclude={"payload"})
data[value.kind] = value.payload.model_dump(mode="json")
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
return f"---\n{frontmatter}---\n"
def load_curated_tree(root: Path) -> list[CuratedEvidence]:
"""Load canonical Evidence units in stable path order from a curated root."""
if not root.is_dir():
return []
documents: list[CuratedEvidence] = []
for path in sorted(root.rglob("*.md")):
if path.name.upper().startswith("README"):
continue
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
raise ValueError("curated evidence exceeds the size limit")
try:
text = path.read_text(encoding="utf-8")
except UnicodeDecodeError as error:
raise ValueError("curated evidence must be UTF-8") from error
documents.append(parse_curated_markdown(text, path=path))
return documents
def _validate_kind_directory(path: Path, kind: EvidenceKind) -> None:
parts = path.parts
try:
curated_index = parts.index("curated")
except ValueError:
return
if len(parts) <= curated_index + 1 or parts[curated_index + 1] != kind:
raise ValueError("curated evidence kind must match its directory")
def _validate_identifiers(
values: tuple[str, ...], pattern: re.Pattern[str], field: str,
) -> tuple[str, ...]:
if any(pattern.fullmatch(value) is None for value in values):
raise ValueError(f"{field} must use canonical schema identifiers")
return values
def validate_source_file(value: str) -> str:
"""Validate a repository-relative, credential-free Source Evidence path."""
path = PurePosixPath(value)
if (
path.is_absolute()
or ".." in path.parts
or not path.parts
or path.parts[0] != "source"
or not value.endswith((".md", ".txt", ".sql.md"))
):
raise ValueError("source_file must be a supported path below source/")
return value
def is_evidence_id(value: str) -> bool:
"""Whether a value uses the stable public Evidence identifier format."""
return _EVIDENCE_ID.fullmatch(value) is not None