"""Typed, reviewable Evidence units stored in the workspace repository.""" from __future__ import annotations import re from pathlib import Path, PurePosixPath from typing import Literal import sqlglot import yaml from pydantic import AnyHttpUrl, BaseModel, ConfigDict, Field, field_validator, model_validator from sqlglot import exp from tht.evidence.contracts import validate_canonical_uri class StrictModel(BaseModel): """Reject undeclared fields in the repository's canonical format.""" model_config = ConfigDict(extra="forbid") EVIDENCE_KINDS = ( "glossary", "domain", "enum", "example", "mapping", "normalization", "formula", "reference", ) EVIDENCE_PURPOSES = ( "disambiguation", "rewriting", "schema_linking", "sql_generation", ) EvidenceKind = Literal[*EVIDENCE_KINDS] EvidencePurpose = Literal[*EVIDENCE_PURPOSES] _IDENTIFIER = r"[A-Za-z_][A-Za-z0-9_$]*" _TABLE_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}$") _COLUMN_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}\.{_IDENTIFIER}$") MAX_CURATED_FILE_BYTES = 10 * 1024 * 1024 class EvidenceScope(StrictModel): concepts: tuple[str, ...] = () tables: tuple[str, ...] = () columns: tuple[str, ...] = () @field_validator("tables") @classmethod def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]: return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables") @field_validator("columns") @classmethod def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]: return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns") class EvidenceProvenance(StrictModel): model_config = ConfigDict(extra="forbid", frozen=True) source_file: str source_sha256: str supporting_excerpts: tuple[str, ...] @field_validator("source_file") @classmethod def _validate_source_file(cls, value: str) -> str: return validate_source_file(value) @field_validator("source_sha256") @classmethod def _validate_sha256(cls, value: str) -> str: if not re.fullmatch(r"sha256:[0-9a-f]{64}", value): raise ValueError("source_sha256 must be a sha256 digest") return value @field_validator("supporting_excerpts") @classmethod def _validate_excerpts(cls, value: tuple[str, ...]) -> tuple[str, ...]: if not 1 <= len(value) <= 5: raise ValueError("supporting_excerpts must contain one to five items") if any(not excerpt.strip() or len(excerpt) > 1000 for excerpt in value): raise ValueError("supporting excerpts must be nonempty and at most 1000 characters") return value class ReviewItem(StrictModel): code: str message: str field: str | None = None class FormulaPayload(StrictModel): concept: str columns: tuple[str, ...] sql: str @field_validator("columns") @classmethod def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]: return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns") @field_validator("sql") @classmethod def _validate_expression(cls, value: str) -> str: try: statements = [statement for statement in sqlglot.parse(value, read="postgres") if statement] except sqlglot.errors.ParseError as error: raise ValueError("formula.sql must be valid PostgreSQL") from error if len(statements) != 1: raise ValueError("formula.sql must contain exactly one expression") expression = statements[0] if expression.find(exp.Select) is not None or expression.find(exp.With) is not None: raise ValueError("formula.sql must not contain a query") if any( expression.find(statement_type) is not None for statement_type in ( exp.Insert, exp.Update, exp.Delete, exp.Create, exp.Drop, exp.Alter, exp.Merge, exp.TruncateTable, exp.Grant, exp.Revoke, exp.Command, exp.Values, exp.Set, exp.Table, ) ): raise ValueError("formula.sql must not contain DDL or DML") return value class ReferencePayload(StrictModel): url: AnyHttpUrl label: str description: str @field_validator("url") @classmethod def _reject_credentials(cls, value: AnyHttpUrl) -> AnyHttpUrl: validate_canonical_uri(str(value)) return value class GlossaryPayload(StrictModel): definition: str synonyms: tuple[str, ...] = () variants: tuple[str, ...] = () class DomainPayload(StrictModel): rule: str class EnumPayload(StrictModel): column: str values: dict[str, str] @field_validator("column") @classmethod def _validate_column(cls, value: str) -> str: _validate_identifiers((value,), _COLUMN_IDENTIFIER, "column") return value class ExamplePayload(StrictModel): question: str interpretation: str class MappingPayload(StrictModel): concept: str tables: tuple[str, ...] columns: tuple[str, ...] @field_validator("tables") @classmethod def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]: return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables") @field_validator("columns") @classmethod def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]: return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns") class NormalizationPayload(StrictModel): input: str output: str rule: str EvidencePayload = ( GlossaryPayload | DomainPayload | EnumPayload | ExamplePayload | MappingPayload | NormalizationPayload | FormulaPayload | ReferencePayload ) _PAYLOAD_TYPE_BY_KIND = { "glossary": GlossaryPayload, "domain": DomainPayload, "enum": EnumPayload, "example": ExamplePayload, "mapping": MappingPayload, "normalization": NormalizationPayload, "formula": FormulaPayload, "reference": ReferencePayload, } _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$") class CuratedEvidence(StrictModel): schema_version: Literal[1, 2] id: str title: str kind: EvidenceKind purposes: tuple[EvidencePurpose, ...] applies_to: EvidenceScope = Field(default_factory=EvidenceScope) language: str provenance: EvidenceProvenance review_items: tuple[ReviewItem, ...] = () payload: EvidencePayload @model_validator(mode="after") def _validate_kind_payload(self) -> CuratedEvidence: if not is_evidence_id(self.id): raise ValueError("id must use the evidence: form") expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind) if expected is not None and not isinstance(self.payload, expected): raise ValueError(f"{self.kind} requires its typed payload") return self _V2_LABELS = { "en": { "column": "Column", "columns": "Columns", "concept": "Concept", "definition": "Definition", "description": "Description", "empty": "No items", "input": "Input", "interpretation": "Interpretation", "label": "Label", "output": "Output", "question": "Question", "review_items": "Review items", "field": "Field", "rule": "Rule", "sql": "SQL", "supporting_excerpts": "Supporting excerpts", "synonyms": "Synonyms", "tables": "Tables", "url": "URL", "value": "Value", "values": "Values", "meaning": "Meaning", "variants": "Variants", }, "it": { "column": "Colonna", "columns": "Colonne", "concept": "Concetto", "definition": "Definizione", "description": "Descrizione", "empty": "Nessun elemento", "input": "Input", "interpretation": "Interpretazione", "label": "Etichetta", "output": "Output", "question": "Domanda", "review_items": "Elementi da rivedere", "field": "Campo", "rule": "Regola", "sql": "SQL", "supporting_excerpts": "Estratti di supporto", "synonyms": "Sinonimi", "tables": "Tabelle", "url": "URL", "value": "Valore", "values": "Valori", "meaning": "Significato", "variants": "Varianti", }, } _V2_FIELD = re.compile( r"\n(.*?)\n", re.DOTALL, ) _V2_EXCERPT_SEPARATOR = "" _V2_EMPTY_LIST = "" _V2_REVIEW_SEPARATOR = "" _V2_REVIEW_FIELD = "" def _v2_labels(language: str) -> dict[str, str]: return _V2_LABELS["it" if language.lower().startswith("it") else "en"] def _render_v2_field(name: str, label: str, content: str) -> str: closing_marker = f"" if "\n" f"## {label}\n\n" f"{content}\n" f"{closing_marker}" ) def _render_v2_excerpt(value: str) -> str: if _V2_EXCERPT_SEPARATOR in value: raise ValueError("supporting excerpt contains a reserved marker") quoted = "\n".join(">" if not line else f"> {line}" for line in value.split("\n")) return quoted def _render_v2_list( values: tuple[str, ...], *, code: bool = False, empty_label: str = "No items", ) -> str: if not values: return f"{_V2_EMPTY_LIST}\n_{empty_label}._" if any("\n" in value for value in values): raise ValueError("curated evidence list values must be single-line") if code and any("`" in value for value in values): raise ValueError("curated evidence code values must not contain backticks") return "\n".join(f"- `{value}`" if code else f"- {value}" for value in values) def _parse_v2_list( value: str, *, code: bool = False, empty_label: str = "No items", ) -> tuple[str, ...]: if value == f"{_V2_EMPTY_LIST}\n_{empty_label}._": return () parsed: list[str] = [] for line in value.split("\n"): if not line.startswith("- "): raise ValueError("curated evidence list is malformed") item = line[2:] if code: if len(item) < 2 or not item.startswith("`") or not item.endswith("`"): raise ValueError("curated evidence code list is malformed") item = item[1:-1] parsed.append(item) return tuple(parsed) def _escape_v2_table_value(value: str) -> str: return value.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "\\n") def _unescape_v2_table_value(value: str) -> str: output: list[str] = [] index = 0 while index < len(value): if value[index] != "\\": output.append(value[index]) index += 1 continue if index + 1 >= len(value): raise ValueError("curated evidence table escape is malformed") escaped = value[index + 1] if escaped not in {"\\", "|", "n"}: raise ValueError("curated evidence table escape is malformed") output.append("\n" if escaped == "n" else escaped) index += 2 return "".join(output) def _render_v2_values(values: dict[str, str], labels: dict[str, str]) -> str: if any("`" in value for value in values): raise ValueError("curated evidence enum values must not contain backticks") rows = [ f"| {labels['value']} | {labels['meaning']} |", "| --- | --- |", ] if not values: rows.append(f"| _{labels['empty']}._ | |") return "\n".join(rows) rows.extend( f"| `{_escape_v2_table_value(value)}` | {_escape_v2_table_value(meaning)} |" for value, meaning in sorted(values.items()) ) return "\n".join(rows) def _parse_v2_values(value: str, labels: dict[str, str]) -> dict[str, str]: lines = value.split("\n") if ( len(lines) < 3 or lines[0] != f"| {labels['value']} | {labels['meaning']} |" or lines[1] != "| --- | --- |" ): raise ValueError("curated evidence values table is malformed") if lines[2:] == [f"| _{labels['empty']}._ | |"]: return {} parsed: dict[str, str] = {} for line in lines[2:]: if not line.startswith("| ") or not line.endswith(" |"): raise ValueError("curated evidence values table is malformed") cells = re.split(r"(? list[str]: payload = value.payload if value.kind == "glossary": fields = [_render_v2_field("definition", labels["definition"], payload.definition)] if payload.synonyms: fields.append(_render_v2_field( "synonyms", labels["synonyms"], _render_v2_list(payload.synonyms), )) if payload.variants: fields.append(_render_v2_field( "variants", labels["variants"], _render_v2_list(payload.variants), )) return fields if value.kind == "domain": return [_render_v2_field("rule", labels["rule"], payload.rule)] if value.kind == "enum": return [ _render_v2_field("column", labels["column"], f"`{payload.column}`"), _render_v2_field("values", labels["values"], _render_v2_values(payload.values, labels)), ] if value.kind == "example": return [ _render_v2_field("question", labels["question"], payload.question), _render_v2_field("interpretation", labels["interpretation"], payload.interpretation), ] if value.kind == "mapping": return [ _render_v2_field("concept", labels["concept"], payload.concept), _render_v2_field("tables", labels["tables"], _render_v2_list( payload.tables, code=True, empty_label=labels["empty"], )), _render_v2_field("columns", labels["columns"], _render_v2_list( payload.columns, code=True, empty_label=labels["empty"], )), ] if value.kind == "normalization": return [ _render_v2_field("input", labels["input"], payload.input), _render_v2_field("output", labels["output"], payload.output), _render_v2_field("rule", labels["rule"], payload.rule), ] if value.kind == "formula": if "```" in payload.sql: raise ValueError("curated evidence SQL contains a reserved Markdown fence") return [ _render_v2_field("concept", labels["concept"], payload.concept), _render_v2_field("columns", labels["columns"], _render_v2_list( payload.columns, code=True, empty_label=labels["empty"], )), _render_v2_field("sql", labels["sql"], f"```sql\n{payload.sql}\n```"), ] if value.kind == "reference": return [ _render_v2_field("label", labels["label"], payload.label), _render_v2_field("url", labels["url"], f"<{payload.url}>"), _render_v2_field("description", labels["description"], payload.description), ] raise ValueError(f"unsupported curated evidence kind {value.kind}") def _render_v2_review_items(value: CuratedEvidence, labels: dict[str, str]) -> str: rendered: list[str] = [] for item in value.review_items: if "`" in item.code or (item.field is not None and "`" in item.field): raise ValueError("curated evidence review identifiers must not contain backticks") if _V2_REVIEW_SEPARATOR in item.message or _V2_REVIEW_FIELD in item.message: raise ValueError("curated evidence review message contains a reserved marker") block = f"### `{item.code}`\n\n{item.message}" if item.field is not None: block += f"\n\n{_V2_REVIEW_FIELD}\n**{labels['field']}:** `{item.field}`" rendered.append(block) return f"\n{_V2_REVIEW_SEPARATOR}\n".join(rendered) def _render_v2_body(value: CuratedEvidence) -> str: if "\n" in value.title: raise ValueError("curated evidence title must be single-line in v2") labels = _v2_labels(value.language) fields = [ *_render_v2_payload(value, labels), _render_v2_field( "supporting_excerpts", labels["supporting_excerpts"], f"\n{_V2_EXCERPT_SEPARATOR}\n".join( _render_v2_excerpt(excerpt) for excerpt in value.provenance.supporting_excerpts ), ), ] if value.review_items: fields.append(_render_v2_field( "review_items", labels["review_items"], _render_v2_review_items(value, labels), )) return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n" def _parse_v2_field_content(name: str, block: str) -> str: try: heading, content = block.split("\n\n", 1) except ValueError as error: raise ValueError(f"curated evidence field {name} is malformed") from error if not heading.startswith("## ") or not content: raise ValueError(f"curated evidence field {name} is malformed") return content def _parse_v2_excerpts(block: str) -> tuple[str, ...]: excerpts: list[str] = [] for raw_excerpt in block.split(f"\n{_V2_EXCERPT_SEPARATOR}\n"): lines = raw_excerpt.split("\n") if any(line != ">" and not line.startswith("> ") for line in lines): raise ValueError("curated evidence supporting excerpt is malformed") excerpts.append("\n".join(line[2:] if line.startswith("> ") else "" for line in lines)) if not excerpts: raise ValueError("curated evidence supporting excerpts are malformed") return tuple(excerpts) def _parse_inline_code(value: str, name: str) -> str: if len(value) < 2 or not value.startswith("`") or not value.endswith("`"): raise ValueError(f"curated evidence field {name} must be inline code") return value[1:-1] def _parse_v2_payload( kind: str, fields: dict[str, str], labels: dict[str, str], ) -> tuple[dict, set[str]]: if kind == "glossary": expected = {"definition"} payload: dict = {"definition": fields.get("definition")} for name in ("synonyms", "variants"): if name in fields: expected.add(name) payload[name] = _parse_v2_list(fields[name], empty_label=labels["empty"]) else: payload[name] = () return payload, expected if kind == "domain": return {"rule": fields.get("rule")}, {"rule"} if kind == "enum": return { "column": _parse_inline_code(fields.get("column", ""), "column"), "values": _parse_v2_values(fields.get("values", ""), labels), }, {"column", "values"} if kind == "example": return { "question": fields.get("question"), "interpretation": fields.get("interpretation"), }, {"question", "interpretation"} if kind == "mapping": return { "concept": fields.get("concept"), "tables": _parse_v2_list( fields.get("tables", ""), code=True, empty_label=labels["empty"], ), "columns": _parse_v2_list( fields.get("columns", ""), code=True, empty_label=labels["empty"], ), }, {"concept", "tables", "columns"} if kind == "normalization": return { "input": fields.get("input"), "output": fields.get("output"), "rule": fields.get("rule"), }, {"input", "output", "rule"} if kind == "formula": sql = fields.get("sql", "") if not sql.startswith("```sql\n") or not sql.endswith("\n```"): raise ValueError("curated evidence SQL block is malformed") return { "concept": fields.get("concept"), "columns": _parse_v2_list( fields.get("columns", ""), code=True, empty_label=labels["empty"], ), "sql": sql.removeprefix("```sql\n").removesuffix("\n```"), }, {"concept", "columns", "sql"} if kind == "reference": url = fields.get("url", "") if not url.startswith("<") or not url.endswith(">"): raise ValueError("curated evidence reference URL is malformed") return { "label": fields.get("label"), "url": url[1:-1], "description": fields.get("description"), }, {"label", "url", "description"} raise ValueError("curated evidence body kind is unsupported") def _parse_v2_review_items(value: str) -> tuple[ReviewItem, ...]: items: list[ReviewItem] = [] for raw_item in value.split(f"\n{_V2_REVIEW_SEPARATOR}\n"): try: heading, detail = raw_item.split("\n\n", 1) except ValueError as error: raise ValueError("curated evidence review item is malformed") from error if not heading.startswith("### `") or not heading.endswith("`"): raise ValueError("curated evidence review item code is malformed") code = heading.removeprefix("### `").removesuffix("`") field = None marker = f"\n\n{_V2_REVIEW_FIELD}\n" if marker in detail: message, rendered_field = detail.split(marker, 1) match = re.fullmatch(r"\*\*[^*]+:\*\* `([^`]+)`", rendered_field) if match is None: raise ValueError("curated evidence review item field is malformed") field = match.group(1) else: message = detail if not code or not message: raise ValueError("curated evidence review item is malformed") items.append(ReviewItem(code=code, message=message, field=field)) return tuple(items) def _parse_v2_body(data: dict, body: str) -> dict: kind = data.get("kind") body_owned = {"payload", "review_items"} if isinstance(kind, str): body_owned.add(kind) if body_owned.intersection(data): raise ValueError("curated evidence v2 frontmatter contains body-owned fields") title = data.get("title") if not isinstance(title, str) or not body.startswith(f"# {title}\n"): raise ValueError("curated evidence body title must match its metadata") fields: dict[str, str] = {} for match in _V2_FIELD.finditer(body): name = match.group(1) if name in fields: raise ValueError(f"curated evidence field {name} appears more than once") fields[name] = _parse_v2_field_content(name, match.group(2)) skeleton = _V2_FIELD.sub("", body).strip() if skeleton != f"# {title}": raise ValueError("curated evidence body contains unstructured content") labels = _v2_labels(str(data.get("language", ""))) payload, payload_fields = _parse_v2_payload(kind, fields, labels) common_fields = {"supporting_excerpts"} if "review_items" in fields: common_fields.add("review_items") if set(fields) != payload_fields | common_fields: raise ValueError("curated evidence body fields do not match its kind") provenance = data.get("provenance") if not isinstance(provenance, dict) or "supporting_excerpts" in provenance: raise ValueError("curated evidence v2 provenance is malformed") provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"]) data["review_items"] = ( _parse_v2_review_items(fields["review_items"]) if "review_items" in fields else [] ) data["payload"] = payload return data def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence: """Parse the canonical frontmatter representation of one Curated Evidence unit.""" if not text.startswith("---\n"): raise ValueError("curated evidence requires YAML frontmatter") try: _, frontmatter, body = text.split("---\n", 2) except ValueError as error: raise ValueError("curated evidence frontmatter is malformed") from error raw = yaml.safe_load(frontmatter) try: data = dict(raw) except (TypeError, ValueError) as error: raise ValueError("curated evidence frontmatter must be a mapping") from error if data.get("schema_version") == 2: data = _parse_v2_body(data, body) else: if body.strip(): raise ValueError("curated evidence must not contain an ignored body") kind = data.get("kind") if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND: data["payload"] = data.pop(kind, None) evidence = CuratedEvidence.model_validate(data) if path is not None: _validate_kind_directory(path, evidence.kind) return evidence def dump_curated_markdown(value: CuratedEvidence) -> str: """Render canonical frontmatter with a human-readable kind-specific payload key.""" if value.schema_version == 2: data = value.model_dump(mode="json", exclude={"payload", "review_items"}) data["provenance"].pop("supporting_excerpts") frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False) return f"---\n{frontmatter}---\n{_render_v2_body(value)}" data = value.model_dump(mode="json", exclude={"payload"}) data[value.kind] = value.payload.model_dump(mode="json") frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False) return f"---\n{frontmatter}---\n" def load_curated_tree(root: Path) -> list[CuratedEvidence]: """Load canonical Evidence units in stable path order from a curated root.""" if not root.is_dir(): return [] documents: list[CuratedEvidence] = [] for path in sorted(root.rglob("*.md")): if path.name.upper().startswith("README"): continue if path.stat().st_size > MAX_CURATED_FILE_BYTES: raise ValueError("curated evidence exceeds the size limit") try: text = path.read_text(encoding="utf-8") except UnicodeDecodeError as error: raise ValueError("curated evidence must be UTF-8") from error documents.append(parse_curated_markdown(text, path=path)) return documents def _validate_kind_directory(path: Path, kind: EvidenceKind) -> None: parts = path.parts try: curated_index = parts.index("curated") except ValueError: return if len(parts) <= curated_index + 1 or parts[curated_index + 1] != kind: raise ValueError("curated evidence kind must match its directory") def _validate_identifiers( values: tuple[str, ...], pattern: re.Pattern[str], field: str, ) -> tuple[str, ...]: if any(pattern.fullmatch(value) is None for value in values): raise ValueError(f"{field} must use canonical schema identifiers") return values def validate_source_file(value: str) -> str: """Validate a repository-relative, credential-free Source Evidence path.""" path = PurePosixPath(value) if ( path.is_absolute() or ".." in path.parts or not path.parts or path.parts[0] != "source" or not value.endswith((".md", ".txt", ".sql.md")) ): raise ValueError("source_file must be a supported path below source/") return value def is_evidence_id(value: str) -> bool: """Whether a value uses the stable public Evidence identifier format.""" return _EVIDENCE_ID.fullmatch(value) is not None