1062 lines
38 KiB
Python
1062 lines
38 KiB
Python
"""Typed, reviewable Evidence units stored in the workspace repository."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import base64
|
|
import binascii
|
|
import json
|
|
import re
|
|
from pathlib import Path, PurePosixPath
|
|
from typing import Literal
|
|
|
|
import sqlglot
|
|
import yaml
|
|
from pydantic import AnyHttpUrl, BaseModel, ConfigDict, Field, field_validator, model_validator
|
|
from sqlglot import exp
|
|
|
|
from tht.evidence.contracts import validate_canonical_uri
|
|
|
|
|
|
class StrictModel(BaseModel):
|
|
"""Reject undeclared fields in the repository's canonical format."""
|
|
|
|
model_config = ConfigDict(extra="forbid")
|
|
|
|
|
|
EVIDENCE_KINDS = (
|
|
"glossary",
|
|
"domain",
|
|
"enum",
|
|
"example",
|
|
"mapping",
|
|
"normalization",
|
|
"formula",
|
|
"reference",
|
|
)
|
|
EVIDENCE_PURPOSES = (
|
|
"disambiguation",
|
|
"rewriting",
|
|
"schema_linking",
|
|
"sql_generation",
|
|
)
|
|
EvidenceKind = Literal[*EVIDENCE_KINDS]
|
|
EvidencePurpose = Literal[*EVIDENCE_PURPOSES]
|
|
_IDENTIFIER = r"[A-Za-z_][A-Za-z0-9_$]*"
|
|
_TABLE_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}$")
|
|
_COLUMN_IDENTIFIER = re.compile(rf"^{_IDENTIFIER}\.{_IDENTIFIER}\.{_IDENTIFIER}$")
|
|
MAX_CURATED_FILE_BYTES = 10 * 1024 * 1024
|
|
|
|
|
|
class EvidenceScope(StrictModel):
|
|
concepts: tuple[str, ...] = ()
|
|
tables: tuple[str, ...] = ()
|
|
columns: tuple[str, ...] = ()
|
|
|
|
@field_validator("tables")
|
|
@classmethod
|
|
def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables")
|
|
|
|
@field_validator("columns")
|
|
@classmethod
|
|
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
|
|
|
|
|
|
class EvidenceProvenance(StrictModel):
|
|
model_config = ConfigDict(extra="forbid", frozen=True)
|
|
|
|
source_file: str
|
|
source_sha256: str
|
|
supporting_excerpts: tuple[str, ...]
|
|
|
|
@field_validator("source_file")
|
|
@classmethod
|
|
def _validate_source_file(cls, value: str) -> str:
|
|
return validate_source_file(value)
|
|
|
|
@field_validator("source_sha256")
|
|
@classmethod
|
|
def _validate_sha256(cls, value: str) -> str:
|
|
if not re.fullmatch(r"sha256:[0-9a-f]{64}", value):
|
|
raise ValueError("source_sha256 must be a sha256 digest")
|
|
return value
|
|
|
|
@field_validator("supporting_excerpts")
|
|
@classmethod
|
|
def _validate_excerpts(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
if not 1 <= len(value) <= 5:
|
|
raise ValueError("supporting_excerpts must contain one to five items")
|
|
if any(not excerpt.strip() or len(excerpt) > 1000 for excerpt in value):
|
|
raise ValueError("supporting excerpts must be nonempty and at most 1000 characters")
|
|
return value
|
|
|
|
|
|
class ReviewItem(StrictModel):
|
|
code: str
|
|
message: str
|
|
field: str | None = None
|
|
|
|
|
|
class FormulaPayload(StrictModel):
|
|
concept: str
|
|
columns: tuple[str, ...]
|
|
sql: str
|
|
|
|
@field_validator("columns")
|
|
@classmethod
|
|
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
|
|
|
|
@field_validator("sql")
|
|
@classmethod
|
|
def _validate_expression(cls, value: str) -> str:
|
|
try:
|
|
statements = [statement for statement in sqlglot.parse(value, read="postgres") if statement]
|
|
except sqlglot.errors.ParseError as error:
|
|
raise ValueError("formula.sql must be valid PostgreSQL") from error
|
|
if len(statements) != 1:
|
|
raise ValueError("formula.sql must contain exactly one expression")
|
|
expression = statements[0]
|
|
if expression.find(exp.Select) is not None or expression.find(exp.With) is not None:
|
|
raise ValueError("formula.sql must not contain a query")
|
|
if any(
|
|
expression.find(statement_type) is not None
|
|
for statement_type in (
|
|
exp.Insert,
|
|
exp.Update,
|
|
exp.Delete,
|
|
exp.Create,
|
|
exp.Drop,
|
|
exp.Alter,
|
|
exp.Merge,
|
|
exp.TruncateTable,
|
|
exp.Grant,
|
|
exp.Revoke,
|
|
exp.Command,
|
|
exp.Values,
|
|
exp.Set,
|
|
exp.Table,
|
|
)
|
|
):
|
|
raise ValueError("formula.sql must not contain DDL or DML")
|
|
return value
|
|
|
|
|
|
class ReferencePayload(StrictModel):
|
|
url: AnyHttpUrl
|
|
label: str
|
|
description: str
|
|
|
|
@field_validator("url")
|
|
@classmethod
|
|
def _reject_credentials(cls, value: AnyHttpUrl) -> AnyHttpUrl:
|
|
validate_canonical_uri(str(value))
|
|
return value
|
|
|
|
|
|
class GlossaryPayload(StrictModel):
|
|
definition: str
|
|
synonyms: tuple[str, ...] = ()
|
|
variants: tuple[str, ...] = ()
|
|
|
|
|
|
class DomainPayload(StrictModel):
|
|
rule: str
|
|
|
|
|
|
class EnumPayload(StrictModel):
|
|
column: str
|
|
values: dict[str, str]
|
|
|
|
@field_validator("column")
|
|
@classmethod
|
|
def _validate_column(cls, value: str) -> str:
|
|
_validate_identifiers((value,), _COLUMN_IDENTIFIER, "column")
|
|
return value
|
|
|
|
|
|
class ExamplePayload(StrictModel):
|
|
question: str
|
|
interpretation: str
|
|
|
|
|
|
class MappingPayload(StrictModel):
|
|
concept: str
|
|
tables: tuple[str, ...]
|
|
columns: tuple[str, ...]
|
|
|
|
@field_validator("tables")
|
|
@classmethod
|
|
def _validate_tables(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
return _validate_identifiers(value, _TABLE_IDENTIFIER, "tables")
|
|
|
|
@field_validator("columns")
|
|
@classmethod
|
|
def _validate_columns(cls, value: tuple[str, ...]) -> tuple[str, ...]:
|
|
return _validate_identifiers(value, _COLUMN_IDENTIFIER, "columns")
|
|
|
|
|
|
class NormalizationPayload(StrictModel):
|
|
input: str
|
|
output: str
|
|
rule: str
|
|
|
|
|
|
EvidencePayload = (
|
|
GlossaryPayload
|
|
| DomainPayload
|
|
| EnumPayload
|
|
| ExamplePayload
|
|
| MappingPayload
|
|
| NormalizationPayload
|
|
| FormulaPayload
|
|
| ReferencePayload
|
|
)
|
|
|
|
|
|
_PAYLOAD_TYPE_BY_KIND = {
|
|
"glossary": GlossaryPayload,
|
|
"domain": DomainPayload,
|
|
"enum": EnumPayload,
|
|
"example": ExamplePayload,
|
|
"mapping": MappingPayload,
|
|
"normalization": NormalizationPayload,
|
|
"formula": FormulaPayload,
|
|
"reference": ReferencePayload,
|
|
}
|
|
_EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
|
|
|
|
|
|
class CuratedEvidence(StrictModel):
|
|
schema_version: Literal[1, 2, 3]
|
|
id: str
|
|
title: str
|
|
kind: EvidenceKind
|
|
purposes: tuple[EvidencePurpose, ...]
|
|
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
|
|
language: str
|
|
provenance: EvidenceProvenance
|
|
review_items: tuple[ReviewItem, ...] = ()
|
|
payload: EvidencePayload
|
|
|
|
@model_validator(mode="after")
|
|
def _validate_kind_payload(self) -> CuratedEvidence:
|
|
if not is_evidence_id(self.id):
|
|
raise ValueError("id must use the evidence:<slug> form")
|
|
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
|
|
if expected is not None and not isinstance(self.payload, expected):
|
|
raise ValueError(f"{self.kind} requires its typed payload")
|
|
return self
|
|
|
|
|
|
_V2_LABELS = {
|
|
"en": {
|
|
"column": "Column",
|
|
"columns": "Columns",
|
|
"concept": "Concept",
|
|
"definition": "Definition",
|
|
"description": "Description",
|
|
"empty": "No items",
|
|
"input": "Input",
|
|
"interpretation": "Interpretation",
|
|
"label": "Label",
|
|
"output": "Output",
|
|
"question": "Question",
|
|
"review_items": "Review items",
|
|
"field": "Field",
|
|
"rule": "Rule",
|
|
"sql": "SQL",
|
|
"supporting_excerpts": "Supporting excerpts",
|
|
"synonyms": "Synonyms",
|
|
"tables": "Tables",
|
|
"url": "URL",
|
|
"value": "Value",
|
|
"values": "Values",
|
|
"meaning": "Meaning",
|
|
"variants": "Variants",
|
|
"applies_to": "Applies to",
|
|
"concepts": "Concepts",
|
|
"technical_details": "Technical details and provenance",
|
|
"purposes": "Purposes",
|
|
},
|
|
"it": {
|
|
"column": "Colonna",
|
|
"columns": "Colonne",
|
|
"concept": "Concetto",
|
|
"definition": "Definizione",
|
|
"description": "Descrizione",
|
|
"empty": "Nessun elemento",
|
|
"input": "Input",
|
|
"interpretation": "Interpretazione",
|
|
"label": "Etichetta",
|
|
"output": "Output",
|
|
"question": "Domanda",
|
|
"review_items": "Elementi da rivedere",
|
|
"field": "Campo",
|
|
"rule": "Regola",
|
|
"sql": "SQL",
|
|
"supporting_excerpts": "Estratti di supporto",
|
|
"synonyms": "Sinonimi",
|
|
"tables": "Tabelle",
|
|
"url": "URL",
|
|
"value": "Valore",
|
|
"values": "Valori",
|
|
"meaning": "Significato",
|
|
"variants": "Varianti",
|
|
"applies_to": "Ambito di applicazione",
|
|
"concepts": "Concetti",
|
|
"technical_details": "Dettagli tecnici e provenienza",
|
|
"purposes": "Scopi",
|
|
},
|
|
}
|
|
_V2_FIELD = re.compile(
|
|
r"<!-- tht:field:([a-z_]+) -->\n(.*?)\n<!-- /tht:field:\1 -->",
|
|
re.DOTALL,
|
|
)
|
|
_V2_EXCERPT_SEPARATOR = "<!-- tht:excerpt-separator -->"
|
|
_V2_EMPTY_LIST = "<!-- tht:empty-list -->"
|
|
_V2_REVIEW_SEPARATOR = "<!-- tht:review-separator -->"
|
|
_V2_REVIEW_FIELD = "<!-- tht:review-field -->"
|
|
_V3_METADATA = re.compile(r"\A<!-- tht:metadata:([A-Za-z0-9+/=]+) -->\n")
|
|
_V3_KIND_LABELS = {
|
|
"en": {
|
|
"glossary": "Glossary",
|
|
"domain": "Domain",
|
|
"enum": "Enumeration",
|
|
"example": "Example",
|
|
"mapping": "Mapping",
|
|
"normalization": "Normalization",
|
|
"formula": "Formula",
|
|
"reference": "Reference",
|
|
},
|
|
"it": {
|
|
"glossary": "Glossario",
|
|
"domain": "Dominio",
|
|
"enum": "Enumerazione",
|
|
"example": "Esempio",
|
|
"mapping": "Mappatura",
|
|
"normalization": "Normalizzazione",
|
|
"formula": "Formula",
|
|
"reference": "Riferimento",
|
|
},
|
|
}
|
|
_V3_PURPOSE_LABELS = {
|
|
"en": {
|
|
"disambiguation": "Disambiguation",
|
|
"rewriting": "Rewriting",
|
|
"schema_linking": "Schema linking",
|
|
"sql_generation": "SQL generation",
|
|
},
|
|
"it": {
|
|
"disambiguation": "Disambiguazione",
|
|
"rewriting": "Riscrittura",
|
|
"schema_linking": "Collegamento allo schema",
|
|
"sql_generation": "Generazione SQL",
|
|
},
|
|
}
|
|
|
|
|
|
def _v2_labels(language: str) -> dict[str, str]:
|
|
return _V2_LABELS["it" if language.lower().startswith("it") else "en"]
|
|
|
|
|
|
def _render_v2_field(name: str, label: str, content: str) -> str:
|
|
closing_marker = f"<!-- /tht:field:{name} -->"
|
|
if "<!-- tht:field:" in content or "<!-- /tht:field:" in content:
|
|
raise ValueError(f"curated evidence {name} contains a reserved marker")
|
|
return (
|
|
f"<!-- tht:field:{name} -->\n"
|
|
f"## {label}\n\n"
|
|
f"{content}\n"
|
|
f"{closing_marker}"
|
|
)
|
|
|
|
|
|
def _render_v2_excerpt(value: str) -> str:
|
|
if _V2_EXCERPT_SEPARATOR in value:
|
|
raise ValueError("supporting excerpt contains a reserved marker")
|
|
quoted = "\n".join(">" if not line else f"> {line}" for line in value.split("\n"))
|
|
return quoted
|
|
|
|
|
|
def _render_v2_list(
|
|
values: tuple[str, ...], *, code: bool = False, empty_label: str = "No items",
|
|
) -> str:
|
|
if not values:
|
|
return f"{_V2_EMPTY_LIST}\n_{empty_label}._"
|
|
if any("\n" in value for value in values):
|
|
raise ValueError("curated evidence list values must be single-line")
|
|
if code and any("`" in value for value in values):
|
|
raise ValueError("curated evidence code values must not contain backticks")
|
|
return "\n".join(f"- `{value}`" if code else f"- {value}" for value in values)
|
|
|
|
|
|
def _parse_v2_list(
|
|
value: str, *, code: bool = False, empty_label: str = "No items",
|
|
) -> tuple[str, ...]:
|
|
if value == f"{_V2_EMPTY_LIST}\n_{empty_label}._":
|
|
return ()
|
|
parsed: list[str] = []
|
|
for line in value.split("\n"):
|
|
if not line.startswith("- "):
|
|
raise ValueError("curated evidence list is malformed")
|
|
item = line[2:]
|
|
if code:
|
|
if len(item) < 2 or not item.startswith("`") or not item.endswith("`"):
|
|
raise ValueError("curated evidence code list is malformed")
|
|
item = item[1:-1]
|
|
parsed.append(item)
|
|
return tuple(parsed)
|
|
|
|
|
|
def _escape_v2_table_value(value: str) -> str:
|
|
return value.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "\\n")
|
|
|
|
|
|
def _unescape_v2_table_value(value: str) -> str:
|
|
output: list[str] = []
|
|
index = 0
|
|
while index < len(value):
|
|
if value[index] != "\\":
|
|
output.append(value[index])
|
|
index += 1
|
|
continue
|
|
if index + 1 >= len(value):
|
|
raise ValueError("curated evidence table escape is malformed")
|
|
escaped = value[index + 1]
|
|
if escaped not in {"\\", "|", "n"}:
|
|
raise ValueError("curated evidence table escape is malformed")
|
|
output.append("\n" if escaped == "n" else escaped)
|
|
index += 2
|
|
return "".join(output)
|
|
|
|
|
|
def _render_v2_values(values: dict[str, str], labels: dict[str, str]) -> str:
|
|
if any("`" in value for value in values):
|
|
raise ValueError("curated evidence enum values must not contain backticks")
|
|
rows = [
|
|
f"| {labels['value']} | {labels['meaning']} |",
|
|
"| --- | --- |",
|
|
]
|
|
if not values:
|
|
rows.append(f"| _{labels['empty']}._ | |")
|
|
return "\n".join(rows)
|
|
rows.extend(
|
|
f"| `{_escape_v2_table_value(value)}` | {_escape_v2_table_value(meaning)} |"
|
|
for value, meaning in sorted(values.items())
|
|
)
|
|
return "\n".join(rows)
|
|
|
|
|
|
def _parse_v2_values(value: str, labels: dict[str, str]) -> dict[str, str]:
|
|
lines = value.split("\n")
|
|
if (
|
|
len(lines) < 3
|
|
or lines[0] != f"| {labels['value']} | {labels['meaning']} |"
|
|
or lines[1] != "| --- | --- |"
|
|
):
|
|
raise ValueError("curated evidence values table is malformed")
|
|
if lines[2:] == [f"| _{labels['empty']}._ | |"]:
|
|
return {}
|
|
parsed: dict[str, str] = {}
|
|
for line in lines[2:]:
|
|
if not line.startswith("| ") or not line.endswith(" |"):
|
|
raise ValueError("curated evidence values table is malformed")
|
|
cells = re.split(r"(?<!\\)\s\|\s", line[2:-2], maxsplit=1)
|
|
if len(cells) != 2 or not cells[0].startswith("`") or not cells[0].endswith("`"):
|
|
raise ValueError("curated evidence values table is malformed")
|
|
key = _unescape_v2_table_value(cells[0][1:-1])
|
|
if key in parsed:
|
|
raise ValueError("curated evidence enum value appears more than once")
|
|
parsed[key] = _unescape_v2_table_value(cells[1])
|
|
return parsed
|
|
|
|
|
|
def _render_v3_values(values: dict[str, str], labels: dict[str, str]) -> str:
|
|
if not values:
|
|
return f"{_V2_EMPTY_LIST}\n_{labels['empty']}._"
|
|
if any("`" in value for value in values):
|
|
raise ValueError("curated evidence enum values must not contain backticks")
|
|
rendered: list[str] = []
|
|
for value, meaning in sorted(values.items()):
|
|
lines = meaning.split("\n")
|
|
rendered.append(f"- `{value}`: {lines[0]}")
|
|
rendered.extend(f" {line}" for line in lines[1:])
|
|
return "\n".join(rendered)
|
|
|
|
|
|
def _parse_v3_values(value: str, labels: dict[str, str]) -> dict[str, str]:
|
|
if value == f"{_V2_EMPTY_LIST}\n_{labels['empty']}._":
|
|
return {}
|
|
parsed: dict[str, list[str]] = {}
|
|
current: str | None = None
|
|
for line in value.split("\n"):
|
|
match = re.fullmatch(r"- `([^`]+)`: ?(.*)", line)
|
|
if match is not None:
|
|
current = match.group(1)
|
|
if current in parsed:
|
|
raise ValueError("curated evidence enum value appears more than once")
|
|
parsed[current] = [match.group(2)]
|
|
continue
|
|
if current is None or not line.startswith(" "):
|
|
raise ValueError("curated evidence values list is malformed")
|
|
parsed[current].append(line[2:])
|
|
if not parsed:
|
|
raise ValueError("curated evidence values list is malformed")
|
|
return {key: "\n".join(lines) for key, lines in parsed.items()}
|
|
|
|
|
|
def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
|
|
payload = value.payload
|
|
if value.kind == "glossary":
|
|
fields = [_render_v2_field("definition", labels["definition"], payload.definition)]
|
|
if payload.synonyms:
|
|
fields.append(_render_v2_field(
|
|
"synonyms", labels["synonyms"], _render_v2_list(payload.synonyms),
|
|
))
|
|
if payload.variants:
|
|
fields.append(_render_v2_field(
|
|
"variants", labels["variants"], _render_v2_list(payload.variants),
|
|
))
|
|
return fields
|
|
if value.kind == "domain":
|
|
return [_render_v2_field("rule", labels["rule"], payload.rule)]
|
|
if value.kind == "enum":
|
|
return [
|
|
_render_v2_field("column", labels["column"], f"`{payload.column}`"),
|
|
_render_v2_field("values", labels["values"], _render_v2_values(payload.values, labels)),
|
|
]
|
|
if value.kind == "example":
|
|
return [
|
|
_render_v2_field("question", labels["question"], payload.question),
|
|
_render_v2_field("interpretation", labels["interpretation"], payload.interpretation),
|
|
]
|
|
if value.kind == "mapping":
|
|
return [
|
|
_render_v2_field("concept", labels["concept"], payload.concept),
|
|
_render_v2_field("tables", labels["tables"], _render_v2_list(
|
|
payload.tables, code=True, empty_label=labels["empty"],
|
|
)),
|
|
_render_v2_field("columns", labels["columns"], _render_v2_list(
|
|
payload.columns, code=True, empty_label=labels["empty"],
|
|
)),
|
|
]
|
|
if value.kind == "normalization":
|
|
return [
|
|
_render_v2_field("input", labels["input"], payload.input),
|
|
_render_v2_field("output", labels["output"], payload.output),
|
|
_render_v2_field("rule", labels["rule"], payload.rule),
|
|
]
|
|
if value.kind == "formula":
|
|
if "```" in payload.sql:
|
|
raise ValueError("curated evidence SQL contains a reserved Markdown fence")
|
|
return [
|
|
_render_v2_field("concept", labels["concept"], payload.concept),
|
|
_render_v2_field("columns", labels["columns"], _render_v2_list(
|
|
payload.columns, code=True, empty_label=labels["empty"],
|
|
)),
|
|
_render_v2_field("sql", labels["sql"], f"```sql\n{payload.sql}\n```"),
|
|
]
|
|
if value.kind == "reference":
|
|
return [
|
|
_render_v2_field("label", labels["label"], payload.label),
|
|
_render_v2_field("url", labels["url"], f"<{payload.url}>"),
|
|
_render_v2_field("description", labels["description"], payload.description),
|
|
]
|
|
raise ValueError(f"unsupported curated evidence kind {value.kind}")
|
|
|
|
|
|
def _render_v3_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
|
|
if value.kind != "enum":
|
|
return _render_v2_payload(value, labels)
|
|
payload = value.payload
|
|
return [
|
|
_render_v2_field("column", labels["column"], f"`{payload.column}`"),
|
|
_render_v2_field("values", labels["values"], _render_v3_values(
|
|
payload.values, labels,
|
|
)),
|
|
]
|
|
|
|
|
|
def _render_v2_review_items(value: CuratedEvidence, labels: dict[str, str]) -> str:
|
|
rendered: list[str] = []
|
|
for item in value.review_items:
|
|
if "`" in item.code or (item.field is not None and "`" in item.field):
|
|
raise ValueError("curated evidence review identifiers must not contain backticks")
|
|
if _V2_REVIEW_SEPARATOR in item.message or _V2_REVIEW_FIELD in item.message:
|
|
raise ValueError("curated evidence review message contains a reserved marker")
|
|
block = f"### `{item.code}`\n\n{item.message}"
|
|
if item.field is not None:
|
|
block += f"\n\n{_V2_REVIEW_FIELD}\n**{labels['field']}:** `{item.field}`"
|
|
rendered.append(block)
|
|
return f"\n{_V2_REVIEW_SEPARATOR}\n".join(rendered)
|
|
|
|
|
|
def _render_v2_body(value: CuratedEvidence) -> str:
|
|
if "\n" in value.title:
|
|
raise ValueError("curated evidence title must be single-line in v2")
|
|
labels = _v2_labels(value.language)
|
|
fields = [
|
|
*_render_v2_payload(value, labels),
|
|
_render_v2_field(
|
|
"supporting_excerpts",
|
|
labels["supporting_excerpts"],
|
|
f"\n{_V2_EXCERPT_SEPARATOR}\n".join(
|
|
_render_v2_excerpt(excerpt)
|
|
for excerpt in value.provenance.supporting_excerpts
|
|
),
|
|
),
|
|
]
|
|
if value.review_items:
|
|
fields.append(_render_v2_field(
|
|
"review_items",
|
|
labels["review_items"],
|
|
_render_v2_review_items(value, labels),
|
|
))
|
|
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
|
|
|
|
|
|
def _render_v3_block(name: str, content: str) -> str:
|
|
if "<!-- tht:field:" in content or "<!-- /tht:field:" in content:
|
|
raise ValueError(f"curated evidence {name} contains a reserved marker")
|
|
return (
|
|
f"<!-- tht:field:{name} -->\n"
|
|
f"{content}\n"
|
|
f"<!-- /tht:field:{name} -->"
|
|
)
|
|
|
|
|
|
def _v3_locale(value: CuratedEvidence) -> str:
|
|
return "it" if value.language.lower().startswith("it") else "en"
|
|
|
|
|
|
def _render_v3_overview(value: CuratedEvidence, labels: dict[str, str]) -> str:
|
|
locale = _v3_locale(value)
|
|
language = "Italiano" if locale == "it" else "English"
|
|
kind = _V3_KIND_LABELS[locale][value.kind]
|
|
purposes = " · ".join(_V3_PURPOSE_LABELS[locale][purpose] for purpose in value.purposes)
|
|
if not purposes:
|
|
purposes = labels["empty"]
|
|
return _render_v3_block(
|
|
"overview",
|
|
f"> **{kind}** · {language}\n>\n> **{labels['purposes']}:** {purposes}",
|
|
)
|
|
|
|
|
|
def _render_v3_scope(value: CuratedEvidence, labels: dict[str, str]) -> str:
|
|
sections: list[str] = []
|
|
for label, values, code in (
|
|
(labels["concepts"], value.applies_to.concepts, False),
|
|
(labels["tables"], value.applies_to.tables, True),
|
|
(labels["columns"], value.applies_to.columns, True),
|
|
):
|
|
if values:
|
|
sections.append(
|
|
f"### {label}\n\n"
|
|
f"{_render_v2_list(values, code=code, empty_label=labels['empty'])}"
|
|
)
|
|
content = "\n\n".join(sections) if sections else f"_{labels['empty']}._"
|
|
return _render_v3_block(
|
|
"applies_to",
|
|
f"## {labels['applies_to']}\n\n{content}",
|
|
)
|
|
|
|
|
|
def _render_v3_provenance(value: CuratedEvidence, labels: dict[str, str]) -> str:
|
|
locale = _v3_locale(value)
|
|
technical_labels = {
|
|
"en": {
|
|
"schema": "Schema version",
|
|
"kind": "Kind",
|
|
"language": "Language",
|
|
"source": "Source file",
|
|
},
|
|
"it": {
|
|
"schema": "Versione schema",
|
|
"kind": "Tipo",
|
|
"language": "Lingua",
|
|
"source": "File sorgente",
|
|
},
|
|
}[locale]
|
|
content = (
|
|
"<details>\n"
|
|
f"<summary>{labels['technical_details']}</summary>\n\n"
|
|
f"- **ID:** `{value.id}`\n"
|
|
f"- **{technical_labels['schema']}:** `{value.schema_version}`\n"
|
|
f"- **{technical_labels['kind']}:** `{value.kind}`\n"
|
|
f"- **{technical_labels['language']}:** `{value.language}`\n"
|
|
f"- **{technical_labels['source']}:** `{value.provenance.source_file}`\n"
|
|
f"- **SHA-256:** `{value.provenance.source_sha256}`\n\n"
|
|
"</details>"
|
|
)
|
|
return _render_v3_block("provenance", content)
|
|
|
|
|
|
def _render_v3_metadata(value: CuratedEvidence) -> str:
|
|
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
|
|
data["provenance"].pop("supporting_excerpts")
|
|
encoded = base64.b64encode(json.dumps(
|
|
data,
|
|
ensure_ascii=False,
|
|
separators=(",", ":"),
|
|
sort_keys=True,
|
|
).encode("utf-8")).decode("ascii")
|
|
return f"<!-- tht:metadata:{encoded} -->"
|
|
|
|
|
|
def _render_v3_body(value: CuratedEvidence) -> str:
|
|
if "\n" in value.title:
|
|
raise ValueError("curated evidence title must be single-line in v3")
|
|
labels = _v2_labels(value.language)
|
|
fields = [
|
|
_render_v3_overview(value, labels),
|
|
_render_v3_scope(value, labels),
|
|
*_render_v3_payload(value, labels),
|
|
_render_v2_field(
|
|
"supporting_excerpts",
|
|
labels["supporting_excerpts"],
|
|
f"\n{_V2_EXCERPT_SEPARATOR}\n".join(
|
|
_render_v2_excerpt(excerpt)
|
|
for excerpt in value.provenance.supporting_excerpts
|
|
),
|
|
),
|
|
]
|
|
if value.review_items:
|
|
fields.append(_render_v2_field(
|
|
"review_items",
|
|
labels["review_items"],
|
|
_render_v2_review_items(value, labels),
|
|
))
|
|
fields.append(_render_v3_provenance(value, labels))
|
|
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
|
|
|
|
|
|
def _parse_v2_field_content(name: str, block: str) -> str:
|
|
try:
|
|
heading, content = block.split("\n\n", 1)
|
|
except ValueError as error:
|
|
raise ValueError(f"curated evidence field {name} is malformed") from error
|
|
if not heading.startswith("## ") or not content:
|
|
raise ValueError(f"curated evidence field {name} is malformed")
|
|
return content
|
|
|
|
|
|
def _parse_v2_excerpts(block: str) -> tuple[str, ...]:
|
|
excerpts: list[str] = []
|
|
for raw_excerpt in block.split(f"\n{_V2_EXCERPT_SEPARATOR}\n"):
|
|
lines = raw_excerpt.split("\n")
|
|
if any(line != ">" and not line.startswith("> ") for line in lines):
|
|
raise ValueError("curated evidence supporting excerpt is malformed")
|
|
excerpts.append("\n".join(line[2:] if line.startswith("> ") else "" for line in lines))
|
|
if not excerpts:
|
|
raise ValueError("curated evidence supporting excerpts are malformed")
|
|
return tuple(excerpts)
|
|
|
|
|
|
def _parse_inline_code(value: str, name: str) -> str:
|
|
if len(value) < 2 or not value.startswith("`") or not value.endswith("`"):
|
|
raise ValueError(f"curated evidence field {name} must be inline code")
|
|
return value[1:-1]
|
|
|
|
|
|
def _parse_v2_payload(
|
|
kind: str, fields: dict[str, str], labels: dict[str, str],
|
|
) -> tuple[dict, set[str]]:
|
|
if kind == "glossary":
|
|
expected = {"definition"}
|
|
payload: dict = {"definition": fields.get("definition")}
|
|
for name in ("synonyms", "variants"):
|
|
if name in fields:
|
|
expected.add(name)
|
|
payload[name] = _parse_v2_list(fields[name], empty_label=labels["empty"])
|
|
else:
|
|
payload[name] = ()
|
|
return payload, expected
|
|
if kind == "domain":
|
|
return {"rule": fields.get("rule")}, {"rule"}
|
|
if kind == "enum":
|
|
return {
|
|
"column": _parse_inline_code(fields.get("column", ""), "column"),
|
|
"values": _parse_v2_values(fields.get("values", ""), labels),
|
|
}, {"column", "values"}
|
|
if kind == "example":
|
|
return {
|
|
"question": fields.get("question"),
|
|
"interpretation": fields.get("interpretation"),
|
|
}, {"question", "interpretation"}
|
|
if kind == "mapping":
|
|
return {
|
|
"concept": fields.get("concept"),
|
|
"tables": _parse_v2_list(
|
|
fields.get("tables", ""), code=True, empty_label=labels["empty"],
|
|
),
|
|
"columns": _parse_v2_list(
|
|
fields.get("columns", ""), code=True, empty_label=labels["empty"],
|
|
),
|
|
}, {"concept", "tables", "columns"}
|
|
if kind == "normalization":
|
|
return {
|
|
"input": fields.get("input"),
|
|
"output": fields.get("output"),
|
|
"rule": fields.get("rule"),
|
|
}, {"input", "output", "rule"}
|
|
if kind == "formula":
|
|
sql = fields.get("sql", "")
|
|
if not sql.startswith("```sql\n") or not sql.endswith("\n```"):
|
|
raise ValueError("curated evidence SQL block is malformed")
|
|
return {
|
|
"concept": fields.get("concept"),
|
|
"columns": _parse_v2_list(
|
|
fields.get("columns", ""), code=True, empty_label=labels["empty"],
|
|
),
|
|
"sql": sql.removeprefix("```sql\n").removesuffix("\n```"),
|
|
}, {"concept", "columns", "sql"}
|
|
if kind == "reference":
|
|
url = fields.get("url", "")
|
|
if not url.startswith("<") or not url.endswith(">"):
|
|
raise ValueError("curated evidence reference URL is malformed")
|
|
return {
|
|
"label": fields.get("label"),
|
|
"url": url[1:-1],
|
|
"description": fields.get("description"),
|
|
}, {"label", "url", "description"}
|
|
raise ValueError("curated evidence body kind is unsupported")
|
|
|
|
|
|
def _parse_v3_payload(
|
|
kind: str, fields: dict[str, str], labels: dict[str, str],
|
|
) -> tuple[dict, set[str]]:
|
|
if kind != "enum":
|
|
return _parse_v2_payload(kind, fields, labels)
|
|
return {
|
|
"column": _parse_inline_code(fields.get("column", ""), "column"),
|
|
"values": _parse_v3_values(fields.get("values", ""), labels),
|
|
}, {"column", "values"}
|
|
|
|
|
|
def _parse_v2_review_items(value: str) -> tuple[ReviewItem, ...]:
|
|
items: list[ReviewItem] = []
|
|
for raw_item in value.split(f"\n{_V2_REVIEW_SEPARATOR}\n"):
|
|
try:
|
|
heading, detail = raw_item.split("\n\n", 1)
|
|
except ValueError as error:
|
|
raise ValueError("curated evidence review item is malformed") from error
|
|
if not heading.startswith("### `") or not heading.endswith("`"):
|
|
raise ValueError("curated evidence review item code is malformed")
|
|
code = heading.removeprefix("### `").removesuffix("`")
|
|
field = None
|
|
marker = f"\n\n{_V2_REVIEW_FIELD}\n"
|
|
if marker in detail:
|
|
message, rendered_field = detail.split(marker, 1)
|
|
match = re.fullmatch(r"\*\*[^*]+:\*\* `([^`]+)`", rendered_field)
|
|
if match is None:
|
|
raise ValueError("curated evidence review item field is malformed")
|
|
field = match.group(1)
|
|
else:
|
|
message = detail
|
|
if not code or not message:
|
|
raise ValueError("curated evidence review item is malformed")
|
|
items.append(ReviewItem(code=code, message=message, field=field))
|
|
return tuple(items)
|
|
|
|
|
|
def _parse_v2_body(data: dict, body: str) -> dict:
|
|
kind = data.get("kind")
|
|
body_owned = {"payload", "review_items"}
|
|
if isinstance(kind, str):
|
|
body_owned.add(kind)
|
|
if body_owned.intersection(data):
|
|
raise ValueError("curated evidence v2 frontmatter contains body-owned fields")
|
|
title = data.get("title")
|
|
if not isinstance(title, str) or not body.startswith(f"# {title}\n"):
|
|
raise ValueError("curated evidence body title must match its metadata")
|
|
fields: dict[str, str] = {}
|
|
for match in _V2_FIELD.finditer(body):
|
|
name = match.group(1)
|
|
if name in fields:
|
|
raise ValueError(f"curated evidence field {name} appears more than once")
|
|
fields[name] = _parse_v2_field_content(name, match.group(2))
|
|
skeleton = _V2_FIELD.sub("", body).strip()
|
|
if skeleton != f"# {title}":
|
|
raise ValueError("curated evidence body contains unstructured content")
|
|
labels = _v2_labels(str(data.get("language", "")))
|
|
payload, payload_fields = _parse_v2_payload(kind, fields, labels)
|
|
common_fields = {"supporting_excerpts"}
|
|
if "review_items" in fields:
|
|
common_fields.add("review_items")
|
|
if set(fields) != payload_fields | common_fields:
|
|
raise ValueError("curated evidence body fields do not match its kind")
|
|
provenance = data.get("provenance")
|
|
if not isinstance(provenance, dict) or "supporting_excerpts" in provenance:
|
|
raise ValueError("curated evidence v2 provenance is malformed")
|
|
provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"])
|
|
data["review_items"] = (
|
|
_parse_v2_review_items(fields["review_items"])
|
|
if "review_items" in fields
|
|
else []
|
|
)
|
|
data["payload"] = payload
|
|
return data
|
|
|
|
|
|
def _parse_v3_body(data: dict, body: str) -> dict:
|
|
kind = data.get("kind")
|
|
body_owned = {"payload", "review_items"}
|
|
if isinstance(kind, str):
|
|
body_owned.add(kind)
|
|
if body_owned.intersection(data):
|
|
raise ValueError("curated evidence v3 metadata contains body-owned fields")
|
|
title = data.get("title")
|
|
if not isinstance(title, str) or not body.startswith(f"# {title}\n"):
|
|
raise ValueError("curated evidence body title must match its metadata")
|
|
fields: dict[str, str] = {}
|
|
for match in _V2_FIELD.finditer(body):
|
|
name = match.group(1)
|
|
if name in fields:
|
|
raise ValueError(f"curated evidence field {name} appears more than once")
|
|
raw_content = match.group(2)
|
|
fields[name] = (
|
|
raw_content
|
|
if name in {"overview", "applies_to", "provenance"}
|
|
else _parse_v2_field_content(name, raw_content)
|
|
)
|
|
skeleton = _V2_FIELD.sub("", body).strip()
|
|
if skeleton != f"# {title}":
|
|
raise ValueError("curated evidence body contains unstructured content")
|
|
labels = _v2_labels(str(data.get("language", "")))
|
|
payload, payload_fields = _parse_v3_payload(kind, fields, labels)
|
|
common_fields = {"overview", "applies_to", "supporting_excerpts", "provenance"}
|
|
if "review_items" in fields:
|
|
common_fields.add("review_items")
|
|
if set(fields) != payload_fields | common_fields:
|
|
raise ValueError("curated evidence body fields do not match its kind")
|
|
provenance = data.get("provenance")
|
|
if not isinstance(provenance, dict) or "supporting_excerpts" in provenance:
|
|
raise ValueError("curated evidence v3 provenance is malformed")
|
|
provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"])
|
|
data["review_items"] = (
|
|
_parse_v2_review_items(fields["review_items"])
|
|
if "review_items" in fields
|
|
else []
|
|
)
|
|
data["payload"] = payload
|
|
return data
|
|
|
|
|
|
def _parse_v3_document(text: str) -> dict:
|
|
match = _V3_METADATA.match(text)
|
|
if match is None:
|
|
raise ValueError("curated evidence v3 metadata is malformed")
|
|
try:
|
|
decoded = base64.b64decode(match.group(1), validate=True).decode("utf-8")
|
|
raw = json.loads(decoded)
|
|
data = dict(raw)
|
|
except (binascii.Error, UnicodeDecodeError, json.JSONDecodeError, TypeError, ValueError) as error:
|
|
raise ValueError("curated evidence v3 metadata is malformed") from error
|
|
if data.get("schema_version") != 3:
|
|
raise ValueError("curated evidence v3 metadata has the wrong schema version")
|
|
return _parse_v3_body(data, text[match.end():])
|
|
|
|
|
|
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
|
|
"""Parse one canonical Curated Evidence Markdown document."""
|
|
if text.startswith("<!-- tht:metadata:"):
|
|
data = _parse_v3_document(text)
|
|
else:
|
|
if not text.startswith("---\n"):
|
|
raise ValueError("curated evidence requires canonical metadata")
|
|
try:
|
|
_, frontmatter, body = text.split("---\n", 2)
|
|
except ValueError as error:
|
|
raise ValueError("curated evidence frontmatter is malformed") from error
|
|
raw = yaml.safe_load(frontmatter)
|
|
try:
|
|
data = dict(raw)
|
|
except (TypeError, ValueError) as error:
|
|
raise ValueError("curated evidence frontmatter must be a mapping") from error
|
|
if data.get("schema_version") == 2:
|
|
data = _parse_v2_body(data, body)
|
|
else:
|
|
if body.strip():
|
|
raise ValueError("curated evidence must not contain an ignored body")
|
|
kind = data.get("kind")
|
|
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
|
|
data["payload"] = data.pop(kind, None)
|
|
evidence = CuratedEvidence.model_validate(data)
|
|
if evidence.schema_version == 3 and dump_curated_markdown(evidence) != text:
|
|
raise ValueError("curated evidence v3 presentation is not canonical")
|
|
if path is not None:
|
|
_validate_kind_directory(path, evidence.kind)
|
|
return evidence
|
|
|
|
|
|
def dump_curated_markdown(value: CuratedEvidence) -> str:
|
|
"""Render one canonical Curated Evidence Markdown document."""
|
|
if value.schema_version == 3:
|
|
return f"{_render_v3_metadata(value)}\n{_render_v3_body(value)}"
|
|
if value.schema_version == 2:
|
|
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
|
|
data["provenance"].pop("supporting_excerpts")
|
|
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
|
|
return f"---\n{frontmatter}---\n{_render_v2_body(value)}"
|
|
data = value.model_dump(mode="json", exclude={"payload"})
|
|
data[value.kind] = value.payload.model_dump(mode="json")
|
|
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
|
|
return f"---\n{frontmatter}---\n"
|
|
|
|
|
|
def load_curated_tree(root: Path) -> list[CuratedEvidence]:
|
|
"""Load canonical Evidence units in stable path order from a curated root."""
|
|
if not root.is_dir():
|
|
return []
|
|
documents: list[CuratedEvidence] = []
|
|
for path in sorted(root.rglob("*.md")):
|
|
if path.name.upper().startswith("README"):
|
|
continue
|
|
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
|
raise ValueError("curated evidence exceeds the size limit")
|
|
try:
|
|
text = path.read_text(encoding="utf-8")
|
|
except UnicodeDecodeError as error:
|
|
raise ValueError("curated evidence must be UTF-8") from error
|
|
documents.append(parse_curated_markdown(text, path=path))
|
|
return documents
|
|
|
|
|
|
def _validate_kind_directory(path: Path, kind: EvidenceKind) -> None:
|
|
parts = path.parts
|
|
try:
|
|
curated_index = parts.index("curated")
|
|
except ValueError:
|
|
return
|
|
if len(parts) <= curated_index + 1 or parts[curated_index + 1] != kind:
|
|
raise ValueError("curated evidence kind must match its directory")
|
|
|
|
|
|
def _validate_identifiers(
|
|
values: tuple[str, ...], pattern: re.Pattern[str], field: str,
|
|
) -> tuple[str, ...]:
|
|
if any(pattern.fullmatch(value) is None for value in values):
|
|
raise ValueError(f"{field} must use canonical schema identifiers")
|
|
return values
|
|
|
|
|
|
def validate_source_file(value: str) -> str:
|
|
"""Validate a repository-relative, credential-free Source Evidence path."""
|
|
path = PurePosixPath(value)
|
|
if (
|
|
path.is_absolute()
|
|
or ".." in path.parts
|
|
or not path.parts
|
|
or path.parts[0] != "source"
|
|
or not value.endswith((".md", ".txt", ".sql.md"))
|
|
):
|
|
raise ValueError("source_file must be a supported path below source/")
|
|
return value
|
|
|
|
|
|
def is_evidence_id(value: str) -> bool:
|
|
"""Whether a value uses the stable public Evidence identifier format."""
|
|
return _EVIDENCE_ID.fullmatch(value) is not None
|