feat: complete evidence restructuring worktree

This commit is contained in:
Codex
2026-08-26 11:39:02 +02:00
parent a54d4769dd
commit 38f02cfd08
56 changed files with 1981 additions and 1801 deletions
+4
View File
@@ -3,6 +3,7 @@
from tht.evidence.acquisition import acquire, discover
from tht.evidence.authoring import (
EvidenceManifest,
EvidenceMigrationReport,
EvidencePreparationError,
EvidencePreparationReport,
EvidenceResolutionReport,
@@ -14,6 +15,7 @@ from tht.evidence.authoring import (
ValidationReport,
dump_manifest,
load_manifest,
migrate_workspace_evidence,
prepare_workspace_evidence,
resolve_workspace_evidence,
validate_workspace_evidence,
@@ -60,6 +62,7 @@ __all__ = [
"CuratedEvidence",
"EvidenceEmbedder",
"EvidenceManifest",
"EvidenceMigrationReport",
"EvidencePreparationError",
"EvidencePreparationReport",
"EvidenceQueryEmbedder",
@@ -89,6 +92,7 @@ __all__ = [
"dump_manifest",
"load_curated_tree",
"load_manifest",
"migrate_workspace_evidence",
"normalize_aware_datetime",
"parse_curated_markdown",
"prepare_workspace_evidence",
+77 -3
View File
@@ -293,6 +293,15 @@ class EvidencePreparationReport:
model_calls: int
@dataclass(frozen=True)
class EvidenceMigrationReport:
"""Result of a deterministic Curated Evidence presentation-format migration."""
migrated: tuple[str, ...]
unchanged: tuple[str, ...]
findings: tuple[ValidationFinding, ...]
@dataclass(frozen=True)
class EvidenceResolutionReport:
"""The result of one curator-directed Evidence resolution."""
@@ -680,6 +689,48 @@ def prepare_workspace_evidence(
)
def migrate_workspace_evidence(
workspace_root: Path,
*,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
) -> EvidenceMigrationReport:
"""Rewrite v1 Curated units as readable v2 Markdown without changing semantics."""
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
try:
manifest = load_manifest(evidence_root / "manifest.yaml")
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
documents_by_id = {document.id: document for document in documents}
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
migrated = tuple(sorted(
document.id for document in documents if document.schema_version == 1
))
unchanged = tuple(sorted(
document.id for document in documents if document.schema_version == 2
))
if not migrated:
return EvidenceMigrationReport(
migrated=(),
unchanged=unchanged,
findings=validate_workspace_evidence(workspace_root).findings,
)
upgraded = {
evidence_id: document.model_copy(update={"schema_version": 2})
for evidence_id, document in documents_by_id.items()
}
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
return EvidenceMigrationReport(
migrated=migrated,
unchanged=unchanged,
findings=findings,
)
def resolve_workspace_evidence(
workspace_root: Path,
evidence_id: str,
@@ -827,7 +878,7 @@ def _load_resolution_source(evidence_root: Path, source_file: str) -> str:
def _git_status(workspace_root: Path) -> tuple[str, ...]:
result = subprocess.run(
["git", "status", "--porcelain"],
["git", "status", "--porcelain", "--untracked-files=all"],
cwd=workspace_root,
check=False,
capture_output=True,
@@ -835,7 +886,25 @@ def _git_status(workspace_root: Path) -> tuple[str, ...]:
)
if result.returncode != 0:
raise EvidencePreparationError("canonical_git_worktree_required")
return tuple(line for line in result.stdout.splitlines() if line)
prefix_result = subprocess.run(
["git", "rev-parse", "--show-prefix"],
cwd=workspace_root,
check=False,
capture_output=True,
text=True,
)
if prefix_result.returncode != 0:
raise EvidencePreparationError("canonical_git_worktree_required")
prefix = prefix_result.stdout.strip()
entries: list[str] = []
for line in result.stdout.splitlines():
if not line:
continue
path = line[3:] if len(line) > 3 else line
if prefix and path.startswith(prefix):
line = line[:3] + path.removeprefix(prefix)
entries.append(line)
return tuple(entries)
def _reject_dirty_authoring_state(
@@ -934,7 +1003,11 @@ def _candidate_to_evidence(
source_file: str,
source_hash: str,
) -> CuratedEvidence:
data = candidate.model_dump(mode="json", exclude={"existing_id", "supporting_excerpts"})
data = candidate.model_dump(
mode="json",
exclude={"schema_version", "existing_id", "supporting_excerpts"},
)
data["schema_version"] = 2
data["id"] = evidence_id
data["provenance"] = {
"source_file": source_file,
@@ -955,6 +1028,7 @@ def _unsupported_unit(
message="The current source no longer supports this Evidence unit.",
),)
return evidence.model_copy(update={
"schema_version": 2,
"provenance": evidence.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": source_hash,
+447 -6
View File
@@ -226,7 +226,7 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
class CuratedEvidence(StrictModel):
schema_version: Literal[1]
schema_version: Literal[1, 2]
id: str
title: str
kind: EvidenceKind
@@ -247,6 +247,439 @@ class CuratedEvidence(StrictModel):
return self
_V2_LABELS = {
"en": {
"column": "Column",
"columns": "Columns",
"concept": "Concept",
"definition": "Definition",
"description": "Description",
"empty": "No items",
"input": "Input",
"interpretation": "Interpretation",
"label": "Label",
"output": "Output",
"question": "Question",
"review_items": "Review items",
"field": "Field",
"rule": "Rule",
"sql": "SQL",
"supporting_excerpts": "Supporting excerpts",
"synonyms": "Synonyms",
"tables": "Tables",
"url": "URL",
"value": "Value",
"values": "Values",
"meaning": "Meaning",
"variants": "Variants",
},
"it": {
"column": "Colonna",
"columns": "Colonne",
"concept": "Concetto",
"definition": "Definizione",
"description": "Descrizione",
"empty": "Nessun elemento",
"input": "Input",
"interpretation": "Interpretazione",
"label": "Etichetta",
"output": "Output",
"question": "Domanda",
"review_items": "Elementi da rivedere",
"field": "Campo",
"rule": "Regola",
"sql": "SQL",
"supporting_excerpts": "Estratti di supporto",
"synonyms": "Sinonimi",
"tables": "Tabelle",
"url": "URL",
"value": "Valore",
"values": "Valori",
"meaning": "Significato",
"variants": "Varianti",
},
}
_V2_FIELD = re.compile(
r"<!-- tht:field:([a-z_]+) -->\n(.*?)\n<!-- /tht:field:\1 -->",
re.DOTALL,
)
_V2_EXCERPT_SEPARATOR = "<!-- tht:excerpt-separator -->"
_V2_EMPTY_LIST = "<!-- tht:empty-list -->"
_V2_REVIEW_SEPARATOR = "<!-- tht:review-separator -->"
_V2_REVIEW_FIELD = "<!-- tht:review-field -->"
def _v2_labels(language: str) -> dict[str, str]:
return _V2_LABELS["it" if language.lower().startswith("it") else "en"]
def _render_v2_field(name: str, label: str, content: str) -> str:
closing_marker = f"<!-- /tht:field:{name} -->"
if "<!-- tht:field:" in content or "<!-- /tht:field:" in content:
raise ValueError(f"curated evidence {name} contains a reserved marker")
return (
f"<!-- tht:field:{name} -->\n"
f"## {label}\n\n"
f"{content}\n"
f"{closing_marker}"
)
def _render_v2_excerpt(value: str) -> str:
if _V2_EXCERPT_SEPARATOR in value:
raise ValueError("supporting excerpt contains a reserved marker")
quoted = "\n".join(">" if not line else f"> {line}" for line in value.split("\n"))
return quoted
def _render_v2_list(
values: tuple[str, ...], *, code: bool = False, empty_label: str = "No items",
) -> str:
if not values:
return f"{_V2_EMPTY_LIST}\n_{empty_label}._"
if any("\n" in value for value in values):
raise ValueError("curated evidence list values must be single-line")
if code and any("`" in value for value in values):
raise ValueError("curated evidence code values must not contain backticks")
return "\n".join(f"- `{value}`" if code else f"- {value}" for value in values)
def _parse_v2_list(
value: str, *, code: bool = False, empty_label: str = "No items",
) -> tuple[str, ...]:
if value == f"{_V2_EMPTY_LIST}\n_{empty_label}._":
return ()
parsed: list[str] = []
for line in value.split("\n"):
if not line.startswith("- "):
raise ValueError("curated evidence list is malformed")
item = line[2:]
if code:
if len(item) < 2 or not item.startswith("`") or not item.endswith("`"):
raise ValueError("curated evidence code list is malformed")
item = item[1:-1]
parsed.append(item)
return tuple(parsed)
def _escape_v2_table_value(value: str) -> str:
return value.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "\\n")
def _unescape_v2_table_value(value: str) -> str:
output: list[str] = []
index = 0
while index < len(value):
if value[index] != "\\":
output.append(value[index])
index += 1
continue
if index + 1 >= len(value):
raise ValueError("curated evidence table escape is malformed")
escaped = value[index + 1]
if escaped not in {"\\", "|", "n"}:
raise ValueError("curated evidence table escape is malformed")
output.append("\n" if escaped == "n" else escaped)
index += 2
return "".join(output)
def _render_v2_values(values: dict[str, str], labels: dict[str, str]) -> str:
if any("`" in value for value in values):
raise ValueError("curated evidence enum values must not contain backticks")
rows = [
f"| {labels['value']} | {labels['meaning']} |",
"| --- | --- |",
]
if not values:
rows.append(f"| _{labels['empty']}._ | |")
return "\n".join(rows)
rows.extend(
f"| `{_escape_v2_table_value(value)}` | {_escape_v2_table_value(meaning)} |"
for value, meaning in sorted(values.items())
)
return "\n".join(rows)
def _parse_v2_values(value: str, labels: dict[str, str]) -> dict[str, str]:
lines = value.split("\n")
if (
len(lines) < 3
or lines[0] != f"| {labels['value']} | {labels['meaning']} |"
or lines[1] != "| --- | --- |"
):
raise ValueError("curated evidence values table is malformed")
if lines[2:] == [f"| _{labels['empty']}._ | |"]:
return {}
parsed: dict[str, str] = {}
for line in lines[2:]:
if not line.startswith("| ") or not line.endswith(" |"):
raise ValueError("curated evidence values table is malformed")
cells = re.split(r"(?<!\\)\s\|\s", line[2:-2], maxsplit=1)
if len(cells) != 2 or not cells[0].startswith("`") or not cells[0].endswith("`"):
raise ValueError("curated evidence values table is malformed")
key = _unescape_v2_table_value(cells[0][1:-1])
if key in parsed:
raise ValueError("curated evidence enum value appears more than once")
parsed[key] = _unescape_v2_table_value(cells[1])
return parsed
def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
payload = value.payload
if value.kind == "glossary":
fields = [_render_v2_field("definition", labels["definition"], payload.definition)]
if payload.synonyms:
fields.append(_render_v2_field(
"synonyms", labels["synonyms"], _render_v2_list(payload.synonyms),
))
if payload.variants:
fields.append(_render_v2_field(
"variants", labels["variants"], _render_v2_list(payload.variants),
))
return fields
if value.kind == "domain":
return [_render_v2_field("rule", labels["rule"], payload.rule)]
if value.kind == "enum":
return [
_render_v2_field("column", labels["column"], f"`{payload.column}`"),
_render_v2_field("values", labels["values"], _render_v2_values(payload.values, labels)),
]
if value.kind == "example":
return [
_render_v2_field("question", labels["question"], payload.question),
_render_v2_field("interpretation", labels["interpretation"], payload.interpretation),
]
if value.kind == "mapping":
return [
_render_v2_field("concept", labels["concept"], payload.concept),
_render_v2_field("tables", labels["tables"], _render_v2_list(
payload.tables, code=True, empty_label=labels["empty"],
)),
_render_v2_field("columns", labels["columns"], _render_v2_list(
payload.columns, code=True, empty_label=labels["empty"],
)),
]
if value.kind == "normalization":
return [
_render_v2_field("input", labels["input"], payload.input),
_render_v2_field("output", labels["output"], payload.output),
_render_v2_field("rule", labels["rule"], payload.rule),
]
if value.kind == "formula":
if "```" in payload.sql:
raise ValueError("curated evidence SQL contains a reserved Markdown fence")
return [
_render_v2_field("concept", labels["concept"], payload.concept),
_render_v2_field("columns", labels["columns"], _render_v2_list(
payload.columns, code=True, empty_label=labels["empty"],
)),
_render_v2_field("sql", labels["sql"], f"```sql\n{payload.sql}\n```"),
]
if value.kind == "reference":
return [
_render_v2_field("label", labels["label"], payload.label),
_render_v2_field("url", labels["url"], f"<{payload.url}>"),
_render_v2_field("description", labels["description"], payload.description),
]
raise ValueError(f"unsupported curated evidence kind {value.kind}")
def _render_v2_review_items(value: CuratedEvidence, labels: dict[str, str]) -> str:
rendered: list[str] = []
for item in value.review_items:
if "`" in item.code or (item.field is not None and "`" in item.field):
raise ValueError("curated evidence review identifiers must not contain backticks")
if _V2_REVIEW_SEPARATOR in item.message or _V2_REVIEW_FIELD in item.message:
raise ValueError("curated evidence review message contains a reserved marker")
block = f"### `{item.code}`\n\n{item.message}"
if item.field is not None:
block += f"\n\n{_V2_REVIEW_FIELD}\n**{labels['field']}:** `{item.field}`"
rendered.append(block)
return f"\n{_V2_REVIEW_SEPARATOR}\n".join(rendered)
def _render_v2_body(value: CuratedEvidence) -> str:
if "\n" in value.title:
raise ValueError("curated evidence title must be single-line in v2")
labels = _v2_labels(value.language)
fields = [
*_render_v2_payload(value, labels),
_render_v2_field(
"supporting_excerpts",
labels["supporting_excerpts"],
f"\n{_V2_EXCERPT_SEPARATOR}\n".join(
_render_v2_excerpt(excerpt)
for excerpt in value.provenance.supporting_excerpts
),
),
]
if value.review_items:
fields.append(_render_v2_field(
"review_items",
labels["review_items"],
_render_v2_review_items(value, labels),
))
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
def _parse_v2_field_content(name: str, block: str) -> str:
try:
heading, content = block.split("\n\n", 1)
except ValueError as error:
raise ValueError(f"curated evidence field {name} is malformed") from error
if not heading.startswith("## ") or not content:
raise ValueError(f"curated evidence field {name} is malformed")
return content
def _parse_v2_excerpts(block: str) -> tuple[str, ...]:
excerpts: list[str] = []
for raw_excerpt in block.split(f"\n{_V2_EXCERPT_SEPARATOR}\n"):
lines = raw_excerpt.split("\n")
if any(line != ">" and not line.startswith("> ") for line in lines):
raise ValueError("curated evidence supporting excerpt is malformed")
excerpts.append("\n".join(line[2:] if line.startswith("> ") else "" for line in lines))
if not excerpts:
raise ValueError("curated evidence supporting excerpts are malformed")
return tuple(excerpts)
def _parse_inline_code(value: str, name: str) -> str:
if len(value) < 2 or not value.startswith("`") or not value.endswith("`"):
raise ValueError(f"curated evidence field {name} must be inline code")
return value[1:-1]
def _parse_v2_payload(
kind: str, fields: dict[str, str], labels: dict[str, str],
) -> tuple[dict, set[str]]:
if kind == "glossary":
expected = {"definition"}
payload: dict = {"definition": fields.get("definition")}
for name in ("synonyms", "variants"):
if name in fields:
expected.add(name)
payload[name] = _parse_v2_list(fields[name], empty_label=labels["empty"])
else:
payload[name] = ()
return payload, expected
if kind == "domain":
return {"rule": fields.get("rule")}, {"rule"}
if kind == "enum":
return {
"column": _parse_inline_code(fields.get("column", ""), "column"),
"values": _parse_v2_values(fields.get("values", ""), labels),
}, {"column", "values"}
if kind == "example":
return {
"question": fields.get("question"),
"interpretation": fields.get("interpretation"),
}, {"question", "interpretation"}
if kind == "mapping":
return {
"concept": fields.get("concept"),
"tables": _parse_v2_list(
fields.get("tables", ""), code=True, empty_label=labels["empty"],
),
"columns": _parse_v2_list(
fields.get("columns", ""), code=True, empty_label=labels["empty"],
),
}, {"concept", "tables", "columns"}
if kind == "normalization":
return {
"input": fields.get("input"),
"output": fields.get("output"),
"rule": fields.get("rule"),
}, {"input", "output", "rule"}
if kind == "formula":
sql = fields.get("sql", "")
if not sql.startswith("```sql\n") or not sql.endswith("\n```"):
raise ValueError("curated evidence SQL block is malformed")
return {
"concept": fields.get("concept"),
"columns": _parse_v2_list(
fields.get("columns", ""), code=True, empty_label=labels["empty"],
),
"sql": sql.removeprefix("```sql\n").removesuffix("\n```"),
}, {"concept", "columns", "sql"}
if kind == "reference":
url = fields.get("url", "")
if not url.startswith("<") or not url.endswith(">"):
raise ValueError("curated evidence reference URL is malformed")
return {
"label": fields.get("label"),
"url": url[1:-1],
"description": fields.get("description"),
}, {"label", "url", "description"}
raise ValueError("curated evidence body kind is unsupported")
def _parse_v2_review_items(value: str) -> tuple[ReviewItem, ...]:
items: list[ReviewItem] = []
for raw_item in value.split(f"\n{_V2_REVIEW_SEPARATOR}\n"):
try:
heading, detail = raw_item.split("\n\n", 1)
except ValueError as error:
raise ValueError("curated evidence review item is malformed") from error
if not heading.startswith("### `") or not heading.endswith("`"):
raise ValueError("curated evidence review item code is malformed")
code = heading.removeprefix("### `").removesuffix("`")
field = None
marker = f"\n\n{_V2_REVIEW_FIELD}\n"
if marker in detail:
message, rendered_field = detail.split(marker, 1)
match = re.fullmatch(r"\*\*[^*]+:\*\* `([^`]+)`", rendered_field)
if match is None:
raise ValueError("curated evidence review item field is malformed")
field = match.group(1)
else:
message = detail
if not code or not message:
raise ValueError("curated evidence review item is malformed")
items.append(ReviewItem(code=code, message=message, field=field))
return tuple(items)
def _parse_v2_body(data: dict, body: str) -> dict:
kind = data.get("kind")
body_owned = {"payload", "review_items"}
if isinstance(kind, str):
body_owned.add(kind)
if body_owned.intersection(data):
raise ValueError("curated evidence v2 frontmatter contains body-owned fields")
title = data.get("title")
if not isinstance(title, str) or not body.startswith(f"# {title}\n"):
raise ValueError("curated evidence body title must match its metadata")
fields: dict[str, str] = {}
for match in _V2_FIELD.finditer(body):
name = match.group(1)
if name in fields:
raise ValueError(f"curated evidence field {name} appears more than once")
fields[name] = _parse_v2_field_content(name, match.group(2))
skeleton = _V2_FIELD.sub("", body).strip()
if skeleton != f"# {title}":
raise ValueError("curated evidence body contains unstructured content")
labels = _v2_labels(str(data.get("language", "")))
payload, payload_fields = _parse_v2_payload(kind, fields, labels)
common_fields = {"supporting_excerpts"}
if "review_items" in fields:
common_fields.add("review_items")
if set(fields) != payload_fields | common_fields:
raise ValueError("curated evidence body fields do not match its kind")
provenance = data.get("provenance")
if not isinstance(provenance, dict) or "supporting_excerpts" in provenance:
raise ValueError("curated evidence v2 provenance is malformed")
provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"])
data["review_items"] = (
_parse_v2_review_items(fields["review_items"])
if "review_items" in fields
else []
)
data["payload"] = payload
return data
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
"""Parse the canonical frontmatter representation of one Curated Evidence unit."""
if not text.startswith("---\n"):
@@ -260,11 +693,14 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
data = dict(raw)
except (TypeError, ValueError) as error:
raise ValueError("curated evidence frontmatter must be a mapping") from error
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
kind = data.get("kind")
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
data["payload"] = data.pop(kind, None)
if data.get("schema_version") == 2:
data = _parse_v2_body(data, body)
else:
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
kind = data.get("kind")
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
data["payload"] = data.pop(kind, None)
evidence = CuratedEvidence.model_validate(data)
if path is not None:
_validate_kind_directory(path, evidence.kind)
@@ -273,6 +709,11 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
def dump_curated_markdown(value: CuratedEvidence) -> str:
"""Render canonical frontmatter with a human-readable kind-specific payload key."""
if value.schema_version == 2:
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
data["provenance"].pop("supporting_excerpts")
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
return f"---\n{frontmatter}---\n{_render_v2_body(value)}"
data = value.model_dump(mode="json", exclude={"payload"})
data[value.kind] = value.payload.model_dump(mode="json")
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
-202
View File
@@ -1,202 +0,0 @@
"""Legacy ConceptFormula reader and one-way migration into Curated Evidence.
The ``formulas/*.sql.md`` store is retained only for the migration window. Runtime
lookup uses typed, published ``kind=formula`` Evidence instead. A session reviewer
may still approve a formula locally; that is a proposal, not publication.
"""
from __future__ import annotations
import hashlib
import re
import unicodedata
from dataclasses import dataclass
from pathlib import Path
from typing import Literal
import yaml
from pydantic import BaseModel, ValidationError
from tht.evidence.canonical import CuratedEvidence
FORMULAS_SUBDIR = "formulas"
_SUFFIX_RE = re.compile(r"^(.*?)-(\d+)\.sql\.md$")
class ConceptFormula(BaseModel):
concept: str
columns: list[str] = []
sql: str
# auto = sintetizzata dal modello (non ancora rivista); draft = bozza umana;
# reviewed = approvata da un revisore. (spec §4.7.2: status auto/draft/reviewed)
status: Literal["auto", "draft", "reviewed"] = "draft"
sources: list[str] = []
@property
def _slug(self) -> str:
"""ASCII slug for the filename (matches textutil.slugify shape)."""
import unicodedata
text = unicodedata.normalize("NFKD", self.concept).encode("ascii", "ignore").decode()
return re.sub(r"[^a-z0-9_]+", "-", text.lower()).strip("-") or "formula"
def dump(self) -> str:
meta = self.model_dump(exclude={"sql"}, mode="json")
fm = yaml.safe_dump(meta, sort_keys=False, allow_unicode=True)
return f"---\n{fm}---\n{self.sql}\n"
@classmethod
def parse(cls, text: str) -> ConceptFormula:
if not text.startswith("---\n"):
raise ValueError("frontmatter mancante (atteso '---\\n' iniziale)")
try:
_, fm, body = text.split("---\n", 2)
except ValueError as e:
raise ValueError("frontmatter malformato") from e
meta = yaml.safe_load(fm)
if not isinstance(meta, dict):
raise TypeError("frontmatter non valido")
return cls.model_validate({**meta, "sql": body.strip("\n")})
@dataclass(frozen=True)
class LegacyFormulaMigrationFailure:
"""A reviewed legacy formula that must be resolved manually before publication."""
code: Literal["legacy_formula_requires_manual_review"]
legacy_path: str
formula: ConceptFormula
problems: tuple[str, ...]
def _next_path(root: Path, slug: str) -> Path:
"""First free <slug>-<n>.sql.md path under root (n starts at 1)."""
root.mkdir(parents=True, exist_ok=True)
existing = sorted(root.glob(f"{slug}-*.sql.md"))
n = 0
for p in existing:
m = _SUFFIX_RE.match(p.name)
if m:
n = max(n, int(m.group(2)))
return root / f"{slug}-{n + 1}.sql.md"
def save_formula(root: Path | str, formula: ConceptFormula) -> Path:
"""Persist a single concept->formula unit under <root>/formulas/. Returns the
written path. Append-only: each save writes a new file (so competing drafts and
reviewed versions coexist until a curator prunes)."""
root = Path(root)
formulas_dir = root / FORMULAS_SUBDIR
path = _next_path(formulas_dir, formula._slug)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(formula.dump())
return path
def _load_all(root: Path) -> list[ConceptFormula]:
formulas_dir = root / FORMULAS_SUBDIR
if not formulas_dir.is_dir():
return []
out: list[ConceptFormula] = []
for f in sorted(formulas_dir.glob("*.sql.md")):
try:
out.append(ConceptFormula.parse(f.read_text()))
except ValueError:
continue # malformed file: skip, don't crash retrieval
return out
def retrieve_formula(root: Path | str, concept: str) -> list[ConceptFormula]:
"""All formulas matching `concept` exactly under <root>/formulas/. Empty list if
none (or if the dir is absent). Multiple results mean competing drafts/versions for
the same concept -- the caller (gate) lets the reviewer pick."""
return [f for f in _load_all(Path(root)) if f.concept == concept]
def search_formulas(root: Path | str, query: str) -> list[ConceptFormula]:
"""Read legacy formulas for migration tooling only (case-insensitive concept match)."""
q = query.strip().lower()
return [f for f in _load_all(Path(root)) if q in f.concept.lower()]
def legacy_formula_to_curated(
formula: ConceptFormula,
*,
legacy_path: str,
source_content: str | None = None,
) -> CuratedEvidence | LegacyFormulaMigrationFailure | None:
"""Convert one reviewed legacy formula into its deterministic curated counterpart.
Drafts and model-generated formulas have no global publication status. Their
caller must project them as session-local Formula proposals instead.
"""
if formula.status != "reviewed":
return None
problems: list[str] = []
normalized_source: str | None = None
if source_content is None:
problems.append("original_source_required")
else:
from tht.evidence.authoring import normalize_source_text
normalized_source = normalize_source_text(source_content)
try:
original_formula = ConceptFormula.parse(source_content)
except (TypeError, ValidationError, ValueError, yaml.YAMLError):
problems.append("original_source_invalid")
else:
if original_formula != formula:
problems.append("original_source_mismatch")
source_notes = tuple(formula.sources)
if not source_notes:
problems.append("supporting_excerpts_required")
elif normalized_source is not None and any(
unicodedata.normalize("NFC", note.replace("\r\n", "\n").replace("\r", "\n"))
not in normalized_source
for note in source_notes
):
problems.append("supporting_excerpt_unverified")
if problems:
return LegacyFormulaMigrationFailure(
code="legacy_formula_requires_manual_review",
legacy_path=legacy_path,
formula=formula,
problems=tuple(sorted(problems)),
)
assert normalized_source is not None
source_sha256 = hashlib.sha256(normalized_source.encode("utf-8")).hexdigest()
source_file = legacy_path if legacy_path.startswith("source/") else f"source/{legacy_path}"
# The legacy path is the immutable identity of this unit during migration. Keeping
# its full digest avoids a duplicate public ID when the same concept has reviewed
# competing formulas, while retaining the readable concept slug as the prefix.
legacy_identity = hashlib.sha256(legacy_path.encode("utf-8")).hexdigest()
try:
return CuratedEvidence.model_validate({
"schema_version": 1,
"id": f"evidence:{formula._slug}-{legacy_identity}",
"title": formula.concept[:1].upper() + formula.concept[1:],
"kind": "formula",
"purposes": ["schema_linking", "sql_generation"],
"applies_to": {"concepts": [formula.concept], "columns": formula.columns},
"language": "it",
"provenance": {
"source_file": source_file,
"source_sha256": f"sha256:{source_sha256}",
"supporting_excerpts": source_notes,
},
"review_items": [],
"payload": {
"concept": formula.concept,
"columns": formula.columns,
"sql": formula.sql,
},
})
except ValidationError as error:
return LegacyFormulaMigrationFailure(
code="legacy_formula_requires_manual_review",
legacy_path=legacy_path,
formula=formula,
problems=tuple(sorted(
".".join(str(part) for part in issue["loc"])
for issue in error.errors()
)),
)
-4
View File
@@ -58,9 +58,5 @@ def load_evidence_dir(root: Path) -> list[EvidenceDoc]:
for f in sorted(root.rglob("*.md")):
if f.name.upper().startswith("README"):
continue
# I file formula (concept->SQL, frontmatter diverso) vivono sotto formulas/ con
# estensione .sql.md: non sono EvidenceDoc, li gestisce formula_store (D14b).
if f.name.endswith(".sql.md"):
continue
docs.append(EvidenceDoc.parse(f.read_text(), path=f))
return docs