feat: complete evidence restructuring worktree

This commit is contained in:
Codex
2026-08-26 11:39:02 +02:00
parent a54d4769dd
commit 38f02cfd08
56 changed files with 1981 additions and 1801 deletions
+33
View File
@@ -14,6 +14,7 @@ from tht.cli.config_cmd import CONFIG_OPT
from tht.evidence import (
EvidencePreparationError,
PiEvidenceRestructurer,
migrate_workspace_evidence,
prepare_workspace_evidence,
resolve_workspace_evidence,
validate_workspace_evidence,
@@ -174,6 +175,38 @@ def validate_cmd(
raise typer.Exit(code=1)
@evidence_app.command("migrate")
def migrate_cmd(
workspace_root: Path,
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
) -> None:
"""Rewrite legacy Curated units as readable Markdown without model calls."""
root = _canonical_worktree(workspace_root)
try:
report = migrate_workspace_evidence(root)
except EvidencePreparationError as error:
_emit({
"schemaVersion": 1,
"operation": "evidence_migrate",
"status": "failed",
"code": error.code,
}, json_output)
raise typer.Exit(code=1) from error
_emit({
"schemaVersion": 1,
"operation": "evidence_migrate",
"status": "migrated",
"migrated": list(report.migrated),
"unchanged": list(report.unchanged),
"findings": _findings_payload(report.findings),
}, json_output)
if report.findings:
if all(finding.code in {"orphaned_unit", "unresolved_review_item"}
for finding in report.findings):
raise typer.Exit(code=3)
raise typer.Exit(code=1)
@evidence_app.command("evaluate")
def evaluate_cmd(
workspace_root: Path,
-6
View File
@@ -196,12 +196,6 @@ def search_cmd(
for result in outcome.results
], ensure_ascii=False, indent=2))
return
typer.secho(
"ATTENZIONE: lo store formule legacy non viene più consultato; "
"sono disponibili solo Formula Evidence pubblicate.",
fg=typer.colors.YELLOW,
err=True,
)
if not outcome.results:
typer.secho(f"Nessuna formula per '{keyword}'.", fg=typer.colors.YELLOW)
return
+2 -1
View File
@@ -52,7 +52,8 @@ DecisionType = Literal[
"value_grounded",
# D14b: formula di concetto approvata/rifiutata dal reviewer. subject = "phase:4",
# detail = il concetto (es. "fascia pediatrica"), rationale = la/e colonna/e o il motivo.
# retrieve_formula restituisce i candidati; queste decisioni registrano la scelta.
# La ricerca nelle Formula Evidence pubblicate restituisce i candidati; queste decisioni
# registrano la scelta della proposta nella sessione.
"concept_formula_approved",
"concept_formula_rejected",
]
+4
View File
@@ -3,6 +3,7 @@
from tht.evidence.acquisition import acquire, discover
from tht.evidence.authoring import (
EvidenceManifest,
EvidenceMigrationReport,
EvidencePreparationError,
EvidencePreparationReport,
EvidenceResolutionReport,
@@ -14,6 +15,7 @@ from tht.evidence.authoring import (
ValidationReport,
dump_manifest,
load_manifest,
migrate_workspace_evidence,
prepare_workspace_evidence,
resolve_workspace_evidence,
validate_workspace_evidence,
@@ -60,6 +62,7 @@ __all__ = [
"CuratedEvidence",
"EvidenceEmbedder",
"EvidenceManifest",
"EvidenceMigrationReport",
"EvidencePreparationError",
"EvidencePreparationReport",
"EvidenceQueryEmbedder",
@@ -89,6 +92,7 @@ __all__ = [
"dump_manifest",
"load_curated_tree",
"load_manifest",
"migrate_workspace_evidence",
"normalize_aware_datetime",
"parse_curated_markdown",
"prepare_workspace_evidence",
+77 -3
View File
@@ -293,6 +293,15 @@ class EvidencePreparationReport:
model_calls: int
@dataclass(frozen=True)
class EvidenceMigrationReport:
"""Result of a deterministic Curated Evidence presentation-format migration."""
migrated: tuple[str, ...]
unchanged: tuple[str, ...]
findings: tuple[ValidationFinding, ...]
@dataclass(frozen=True)
class EvidenceResolutionReport:
"""The result of one curator-directed Evidence resolution."""
@@ -680,6 +689,48 @@ def prepare_workspace_evidence(
)
def migrate_workspace_evidence(
workspace_root: Path,
*,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
) -> EvidenceMigrationReport:
"""Rewrite v1 Curated units as readable v2 Markdown without changing semantics."""
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
try:
manifest = load_manifest(evidence_root / "manifest.yaml")
documents = load_curated_tree(evidence_root / "curated")
except (OSError, ValidationError, ValueError) as error:
raise EvidencePreparationError("authoring_state_invalid") from error
documents_by_id = {document.id: document for document in documents}
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
migrated = tuple(sorted(
document.id for document in documents if document.schema_version == 1
))
unchanged = tuple(sorted(
document.id for document in documents if document.schema_version == 2
))
if not migrated:
return EvidenceMigrationReport(
migrated=(),
unchanged=unchanged,
findings=validate_workspace_evidence(workspace_root).findings,
)
upgraded = {
evidence_id: document.model_copy(update={"schema_version": 2})
for evidence_id, document in documents_by_id.items()
}
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
return EvidenceMigrationReport(
migrated=migrated,
unchanged=unchanged,
findings=findings,
)
def resolve_workspace_evidence(
workspace_root: Path,
evidence_id: str,
@@ -827,7 +878,7 @@ def _load_resolution_source(evidence_root: Path, source_file: str) -> str:
def _git_status(workspace_root: Path) -> tuple[str, ...]:
result = subprocess.run(
["git", "status", "--porcelain"],
["git", "status", "--porcelain", "--untracked-files=all"],
cwd=workspace_root,
check=False,
capture_output=True,
@@ -835,7 +886,25 @@ def _git_status(workspace_root: Path) -> tuple[str, ...]:
)
if result.returncode != 0:
raise EvidencePreparationError("canonical_git_worktree_required")
return tuple(line for line in result.stdout.splitlines() if line)
prefix_result = subprocess.run(
["git", "rev-parse", "--show-prefix"],
cwd=workspace_root,
check=False,
capture_output=True,
text=True,
)
if prefix_result.returncode != 0:
raise EvidencePreparationError("canonical_git_worktree_required")
prefix = prefix_result.stdout.strip()
entries: list[str] = []
for line in result.stdout.splitlines():
if not line:
continue
path = line[3:] if len(line) > 3 else line
if prefix and path.startswith(prefix):
line = line[:3] + path.removeprefix(prefix)
entries.append(line)
return tuple(entries)
def _reject_dirty_authoring_state(
@@ -934,7 +1003,11 @@ def _candidate_to_evidence(
source_file: str,
source_hash: str,
) -> CuratedEvidence:
data = candidate.model_dump(mode="json", exclude={"existing_id", "supporting_excerpts"})
data = candidate.model_dump(
mode="json",
exclude={"schema_version", "existing_id", "supporting_excerpts"},
)
data["schema_version"] = 2
data["id"] = evidence_id
data["provenance"] = {
"source_file": source_file,
@@ -955,6 +1028,7 @@ def _unsupported_unit(
message="The current source no longer supports this Evidence unit.",
),)
return evidence.model_copy(update={
"schema_version": 2,
"provenance": evidence.provenance.model_copy(update={
"source_file": source_file,
"source_sha256": source_hash,
+447 -6
View File
@@ -226,7 +226,7 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
class CuratedEvidence(StrictModel):
schema_version: Literal[1]
schema_version: Literal[1, 2]
id: str
title: str
kind: EvidenceKind
@@ -247,6 +247,439 @@ class CuratedEvidence(StrictModel):
return self
_V2_LABELS = {
"en": {
"column": "Column",
"columns": "Columns",
"concept": "Concept",
"definition": "Definition",
"description": "Description",
"empty": "No items",
"input": "Input",
"interpretation": "Interpretation",
"label": "Label",
"output": "Output",
"question": "Question",
"review_items": "Review items",
"field": "Field",
"rule": "Rule",
"sql": "SQL",
"supporting_excerpts": "Supporting excerpts",
"synonyms": "Synonyms",
"tables": "Tables",
"url": "URL",
"value": "Value",
"values": "Values",
"meaning": "Meaning",
"variants": "Variants",
},
"it": {
"column": "Colonna",
"columns": "Colonne",
"concept": "Concetto",
"definition": "Definizione",
"description": "Descrizione",
"empty": "Nessun elemento",
"input": "Input",
"interpretation": "Interpretazione",
"label": "Etichetta",
"output": "Output",
"question": "Domanda",
"review_items": "Elementi da rivedere",
"field": "Campo",
"rule": "Regola",
"sql": "SQL",
"supporting_excerpts": "Estratti di supporto",
"synonyms": "Sinonimi",
"tables": "Tabelle",
"url": "URL",
"value": "Valore",
"values": "Valori",
"meaning": "Significato",
"variants": "Varianti",
},
}
_V2_FIELD = re.compile(
r"<!-- tht:field:([a-z_]+) -->\n(.*?)\n<!-- /tht:field:\1 -->",
re.DOTALL,
)
_V2_EXCERPT_SEPARATOR = "<!-- tht:excerpt-separator -->"
_V2_EMPTY_LIST = "<!-- tht:empty-list -->"
_V2_REVIEW_SEPARATOR = "<!-- tht:review-separator -->"
_V2_REVIEW_FIELD = "<!-- tht:review-field -->"
def _v2_labels(language: str) -> dict[str, str]:
return _V2_LABELS["it" if language.lower().startswith("it") else "en"]
def _render_v2_field(name: str, label: str, content: str) -> str:
closing_marker = f"<!-- /tht:field:{name} -->"
if "<!-- tht:field:" in content or "<!-- /tht:field:" in content:
raise ValueError(f"curated evidence {name} contains a reserved marker")
return (
f"<!-- tht:field:{name} -->\n"
f"## {label}\n\n"
f"{content}\n"
f"{closing_marker}"
)
def _render_v2_excerpt(value: str) -> str:
if _V2_EXCERPT_SEPARATOR in value:
raise ValueError("supporting excerpt contains a reserved marker")
quoted = "\n".join(">" if not line else f"> {line}" for line in value.split("\n"))
return quoted
def _render_v2_list(
values: tuple[str, ...], *, code: bool = False, empty_label: str = "No items",
) -> str:
if not values:
return f"{_V2_EMPTY_LIST}\n_{empty_label}._"
if any("\n" in value for value in values):
raise ValueError("curated evidence list values must be single-line")
if code and any("`" in value for value in values):
raise ValueError("curated evidence code values must not contain backticks")
return "\n".join(f"- `{value}`" if code else f"- {value}" for value in values)
def _parse_v2_list(
value: str, *, code: bool = False, empty_label: str = "No items",
) -> tuple[str, ...]:
if value == f"{_V2_EMPTY_LIST}\n_{empty_label}._":
return ()
parsed: list[str] = []
for line in value.split("\n"):
if not line.startswith("- "):
raise ValueError("curated evidence list is malformed")
item = line[2:]
if code:
if len(item) < 2 or not item.startswith("`") or not item.endswith("`"):
raise ValueError("curated evidence code list is malformed")
item = item[1:-1]
parsed.append(item)
return tuple(parsed)
def _escape_v2_table_value(value: str) -> str:
return value.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "\\n")
def _unescape_v2_table_value(value: str) -> str:
output: list[str] = []
index = 0
while index < len(value):
if value[index] != "\\":
output.append(value[index])
index += 1
continue
if index + 1 >= len(value):
raise ValueError("curated evidence table escape is malformed")
escaped = value[index + 1]
if escaped not in {"\\", "|", "n"}:
raise ValueError("curated evidence table escape is malformed")
output.append("\n" if escaped == "n" else escaped)
index += 2
return "".join(output)
def _render_v2_values(values: dict[str, str], labels: dict[str, str]) -> str:
if any("`" in value for value in values):
raise ValueError("curated evidence enum values must not contain backticks")
rows = [
f"| {labels['value']} | {labels['meaning']} |",
"| --- | --- |",
]
if not values:
rows.append(f"| _{labels['empty']}._ | |")
return "\n".join(rows)
rows.extend(
f"| `{_escape_v2_table_value(value)}` | {_escape_v2_table_value(meaning)} |"
for value, meaning in sorted(values.items())
)
return "\n".join(rows)
def _parse_v2_values(value: str, labels: dict[str, str]) -> dict[str, str]:
lines = value.split("\n")
if (
len(lines) < 3
or lines[0] != f"| {labels['value']} | {labels['meaning']} |"
or lines[1] != "| --- | --- |"
):
raise ValueError("curated evidence values table is malformed")
if lines[2:] == [f"| _{labels['empty']}._ | |"]:
return {}
parsed: dict[str, str] = {}
for line in lines[2:]:
if not line.startswith("| ") or not line.endswith(" |"):
raise ValueError("curated evidence values table is malformed")
cells = re.split(r"(?<!\\)\s\|\s", line[2:-2], maxsplit=1)
if len(cells) != 2 or not cells[0].startswith("`") or not cells[0].endswith("`"):
raise ValueError("curated evidence values table is malformed")
key = _unescape_v2_table_value(cells[0][1:-1])
if key in parsed:
raise ValueError("curated evidence enum value appears more than once")
parsed[key] = _unescape_v2_table_value(cells[1])
return parsed
def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
payload = value.payload
if value.kind == "glossary":
fields = [_render_v2_field("definition", labels["definition"], payload.definition)]
if payload.synonyms:
fields.append(_render_v2_field(
"synonyms", labels["synonyms"], _render_v2_list(payload.synonyms),
))
if payload.variants:
fields.append(_render_v2_field(
"variants", labels["variants"], _render_v2_list(payload.variants),
))
return fields
if value.kind == "domain":
return [_render_v2_field("rule", labels["rule"], payload.rule)]
if value.kind == "enum":
return [
_render_v2_field("column", labels["column"], f"`{payload.column}`"),
_render_v2_field("values", labels["values"], _render_v2_values(payload.values, labels)),
]
if value.kind == "example":
return [
_render_v2_field("question", labels["question"], payload.question),
_render_v2_field("interpretation", labels["interpretation"], payload.interpretation),
]
if value.kind == "mapping":
return [
_render_v2_field("concept", labels["concept"], payload.concept),
_render_v2_field("tables", labels["tables"], _render_v2_list(
payload.tables, code=True, empty_label=labels["empty"],
)),
_render_v2_field("columns", labels["columns"], _render_v2_list(
payload.columns, code=True, empty_label=labels["empty"],
)),
]
if value.kind == "normalization":
return [
_render_v2_field("input", labels["input"], payload.input),
_render_v2_field("output", labels["output"], payload.output),
_render_v2_field("rule", labels["rule"], payload.rule),
]
if value.kind == "formula":
if "```" in payload.sql:
raise ValueError("curated evidence SQL contains a reserved Markdown fence")
return [
_render_v2_field("concept", labels["concept"], payload.concept),
_render_v2_field("columns", labels["columns"], _render_v2_list(
payload.columns, code=True, empty_label=labels["empty"],
)),
_render_v2_field("sql", labels["sql"], f"```sql\n{payload.sql}\n```"),
]
if value.kind == "reference":
return [
_render_v2_field("label", labels["label"], payload.label),
_render_v2_field("url", labels["url"], f"<{payload.url}>"),
_render_v2_field("description", labels["description"], payload.description),
]
raise ValueError(f"unsupported curated evidence kind {value.kind}")
def _render_v2_review_items(value: CuratedEvidence, labels: dict[str, str]) -> str:
rendered: list[str] = []
for item in value.review_items:
if "`" in item.code or (item.field is not None and "`" in item.field):
raise ValueError("curated evidence review identifiers must not contain backticks")
if _V2_REVIEW_SEPARATOR in item.message or _V2_REVIEW_FIELD in item.message:
raise ValueError("curated evidence review message contains a reserved marker")
block = f"### `{item.code}`\n\n{item.message}"
if item.field is not None:
block += f"\n\n{_V2_REVIEW_FIELD}\n**{labels['field']}:** `{item.field}`"
rendered.append(block)
return f"\n{_V2_REVIEW_SEPARATOR}\n".join(rendered)
def _render_v2_body(value: CuratedEvidence) -> str:
if "\n" in value.title:
raise ValueError("curated evidence title must be single-line in v2")
labels = _v2_labels(value.language)
fields = [
*_render_v2_payload(value, labels),
_render_v2_field(
"supporting_excerpts",
labels["supporting_excerpts"],
f"\n{_V2_EXCERPT_SEPARATOR}\n".join(
_render_v2_excerpt(excerpt)
for excerpt in value.provenance.supporting_excerpts
),
),
]
if value.review_items:
fields.append(_render_v2_field(
"review_items",
labels["review_items"],
_render_v2_review_items(value, labels),
))
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
def _parse_v2_field_content(name: str, block: str) -> str:
try:
heading, content = block.split("\n\n", 1)
except ValueError as error:
raise ValueError(f"curated evidence field {name} is malformed") from error
if not heading.startswith("## ") or not content:
raise ValueError(f"curated evidence field {name} is malformed")
return content
def _parse_v2_excerpts(block: str) -> tuple[str, ...]:
excerpts: list[str] = []
for raw_excerpt in block.split(f"\n{_V2_EXCERPT_SEPARATOR}\n"):
lines = raw_excerpt.split("\n")
if any(line != ">" and not line.startswith("> ") for line in lines):
raise ValueError("curated evidence supporting excerpt is malformed")
excerpts.append("\n".join(line[2:] if line.startswith("> ") else "" for line in lines))
if not excerpts:
raise ValueError("curated evidence supporting excerpts are malformed")
return tuple(excerpts)
def _parse_inline_code(value: str, name: str) -> str:
if len(value) < 2 or not value.startswith("`") or not value.endswith("`"):
raise ValueError(f"curated evidence field {name} must be inline code")
return value[1:-1]
def _parse_v2_payload(
kind: str, fields: dict[str, str], labels: dict[str, str],
) -> tuple[dict, set[str]]:
if kind == "glossary":
expected = {"definition"}
payload: dict = {"definition": fields.get("definition")}
for name in ("synonyms", "variants"):
if name in fields:
expected.add(name)
payload[name] = _parse_v2_list(fields[name], empty_label=labels["empty"])
else:
payload[name] = ()
return payload, expected
if kind == "domain":
return {"rule": fields.get("rule")}, {"rule"}
if kind == "enum":
return {
"column": _parse_inline_code(fields.get("column", ""), "column"),
"values": _parse_v2_values(fields.get("values", ""), labels),
}, {"column", "values"}
if kind == "example":
return {
"question": fields.get("question"),
"interpretation": fields.get("interpretation"),
}, {"question", "interpretation"}
if kind == "mapping":
return {
"concept": fields.get("concept"),
"tables": _parse_v2_list(
fields.get("tables", ""), code=True, empty_label=labels["empty"],
),
"columns": _parse_v2_list(
fields.get("columns", ""), code=True, empty_label=labels["empty"],
),
}, {"concept", "tables", "columns"}
if kind == "normalization":
return {
"input": fields.get("input"),
"output": fields.get("output"),
"rule": fields.get("rule"),
}, {"input", "output", "rule"}
if kind == "formula":
sql = fields.get("sql", "")
if not sql.startswith("```sql\n") or not sql.endswith("\n```"):
raise ValueError("curated evidence SQL block is malformed")
return {
"concept": fields.get("concept"),
"columns": _parse_v2_list(
fields.get("columns", ""), code=True, empty_label=labels["empty"],
),
"sql": sql.removeprefix("```sql\n").removesuffix("\n```"),
}, {"concept", "columns", "sql"}
if kind == "reference":
url = fields.get("url", "")
if not url.startswith("<") or not url.endswith(">"):
raise ValueError("curated evidence reference URL is malformed")
return {
"label": fields.get("label"),
"url": url[1:-1],
"description": fields.get("description"),
}, {"label", "url", "description"}
raise ValueError("curated evidence body kind is unsupported")
def _parse_v2_review_items(value: str) -> tuple[ReviewItem, ...]:
items: list[ReviewItem] = []
for raw_item in value.split(f"\n{_V2_REVIEW_SEPARATOR}\n"):
try:
heading, detail = raw_item.split("\n\n", 1)
except ValueError as error:
raise ValueError("curated evidence review item is malformed") from error
if not heading.startswith("### `") or not heading.endswith("`"):
raise ValueError("curated evidence review item code is malformed")
code = heading.removeprefix("### `").removesuffix("`")
field = None
marker = f"\n\n{_V2_REVIEW_FIELD}\n"
if marker in detail:
message, rendered_field = detail.split(marker, 1)
match = re.fullmatch(r"\*\*[^*]+:\*\* `([^`]+)`", rendered_field)
if match is None:
raise ValueError("curated evidence review item field is malformed")
field = match.group(1)
else:
message = detail
if not code or not message:
raise ValueError("curated evidence review item is malformed")
items.append(ReviewItem(code=code, message=message, field=field))
return tuple(items)
def _parse_v2_body(data: dict, body: str) -> dict:
kind = data.get("kind")
body_owned = {"payload", "review_items"}
if isinstance(kind, str):
body_owned.add(kind)
if body_owned.intersection(data):
raise ValueError("curated evidence v2 frontmatter contains body-owned fields")
title = data.get("title")
if not isinstance(title, str) or not body.startswith(f"# {title}\n"):
raise ValueError("curated evidence body title must match its metadata")
fields: dict[str, str] = {}
for match in _V2_FIELD.finditer(body):
name = match.group(1)
if name in fields:
raise ValueError(f"curated evidence field {name} appears more than once")
fields[name] = _parse_v2_field_content(name, match.group(2))
skeleton = _V2_FIELD.sub("", body).strip()
if skeleton != f"# {title}":
raise ValueError("curated evidence body contains unstructured content")
labels = _v2_labels(str(data.get("language", "")))
payload, payload_fields = _parse_v2_payload(kind, fields, labels)
common_fields = {"supporting_excerpts"}
if "review_items" in fields:
common_fields.add("review_items")
if set(fields) != payload_fields | common_fields:
raise ValueError("curated evidence body fields do not match its kind")
provenance = data.get("provenance")
if not isinstance(provenance, dict) or "supporting_excerpts" in provenance:
raise ValueError("curated evidence v2 provenance is malformed")
provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"])
data["review_items"] = (
_parse_v2_review_items(fields["review_items"])
if "review_items" in fields
else []
)
data["payload"] = payload
return data
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
"""Parse the canonical frontmatter representation of one Curated Evidence unit."""
if not text.startswith("---\n"):
@@ -260,11 +693,14 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
data = dict(raw)
except (TypeError, ValueError) as error:
raise ValueError("curated evidence frontmatter must be a mapping") from error
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
kind = data.get("kind")
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
data["payload"] = data.pop(kind, None)
if data.get("schema_version") == 2:
data = _parse_v2_body(data, body)
else:
if body.strip():
raise ValueError("curated evidence must not contain an ignored body")
kind = data.get("kind")
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
data["payload"] = data.pop(kind, None)
evidence = CuratedEvidence.model_validate(data)
if path is not None:
_validate_kind_directory(path, evidence.kind)
@@ -273,6 +709,11 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
def dump_curated_markdown(value: CuratedEvidence) -> str:
"""Render canonical frontmatter with a human-readable kind-specific payload key."""
if value.schema_version == 2:
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
data["provenance"].pop("supporting_excerpts")
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
return f"---\n{frontmatter}---\n{_render_v2_body(value)}"
data = value.model_dump(mode="json", exclude={"payload"})
data[value.kind] = value.payload.model_dump(mode="json")
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
-202
View File
@@ -1,202 +0,0 @@
"""Legacy ConceptFormula reader and one-way migration into Curated Evidence.
The ``formulas/*.sql.md`` store is retained only for the migration window. Runtime
lookup uses typed, published ``kind=formula`` Evidence instead. A session reviewer
may still approve a formula locally; that is a proposal, not publication.
"""
from __future__ import annotations
import hashlib
import re
import unicodedata
from dataclasses import dataclass
from pathlib import Path
from typing import Literal
import yaml
from pydantic import BaseModel, ValidationError
from tht.evidence.canonical import CuratedEvidence
FORMULAS_SUBDIR = "formulas"
_SUFFIX_RE = re.compile(r"^(.*?)-(\d+)\.sql\.md$")
class ConceptFormula(BaseModel):
concept: str
columns: list[str] = []
sql: str
# auto = sintetizzata dal modello (non ancora rivista); draft = bozza umana;
# reviewed = approvata da un revisore. (spec §4.7.2: status auto/draft/reviewed)
status: Literal["auto", "draft", "reviewed"] = "draft"
sources: list[str] = []
@property
def _slug(self) -> str:
"""ASCII slug for the filename (matches textutil.slugify shape)."""
import unicodedata
text = unicodedata.normalize("NFKD", self.concept).encode("ascii", "ignore").decode()
return re.sub(r"[^a-z0-9_]+", "-", text.lower()).strip("-") or "formula"
def dump(self) -> str:
meta = self.model_dump(exclude={"sql"}, mode="json")
fm = yaml.safe_dump(meta, sort_keys=False, allow_unicode=True)
return f"---\n{fm}---\n{self.sql}\n"
@classmethod
def parse(cls, text: str) -> ConceptFormula:
if not text.startswith("---\n"):
raise ValueError("frontmatter mancante (atteso '---\\n' iniziale)")
try:
_, fm, body = text.split("---\n", 2)
except ValueError as e:
raise ValueError("frontmatter malformato") from e
meta = yaml.safe_load(fm)
if not isinstance(meta, dict):
raise TypeError("frontmatter non valido")
return cls.model_validate({**meta, "sql": body.strip("\n")})
@dataclass(frozen=True)
class LegacyFormulaMigrationFailure:
"""A reviewed legacy formula that must be resolved manually before publication."""
code: Literal["legacy_formula_requires_manual_review"]
legacy_path: str
formula: ConceptFormula
problems: tuple[str, ...]
def _next_path(root: Path, slug: str) -> Path:
"""First free <slug>-<n>.sql.md path under root (n starts at 1)."""
root.mkdir(parents=True, exist_ok=True)
existing = sorted(root.glob(f"{slug}-*.sql.md"))
n = 0
for p in existing:
m = _SUFFIX_RE.match(p.name)
if m:
n = max(n, int(m.group(2)))
return root / f"{slug}-{n + 1}.sql.md"
def save_formula(root: Path | str, formula: ConceptFormula) -> Path:
"""Persist a single concept->formula unit under <root>/formulas/. Returns the
written path. Append-only: each save writes a new file (so competing drafts and
reviewed versions coexist until a curator prunes)."""
root = Path(root)
formulas_dir = root / FORMULAS_SUBDIR
path = _next_path(formulas_dir, formula._slug)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(formula.dump())
return path
def _load_all(root: Path) -> list[ConceptFormula]:
formulas_dir = root / FORMULAS_SUBDIR
if not formulas_dir.is_dir():
return []
out: list[ConceptFormula] = []
for f in sorted(formulas_dir.glob("*.sql.md")):
try:
out.append(ConceptFormula.parse(f.read_text()))
except ValueError:
continue # malformed file: skip, don't crash retrieval
return out
def retrieve_formula(root: Path | str, concept: str) -> list[ConceptFormula]:
"""All formulas matching `concept` exactly under <root>/formulas/. Empty list if
none (or if the dir is absent). Multiple results mean competing drafts/versions for
the same concept -- the caller (gate) lets the reviewer pick."""
return [f for f in _load_all(Path(root)) if f.concept == concept]
def search_formulas(root: Path | str, query: str) -> list[ConceptFormula]:
"""Read legacy formulas for migration tooling only (case-insensitive concept match)."""
q = query.strip().lower()
return [f for f in _load_all(Path(root)) if q in f.concept.lower()]
def legacy_formula_to_curated(
formula: ConceptFormula,
*,
legacy_path: str,
source_content: str | None = None,
) -> CuratedEvidence | LegacyFormulaMigrationFailure | None:
"""Convert one reviewed legacy formula into its deterministic curated counterpart.
Drafts and model-generated formulas have no global publication status. Their
caller must project them as session-local Formula proposals instead.
"""
if formula.status != "reviewed":
return None
problems: list[str] = []
normalized_source: str | None = None
if source_content is None:
problems.append("original_source_required")
else:
from tht.evidence.authoring import normalize_source_text
normalized_source = normalize_source_text(source_content)
try:
original_formula = ConceptFormula.parse(source_content)
except (TypeError, ValidationError, ValueError, yaml.YAMLError):
problems.append("original_source_invalid")
else:
if original_formula != formula:
problems.append("original_source_mismatch")
source_notes = tuple(formula.sources)
if not source_notes:
problems.append("supporting_excerpts_required")
elif normalized_source is not None and any(
unicodedata.normalize("NFC", note.replace("\r\n", "\n").replace("\r", "\n"))
not in normalized_source
for note in source_notes
):
problems.append("supporting_excerpt_unverified")
if problems:
return LegacyFormulaMigrationFailure(
code="legacy_formula_requires_manual_review",
legacy_path=legacy_path,
formula=formula,
problems=tuple(sorted(problems)),
)
assert normalized_source is not None
source_sha256 = hashlib.sha256(normalized_source.encode("utf-8")).hexdigest()
source_file = legacy_path if legacy_path.startswith("source/") else f"source/{legacy_path}"
# The legacy path is the immutable identity of this unit during migration. Keeping
# its full digest avoids a duplicate public ID when the same concept has reviewed
# competing formulas, while retaining the readable concept slug as the prefix.
legacy_identity = hashlib.sha256(legacy_path.encode("utf-8")).hexdigest()
try:
return CuratedEvidence.model_validate({
"schema_version": 1,
"id": f"evidence:{formula._slug}-{legacy_identity}",
"title": formula.concept[:1].upper() + formula.concept[1:],
"kind": "formula",
"purposes": ["schema_linking", "sql_generation"],
"applies_to": {"concepts": [formula.concept], "columns": formula.columns},
"language": "it",
"provenance": {
"source_file": source_file,
"source_sha256": f"sha256:{source_sha256}",
"supporting_excerpts": source_notes,
},
"review_items": [],
"payload": {
"concept": formula.concept,
"columns": formula.columns,
"sql": formula.sql,
},
})
except ValidationError as error:
return LegacyFormulaMigrationFailure(
code="legacy_formula_requires_manual_review",
legacy_path=legacy_path,
formula=formula,
problems=tuple(sorted(
".".join(str(part) for part in issue["loc"])
for issue in error.errors()
)),
)
-4
View File
@@ -58,9 +58,5 @@ def load_evidence_dir(root: Path) -> list[EvidenceDoc]:
for f in sorted(root.rglob("*.md")):
if f.name.upper().startswith("README"):
continue
# I file formula (concept->SQL, frontmatter diverso) vivono sotto formulas/ con
# estensione .sql.md: non sono EvidenceDoc, li gestisce formula_store (D14b).
if f.name.endswith(".sql.md"):
continue
docs.append(EvidenceDoc.parse(f.read_text(), path=f))
return docs
+15 -2
View File
@@ -1,6 +1,19 @@
from typing import Protocol
from pydantic import BaseModel
from tht.vectorstore.store import VectorStore
from tht.vectorstore.store import VectorHit
class SemanticSearcher(Protocol):
def search(
self,
embedding: list[float],
*,
top_n: int,
kinds: list[str] | None,
query_text: str | None = None,
) -> list[VectorHit]: ...
class SearchResult(BaseModel):
@@ -92,7 +105,7 @@ def combined_search(
keyword: str,
*,
lsh_hits: list[tuple[str, str, str, float]] | None,
store: VectorStore,
store: SemanticSearcher,
embedder,
top: int,
rrf_k: int,
+1 -155
View File
@@ -1,20 +1,11 @@
import hashlib
import json
from dataclasses import dataclass
from sqlalchemy import Engine, text
from tht.vectorstore.records import VectorRecord
def content_hash(content: str) -> str:
return hashlib.sha256(content.encode()).hexdigest()
def _to_vector_literal(vec: list[float]) -> str:
return "[" + ",".join(f"{x:.8f}" for x in vec) + "]"
@dataclass
class SyncStats:
added: int = 0
@@ -35,10 +26,7 @@ class VectorHit:
def hit_from_metadata(similarity: float, metadata: dict | None) -> VectorHit:
"""Ricostruisce un VectorHit dal solo `metadata` (più la similarity). È l'unico modo
disponibile leggendo via REST (`search_similar` ritorna id/similarity/metadata), e viene
usato anche dalla lettura diretta per avere un'unica logica. Tollerante: usa default sui
campi assenti (es. metadata estranei della tabella fake remota)."""
"""Build a transport-neutral hit from the metadata returned by a vector adapter."""
md = metadata or {}
return VectorHit(
id=md.get("record_key", ""),
@@ -49,145 +37,3 @@ def hit_from_metadata(similarity: float, metadata: dict | None) -> VectorHit:
metadata=md,
similarity=float(similarity),
)
class VectorStore:
"""Legacy table-scoped vector store retained for compatibility fixtures.
New operational semantic storage is handled by the Qdrant adapter. This class preserves
the older SQL-table contract used by historical tests and migration checks: `id`
(BIGSERIAL), `embedding vector(N)`, and `metadata jsonb`; the extra columns
(`record_key`, `kind`, `content_hash`) serve only the loader and are not exposed by REST.
"""
def __init__(
self, engine: Engine, schema: str = "vectors", table: str = "records", dim: int = 768
):
self.engine = engine
self.schema = schema
self.dim = dim
self._table = f"{schema}.{table}"
def init_schema(self) -> None:
with self.engine.begin() as conn:
conn.execute(text("CREATE EXTENSION IF NOT EXISTS vector"))
conn.execute(text(f"CREATE SCHEMA IF NOT EXISTS {self.schema}"))
conn.execute(text(f"""
CREATE TABLE IF NOT EXISTS {self._table} (
id bigserial PRIMARY KEY,
record_key text UNIQUE NOT NULL,
kind text NOT NULL,
content_hash text NOT NULL,
metadata jsonb NOT NULL DEFAULT '{{}}',
embedding vector({self.dim}) NOT NULL,
indexed_at timestamptz NOT NULL DEFAULT now()
)
"""))
conn.execute(text(
f"CREATE INDEX IF NOT EXISTS {self._idx('embedding')} ON {self._table} "
f"USING hnsw (embedding vector_cosine_ops) WITH (m = 16, ef_construction = 200)"
))
conn.execute(text(
f"CREATE INDEX IF NOT EXISTS {self._idx('kind')} ON {self._table} (kind)"
))
# GRANT al ruolo di sola lettura della REST, solo se esiste (assente in test/locale).
conn.execute(text(f"""
DO $$ BEGIN
IF EXISTS (SELECT 1 FROM pg_roles WHERE rolname = 'vector_reader') THEN
EXECUTE 'GRANT SELECT ON {self._table} TO vector_reader';
END IF;
END $$;
"""))
def _idx(self, suffix: str) -> str:
return f"{self._table.replace('.', '_')}_{suffix}_idx"
def clear(self) -> None:
with self.engine.begin() as conn:
conn.execute(text(f"DELETE FROM {self._table}"))
def existing_hashes(self, kinds: set[str]) -> dict[str, str]:
q = text(
f"SELECT record_key, content_hash FROM {self._table} WHERE kind = ANY(:kinds)"
)
with self.engine.connect() as conn:
return dict(conn.execute(q, {"kinds": list(kinds)}).fetchall())
def sync(self, records: list[VectorRecord], embedder, kinds: set[str]) -> SyncStats:
"""Allinea l'indice ai record correnti (per i kind dati): embedda solo il nuovo
o il modificato, elimina cio' che non esiste piu'."""
stats = SyncStats()
existing = self.existing_hashes(kinds)
current_ids = {r.id for r in records}
to_embed: list[VectorRecord] = []
for r in records:
h = content_hash(r.content)
if r.id not in existing:
to_embed.append(r)
stats.added += 1
elif existing[r.id] != h:
to_embed.append(r)
stats.updated += 1
else:
stats.unchanged += 1
vectors = embedder.embed_documents([r.content for r in to_embed]) if to_embed else []
upsert = text(f"""
INSERT INTO {self._table}
(record_key, kind, content_hash, metadata, embedding)
VALUES
(:record_key, :kind, :content_hash, CAST(:metadata AS jsonb),
CAST(:embedding AS vector))
ON CONFLICT (record_key) DO UPDATE SET
kind = EXCLUDED.kind, content_hash = EXCLUDED.content_hash,
metadata = EXCLUDED.metadata, embedding = EXCLUDED.embedding,
indexed_at = now()
""")
stale = [i for i in existing if i not in current_ids]
with self.engine.begin() as conn:
for r, vec in zip(to_embed, vectors):
conn.execute(upsert, {
"record_key": r.id, "kind": r.kind,
"content_hash": content_hash(r.content),
"metadata": json.dumps(_pack_metadata(r)),
"embedding": _to_vector_literal(vec),
})
if stale:
conn.execute(
text(f"DELETE FROM {self._table} WHERE record_key = ANY(:ids)"),
{"ids": stale},
)
stats.deleted = len(stale)
return stats
def search(
self, query_vec: list[float], top_n: int = 10, kinds: list[str] | None = None
) -> list[VectorHit]:
where = "WHERE kind = ANY(:kinds)" if kinds else ""
q = text(f"""
SELECT metadata, 1 - (embedding <=> CAST(:q AS vector)) AS similarity
FROM {self._table}
{where}
ORDER BY embedding <=> CAST(:q AS vector)
LIMIT :top_n
""")
params: dict = {"q": _to_vector_literal(query_vec), "top_n": top_n}
if kinds:
params["kinds"] = kinds
with self.engine.connect() as conn:
rows = conn.execute(q, params).fetchall()
return [hit_from_metadata(r.similarity, r.metadata) for r in rows]
def _pack_metadata(r: VectorRecord) -> dict:
"""Impacchetta nel `metadata` (unica colonna letta via REST) tutta la semantica Thoth."""
return {
"kind": r.kind,
"ref": r.ref,
"record_key": r.id,
"title": r.title,
"content": r.content,
**r.metadata,
}