feat: complete evidence restructuring worktree
This commit is contained in:
@@ -14,6 +14,7 @@ from tht.cli.config_cmd import CONFIG_OPT
|
||||
from tht.evidence import (
|
||||
EvidencePreparationError,
|
||||
PiEvidenceRestructurer,
|
||||
migrate_workspace_evidence,
|
||||
prepare_workspace_evidence,
|
||||
resolve_workspace_evidence,
|
||||
validate_workspace_evidence,
|
||||
@@ -174,6 +175,38 @@ def validate_cmd(
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
|
||||
@evidence_app.command("migrate")
|
||||
def migrate_cmd(
|
||||
workspace_root: Path,
|
||||
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
|
||||
) -> None:
|
||||
"""Rewrite legacy Curated units as readable Markdown without model calls."""
|
||||
root = _canonical_worktree(workspace_root)
|
||||
try:
|
||||
report = migrate_workspace_evidence(root)
|
||||
except EvidencePreparationError as error:
|
||||
_emit({
|
||||
"schemaVersion": 1,
|
||||
"operation": "evidence_migrate",
|
||||
"status": "failed",
|
||||
"code": error.code,
|
||||
}, json_output)
|
||||
raise typer.Exit(code=1) from error
|
||||
_emit({
|
||||
"schemaVersion": 1,
|
||||
"operation": "evidence_migrate",
|
||||
"status": "migrated",
|
||||
"migrated": list(report.migrated),
|
||||
"unchanged": list(report.unchanged),
|
||||
"findings": _findings_payload(report.findings),
|
||||
}, json_output)
|
||||
if report.findings:
|
||||
if all(finding.code in {"orphaned_unit", "unresolved_review_item"}
|
||||
for finding in report.findings):
|
||||
raise typer.Exit(code=3)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
|
||||
@evidence_app.command("evaluate")
|
||||
def evaluate_cmd(
|
||||
workspace_root: Path,
|
||||
|
||||
@@ -196,12 +196,6 @@ def search_cmd(
|
||||
for result in outcome.results
|
||||
], ensure_ascii=False, indent=2))
|
||||
return
|
||||
typer.secho(
|
||||
"ATTENZIONE: lo store formule legacy non viene più consultato; "
|
||||
"sono disponibili solo Formula Evidence pubblicate.",
|
||||
fg=typer.colors.YELLOW,
|
||||
err=True,
|
||||
)
|
||||
if not outcome.results:
|
||||
typer.secho(f"Nessuna formula per '{keyword}'.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
|
||||
@@ -52,7 +52,8 @@ DecisionType = Literal[
|
||||
"value_grounded",
|
||||
# D14b: formula di concetto approvata/rifiutata dal reviewer. subject = "phase:4",
|
||||
# detail = il concetto (es. "fascia pediatrica"), rationale = la/e colonna/e o il motivo.
|
||||
# retrieve_formula restituisce i candidati; queste decisioni registrano la scelta.
|
||||
# La ricerca nelle Formula Evidence pubblicate restituisce i candidati; queste decisioni
|
||||
# registrano la scelta della proposta nella sessione.
|
||||
"concept_formula_approved",
|
||||
"concept_formula_rejected",
|
||||
]
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
from tht.evidence.acquisition import acquire, discover
|
||||
from tht.evidence.authoring import (
|
||||
EvidenceManifest,
|
||||
EvidenceMigrationReport,
|
||||
EvidencePreparationError,
|
||||
EvidencePreparationReport,
|
||||
EvidenceResolutionReport,
|
||||
@@ -14,6 +15,7 @@ from tht.evidence.authoring import (
|
||||
ValidationReport,
|
||||
dump_manifest,
|
||||
load_manifest,
|
||||
migrate_workspace_evidence,
|
||||
prepare_workspace_evidence,
|
||||
resolve_workspace_evidence,
|
||||
validate_workspace_evidence,
|
||||
@@ -60,6 +62,7 @@ __all__ = [
|
||||
"CuratedEvidence",
|
||||
"EvidenceEmbedder",
|
||||
"EvidenceManifest",
|
||||
"EvidenceMigrationReport",
|
||||
"EvidencePreparationError",
|
||||
"EvidencePreparationReport",
|
||||
"EvidenceQueryEmbedder",
|
||||
@@ -89,6 +92,7 @@ __all__ = [
|
||||
"dump_manifest",
|
||||
"load_curated_tree",
|
||||
"load_manifest",
|
||||
"migrate_workspace_evidence",
|
||||
"normalize_aware_datetime",
|
||||
"parse_curated_markdown",
|
||||
"prepare_workspace_evidence",
|
||||
|
||||
@@ -293,6 +293,15 @@ class EvidencePreparationReport:
|
||||
model_calls: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EvidenceMigrationReport:
|
||||
"""Result of a deterministic Curated Evidence presentation-format migration."""
|
||||
|
||||
migrated: tuple[str, ...]
|
||||
unchanged: tuple[str, ...]
|
||||
findings: tuple[ValidationFinding, ...]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EvidenceResolutionReport:
|
||||
"""The result of one curator-directed Evidence resolution."""
|
||||
@@ -680,6 +689,48 @@ def prepare_workspace_evidence(
|
||||
)
|
||||
|
||||
|
||||
def migrate_workspace_evidence(
|
||||
workspace_root: Path,
|
||||
*,
|
||||
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
||||
) -> EvidenceMigrationReport:
|
||||
"""Rewrite v1 Curated units as readable v2 Markdown without changing semantics."""
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
try:
|
||||
manifest = load_manifest(evidence_root / "manifest.yaml")
|
||||
documents = load_curated_tree(evidence_root / "curated")
|
||||
except (OSError, ValidationError, ValueError) as error:
|
||||
raise EvidencePreparationError("authoring_state_invalid") from error
|
||||
|
||||
documents_by_id = {document.id: document for document in documents}
|
||||
if len(documents_by_id) != len(documents):
|
||||
raise EvidencePreparationError("duplicate_evidence_id")
|
||||
migrated = tuple(sorted(
|
||||
document.id for document in documents if document.schema_version == 1
|
||||
))
|
||||
unchanged = tuple(sorted(
|
||||
document.id for document in documents if document.schema_version == 2
|
||||
))
|
||||
if not migrated:
|
||||
return EvidenceMigrationReport(
|
||||
migrated=(),
|
||||
unchanged=unchanged,
|
||||
findings=validate_workspace_evidence(workspace_root).findings,
|
||||
)
|
||||
upgraded = {
|
||||
evidence_id: document.model_copy(update={"schema_version": 2})
|
||||
for evidence_id, document in documents_by_id.items()
|
||||
}
|
||||
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
|
||||
return EvidenceMigrationReport(
|
||||
migrated=migrated,
|
||||
unchanged=unchanged,
|
||||
findings=findings,
|
||||
)
|
||||
|
||||
|
||||
def resolve_workspace_evidence(
|
||||
workspace_root: Path,
|
||||
evidence_id: str,
|
||||
@@ -827,7 +878,7 @@ def _load_resolution_source(evidence_root: Path, source_file: str) -> str:
|
||||
|
||||
def _git_status(workspace_root: Path) -> tuple[str, ...]:
|
||||
result = subprocess.run(
|
||||
["git", "status", "--porcelain"],
|
||||
["git", "status", "--porcelain", "--untracked-files=all"],
|
||||
cwd=workspace_root,
|
||||
check=False,
|
||||
capture_output=True,
|
||||
@@ -835,7 +886,25 @@ def _git_status(workspace_root: Path) -> tuple[str, ...]:
|
||||
)
|
||||
if result.returncode != 0:
|
||||
raise EvidencePreparationError("canonical_git_worktree_required")
|
||||
return tuple(line for line in result.stdout.splitlines() if line)
|
||||
prefix_result = subprocess.run(
|
||||
["git", "rev-parse", "--show-prefix"],
|
||||
cwd=workspace_root,
|
||||
check=False,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
)
|
||||
if prefix_result.returncode != 0:
|
||||
raise EvidencePreparationError("canonical_git_worktree_required")
|
||||
prefix = prefix_result.stdout.strip()
|
||||
entries: list[str] = []
|
||||
for line in result.stdout.splitlines():
|
||||
if not line:
|
||||
continue
|
||||
path = line[3:] if len(line) > 3 else line
|
||||
if prefix and path.startswith(prefix):
|
||||
line = line[:3] + path.removeprefix(prefix)
|
||||
entries.append(line)
|
||||
return tuple(entries)
|
||||
|
||||
|
||||
def _reject_dirty_authoring_state(
|
||||
@@ -934,7 +1003,11 @@ def _candidate_to_evidence(
|
||||
source_file: str,
|
||||
source_hash: str,
|
||||
) -> CuratedEvidence:
|
||||
data = candidate.model_dump(mode="json", exclude={"existing_id", "supporting_excerpts"})
|
||||
data = candidate.model_dump(
|
||||
mode="json",
|
||||
exclude={"schema_version", "existing_id", "supporting_excerpts"},
|
||||
)
|
||||
data["schema_version"] = 2
|
||||
data["id"] = evidence_id
|
||||
data["provenance"] = {
|
||||
"source_file": source_file,
|
||||
@@ -955,6 +1028,7 @@ def _unsupported_unit(
|
||||
message="The current source no longer supports this Evidence unit.",
|
||||
),)
|
||||
return evidence.model_copy(update={
|
||||
"schema_version": 2,
|
||||
"provenance": evidence.provenance.model_copy(update={
|
||||
"source_file": source_file,
|
||||
"source_sha256": source_hash,
|
||||
|
||||
@@ -226,7 +226,7 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
|
||||
|
||||
|
||||
class CuratedEvidence(StrictModel):
|
||||
schema_version: Literal[1]
|
||||
schema_version: Literal[1, 2]
|
||||
id: str
|
||||
title: str
|
||||
kind: EvidenceKind
|
||||
@@ -247,6 +247,439 @@ class CuratedEvidence(StrictModel):
|
||||
return self
|
||||
|
||||
|
||||
_V2_LABELS = {
|
||||
"en": {
|
||||
"column": "Column",
|
||||
"columns": "Columns",
|
||||
"concept": "Concept",
|
||||
"definition": "Definition",
|
||||
"description": "Description",
|
||||
"empty": "No items",
|
||||
"input": "Input",
|
||||
"interpretation": "Interpretation",
|
||||
"label": "Label",
|
||||
"output": "Output",
|
||||
"question": "Question",
|
||||
"review_items": "Review items",
|
||||
"field": "Field",
|
||||
"rule": "Rule",
|
||||
"sql": "SQL",
|
||||
"supporting_excerpts": "Supporting excerpts",
|
||||
"synonyms": "Synonyms",
|
||||
"tables": "Tables",
|
||||
"url": "URL",
|
||||
"value": "Value",
|
||||
"values": "Values",
|
||||
"meaning": "Meaning",
|
||||
"variants": "Variants",
|
||||
},
|
||||
"it": {
|
||||
"column": "Colonna",
|
||||
"columns": "Colonne",
|
||||
"concept": "Concetto",
|
||||
"definition": "Definizione",
|
||||
"description": "Descrizione",
|
||||
"empty": "Nessun elemento",
|
||||
"input": "Input",
|
||||
"interpretation": "Interpretazione",
|
||||
"label": "Etichetta",
|
||||
"output": "Output",
|
||||
"question": "Domanda",
|
||||
"review_items": "Elementi da rivedere",
|
||||
"field": "Campo",
|
||||
"rule": "Regola",
|
||||
"sql": "SQL",
|
||||
"supporting_excerpts": "Estratti di supporto",
|
||||
"synonyms": "Sinonimi",
|
||||
"tables": "Tabelle",
|
||||
"url": "URL",
|
||||
"value": "Valore",
|
||||
"values": "Valori",
|
||||
"meaning": "Significato",
|
||||
"variants": "Varianti",
|
||||
},
|
||||
}
|
||||
_V2_FIELD = re.compile(
|
||||
r"<!-- tht:field:([a-z_]+) -->\n(.*?)\n<!-- /tht:field:\1 -->",
|
||||
re.DOTALL,
|
||||
)
|
||||
_V2_EXCERPT_SEPARATOR = "<!-- tht:excerpt-separator -->"
|
||||
_V2_EMPTY_LIST = "<!-- tht:empty-list -->"
|
||||
_V2_REVIEW_SEPARATOR = "<!-- tht:review-separator -->"
|
||||
_V2_REVIEW_FIELD = "<!-- tht:review-field -->"
|
||||
|
||||
|
||||
def _v2_labels(language: str) -> dict[str, str]:
|
||||
return _V2_LABELS["it" if language.lower().startswith("it") else "en"]
|
||||
|
||||
|
||||
def _render_v2_field(name: str, label: str, content: str) -> str:
|
||||
closing_marker = f"<!-- /tht:field:{name} -->"
|
||||
if "<!-- tht:field:" in content or "<!-- /tht:field:" in content:
|
||||
raise ValueError(f"curated evidence {name} contains a reserved marker")
|
||||
return (
|
||||
f"<!-- tht:field:{name} -->\n"
|
||||
f"## {label}\n\n"
|
||||
f"{content}\n"
|
||||
f"{closing_marker}"
|
||||
)
|
||||
|
||||
|
||||
def _render_v2_excerpt(value: str) -> str:
|
||||
if _V2_EXCERPT_SEPARATOR in value:
|
||||
raise ValueError("supporting excerpt contains a reserved marker")
|
||||
quoted = "\n".join(">" if not line else f"> {line}" for line in value.split("\n"))
|
||||
return quoted
|
||||
|
||||
|
||||
def _render_v2_list(
|
||||
values: tuple[str, ...], *, code: bool = False, empty_label: str = "No items",
|
||||
) -> str:
|
||||
if not values:
|
||||
return f"{_V2_EMPTY_LIST}\n_{empty_label}._"
|
||||
if any("\n" in value for value in values):
|
||||
raise ValueError("curated evidence list values must be single-line")
|
||||
if code and any("`" in value for value in values):
|
||||
raise ValueError("curated evidence code values must not contain backticks")
|
||||
return "\n".join(f"- `{value}`" if code else f"- {value}" for value in values)
|
||||
|
||||
|
||||
def _parse_v2_list(
|
||||
value: str, *, code: bool = False, empty_label: str = "No items",
|
||||
) -> tuple[str, ...]:
|
||||
if value == f"{_V2_EMPTY_LIST}\n_{empty_label}._":
|
||||
return ()
|
||||
parsed: list[str] = []
|
||||
for line in value.split("\n"):
|
||||
if not line.startswith("- "):
|
||||
raise ValueError("curated evidence list is malformed")
|
||||
item = line[2:]
|
||||
if code:
|
||||
if len(item) < 2 or not item.startswith("`") or not item.endswith("`"):
|
||||
raise ValueError("curated evidence code list is malformed")
|
||||
item = item[1:-1]
|
||||
parsed.append(item)
|
||||
return tuple(parsed)
|
||||
|
||||
|
||||
def _escape_v2_table_value(value: str) -> str:
|
||||
return value.replace("\\", "\\\\").replace("|", "\\|").replace("\n", "\\n")
|
||||
|
||||
|
||||
def _unescape_v2_table_value(value: str) -> str:
|
||||
output: list[str] = []
|
||||
index = 0
|
||||
while index < len(value):
|
||||
if value[index] != "\\":
|
||||
output.append(value[index])
|
||||
index += 1
|
||||
continue
|
||||
if index + 1 >= len(value):
|
||||
raise ValueError("curated evidence table escape is malformed")
|
||||
escaped = value[index + 1]
|
||||
if escaped not in {"\\", "|", "n"}:
|
||||
raise ValueError("curated evidence table escape is malformed")
|
||||
output.append("\n" if escaped == "n" else escaped)
|
||||
index += 2
|
||||
return "".join(output)
|
||||
|
||||
|
||||
def _render_v2_values(values: dict[str, str], labels: dict[str, str]) -> str:
|
||||
if any("`" in value for value in values):
|
||||
raise ValueError("curated evidence enum values must not contain backticks")
|
||||
rows = [
|
||||
f"| {labels['value']} | {labels['meaning']} |",
|
||||
"| --- | --- |",
|
||||
]
|
||||
if not values:
|
||||
rows.append(f"| _{labels['empty']}._ | |")
|
||||
return "\n".join(rows)
|
||||
rows.extend(
|
||||
f"| `{_escape_v2_table_value(value)}` | {_escape_v2_table_value(meaning)} |"
|
||||
for value, meaning in sorted(values.items())
|
||||
)
|
||||
return "\n".join(rows)
|
||||
|
||||
|
||||
def _parse_v2_values(value: str, labels: dict[str, str]) -> dict[str, str]:
|
||||
lines = value.split("\n")
|
||||
if (
|
||||
len(lines) < 3
|
||||
or lines[0] != f"| {labels['value']} | {labels['meaning']} |"
|
||||
or lines[1] != "| --- | --- |"
|
||||
):
|
||||
raise ValueError("curated evidence values table is malformed")
|
||||
if lines[2:] == [f"| _{labels['empty']}._ | |"]:
|
||||
return {}
|
||||
parsed: dict[str, str] = {}
|
||||
for line in lines[2:]:
|
||||
if not line.startswith("| ") or not line.endswith(" |"):
|
||||
raise ValueError("curated evidence values table is malformed")
|
||||
cells = re.split(r"(?<!\\)\s\|\s", line[2:-2], maxsplit=1)
|
||||
if len(cells) != 2 or not cells[0].startswith("`") or not cells[0].endswith("`"):
|
||||
raise ValueError("curated evidence values table is malformed")
|
||||
key = _unescape_v2_table_value(cells[0][1:-1])
|
||||
if key in parsed:
|
||||
raise ValueError("curated evidence enum value appears more than once")
|
||||
parsed[key] = _unescape_v2_table_value(cells[1])
|
||||
return parsed
|
||||
|
||||
|
||||
def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
|
||||
payload = value.payload
|
||||
if value.kind == "glossary":
|
||||
fields = [_render_v2_field("definition", labels["definition"], payload.definition)]
|
||||
if payload.synonyms:
|
||||
fields.append(_render_v2_field(
|
||||
"synonyms", labels["synonyms"], _render_v2_list(payload.synonyms),
|
||||
))
|
||||
if payload.variants:
|
||||
fields.append(_render_v2_field(
|
||||
"variants", labels["variants"], _render_v2_list(payload.variants),
|
||||
))
|
||||
return fields
|
||||
if value.kind == "domain":
|
||||
return [_render_v2_field("rule", labels["rule"], payload.rule)]
|
||||
if value.kind == "enum":
|
||||
return [
|
||||
_render_v2_field("column", labels["column"], f"`{payload.column}`"),
|
||||
_render_v2_field("values", labels["values"], _render_v2_values(payload.values, labels)),
|
||||
]
|
||||
if value.kind == "example":
|
||||
return [
|
||||
_render_v2_field("question", labels["question"], payload.question),
|
||||
_render_v2_field("interpretation", labels["interpretation"], payload.interpretation),
|
||||
]
|
||||
if value.kind == "mapping":
|
||||
return [
|
||||
_render_v2_field("concept", labels["concept"], payload.concept),
|
||||
_render_v2_field("tables", labels["tables"], _render_v2_list(
|
||||
payload.tables, code=True, empty_label=labels["empty"],
|
||||
)),
|
||||
_render_v2_field("columns", labels["columns"], _render_v2_list(
|
||||
payload.columns, code=True, empty_label=labels["empty"],
|
||||
)),
|
||||
]
|
||||
if value.kind == "normalization":
|
||||
return [
|
||||
_render_v2_field("input", labels["input"], payload.input),
|
||||
_render_v2_field("output", labels["output"], payload.output),
|
||||
_render_v2_field("rule", labels["rule"], payload.rule),
|
||||
]
|
||||
if value.kind == "formula":
|
||||
if "```" in payload.sql:
|
||||
raise ValueError("curated evidence SQL contains a reserved Markdown fence")
|
||||
return [
|
||||
_render_v2_field("concept", labels["concept"], payload.concept),
|
||||
_render_v2_field("columns", labels["columns"], _render_v2_list(
|
||||
payload.columns, code=True, empty_label=labels["empty"],
|
||||
)),
|
||||
_render_v2_field("sql", labels["sql"], f"```sql\n{payload.sql}\n```"),
|
||||
]
|
||||
if value.kind == "reference":
|
||||
return [
|
||||
_render_v2_field("label", labels["label"], payload.label),
|
||||
_render_v2_field("url", labels["url"], f"<{payload.url}>"),
|
||||
_render_v2_field("description", labels["description"], payload.description),
|
||||
]
|
||||
raise ValueError(f"unsupported curated evidence kind {value.kind}")
|
||||
|
||||
|
||||
def _render_v2_review_items(value: CuratedEvidence, labels: dict[str, str]) -> str:
|
||||
rendered: list[str] = []
|
||||
for item in value.review_items:
|
||||
if "`" in item.code or (item.field is not None and "`" in item.field):
|
||||
raise ValueError("curated evidence review identifiers must not contain backticks")
|
||||
if _V2_REVIEW_SEPARATOR in item.message or _V2_REVIEW_FIELD in item.message:
|
||||
raise ValueError("curated evidence review message contains a reserved marker")
|
||||
block = f"### `{item.code}`\n\n{item.message}"
|
||||
if item.field is not None:
|
||||
block += f"\n\n{_V2_REVIEW_FIELD}\n**{labels['field']}:** `{item.field}`"
|
||||
rendered.append(block)
|
||||
return f"\n{_V2_REVIEW_SEPARATOR}\n".join(rendered)
|
||||
|
||||
|
||||
def _render_v2_body(value: CuratedEvidence) -> str:
|
||||
if "\n" in value.title:
|
||||
raise ValueError("curated evidence title must be single-line in v2")
|
||||
labels = _v2_labels(value.language)
|
||||
fields = [
|
||||
*_render_v2_payload(value, labels),
|
||||
_render_v2_field(
|
||||
"supporting_excerpts",
|
||||
labels["supporting_excerpts"],
|
||||
f"\n{_V2_EXCERPT_SEPARATOR}\n".join(
|
||||
_render_v2_excerpt(excerpt)
|
||||
for excerpt in value.provenance.supporting_excerpts
|
||||
),
|
||||
),
|
||||
]
|
||||
if value.review_items:
|
||||
fields.append(_render_v2_field(
|
||||
"review_items",
|
||||
labels["review_items"],
|
||||
_render_v2_review_items(value, labels),
|
||||
))
|
||||
return f"# {value.title}\n\n" + "\n\n".join(fields) + "\n"
|
||||
|
||||
|
||||
def _parse_v2_field_content(name: str, block: str) -> str:
|
||||
try:
|
||||
heading, content = block.split("\n\n", 1)
|
||||
except ValueError as error:
|
||||
raise ValueError(f"curated evidence field {name} is malformed") from error
|
||||
if not heading.startswith("## ") or not content:
|
||||
raise ValueError(f"curated evidence field {name} is malformed")
|
||||
return content
|
||||
|
||||
|
||||
def _parse_v2_excerpts(block: str) -> tuple[str, ...]:
|
||||
excerpts: list[str] = []
|
||||
for raw_excerpt in block.split(f"\n{_V2_EXCERPT_SEPARATOR}\n"):
|
||||
lines = raw_excerpt.split("\n")
|
||||
if any(line != ">" and not line.startswith("> ") for line in lines):
|
||||
raise ValueError("curated evidence supporting excerpt is malformed")
|
||||
excerpts.append("\n".join(line[2:] if line.startswith("> ") else "" for line in lines))
|
||||
if not excerpts:
|
||||
raise ValueError("curated evidence supporting excerpts are malformed")
|
||||
return tuple(excerpts)
|
||||
|
||||
|
||||
def _parse_inline_code(value: str, name: str) -> str:
|
||||
if len(value) < 2 or not value.startswith("`") or not value.endswith("`"):
|
||||
raise ValueError(f"curated evidence field {name} must be inline code")
|
||||
return value[1:-1]
|
||||
|
||||
|
||||
def _parse_v2_payload(
|
||||
kind: str, fields: dict[str, str], labels: dict[str, str],
|
||||
) -> tuple[dict, set[str]]:
|
||||
if kind == "glossary":
|
||||
expected = {"definition"}
|
||||
payload: dict = {"definition": fields.get("definition")}
|
||||
for name in ("synonyms", "variants"):
|
||||
if name in fields:
|
||||
expected.add(name)
|
||||
payload[name] = _parse_v2_list(fields[name], empty_label=labels["empty"])
|
||||
else:
|
||||
payload[name] = ()
|
||||
return payload, expected
|
||||
if kind == "domain":
|
||||
return {"rule": fields.get("rule")}, {"rule"}
|
||||
if kind == "enum":
|
||||
return {
|
||||
"column": _parse_inline_code(fields.get("column", ""), "column"),
|
||||
"values": _parse_v2_values(fields.get("values", ""), labels),
|
||||
}, {"column", "values"}
|
||||
if kind == "example":
|
||||
return {
|
||||
"question": fields.get("question"),
|
||||
"interpretation": fields.get("interpretation"),
|
||||
}, {"question", "interpretation"}
|
||||
if kind == "mapping":
|
||||
return {
|
||||
"concept": fields.get("concept"),
|
||||
"tables": _parse_v2_list(
|
||||
fields.get("tables", ""), code=True, empty_label=labels["empty"],
|
||||
),
|
||||
"columns": _parse_v2_list(
|
||||
fields.get("columns", ""), code=True, empty_label=labels["empty"],
|
||||
),
|
||||
}, {"concept", "tables", "columns"}
|
||||
if kind == "normalization":
|
||||
return {
|
||||
"input": fields.get("input"),
|
||||
"output": fields.get("output"),
|
||||
"rule": fields.get("rule"),
|
||||
}, {"input", "output", "rule"}
|
||||
if kind == "formula":
|
||||
sql = fields.get("sql", "")
|
||||
if not sql.startswith("```sql\n") or not sql.endswith("\n```"):
|
||||
raise ValueError("curated evidence SQL block is malformed")
|
||||
return {
|
||||
"concept": fields.get("concept"),
|
||||
"columns": _parse_v2_list(
|
||||
fields.get("columns", ""), code=True, empty_label=labels["empty"],
|
||||
),
|
||||
"sql": sql.removeprefix("```sql\n").removesuffix("\n```"),
|
||||
}, {"concept", "columns", "sql"}
|
||||
if kind == "reference":
|
||||
url = fields.get("url", "")
|
||||
if not url.startswith("<") or not url.endswith(">"):
|
||||
raise ValueError("curated evidence reference URL is malformed")
|
||||
return {
|
||||
"label": fields.get("label"),
|
||||
"url": url[1:-1],
|
||||
"description": fields.get("description"),
|
||||
}, {"label", "url", "description"}
|
||||
raise ValueError("curated evidence body kind is unsupported")
|
||||
|
||||
|
||||
def _parse_v2_review_items(value: str) -> tuple[ReviewItem, ...]:
|
||||
items: list[ReviewItem] = []
|
||||
for raw_item in value.split(f"\n{_V2_REVIEW_SEPARATOR}\n"):
|
||||
try:
|
||||
heading, detail = raw_item.split("\n\n", 1)
|
||||
except ValueError as error:
|
||||
raise ValueError("curated evidence review item is malformed") from error
|
||||
if not heading.startswith("### `") or not heading.endswith("`"):
|
||||
raise ValueError("curated evidence review item code is malformed")
|
||||
code = heading.removeprefix("### `").removesuffix("`")
|
||||
field = None
|
||||
marker = f"\n\n{_V2_REVIEW_FIELD}\n"
|
||||
if marker in detail:
|
||||
message, rendered_field = detail.split(marker, 1)
|
||||
match = re.fullmatch(r"\*\*[^*]+:\*\* `([^`]+)`", rendered_field)
|
||||
if match is None:
|
||||
raise ValueError("curated evidence review item field is malformed")
|
||||
field = match.group(1)
|
||||
else:
|
||||
message = detail
|
||||
if not code or not message:
|
||||
raise ValueError("curated evidence review item is malformed")
|
||||
items.append(ReviewItem(code=code, message=message, field=field))
|
||||
return tuple(items)
|
||||
|
||||
|
||||
def _parse_v2_body(data: dict, body: str) -> dict:
|
||||
kind = data.get("kind")
|
||||
body_owned = {"payload", "review_items"}
|
||||
if isinstance(kind, str):
|
||||
body_owned.add(kind)
|
||||
if body_owned.intersection(data):
|
||||
raise ValueError("curated evidence v2 frontmatter contains body-owned fields")
|
||||
title = data.get("title")
|
||||
if not isinstance(title, str) or not body.startswith(f"# {title}\n"):
|
||||
raise ValueError("curated evidence body title must match its metadata")
|
||||
fields: dict[str, str] = {}
|
||||
for match in _V2_FIELD.finditer(body):
|
||||
name = match.group(1)
|
||||
if name in fields:
|
||||
raise ValueError(f"curated evidence field {name} appears more than once")
|
||||
fields[name] = _parse_v2_field_content(name, match.group(2))
|
||||
skeleton = _V2_FIELD.sub("", body).strip()
|
||||
if skeleton != f"# {title}":
|
||||
raise ValueError("curated evidence body contains unstructured content")
|
||||
labels = _v2_labels(str(data.get("language", "")))
|
||||
payload, payload_fields = _parse_v2_payload(kind, fields, labels)
|
||||
common_fields = {"supporting_excerpts"}
|
||||
if "review_items" in fields:
|
||||
common_fields.add("review_items")
|
||||
if set(fields) != payload_fields | common_fields:
|
||||
raise ValueError("curated evidence body fields do not match its kind")
|
||||
provenance = data.get("provenance")
|
||||
if not isinstance(provenance, dict) or "supporting_excerpts" in provenance:
|
||||
raise ValueError("curated evidence v2 provenance is malformed")
|
||||
provenance["supporting_excerpts"] = _parse_v2_excerpts(fields["supporting_excerpts"])
|
||||
data["review_items"] = (
|
||||
_parse_v2_review_items(fields["review_items"])
|
||||
if "review_items" in fields
|
||||
else []
|
||||
)
|
||||
data["payload"] = payload
|
||||
return data
|
||||
|
||||
|
||||
def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvidence:
|
||||
"""Parse the canonical frontmatter representation of one Curated Evidence unit."""
|
||||
if not text.startswith("---\n"):
|
||||
@@ -260,11 +693,14 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
data = dict(raw)
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError("curated evidence frontmatter must be a mapping") from error
|
||||
if body.strip():
|
||||
raise ValueError("curated evidence must not contain an ignored body")
|
||||
kind = data.get("kind")
|
||||
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
|
||||
data["payload"] = data.pop(kind, None)
|
||||
if data.get("schema_version") == 2:
|
||||
data = _parse_v2_body(data, body)
|
||||
else:
|
||||
if body.strip():
|
||||
raise ValueError("curated evidence must not contain an ignored body")
|
||||
kind = data.get("kind")
|
||||
if "payload" not in data and kind in _PAYLOAD_TYPE_BY_KIND:
|
||||
data["payload"] = data.pop(kind, None)
|
||||
evidence = CuratedEvidence.model_validate(data)
|
||||
if path is not None:
|
||||
_validate_kind_directory(path, evidence.kind)
|
||||
@@ -273,6 +709,11 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
|
||||
def dump_curated_markdown(value: CuratedEvidence) -> str:
|
||||
"""Render canonical frontmatter with a human-readable kind-specific payload key."""
|
||||
if value.schema_version == 2:
|
||||
data = value.model_dump(mode="json", exclude={"payload", "review_items"})
|
||||
data["provenance"].pop("supporting_excerpts")
|
||||
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
|
||||
return f"---\n{frontmatter}---\n{_render_v2_body(value)}"
|
||||
data = value.model_dump(mode="json", exclude={"payload"})
|
||||
data[value.kind] = value.payload.model_dump(mode="json")
|
||||
frontmatter = yaml.safe_dump(data, allow_unicode=True, sort_keys=False)
|
||||
|
||||
@@ -1,202 +0,0 @@
|
||||
"""Legacy ConceptFormula reader and one-way migration into Curated Evidence.
|
||||
|
||||
The ``formulas/*.sql.md`` store is retained only for the migration window. Runtime
|
||||
lookup uses typed, published ``kind=formula`` Evidence instead. A session reviewer
|
||||
may still approve a formula locally; that is a proposal, not publication.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import re
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
||||
import yaml
|
||||
from pydantic import BaseModel, ValidationError
|
||||
|
||||
from tht.evidence.canonical import CuratedEvidence
|
||||
|
||||
FORMULAS_SUBDIR = "formulas"
|
||||
_SUFFIX_RE = re.compile(r"^(.*?)-(\d+)\.sql\.md$")
|
||||
|
||||
|
||||
class ConceptFormula(BaseModel):
|
||||
concept: str
|
||||
columns: list[str] = []
|
||||
sql: str
|
||||
# auto = sintetizzata dal modello (non ancora rivista); draft = bozza umana;
|
||||
# reviewed = approvata da un revisore. (spec §4.7.2: status auto/draft/reviewed)
|
||||
status: Literal["auto", "draft", "reviewed"] = "draft"
|
||||
sources: list[str] = []
|
||||
|
||||
@property
|
||||
def _slug(self) -> str:
|
||||
"""ASCII slug for the filename (matches textutil.slugify shape)."""
|
||||
import unicodedata
|
||||
|
||||
text = unicodedata.normalize("NFKD", self.concept).encode("ascii", "ignore").decode()
|
||||
return re.sub(r"[^a-z0-9_]+", "-", text.lower()).strip("-") or "formula"
|
||||
|
||||
def dump(self) -> str:
|
||||
meta = self.model_dump(exclude={"sql"}, mode="json")
|
||||
fm = yaml.safe_dump(meta, sort_keys=False, allow_unicode=True)
|
||||
return f"---\n{fm}---\n{self.sql}\n"
|
||||
|
||||
@classmethod
|
||||
def parse(cls, text: str) -> ConceptFormula:
|
||||
if not text.startswith("---\n"):
|
||||
raise ValueError("frontmatter mancante (atteso '---\\n' iniziale)")
|
||||
try:
|
||||
_, fm, body = text.split("---\n", 2)
|
||||
except ValueError as e:
|
||||
raise ValueError("frontmatter malformato") from e
|
||||
meta = yaml.safe_load(fm)
|
||||
if not isinstance(meta, dict):
|
||||
raise TypeError("frontmatter non valido")
|
||||
return cls.model_validate({**meta, "sql": body.strip("\n")})
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LegacyFormulaMigrationFailure:
|
||||
"""A reviewed legacy formula that must be resolved manually before publication."""
|
||||
|
||||
code: Literal["legacy_formula_requires_manual_review"]
|
||||
legacy_path: str
|
||||
formula: ConceptFormula
|
||||
problems: tuple[str, ...]
|
||||
|
||||
|
||||
def _next_path(root: Path, slug: str) -> Path:
|
||||
"""First free <slug>-<n>.sql.md path under root (n starts at 1)."""
|
||||
root.mkdir(parents=True, exist_ok=True)
|
||||
existing = sorted(root.glob(f"{slug}-*.sql.md"))
|
||||
n = 0
|
||||
for p in existing:
|
||||
m = _SUFFIX_RE.match(p.name)
|
||||
if m:
|
||||
n = max(n, int(m.group(2)))
|
||||
return root / f"{slug}-{n + 1}.sql.md"
|
||||
|
||||
|
||||
def save_formula(root: Path | str, formula: ConceptFormula) -> Path:
|
||||
"""Persist a single concept->formula unit under <root>/formulas/. Returns the
|
||||
written path. Append-only: each save writes a new file (so competing drafts and
|
||||
reviewed versions coexist until a curator prunes)."""
|
||||
root = Path(root)
|
||||
formulas_dir = root / FORMULAS_SUBDIR
|
||||
path = _next_path(formulas_dir, formula._slug)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(formula.dump())
|
||||
return path
|
||||
|
||||
|
||||
def _load_all(root: Path) -> list[ConceptFormula]:
|
||||
formulas_dir = root / FORMULAS_SUBDIR
|
||||
if not formulas_dir.is_dir():
|
||||
return []
|
||||
out: list[ConceptFormula] = []
|
||||
for f in sorted(formulas_dir.glob("*.sql.md")):
|
||||
try:
|
||||
out.append(ConceptFormula.parse(f.read_text()))
|
||||
except ValueError:
|
||||
continue # malformed file: skip, don't crash retrieval
|
||||
return out
|
||||
|
||||
|
||||
def retrieve_formula(root: Path | str, concept: str) -> list[ConceptFormula]:
|
||||
"""All formulas matching `concept` exactly under <root>/formulas/. Empty list if
|
||||
none (or if the dir is absent). Multiple results mean competing drafts/versions for
|
||||
the same concept -- the caller (gate) lets the reviewer pick."""
|
||||
return [f for f in _load_all(Path(root)) if f.concept == concept]
|
||||
|
||||
|
||||
def search_formulas(root: Path | str, query: str) -> list[ConceptFormula]:
|
||||
"""Read legacy formulas for migration tooling only (case-insensitive concept match)."""
|
||||
q = query.strip().lower()
|
||||
return [f for f in _load_all(Path(root)) if q in f.concept.lower()]
|
||||
|
||||
|
||||
def legacy_formula_to_curated(
|
||||
formula: ConceptFormula,
|
||||
*,
|
||||
legacy_path: str,
|
||||
source_content: str | None = None,
|
||||
) -> CuratedEvidence | LegacyFormulaMigrationFailure | None:
|
||||
"""Convert one reviewed legacy formula into its deterministic curated counterpart.
|
||||
|
||||
Drafts and model-generated formulas have no global publication status. Their
|
||||
caller must project them as session-local Formula proposals instead.
|
||||
"""
|
||||
if formula.status != "reviewed":
|
||||
return None
|
||||
problems: list[str] = []
|
||||
normalized_source: str | None = None
|
||||
if source_content is None:
|
||||
problems.append("original_source_required")
|
||||
else:
|
||||
from tht.evidence.authoring import normalize_source_text
|
||||
|
||||
normalized_source = normalize_source_text(source_content)
|
||||
try:
|
||||
original_formula = ConceptFormula.parse(source_content)
|
||||
except (TypeError, ValidationError, ValueError, yaml.YAMLError):
|
||||
problems.append("original_source_invalid")
|
||||
else:
|
||||
if original_formula != formula:
|
||||
problems.append("original_source_mismatch")
|
||||
source_notes = tuple(formula.sources)
|
||||
if not source_notes:
|
||||
problems.append("supporting_excerpts_required")
|
||||
elif normalized_source is not None and any(
|
||||
unicodedata.normalize("NFC", note.replace("\r\n", "\n").replace("\r", "\n"))
|
||||
not in normalized_source
|
||||
for note in source_notes
|
||||
):
|
||||
problems.append("supporting_excerpt_unverified")
|
||||
if problems:
|
||||
return LegacyFormulaMigrationFailure(
|
||||
code="legacy_formula_requires_manual_review",
|
||||
legacy_path=legacy_path,
|
||||
formula=formula,
|
||||
problems=tuple(sorted(problems)),
|
||||
)
|
||||
assert normalized_source is not None
|
||||
source_sha256 = hashlib.sha256(normalized_source.encode("utf-8")).hexdigest()
|
||||
source_file = legacy_path if legacy_path.startswith("source/") else f"source/{legacy_path}"
|
||||
# The legacy path is the immutable identity of this unit during migration. Keeping
|
||||
# its full digest avoids a duplicate public ID when the same concept has reviewed
|
||||
# competing formulas, while retaining the readable concept slug as the prefix.
|
||||
legacy_identity = hashlib.sha256(legacy_path.encode("utf-8")).hexdigest()
|
||||
try:
|
||||
return CuratedEvidence.model_validate({
|
||||
"schema_version": 1,
|
||||
"id": f"evidence:{formula._slug}-{legacy_identity}",
|
||||
"title": formula.concept[:1].upper() + formula.concept[1:],
|
||||
"kind": "formula",
|
||||
"purposes": ["schema_linking", "sql_generation"],
|
||||
"applies_to": {"concepts": [formula.concept], "columns": formula.columns},
|
||||
"language": "it",
|
||||
"provenance": {
|
||||
"source_file": source_file,
|
||||
"source_sha256": f"sha256:{source_sha256}",
|
||||
"supporting_excerpts": source_notes,
|
||||
},
|
||||
"review_items": [],
|
||||
"payload": {
|
||||
"concept": formula.concept,
|
||||
"columns": formula.columns,
|
||||
"sql": formula.sql,
|
||||
},
|
||||
})
|
||||
except ValidationError as error:
|
||||
return LegacyFormulaMigrationFailure(
|
||||
code="legacy_formula_requires_manual_review",
|
||||
legacy_path=legacy_path,
|
||||
formula=formula,
|
||||
problems=tuple(sorted(
|
||||
".".join(str(part) for part in issue["loc"])
|
||||
for issue in error.errors()
|
||||
)),
|
||||
)
|
||||
@@ -58,9 +58,5 @@ def load_evidence_dir(root: Path) -> list[EvidenceDoc]:
|
||||
for f in sorted(root.rglob("*.md")):
|
||||
if f.name.upper().startswith("README"):
|
||||
continue
|
||||
# I file formula (concept->SQL, frontmatter diverso) vivono sotto formulas/ con
|
||||
# estensione .sql.md: non sono EvidenceDoc, li gestisce formula_store (D14b).
|
||||
if f.name.endswith(".sql.md"):
|
||||
continue
|
||||
docs.append(EvidenceDoc.parse(f.read_text(), path=f))
|
||||
return docs
|
||||
|
||||
@@ -1,6 +1,19 @@
|
||||
from typing import Protocol
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
from tht.vectorstore.store import VectorStore
|
||||
from tht.vectorstore.store import VectorHit
|
||||
|
||||
|
||||
class SemanticSearcher(Protocol):
|
||||
def search(
|
||||
self,
|
||||
embedding: list[float],
|
||||
*,
|
||||
top_n: int,
|
||||
kinds: list[str] | None,
|
||||
query_text: str | None = None,
|
||||
) -> list[VectorHit]: ...
|
||||
|
||||
|
||||
class SearchResult(BaseModel):
|
||||
@@ -92,7 +105,7 @@ def combined_search(
|
||||
keyword: str,
|
||||
*,
|
||||
lsh_hits: list[tuple[str, str, str, float]] | None,
|
||||
store: VectorStore,
|
||||
store: SemanticSearcher,
|
||||
embedder,
|
||||
top: int,
|
||||
rrf_k: int,
|
||||
|
||||
@@ -1,20 +1,11 @@
|
||||
import hashlib
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
|
||||
from sqlalchemy import Engine, text
|
||||
|
||||
from tht.vectorstore.records import VectorRecord
|
||||
|
||||
|
||||
def content_hash(content: str) -> str:
|
||||
return hashlib.sha256(content.encode()).hexdigest()
|
||||
|
||||
|
||||
def _to_vector_literal(vec: list[float]) -> str:
|
||||
return "[" + ",".join(f"{x:.8f}" for x in vec) + "]"
|
||||
|
||||
|
||||
@dataclass
|
||||
class SyncStats:
|
||||
added: int = 0
|
||||
@@ -35,10 +26,7 @@ class VectorHit:
|
||||
|
||||
|
||||
def hit_from_metadata(similarity: float, metadata: dict | None) -> VectorHit:
|
||||
"""Ricostruisce un VectorHit dal solo `metadata` (più la similarity). È l'unico modo
|
||||
disponibile leggendo via REST (`search_similar` ritorna id/similarity/metadata), e viene
|
||||
usato anche dalla lettura diretta per avere un'unica logica. Tollerante: usa default sui
|
||||
campi assenti (es. metadata estranei della tabella fake remota)."""
|
||||
"""Build a transport-neutral hit from the metadata returned by a vector adapter."""
|
||||
md = metadata or {}
|
||||
return VectorHit(
|
||||
id=md.get("record_key", ""),
|
||||
@@ -49,145 +37,3 @@ def hit_from_metadata(similarity: float, metadata: dict | None) -> VectorHit:
|
||||
metadata=md,
|
||||
similarity=float(similarity),
|
||||
)
|
||||
|
||||
|
||||
class VectorStore:
|
||||
"""Legacy table-scoped vector store retained for compatibility fixtures.
|
||||
|
||||
New operational semantic storage is handled by the Qdrant adapter. This class preserves
|
||||
the older SQL-table contract used by historical tests and migration checks: `id`
|
||||
(BIGSERIAL), `embedding vector(N)`, and `metadata jsonb`; the extra columns
|
||||
(`record_key`, `kind`, `content_hash`) serve only the loader and are not exposed by REST.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self, engine: Engine, schema: str = "vectors", table: str = "records", dim: int = 768
|
||||
):
|
||||
self.engine = engine
|
||||
self.schema = schema
|
||||
self.dim = dim
|
||||
self._table = f"{schema}.{table}"
|
||||
|
||||
def init_schema(self) -> None:
|
||||
with self.engine.begin() as conn:
|
||||
conn.execute(text("CREATE EXTENSION IF NOT EXISTS vector"))
|
||||
conn.execute(text(f"CREATE SCHEMA IF NOT EXISTS {self.schema}"))
|
||||
conn.execute(text(f"""
|
||||
CREATE TABLE IF NOT EXISTS {self._table} (
|
||||
id bigserial PRIMARY KEY,
|
||||
record_key text UNIQUE NOT NULL,
|
||||
kind text NOT NULL,
|
||||
content_hash text NOT NULL,
|
||||
metadata jsonb NOT NULL DEFAULT '{{}}',
|
||||
embedding vector({self.dim}) NOT NULL,
|
||||
indexed_at timestamptz NOT NULL DEFAULT now()
|
||||
)
|
||||
"""))
|
||||
conn.execute(text(
|
||||
f"CREATE INDEX IF NOT EXISTS {self._idx('embedding')} ON {self._table} "
|
||||
f"USING hnsw (embedding vector_cosine_ops) WITH (m = 16, ef_construction = 200)"
|
||||
))
|
||||
conn.execute(text(
|
||||
f"CREATE INDEX IF NOT EXISTS {self._idx('kind')} ON {self._table} (kind)"
|
||||
))
|
||||
# GRANT al ruolo di sola lettura della REST, solo se esiste (assente in test/locale).
|
||||
conn.execute(text(f"""
|
||||
DO $$ BEGIN
|
||||
IF EXISTS (SELECT 1 FROM pg_roles WHERE rolname = 'vector_reader') THEN
|
||||
EXECUTE 'GRANT SELECT ON {self._table} TO vector_reader';
|
||||
END IF;
|
||||
END $$;
|
||||
"""))
|
||||
|
||||
def _idx(self, suffix: str) -> str:
|
||||
return f"{self._table.replace('.', '_')}_{suffix}_idx"
|
||||
|
||||
def clear(self) -> None:
|
||||
with self.engine.begin() as conn:
|
||||
conn.execute(text(f"DELETE FROM {self._table}"))
|
||||
|
||||
def existing_hashes(self, kinds: set[str]) -> dict[str, str]:
|
||||
q = text(
|
||||
f"SELECT record_key, content_hash FROM {self._table} WHERE kind = ANY(:kinds)"
|
||||
)
|
||||
with self.engine.connect() as conn:
|
||||
return dict(conn.execute(q, {"kinds": list(kinds)}).fetchall())
|
||||
|
||||
def sync(self, records: list[VectorRecord], embedder, kinds: set[str]) -> SyncStats:
|
||||
"""Allinea l'indice ai record correnti (per i kind dati): embedda solo il nuovo
|
||||
o il modificato, elimina cio' che non esiste piu'."""
|
||||
stats = SyncStats()
|
||||
existing = self.existing_hashes(kinds)
|
||||
current_ids = {r.id for r in records}
|
||||
|
||||
to_embed: list[VectorRecord] = []
|
||||
for r in records:
|
||||
h = content_hash(r.content)
|
||||
if r.id not in existing:
|
||||
to_embed.append(r)
|
||||
stats.added += 1
|
||||
elif existing[r.id] != h:
|
||||
to_embed.append(r)
|
||||
stats.updated += 1
|
||||
else:
|
||||
stats.unchanged += 1
|
||||
|
||||
vectors = embedder.embed_documents([r.content for r in to_embed]) if to_embed else []
|
||||
|
||||
upsert = text(f"""
|
||||
INSERT INTO {self._table}
|
||||
(record_key, kind, content_hash, metadata, embedding)
|
||||
VALUES
|
||||
(:record_key, :kind, :content_hash, CAST(:metadata AS jsonb),
|
||||
CAST(:embedding AS vector))
|
||||
ON CONFLICT (record_key) DO UPDATE SET
|
||||
kind = EXCLUDED.kind, content_hash = EXCLUDED.content_hash,
|
||||
metadata = EXCLUDED.metadata, embedding = EXCLUDED.embedding,
|
||||
indexed_at = now()
|
||||
""")
|
||||
stale = [i for i in existing if i not in current_ids]
|
||||
with self.engine.begin() as conn:
|
||||
for r, vec in zip(to_embed, vectors):
|
||||
conn.execute(upsert, {
|
||||
"record_key": r.id, "kind": r.kind,
|
||||
"content_hash": content_hash(r.content),
|
||||
"metadata": json.dumps(_pack_metadata(r)),
|
||||
"embedding": _to_vector_literal(vec),
|
||||
})
|
||||
if stale:
|
||||
conn.execute(
|
||||
text(f"DELETE FROM {self._table} WHERE record_key = ANY(:ids)"),
|
||||
{"ids": stale},
|
||||
)
|
||||
stats.deleted = len(stale)
|
||||
return stats
|
||||
|
||||
def search(
|
||||
self, query_vec: list[float], top_n: int = 10, kinds: list[str] | None = None
|
||||
) -> list[VectorHit]:
|
||||
where = "WHERE kind = ANY(:kinds)" if kinds else ""
|
||||
q = text(f"""
|
||||
SELECT metadata, 1 - (embedding <=> CAST(:q AS vector)) AS similarity
|
||||
FROM {self._table}
|
||||
{where}
|
||||
ORDER BY embedding <=> CAST(:q AS vector)
|
||||
LIMIT :top_n
|
||||
""")
|
||||
params: dict = {"q": _to_vector_literal(query_vec), "top_n": top_n}
|
||||
if kinds:
|
||||
params["kinds"] = kinds
|
||||
with self.engine.connect() as conn:
|
||||
rows = conn.execute(q, params).fetchall()
|
||||
return [hit_from_metadata(r.similarity, r.metadata) for r in rows]
|
||||
|
||||
|
||||
def _pack_metadata(r: VectorRecord) -> dict:
|
||||
"""Impacchetta nel `metadata` (unica colonna letta via REST) tutta la semantica Thoth."""
|
||||
return {
|
||||
"kind": r.kind,
|
||||
"ref": r.ref,
|
||||
"record_key": r.id,
|
||||
"title": r.title,
|
||||
"content": r.content,
|
||||
**r.metadata,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user