Implementazione del piano di remediation progressiva sui difetti emersi dall'analisi dell'harness. Tutto verificato: 214 test Python (incl. L0 su Postgres reale), 14 test JS del gate, ruff pulito. Blocco 1 (CRITICA, integrazione gate↔CLI): - phase advance: gate usa --auto + exit 6; reviewer_confirm kind:phase fa advance esplicito che applica i prerequisiti (prima non avanzava per le fasi a conferma umana). - cte plan riceve i --name dal gate (param names); set-question con id posizionale; skill `tht search find`; nuovo comando `tht memory save-one` con dedup hash client-side in save_one_memory. Blocco 2 (D15, stato post-rollback): - campo `phase` su DecisionRecord + effective_decisions phase-aware per i subject "a nome" (cte_approved ecc.); _compute_promotions e finalize sulla vista effective; finalize confronta col piano CTE effettivo, non glob; `decision add --retracts` + comando `decision retract`. Blocco 3 (D7 read-only + D6 manifest): - assert_read_only su tutti e quattro i codepath (direct + REST); - manifest author/summary/updated_at/updated_by/schema_version popolati + helper touch_manifest sulle mutazioni. Blocco 4-5 (D14a/D14b): - decision_min_phase data-driven via `emits:` in workflow.yaml; - formula evidence: status auto, search_formulas, gruppo CLI `tht formula`, `search find --kind formula`, load_evidence_dir salta i .sql.md. Blocco 6 (robustezza): - taskdoc slice promoted_tables + bound enforced; report escaping/bound + rsplit note; filtro kind reader REST/direct; conteggio upserted robusto; guard REST run_query non-list; LSH disallineato -> LshIndexError. Blocco 7 (pulizia): - dead code gate e KIND_TO_TABLE morto rimossi; doc Postgres-only (README + connection.py). Blocco 0 (parziale): test di compatibilità firma gate↔CLI (tests/integration). Rinviati: fake-Pi runtime completo, artifact-gate da disco (#23), parità eligibility REST/direct (#28), unificazione reserved-labels (#30), memory_rejected da deselezione (#33). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
112 lines
4.3 KiB
Python
112 lines
4.3 KiB
Python
"""SQL concept->formula evidence store (spec D14b, §4.7).
|
|
|
|
A concept (e.g. 'fascia pediatrica', 'ablazione') maps to a reusable SQL formula
|
|
(a CASE WHEN ...) that derives it from physical columns. These are reviewable
|
|
units: the gate surfaces a candidate formula, the reviewer approves or rejects it
|
|
(decision types concept_formula_approved / concept_formula_rejected), and approved
|
|
formulas travel with the schema-linking artifact.
|
|
|
|
Storage: one file per formula, frontmatter YAML + SQL body (same shape as
|
|
EvidenceDoc.parse). Directory layout: <root>/formulas/<slug>-<n>.sql.md.
|
|
Retrieve is by concept (may return several, e.g. competing drafts vs reviewed).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from pathlib import Path
|
|
from typing import Literal
|
|
|
|
import yaml
|
|
from pydantic import BaseModel
|
|
|
|
FORMULAS_SUBDIR = "formulas"
|
|
_SUFFIX_RE = re.compile(r"^(.*?)-(\d+)\.sql\.md$")
|
|
|
|
|
|
class ConceptFormula(BaseModel):
|
|
concept: str
|
|
columns: list[str] = []
|
|
sql: str
|
|
# auto = sintetizzata dal modello (non ancora rivista); draft = bozza umana;
|
|
# reviewed = approvata da un revisore. (spec §4.7.2: status auto/draft/reviewed)
|
|
status: Literal["auto", "draft", "reviewed"] = "draft"
|
|
sources: list[str] = []
|
|
|
|
@property
|
|
def _slug(self) -> str:
|
|
"""ASCII slug for the filename (matches textutil.slugify shape)."""
|
|
import unicodedata
|
|
|
|
text = unicodedata.normalize("NFKD", self.concept).encode("ascii", "ignore").decode()
|
|
return re.sub(r"[^a-z0-9_]+", "-", text.lower()).strip("-") or "formula"
|
|
|
|
def dump(self) -> str:
|
|
meta = self.model_dump(exclude={"sql"}, mode="json")
|
|
fm = yaml.safe_dump(meta, sort_keys=False, allow_unicode=True)
|
|
return f"---\n{fm}---\n{self.sql}\n"
|
|
|
|
@classmethod
|
|
def parse(cls, text: str) -> "ConceptFormula":
|
|
if not text.startswith("---\n"):
|
|
raise ValueError("frontmatter mancante (atteso '---\\n' iniziale)")
|
|
try:
|
|
_, fm, body = text.split("---\n", 2)
|
|
except ValueError as e:
|
|
raise ValueError("frontmatter malformato") from e
|
|
meta = yaml.safe_load(fm)
|
|
if not isinstance(meta, dict):
|
|
raise ValueError("frontmatter non valido")
|
|
return cls.model_validate({**meta, "sql": body.strip("\n")})
|
|
|
|
|
|
def _next_path(root: Path, slug: str) -> Path:
|
|
"""First free <slug>-<n>.sql.md path under root (n starts at 1)."""
|
|
root.mkdir(parents=True, exist_ok=True)
|
|
existing = sorted(root.glob(f"{slug}-*.sql.md"))
|
|
n = 0
|
|
for p in existing:
|
|
m = _SUFFIX_RE.match(p.name)
|
|
if m:
|
|
n = max(n, int(m.group(2)))
|
|
return root / f"{slug}-{n + 1}.sql.md"
|
|
|
|
|
|
def save_formula(root: Path | str, formula: ConceptFormula) -> Path:
|
|
"""Persist a single concept->formula unit under <root>/formulas/. Returns the
|
|
written path. Append-only: each save writes a new file (so competing drafts and
|
|
reviewed versions coexist until a curator prunes)."""
|
|
root = Path(root)
|
|
formulas_dir = root / FORMULAS_SUBDIR
|
|
path = _next_path(formulas_dir, formula._slug)
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
path.write_text(formula.dump())
|
|
return path
|
|
|
|
|
|
def _load_all(root: Path) -> list[ConceptFormula]:
|
|
formulas_dir = root / FORMULAS_SUBDIR
|
|
if not formulas_dir.is_dir():
|
|
return []
|
|
out: list[ConceptFormula] = []
|
|
for f in sorted(formulas_dir.glob("*.sql.md")):
|
|
try:
|
|
out.append(ConceptFormula.parse(f.read_text()))
|
|
except ValueError:
|
|
continue # malformed file: skip, don't crash retrieval
|
|
return out
|
|
|
|
|
|
def retrieve_formula(root: Path | str, concept: str) -> list[ConceptFormula]:
|
|
"""All formulas matching `concept` exactly under <root>/formulas/. Empty list if
|
|
none (or if the dir is absent). Multiple results mean competing drafts/versions for
|
|
the same concept -- the caller (gate) lets the reviewer pick."""
|
|
return [f for f in _load_all(Path(root)) if f.concept == concept]
|
|
|
|
|
|
def search_formulas(root: Path | str, query: str) -> list[ConceptFormula]:
|
|
"""Formulas whose concept contains `query` (case-insensitive). Used by
|
|
`tht search find --kind formula` (D14b retrieval, §4.7.2): the reviewer searches a
|
|
concept term and gets the candidate formulas to approve before they reach the CTE."""
|
|
q = query.strip().lower()
|
|
return [f for f in _load_all(Path(root)) if q in f.concept.lower()]
|