refactor(harness): renaming prodotto tht (Onda -1)

Thoth (tht) è il prodotto, PSD è il cliente. Nessun riferimento al contesto
clinico nel codice.

Rinomine:
- comando+package nsp→tht (dir nsp/→tht/, 46 import, pyproject entry point)
- gate nsp-gate.js→tht-gate.js (+ rewrite token, relayIfNspFails→relayIfThtFails)
- workspace chirone.{example,test}.yaml→tht.{example,test}.yaml (generici)
- env THOTH_→THT_ (19 var) + NSP_ stragglers (NSP_HARNESS_ROOT, NSP_SESSION)
- commenti/docstring chirone/psdwp3/policlinico neutralizzati ('the reference
  implementation', 'the DWH')

Aggiunto [tool.setuptools.packages.find] include=['tht*'] (necessario: l'auto-
discovery rompeva con tht/ + workspaces/ come top-level multipli).

.env operatore aggiornato in-place (prefissi THT_, valori preservati, gitignored).

Verifica: pytest 109 passed, npm test 14 pass, tht phase meta --json OK, zero
residui nsp/THOTH_/NSP_/chirone nel package.
This commit is contained in:
2026-06-27 10:33:16 +02:00
parent 0dcc0246dc
commit fc5fbe6b65
72 changed files with 309 additions and 306 deletions
View File
+100
View File
@@ -0,0 +1,100 @@
"""SQL concept->formula evidence store (spec D14b, §4.7).
A concept (e.g. 'fascia pediatrica', 'ablazione') maps to a reusable SQL formula
(a CASE WHEN ...) that derives it from physical columns. These are reviewable
units: the gate surfaces a candidate formula, the reviewer approves or rejects it
(decision types concept_formula_approved / concept_formula_rejected), and approved
formulas travel with the schema-linking artifact.
Storage: one file per formula, frontmatter YAML + SQL body (same shape as
EvidenceDoc.parse). Directory layout: <root>/formulas/<slug>-<n>.sql.md.
Retrieve is by concept (may return several, e.g. competing drafts vs reviewed).
"""
from __future__ import annotations
import re
from pathlib import Path
from typing import Literal
import yaml
from pydantic import BaseModel
FORMULAS_SUBDIR = "formulas"
_SUFFIX_RE = re.compile(r"^(.*?)-(\d+)\.sql\.md$")
class ConceptFormula(BaseModel):
concept: str
columns: list[str] = []
sql: str
status: Literal["draft", "reviewed"] = "draft"
sources: list[str] = []
@property
def _slug(self) -> str:
"""ASCII slug for the filename (matches textutil.slugify shape)."""
import unicodedata
text = unicodedata.normalize("NFKD", self.concept).encode("ascii", "ignore").decode()
return re.sub(r"[^a-z0-9_]+", "-", text.lower()).strip("-") or "formula"
def dump(self) -> str:
meta = self.model_dump(exclude={"sql"}, mode="json")
fm = yaml.safe_dump(meta, sort_keys=False, allow_unicode=True)
return f"---\n{fm}---\n{self.sql}\n"
@classmethod
def parse(cls, text: str) -> "ConceptFormula":
if not text.startswith("---\n"):
raise ValueError("frontmatter mancante (atteso '---\\n' iniziale)")
try:
_, fm, body = text.split("---\n", 2)
except ValueError as e:
raise ValueError("frontmatter malformato") from e
meta = yaml.safe_load(fm)
if not isinstance(meta, dict):
raise ValueError("frontmatter non valido")
return cls.model_validate({**meta, "sql": body.strip("\n")})
def _next_path(root: Path, slug: str) -> Path:
"""First free <slug>-<n>.sql.md path under root (n starts at 1)."""
root.mkdir(parents=True, exist_ok=True)
existing = sorted(root.glob(f"{slug}-*.sql.md"))
n = 0
for p in existing:
m = _SUFFIX_RE.match(p.name)
if m:
n = max(n, int(m.group(2)))
return root / f"{slug}-{n + 1}.sql.md"
def save_formula(root: Path | str, formula: ConceptFormula) -> Path:
"""Persist a single concept->formula unit under <root>/formulas/. Returns the
written path. Append-only: each save writes a new file (so competing drafts and
reviewed versions coexist until a curator prunes)."""
root = Path(root)
formulas_dir = root / FORMULAS_SUBDIR
path = _next_path(formulas_dir, formula._slug)
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(formula.dump())
return path
def retrieve_formula(root: Path | str, concept: str) -> list[ConceptFormula]:
"""All formulas for `concept` under <root>/formulas/. Empty list if none (or if
the dir is absent). Multiple results mean competing drafts/versions for the same
concept -- the caller (gate) lets the reviewer pick."""
root = Path(root)
formulas_dir = root / FORMULAS_SUBDIR
if not formulas_dir.is_dir():
return []
out: list[ConceptFormula] = []
for f in sorted(formulas_dir.glob("*.sql.md")):
try:
formula = ConceptFormula.parse(f.read_text())
except ValueError:
continue # malformed file: skip, don't crash retrieval
if formula.concept == concept:
out.append(formula)
return out
+62
View File
@@ -0,0 +1,62 @@
from pathlib import Path
from typing import Literal
import yaml
from pydantic import BaseModel, ValidationError
class EvidenceError(Exception):
pass
class EvidenceDoc(BaseModel):
id: str
title: str
# tier e status sono opzionali: la sola presenza di un documento basta a
# vettorizzarlo, quindi l'autore ETL non e' obbligato a compilarli.
tier: Literal["structural", "concept"] = "structural"
status: Literal["auto", "draft", "reviewed"] = "reviewed"
sources: list[str] = []
tables: list[str] = []
concepts: list[str] = []
body: str = ""
path: Path | None = None # valorizzato al load, escluso dal dump
@classmethod
def parse(cls, text: str, path: Path | None = None) -> "EvidenceDoc":
if not text.startswith("---\n"):
raise EvidenceError(f"frontmatter mancante in {path or '<testo>'}")
try:
_, fm, body = text.split("---\n", 2)
except ValueError as e:
raise EvidenceError(f"frontmatter malformato in {path or '<testo>'}") from e
meta = yaml.safe_load(fm)
if not isinstance(meta, dict):
raise EvidenceError(f"frontmatter non valido in {path or '<testo>'}")
try:
return cls.model_validate({**meta, "body": body.strip("\n"), "path": path})
except ValidationError as e:
raise EvidenceError(f"evidence non valida in {path or '<testo>'}:\n{e}") from e
def dump(self) -> str:
meta = self.model_dump(exclude={"body", "path"}, mode="json")
fm = yaml.safe_dump(meta, sort_keys=False, allow_unicode=True)
return f"---\n{fm}---\n{self.body}\n"
def save(self, path: Path) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(self.dump())
self.path = path
def load_evidence_dir(root: Path) -> list[EvidenceDoc]:
"""Carica ricorsivamente tutte le evidence sotto `root`, preservando la
gerarchia per dominio. I file README (di sola navigazione) sono ignorati."""
docs: list[EvidenceDoc] = []
if not root.is_dir():
return docs
for f in sorted(root.rglob("*.md")):
if f.name.upper().startswith("README"):
continue
docs.append(EvidenceDoc.parse(f.read_text(), path=f))
return docs