refactor(harness): renaming prodotto tht (Onda -1)
Thoth (tht) è il prodotto, PSD è il cliente. Nessun riferimento al contesto
clinico nel codice.
Rinomine:
- comando+package nsp→tht (dir nsp/→tht/, 46 import, pyproject entry point)
- gate nsp-gate.js→tht-gate.js (+ rewrite token, relayIfNspFails→relayIfThtFails)
- workspace chirone.{example,test}.yaml→tht.{example,test}.yaml (generici)
- env THOTH_→THT_ (19 var) + NSP_ stragglers (NSP_HARNESS_ROOT, NSP_SESSION)
- commenti/docstring chirone/psdwp3/policlinico neutralizzati ('the reference
implementation', 'the DWH')
Aggiunto [tool.setuptools.packages.find] include=['tht*'] (necessario: l'auto-
discovery rompeva con tht/ + workspaces/ come top-level multipli).
.env operatore aggiornato in-place (prefissi THT_, valori preservati, gitignored).
Verifica: pytest 109 passed, npm test 14 pass, tht phase meta --json OK, zero
residui nsp/THOTH_/NSP_/chirone nel package.
This commit is contained in:
@@ -0,0 +1,93 @@
|
||||
"""Classificazione di column eligibility (principio trasversale Thoth).
|
||||
|
||||
Vedi docs/superpowers/specs/2026-06-13-tht-column-eligibility-principle.md.
|
||||
Il testo ampio (lettere di dimissione, note, anamnesi) è ignorato ovunque; i dati
|
||||
provengono solo da numerici, enum, temporali, booleani e testo breve.
|
||||
"""
|
||||
|
||||
import re
|
||||
|
||||
from tht.config import EligibilityConfig
|
||||
from tht.db.sampling import is_text_type
|
||||
from tht.mschema.models import ColumnAnnotation, ColumnPhysical, PhysicalSchema
|
||||
|
||||
_NUMERIC_PREFIXES = (
|
||||
"smallint", "integer", "bigint", "numeric", "decimal", "real", "double", "money",
|
||||
)
|
||||
_TEMPORAL_PREFIXES = ("date", "time", "timestamp", "interval")
|
||||
_LEN_RE = re.compile(r"\((\d+)\)")
|
||||
|
||||
|
||||
def _declared_len(pg_type: str) -> int | None:
|
||||
m = _LEN_RE.search(pg_type)
|
||||
return int(m.group(1)) if m else None
|
||||
|
||||
|
||||
def classify_column(
|
||||
pg_type: str,
|
||||
is_enum: bool,
|
||||
sampled_avg: float | None,
|
||||
sampled_max: int | None,
|
||||
cfg: EligibilityConfig,
|
||||
) -> tuple[bool, str]:
|
||||
"""Classifica una colonna come (eligible, reason). Funzione pura, senza I/O."""
|
||||
t = pg_type.strip().lower()
|
||||
if t.endswith("[]"):
|
||||
return False, "wide_text"
|
||||
if is_enum:
|
||||
return True, "enum"
|
||||
if t.startswith(_NUMERIC_PREFIXES):
|
||||
return True, "numeric"
|
||||
if t.startswith("boolean"):
|
||||
return True, "boolean"
|
||||
if t.startswith(_TEMPORAL_PREFIXES):
|
||||
return True, "temporal"
|
||||
if t.startswith("uuid"):
|
||||
return True, "code"
|
||||
if is_text_type(t):
|
||||
declared = _declared_len(t)
|
||||
if declared is not None and declared <= cfg.max_declared_len:
|
||||
return True, "short_text"
|
||||
# bound grande o text/varchar non vincolato: decide il dato campionato
|
||||
if sampled_avg is None or sampled_max is None:
|
||||
return False, "wide_text"
|
||||
if sampled_avg <= cfg.max_avg_length and sampled_max <= cfg.max_sampled_len:
|
||||
return True, "short_text"
|
||||
return False, "wide_text"
|
||||
return False, "wide_text"
|
||||
|
||||
|
||||
def _sampled_stats(examples: list[str]) -> tuple[float | None, int | None]:
|
||||
if not examples:
|
||||
return None, None
|
||||
lengths = [len(v) for v in examples]
|
||||
return sum(lengths) / len(lengths), max(lengths)
|
||||
|
||||
|
||||
def classify_all(physical: PhysicalSchema, cfg: EligibilityConfig) -> None:
|
||||
"""Assegna eligible/eligibility_reason a ogni colonna (in-place) e azzera gli
|
||||
examples delle colonne ignored. Da chiamare DOPO add_examples."""
|
||||
ignore_by_name = {name.lower() for name in cfg.ignore_columns}
|
||||
for table in physical.tables.values():
|
||||
for column_name, column in table.columns.items():
|
||||
if column_name.lower() in ignore_by_name:
|
||||
# colonna di servizio (ETL/audit): ignorata a prescindere dal tipo
|
||||
column.eligible = False
|
||||
column.eligibility_reason = "ignored_by_name"
|
||||
column.examples = []
|
||||
continue
|
||||
avg, mx = _sampled_stats(column.examples)
|
||||
eligible, reason = classify_column(column.type, column.is_enum, avg, mx, cfg)
|
||||
column.eligible = eligible
|
||||
column.eligibility_reason = reason
|
||||
if not eligible:
|
||||
column.examples = []
|
||||
|
||||
|
||||
def effective_eligibility(
|
||||
column: ColumnPhysical, annotation: ColumnAnnotation | None
|
||||
) -> tuple[bool, str]:
|
||||
"""Eligibilità effettiva: l'override in annotations.yaml vince sul fisico."""
|
||||
if annotation is not None and annotation.eligible is not None:
|
||||
return annotation.eligible, "override"
|
||||
return column.eligible, column.eligibility_reason
|
||||
@@ -0,0 +1,15 @@
|
||||
from tht.mschema.models import Annotations, PhysicalSchema
|
||||
|
||||
|
||||
def find_orphans(physical: PhysicalSchema, annotations: Annotations) -> list[str]:
|
||||
"""Annotazioni che puntano a oggetti spariti dal fisico. Non le rimuove mai."""
|
||||
orphans: list[str] = []
|
||||
for table_name, table_ann in annotations.tables.items():
|
||||
table = physical.tables.get(table_name)
|
||||
if table is None:
|
||||
orphans.append(table_name)
|
||||
continue
|
||||
for column_name in table_ann.columns:
|
||||
if column_name not in table.columns:
|
||||
orphans.append(f"{table_name}.{column_name}")
|
||||
return orphans
|
||||
@@ -0,0 +1,90 @@
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Self
|
||||
|
||||
import yaml
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
class _YamlModel(BaseModel):
|
||||
def to_yaml(self, path: Path) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
data = self.model_dump(by_alias=True, mode="json", exclude_defaults=False)
|
||||
path.write_text(
|
||||
yaml.safe_dump(data, sort_keys=False, allow_unicode=True, width=120)
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def from_yaml(cls, path: Path) -> Self:
|
||||
raw = yaml.safe_load(path.read_text())
|
||||
return cls.model_validate(raw)
|
||||
|
||||
|
||||
class ColumnPhysical(BaseModel):
|
||||
type: str
|
||||
nullable: bool = True
|
||||
pk: bool = False
|
||||
default: str | None = None
|
||||
comment: str = ""
|
||||
examples: list[str] = []
|
||||
is_enum: bool = False
|
||||
eligible: bool = True
|
||||
eligibility_reason: str = ""
|
||||
|
||||
|
||||
class ForeignKey(BaseModel):
|
||||
columns: list[str]
|
||||
ref_table: str
|
||||
ref_columns: list[str]
|
||||
name: str = ""
|
||||
|
||||
|
||||
class Index(BaseModel):
|
||||
name: str
|
||||
columns: list[str]
|
||||
unique: bool = False
|
||||
primary: bool = False
|
||||
type: str = "btree"
|
||||
|
||||
|
||||
class TablePhysical(BaseModel):
|
||||
comment: str = ""
|
||||
row_count: int = 0 # stima da pg_class.reltuples
|
||||
columns: dict[str, ColumnPhysical] = {}
|
||||
foreign_keys: list[ForeignKey] = []
|
||||
indexes: list[Index] = []
|
||||
|
||||
|
||||
class PhysicalSchema(_YamlModel):
|
||||
database: str
|
||||
db_schema: str = Field(alias="schema")
|
||||
introspected_at: datetime
|
||||
tables: dict[str, TablePhysical] = {}
|
||||
|
||||
model_config = {"populate_by_name": True}
|
||||
|
||||
|
||||
class ColumnAnnotation(BaseModel):
|
||||
description: str = ""
|
||||
synonyms: list[str] = []
|
||||
concepts: list[str] = []
|
||||
evidence: list[str] = []
|
||||
notes: str = ""
|
||||
eligible: bool | None = None
|
||||
|
||||
|
||||
class TableAnnotation(BaseModel):
|
||||
description: str = ""
|
||||
concepts: list[str] = []
|
||||
notes: str = ""
|
||||
columns: dict[str, ColumnAnnotation] = {}
|
||||
|
||||
|
||||
class Annotations(_YamlModel):
|
||||
tables: dict[str, TableAnnotation] = {}
|
||||
|
||||
@classmethod
|
||||
def from_yaml(cls, path: Path) -> "Annotations":
|
||||
if not path.exists():
|
||||
return cls()
|
||||
return super().from_yaml(path)
|
||||
@@ -0,0 +1,141 @@
|
||||
from typing import Any
|
||||
|
||||
from tht.mschema.eligibility import effective_eligibility
|
||||
from tht.mschema.models import Annotations, ColumnAnnotation, PhysicalSchema
|
||||
|
||||
MAX_EXAMPLES_IN_PROMPT = 5
|
||||
|
||||
|
||||
def _ann_col(annotations: Annotations, table: str, column: str) -> ColumnAnnotation | None:
|
||||
ann = annotations.tables.get(table)
|
||||
if ann is None:
|
||||
return None
|
||||
return ann.columns.get(column)
|
||||
|
||||
|
||||
def _table_description(physical: PhysicalSchema, annotations: Annotations, table: str) -> str:
|
||||
ann = annotations.tables.get(table)
|
||||
if ann and ann.description:
|
||||
return ann.description
|
||||
return physical.tables[table].comment
|
||||
|
||||
|
||||
def _column_description(
|
||||
physical: PhysicalSchema, annotations: Annotations, table: str, column: str
|
||||
) -> str:
|
||||
ann = annotations.tables.get(table)
|
||||
if ann and column in ann.columns and ann.columns[column].description:
|
||||
return ann.columns[column].description
|
||||
return physical.tables[table].columns[column].comment
|
||||
|
||||
|
||||
def to_mschema_text(
|
||||
physical: PhysicalSchema,
|
||||
annotations: Annotations | None = None,
|
||||
tables: list[str] | None = None,
|
||||
) -> str:
|
||||
"""Serializzazione testuale in stile ThothAI (【Schema】/【Foreign keys】)."""
|
||||
annotations = annotations or Annotations()
|
||||
selected = [t for t in physical.tables if tables is None or t in tables]
|
||||
lines: list[str] = ["【Schema】"]
|
||||
fk_lines: list[str] = []
|
||||
for table_name in selected:
|
||||
table = physical.tables[table_name]
|
||||
desc = _table_description(physical, annotations, table_name)
|
||||
if desc:
|
||||
lines.append(f"-- {desc}")
|
||||
lines.append(f"CREATE TABLE {table_name} (")
|
||||
for column_name, column in table.columns.items():
|
||||
if not effective_eligibility(column, _ann_col(annotations, table_name, column_name))[0]:
|
||||
continue
|
||||
line = f" {column_name} {column.type.upper()}"
|
||||
if column.pk:
|
||||
line += " -- PRIMARY KEY"
|
||||
lines.append(line)
|
||||
cdesc = _column_description(physical, annotations, table_name, column_name)
|
||||
if cdesc:
|
||||
lines.append(f" -- {cdesc}")
|
||||
if column.examples:
|
||||
shown = ", ".join(column.examples[:MAX_EXAMPLES_IN_PROMPT])
|
||||
lines.append(f" -- Examples: {shown}")
|
||||
lines.append(");")
|
||||
for fk in table.foreign_keys:
|
||||
for src, dst in zip(fk.columns, fk.ref_columns):
|
||||
fk_lines.append(f"{table_name}.{src}={fk.ref_table}.{dst}")
|
||||
lines.extend(["", "【Foreign keys】", *fk_lines])
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def to_schema_dict(
|
||||
physical: PhysicalSchema, annotations: Annotations | None = None
|
||||
) -> dict[str, Any]:
|
||||
"""Vista compatibile con le logiche AV-SQL (schema_dict)."""
|
||||
annotations = annotations or Annotations()
|
||||
out: dict[str, Any] = {}
|
||||
for table_name, table in physical.tables.items():
|
||||
cols = [
|
||||
c
|
||||
for c in table.columns
|
||||
if effective_eligibility(table.columns[c], _ann_col(annotations, table_name, c))[0]
|
||||
]
|
||||
out[table_name] = {
|
||||
"columns_name": cols,
|
||||
"columns_type": [table.columns[c].type for c in cols],
|
||||
"columns_description": [
|
||||
_column_description(physical, annotations, table_name, c) for c in cols
|
||||
],
|
||||
"example_values": [table.columns[c].examples for c in cols],
|
||||
"table_to_tablefullname": f"{physical.db_schema}.{table_name}",
|
||||
"primary_keys": [c for c in cols if table.columns[c].pk],
|
||||
"foreign_keys": [
|
||||
{"columns": fk.columns, "ref_table": fk.ref_table, "ref_columns": fk.ref_columns}
|
||||
for fk in table.foreign_keys
|
||||
],
|
||||
}
|
||||
return out
|
||||
|
||||
|
||||
def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None) -> str:
|
||||
"""Report leggibile per il reviewer."""
|
||||
annotations = annotations or Annotations()
|
||||
lines = [
|
||||
f"# Schema {physical.db_schema} ({physical.database})",
|
||||
"",
|
||||
f"Introspezione: {physical.introspected_at.isoformat()} — "
|
||||
f"{len(physical.tables)} tabelle",
|
||||
]
|
||||
for table_name, table in physical.tables.items():
|
||||
lines += ["", f"## {table_name}", ""]
|
||||
desc = _table_description(physical, annotations, table_name)
|
||||
if desc:
|
||||
lines += [desc, ""]
|
||||
lines += [
|
||||
f"Righe (stima): {table.row_count}",
|
||||
"",
|
||||
"| Colonna | Tipo | Null | PK | Descrizione | Esempi |",
|
||||
"|---|---|---|---|---|---|",
|
||||
]
|
||||
for column_name, column in table.columns.items():
|
||||
eligible, reason = effective_eligibility(
|
||||
column, _ann_col(annotations, table_name, column_name)
|
||||
)
|
||||
cdesc = _column_description(physical, annotations, table_name, column_name)
|
||||
if not eligible:
|
||||
lines.append(
|
||||
f"| ~~{column_name}~~ | {column.type} | "
|
||||
f"{'sì' if column.nullable else 'no'} | "
|
||||
f"{'sì' if column.pk else ''} | {cdesc} | _ignorata: {reason}_ |"
|
||||
)
|
||||
continue
|
||||
examples = ", ".join(column.examples[:3])
|
||||
lines.append(
|
||||
f"| {column_name} | {column.type} | {'sì' if column.nullable else 'no'} "
|
||||
f"| {'sì' if column.pk else ''} | {cdesc} | {examples} |"
|
||||
)
|
||||
if table.foreign_keys:
|
||||
lines += ["", "Foreign keys:"]
|
||||
for fk in table.foreign_keys:
|
||||
lines.append(
|
||||
f"- ({', '.join(fk.columns)}) → {fk.ref_table} ({', '.join(fk.ref_columns)})"
|
||||
)
|
||||
return "\n".join(lines)
|
||||
Reference in New Issue
Block a user