Files
ThothII/harness/tht/mschema/eligibility.py
T
marcopan fc5fbe6b65 refactor(harness): renaming prodotto tht (Onda -1)
Thoth (tht) è il prodotto, PSD è il cliente. Nessun riferimento al contesto
clinico nel codice.

Rinomine:
- comando+package nsp→tht (dir nsp/→tht/, 46 import, pyproject entry point)
- gate nsp-gate.js→tht-gate.js (+ rewrite token, relayIfNspFails→relayIfThtFails)
- workspace chirone.{example,test}.yaml→tht.{example,test}.yaml (generici)
- env THOTH_→THT_ (19 var) + NSP_ stragglers (NSP_HARNESS_ROOT, NSP_SESSION)
- commenti/docstring chirone/psdwp3/policlinico neutralizzati ('the reference
  implementation', 'the DWH')

Aggiunto [tool.setuptools.packages.find] include=['tht*'] (necessario: l'auto-
discovery rompeva con tht/ + workspaces/ come top-level multipli).

.env operatore aggiornato in-place (prefissi THT_, valori preservati, gitignored).

Verifica: pytest 109 passed, npm test 14 pass, tht phase meta --json OK, zero
residui nsp/THOTH_/NSP_/chirone nel package.
2026-06-27 10:33:16 +02:00

94 lines
3.5 KiB
Python

"""Classificazione di column eligibility (principio trasversale Thoth).
Vedi docs/superpowers/specs/2026-06-13-tht-column-eligibility-principle.md.
Il testo ampio (lettere di dimissione, note, anamnesi) è ignorato ovunque; i dati
provengono solo da numerici, enum, temporali, booleani e testo breve.
"""
import re
from tht.config import EligibilityConfig
from tht.db.sampling import is_text_type
from tht.mschema.models import ColumnAnnotation, ColumnPhysical, PhysicalSchema
_NUMERIC_PREFIXES = (
"smallint", "integer", "bigint", "numeric", "decimal", "real", "double", "money",
)
_TEMPORAL_PREFIXES = ("date", "time", "timestamp", "interval")
_LEN_RE = re.compile(r"\((\d+)\)")
def _declared_len(pg_type: str) -> int | None:
m = _LEN_RE.search(pg_type)
return int(m.group(1)) if m else None
def classify_column(
pg_type: str,
is_enum: bool,
sampled_avg: float | None,
sampled_max: int | None,
cfg: EligibilityConfig,
) -> tuple[bool, str]:
"""Classifica una colonna come (eligible, reason). Funzione pura, senza I/O."""
t = pg_type.strip().lower()
if t.endswith("[]"):
return False, "wide_text"
if is_enum:
return True, "enum"
if t.startswith(_NUMERIC_PREFIXES):
return True, "numeric"
if t.startswith("boolean"):
return True, "boolean"
if t.startswith(_TEMPORAL_PREFIXES):
return True, "temporal"
if t.startswith("uuid"):
return True, "code"
if is_text_type(t):
declared = _declared_len(t)
if declared is not None and declared <= cfg.max_declared_len:
return True, "short_text"
# bound grande o text/varchar non vincolato: decide il dato campionato
if sampled_avg is None or sampled_max is None:
return False, "wide_text"
if sampled_avg <= cfg.max_avg_length and sampled_max <= cfg.max_sampled_len:
return True, "short_text"
return False, "wide_text"
return False, "wide_text"
def _sampled_stats(examples: list[str]) -> tuple[float | None, int | None]:
if not examples:
return None, None
lengths = [len(v) for v in examples]
return sum(lengths) / len(lengths), max(lengths)
def classify_all(physical: PhysicalSchema, cfg: EligibilityConfig) -> None:
"""Assegna eligible/eligibility_reason a ogni colonna (in-place) e azzera gli
examples delle colonne ignored. Da chiamare DOPO add_examples."""
ignore_by_name = {name.lower() for name in cfg.ignore_columns}
for table in physical.tables.values():
for column_name, column in table.columns.items():
if column_name.lower() in ignore_by_name:
# colonna di servizio (ETL/audit): ignorata a prescindere dal tipo
column.eligible = False
column.eligibility_reason = "ignored_by_name"
column.examples = []
continue
avg, mx = _sampled_stats(column.examples)
eligible, reason = classify_column(column.type, column.is_enum, avg, mx, cfg)
column.eligible = eligible
column.eligibility_reason = reason
if not eligible:
column.examples = []
def effective_eligibility(
column: ColumnPhysical, annotation: ColumnAnnotation | None
) -> tuple[bool, str]:
"""Eligibilità effettiva: l'override in annotations.yaml vince sul fisico."""
if annotation is not None and annotation.eligible is not None:
return annotation.eligible, "override"
return column.eligible, column.eligibility_reason