"""Classificazione di column eligibility (principio trasversale Thoth). Vedi docs/superpowers/specs/2026-06-13-tht-column-eligibility-principle.md. Il testo ampio (lettere di dimissione, note, anamnesi) è ignorato ovunque; i dati provengono solo da numerici, enum, temporali, booleani e testo breve. """ import re from tht.config import EligibilityConfig from tht.db.sampling import is_text_type from tht.mschema.models import ColumnAnnotation, ColumnPhysical, PhysicalSchema _NUMERIC_PREFIXES = ( "smallint", "integer", "bigint", "numeric", "decimal", "real", "double", "money", ) _TEMPORAL_PREFIXES = ("date", "time", "timestamp", "interval") _LEN_RE = re.compile(r"\((\d+)\)") def _declared_len(pg_type: str) -> int | None: m = _LEN_RE.search(pg_type) return int(m.group(1)) if m else None def classify_column( pg_type: str, is_enum: bool, sampled_avg: float | None, sampled_max: int | None, cfg: EligibilityConfig, ) -> tuple[bool, str]: """Classifica una colonna come (eligible, reason). Funzione pura, senza I/O.""" t = pg_type.strip().lower() if t.endswith("[]"): return False, "wide_text" if is_enum: return True, "enum" if t.startswith(_NUMERIC_PREFIXES): return True, "numeric" if t.startswith("boolean"): return True, "boolean" if t.startswith(_TEMPORAL_PREFIXES): return True, "temporal" if t.startswith("uuid"): return True, "code" if is_text_type(t): declared = _declared_len(t) if declared is not None and declared <= cfg.max_declared_len: return True, "short_text" # bound grande o text/varchar non vincolato: decide il dato campionato if sampled_avg is None or sampled_max is None: return False, "wide_text" if sampled_avg <= cfg.max_avg_length and sampled_max <= cfg.max_sampled_len: return True, "short_text" return False, "wide_text" return False, "wide_text" def _sampled_stats(examples: list[str]) -> tuple[float | None, int | None]: if not examples: return None, None lengths = [len(v) for v in examples] return sum(lengths) / len(lengths), max(lengths) def classify_all(physical: PhysicalSchema, cfg: EligibilityConfig) -> None: """Assegna eligible/eligibility_reason a ogni colonna (in-place) e azzera gli examples delle colonne ignored. Da chiamare DOPO add_examples.""" ignore_by_name = {name.lower() for name in cfg.ignore_columns} for table in physical.tables.values(): for column_name, column in table.columns.items(): if column_name.lower() in ignore_by_name: # colonna di servizio (ETL/audit): ignorata a prescindere dal tipo column.eligible = False column.eligibility_reason = "ignored_by_name" column.examples = [] continue avg, mx = _sampled_stats(column.examples) eligible, reason = classify_column(column.type, column.is_enum, avg, mx, cfg) column.eligible = eligible column.eligibility_reason = reason if not eligible: column.examples = [] def effective_eligibility( column: ColumnPhysical, annotation: ColumnAnnotation | None ) -> tuple[bool, str]: """Eligibilità effettiva: l'override in annotations.yaml vince sul fisico.""" if annotation is not None and annotation.eligible is not None: return annotation.eligible, "override" return column.eligible, column.eligibility_reason