Thoth (tht) è il prodotto, PSD è il cliente. Nessun riferimento al contesto
clinico nel codice.
Rinomine:
- comando+package nsp→tht (dir nsp/→tht/, 46 import, pyproject entry point)
- gate nsp-gate.js→tht-gate.js (+ rewrite token, relayIfNspFails→relayIfThtFails)
- workspace chirone.{example,test}.yaml→tht.{example,test}.yaml (generici)
- env THOTH_→THT_ (19 var) + NSP_ stragglers (NSP_HARNESS_ROOT, NSP_SESSION)
- commenti/docstring chirone/psdwp3/policlinico neutralizzati ('the reference
implementation', 'the DWH')
Aggiunto [tool.setuptools.packages.find] include=['tht*'] (necessario: l'auto-
discovery rompeva con tht/ + workspaces/ come top-level multipli).
.env operatore aggiornato in-place (prefissi THT_, valori preservati, gitignored).
Verifica: pytest 109 passed, npm test 14 pass, tht phase meta --json OK, zero
residui nsp/THOTH_/NSP_/chirone nel package.
110 lines
3.9 KiB
Python
110 lines
3.9 KiB
Python
"""L1: mschema/eligibility — the column-eligibility principle on fake columns.
|
|
|
|
Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
|
|
comes only from numerics, enums, temporals, booleans, and short text. Annotation
|
|
override wins over the physical classification.
|
|
"""
|
|
from datetime import datetime
|
|
|
|
from tht.config import EligibilityConfig
|
|
from tht.mschema.eligibility import classify_all, classify_column, effective_eligibility
|
|
from tht.mschema.models import (
|
|
Annotations,
|
|
ColumnAnnotation,
|
|
ColumnPhysical,
|
|
PhysicalSchema,
|
|
TablePhysical,
|
|
)
|
|
|
|
CFG = EligibilityConfig()
|
|
|
|
|
|
# --- classify_column (pure) ----------------------------------------------------
|
|
|
|
def test_numeric_eligible():
|
|
ok, reason = classify_column("integer", False, None, None, CFG)
|
|
assert ok and reason == "numeric"
|
|
|
|
|
|
def test_enum_eligible():
|
|
ok, reason = classify_column("tipo_ricovero", True, None, None, CFG)
|
|
assert ok and reason == "enum"
|
|
|
|
|
|
def test_temporal_eligible():
|
|
ok, reason = classify_column("timestamp without time zone", False, None, None, CFG)
|
|
assert ok and reason == "temporal"
|
|
|
|
|
|
def test_boolean_eligible():
|
|
ok, reason = classify_column("boolean", False, None, None, CFG)
|
|
assert ok and reason == "boolean"
|
|
|
|
|
|
def test_short_declared_text_eligible_without_sampling():
|
|
# varchar(100) <= max_declared_len(128) → eligible without sampling the data
|
|
ok, reason = classify_column("varchar(100)", False, None, None, CFG)
|
|
assert ok and reason == "short_text"
|
|
|
|
|
|
def test_unbounded_text_without_sampling_is_wide_text():
|
|
ok, reason = classify_column("text", False, None, None, CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
def test_long_text_sampled_within_bounds_is_short_text():
|
|
ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG)
|
|
assert ok and reason == "short_text"
|
|
|
|
|
|
def test_long_text_sampled_above_bounds_is_wide_text():
|
|
ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
def test_array_type_is_wide_text():
|
|
ok, reason = classify_column("text[]", False, None, None, CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
# --- classify_all (in-place) ---------------------------------------------------
|
|
|
|
def _schema_with(**columns) -> PhysicalSchema:
|
|
return PhysicalSchema(
|
|
database="db", schema="dw", introspected_at=datetime(2025, 1, 1),
|
|
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
|
|
)
|
|
|
|
|
|
def test_classify_all_marks_wide_text_and_clears_examples():
|
|
schema = _schema_with(
|
|
note=dict(type="text", examples=["a" * 500, "b" * 400]),
|
|
cod=dict(type="varchar(10)", examples=["X", "Y"]),
|
|
etl_last_update=dict(type="timestamp", examples=[]),
|
|
)
|
|
classify_all(schema, CFG)
|
|
cols = schema.tables["t"].columns
|
|
assert cols["note"].eligible is False
|
|
assert cols["note"].eligibility_reason == "wide_text"
|
|
assert cols["note"].examples == [] # examples cleared on ignored columns
|
|
assert cols["cod"].eligible is True
|
|
# ignore_columns default includes etl_last_update → ignored_by_name regardless of type
|
|
assert cols["etl_last_update"].eligible is False
|
|
assert cols["etl_last_update"].eligibility_reason == "ignored_by_name"
|
|
|
|
|
|
# --- effective_eligibility (annotation override wins) --------------------------
|
|
|
|
def test_annotation_override_forces_eligible():
|
|
col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text")
|
|
ann = ColumnAnnotation(eligible=True)
|
|
eff, reason = effective_eligibility(col, ann)
|
|
assert eff is True
|
|
assert reason == "override"
|
|
|
|
|
|
def test_no_annotation_falls_back_to_physical():
|
|
col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric")
|
|
eff, reason = effective_eligibility(col, None)
|
|
assert eff is True and reason == "numeric"
|