"""L1: mschema/eligibility — the column-eligibility principle on fake columns. Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data comes only from numerics, enums, temporals, booleans, and short text. Annotation override wins over the physical classification. """ from datetime import datetime from tht.config import EligibilityConfig from tht.mschema.eligibility import classify_all, classify_column, effective_eligibility from tht.mschema.models import ( Annotations, ColumnAnnotation, ColumnPhysical, PhysicalSchema, TablePhysical, ) CFG = EligibilityConfig() # --- classify_column (pure) ---------------------------------------------------- def test_numeric_eligible(): ok, reason = classify_column("integer", False, None, None, CFG) assert ok and reason == "numeric" def test_enum_eligible(): ok, reason = classify_column("tipo_ricovero", True, None, None, CFG) assert ok and reason == "enum" def test_temporal_eligible(): ok, reason = classify_column("timestamp without time zone", False, None, None, CFG) assert ok and reason == "temporal" def test_boolean_eligible(): ok, reason = classify_column("boolean", False, None, None, CFG) assert ok and reason == "boolean" def test_short_declared_text_eligible_without_sampling(): # varchar(100) <= max_declared_len(128) → eligible without sampling the data ok, reason = classify_column("varchar(100)", False, None, None, CFG) assert ok and reason == "short_text" def test_unbounded_text_without_sampling_is_wide_text(): ok, reason = classify_column("text", False, None, None, CFG) assert not ok and reason == "wide_text" def test_long_text_sampled_within_bounds_is_short_text(): ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG) assert ok and reason == "short_text" def test_long_text_sampled_above_bounds_is_wide_text(): ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG) assert not ok and reason == "wide_text" def test_array_type_is_wide_text(): ok, reason = classify_column("text[]", False, None, None, CFG) assert not ok and reason == "wide_text" # --- classify_all (in-place) --------------------------------------------------- def _schema_with(**columns) -> PhysicalSchema: return PhysicalSchema( database="db", schema="dw", introspected_at=datetime(2025, 1, 1), tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})}, ) def test_classify_all_marks_wide_text_and_clears_examples(): schema = _schema_with( note=dict(type="text", examples=["a" * 500, "b" * 400]), cod=dict(type="varchar(10)", examples=["X", "Y"]), etl_last_update=dict(type="timestamp", examples=[]), ) classify_all(schema, CFG) cols = schema.tables["t"].columns assert cols["note"].eligible is False assert cols["note"].eligibility_reason == "wide_text" assert cols["note"].examples == [] # examples cleared on ignored columns assert cols["cod"].eligible is True # ignore_columns default includes etl_last_update → ignored_by_name regardless of type assert cols["etl_last_update"].eligible is False assert cols["etl_last_update"].eligibility_reason == "ignored_by_name" # --- effective_eligibility (annotation override wins) -------------------------- def test_annotation_override_forces_eligible(): col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text") ann = ColumnAnnotation(eligible=True) eff, reason = effective_eligibility(col, ann) assert eff is True assert reason == "override" def test_no_annotation_falls_back_to_physical(): col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric") eff, reason = effective_eligibility(col, None) assert eff is True and reason == "numeric"