109 lines
3.9 KiB
Python
109 lines
3.9 KiB
Python
"""L1: mschema/eligibility — the column-eligibility principle on fake columns.
|
|
|
|
Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
|
|
comes only from numerics, enums, temporals, booleans, and short text. Annotation
|
|
override wins over the physical classification.
|
|
"""
|
|
from datetime import UTC, datetime
|
|
|
|
from tht.config import EligibilityConfig
|
|
from tht.mschema.eligibility import classify_all, classify_column, effective_eligibility
|
|
from tht.mschema.models import (
|
|
ColumnAnnotation,
|
|
ColumnPhysical,
|
|
PhysicalSchema,
|
|
TablePhysical,
|
|
)
|
|
|
|
CFG = EligibilityConfig()
|
|
|
|
|
|
# --- classify_column (pure) ----------------------------------------------------
|
|
|
|
def test_numeric_eligible():
|
|
ok, reason = classify_column("integer", False, None, None, CFG)
|
|
assert ok and reason == "numeric"
|
|
|
|
|
|
def test_enum_eligible():
|
|
ok, reason = classify_column("tipo_ricovero", True, None, None, CFG)
|
|
assert ok and reason == "enum"
|
|
|
|
|
|
def test_temporal_eligible():
|
|
ok, reason = classify_column("timestamp without time zone", False, None, None, CFG)
|
|
assert ok and reason == "temporal"
|
|
|
|
|
|
def test_boolean_eligible():
|
|
ok, reason = classify_column("boolean", False, None, None, CFG)
|
|
assert ok and reason == "boolean"
|
|
|
|
|
|
def test_short_declared_text_eligible_without_sampling():
|
|
# varchar(100) <= max_declared_len(128) → eligible without sampling the data
|
|
ok, reason = classify_column("varchar(100)", False, None, None, CFG)
|
|
assert ok and reason == "short_text"
|
|
|
|
|
|
def test_unbounded_text_without_sampling_is_wide_text():
|
|
ok, reason = classify_column("text", False, None, None, CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
def test_long_text_sampled_within_bounds_is_short_text():
|
|
ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG)
|
|
assert ok and reason == "short_text"
|
|
|
|
|
|
def test_long_text_sampled_above_bounds_is_wide_text():
|
|
ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
def test_array_type_is_wide_text():
|
|
ok, reason = classify_column("text[]", False, None, None, CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
# --- classify_all (in-place) ---------------------------------------------------
|
|
|
|
def _schema_with(**columns) -> PhysicalSchema:
|
|
return PhysicalSchema(
|
|
database="db", schema="dw", introspected_at=datetime(2025, 1, 1, tzinfo=UTC),
|
|
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
|
|
)
|
|
|
|
|
|
def test_classify_all_marks_wide_text_and_clears_examples():
|
|
schema = _schema_with(
|
|
note={"type": "text", "examples": ["a" * 500, "b" * 400]},
|
|
cod={"type": "varchar(10)", "examples": ["X", "Y"]},
|
|
etl_last_update={"type": "timestamp", "examples": []},
|
|
)
|
|
classify_all(schema, CFG)
|
|
cols = schema.tables["t"].columns
|
|
assert cols["note"].eligible is False
|
|
assert cols["note"].eligibility_reason == "wide_text"
|
|
assert cols["note"].examples == [] # examples cleared on ignored columns
|
|
assert cols["cod"].eligible is True
|
|
# ignore_columns default includes etl_last_update → ignored_by_name regardless of type
|
|
assert cols["etl_last_update"].eligible is False
|
|
assert cols["etl_last_update"].eligibility_reason == "ignored_by_name"
|
|
|
|
|
|
# --- effective_eligibility (annotation override wins) --------------------------
|
|
|
|
def test_annotation_override_forces_eligible():
|
|
col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text")
|
|
ann = ColumnAnnotation(eligible=True)
|
|
eff, reason = effective_eligibility(col, ann)
|
|
assert eff is True
|
|
assert reason == "override"
|
|
|
|
|
|
def test_no_annotation_falls_back_to_physical():
|
|
col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric")
|
|
eff, reason = effective_eligibility(col, None)
|
|
assert eff is True and reason == "numeric"
|