Files
ThothII/harness/tests/test_mschema_eligibility.py

109 lines
3.9 KiB
Python

"""L1: mschema/eligibility — the column-eligibility principle on fake columns.
Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
comes only from numerics, enums, temporals, booleans, and short text. Annotation
override wins over the physical classification.
"""
from datetime import UTC, datetime
from tht.config import EligibilityConfig
from tht.mschema.eligibility import classify_all, classify_column, effective_eligibility
from tht.mschema.models import (
ColumnAnnotation,
ColumnPhysical,
PhysicalSchema,
TablePhysical,
)
CFG = EligibilityConfig()
# --- classify_column (pure) ----------------------------------------------------
def test_numeric_eligible():
ok, reason = classify_column("integer", False, None, None, CFG)
assert ok and reason == "numeric"
def test_enum_eligible():
ok, reason = classify_column("tipo_ricovero", True, None, None, CFG)
assert ok and reason == "enum"
def test_temporal_eligible():
ok, reason = classify_column("timestamp without time zone", False, None, None, CFG)
assert ok and reason == "temporal"
def test_boolean_eligible():
ok, reason = classify_column("boolean", False, None, None, CFG)
assert ok and reason == "boolean"
def test_short_declared_text_eligible_without_sampling():
# varchar(100) <= max_declared_len(128) → eligible without sampling the data
ok, reason = classify_column("varchar(100)", False, None, None, CFG)
assert ok and reason == "short_text"
def test_unbounded_text_without_sampling_is_wide_text():
ok, reason = classify_column("text", False, None, None, CFG)
assert not ok and reason == "wide_text"
def test_long_text_sampled_within_bounds_is_short_text():
ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG)
assert ok and reason == "short_text"
def test_long_text_sampled_above_bounds_is_wide_text():
ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG)
assert not ok and reason == "wide_text"
def test_array_type_is_wide_text():
ok, reason = classify_column("text[]", False, None, None, CFG)
assert not ok and reason == "wide_text"
# --- classify_all (in-place) ---------------------------------------------------
def _schema_with(**columns) -> PhysicalSchema:
return PhysicalSchema(
database="db", schema="dw", introspected_at=datetime(2025, 1, 1, tzinfo=UTC),
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
)
def test_classify_all_marks_wide_text_and_clears_examples():
schema = _schema_with(
note={"type": "text", "examples": ["a" * 500, "b" * 400]},
cod={"type": "varchar(10)", "examples": ["X", "Y"]},
etl_last_update={"type": "timestamp", "examples": []},
)
classify_all(schema, CFG)
cols = schema.tables["t"].columns
assert cols["note"].eligible is False
assert cols["note"].eligibility_reason == "wide_text"
assert cols["note"].examples == [] # examples cleared on ignored columns
assert cols["cod"].eligible is True
# ignore_columns default includes etl_last_update → ignored_by_name regardless of type
assert cols["etl_last_update"].eligible is False
assert cols["etl_last_update"].eligibility_reason == "ignored_by_name"
# --- effective_eligibility (annotation override wins) --------------------------
def test_annotation_override_forces_eligible():
col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text")
ann = ColumnAnnotation(eligible=True)
eff, reason = effective_eligibility(col, ann)
assert eff is True
assert reason == "override"
def test_no_annotation_falls_back_to_physical():
col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric")
eff, reason = effective_eligibility(col, None)
assert eff is True and reason == "numeric"