Files
ThothII/harness/tests/test_mschema_eligibility.py
T
marcopanandClaude Fable 5 c4951e2aa5 chore: hygiene pass — ruff clean, docs storage-model truth, replay /me, failSession log
Audit findings 6.1-6.4 + the audit's remediation plan itself
(docs/superpowers/plans/2026-07-20-full-audit-remediation-plan.md).

- ruff: 34 → 0 (unused imports/f-strings auto-fixed; E702 semicolon lines
  split in test files; one unused local dropped). Suite still 819 green.
- CLAUDE.md + PROJECT_STATE.md no longer claim "no database / settings in
  settings.json": the harness selects filesystem OR PostgreSQL session
  storage (repository.py, server mode), and settings flow through harness
  preferences with the JSON file as fallback only.
- tools/replay: stub /me (SPA boot was parsing the SPA's own HTML as JSON)
  and /runtime/prewarm.
- failSession best-effort persistence now logs its failure server-side
  instead of vanishing.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-20 01:55:41 +02:00

109 lines
3.9 KiB
Python

"""L1: mschema/eligibility — the column-eligibility principle on fake columns.
Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
comes only from numerics, enums, temporals, booleans, and short text. Annotation
override wins over the physical classification.
"""
from datetime import datetime
from tht.config import EligibilityConfig
from tht.mschema.eligibility import classify_all, classify_column, effective_eligibility
from tht.mschema.models import (
ColumnAnnotation,
ColumnPhysical,
PhysicalSchema,
TablePhysical,
)
CFG = EligibilityConfig()
# --- classify_column (pure) ----------------------------------------------------
def test_numeric_eligible():
ok, reason = classify_column("integer", False, None, None, CFG)
assert ok and reason == "numeric"
def test_enum_eligible():
ok, reason = classify_column("tipo_ricovero", True, None, None, CFG)
assert ok and reason == "enum"
def test_temporal_eligible():
ok, reason = classify_column("timestamp without time zone", False, None, None, CFG)
assert ok and reason == "temporal"
def test_boolean_eligible():
ok, reason = classify_column("boolean", False, None, None, CFG)
assert ok and reason == "boolean"
def test_short_declared_text_eligible_without_sampling():
# varchar(100) <= max_declared_len(128) → eligible without sampling the data
ok, reason = classify_column("varchar(100)", False, None, None, CFG)
assert ok and reason == "short_text"
def test_unbounded_text_without_sampling_is_wide_text():
ok, reason = classify_column("text", False, None, None, CFG)
assert not ok and reason == "wide_text"
def test_long_text_sampled_within_bounds_is_short_text():
ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG)
assert ok and reason == "short_text"
def test_long_text_sampled_above_bounds_is_wide_text():
ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG)
assert not ok and reason == "wide_text"
def test_array_type_is_wide_text():
ok, reason = classify_column("text[]", False, None, None, CFG)
assert not ok and reason == "wide_text"
# --- classify_all (in-place) ---------------------------------------------------
def _schema_with(**columns) -> PhysicalSchema:
return PhysicalSchema(
database="db", schema="dw", introspected_at=datetime(2025, 1, 1),
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
)
def test_classify_all_marks_wide_text_and_clears_examples():
schema = _schema_with(
note=dict(type="text", examples=["a" * 500, "b" * 400]),
cod=dict(type="varchar(10)", examples=["X", "Y"]),
etl_last_update=dict(type="timestamp", examples=[]),
)
classify_all(schema, CFG)
cols = schema.tables["t"].columns
assert cols["note"].eligible is False
assert cols["note"].eligibility_reason == "wide_text"
assert cols["note"].examples == [] # examples cleared on ignored columns
assert cols["cod"].eligible is True
# ignore_columns default includes etl_last_update → ignored_by_name regardless of type
assert cols["etl_last_update"].eligible is False
assert cols["etl_last_update"].eligibility_reason == "ignored_by_name"
# --- effective_eligibility (annotation override wins) --------------------------
def test_annotation_override_forces_eligible():
col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text")
ann = ColumnAnnotation(eligible=True)
eff, reason = effective_eligibility(col, ann)
assert eff is True
assert reason == "override"
def test_no_annotation_falls_back_to_physical():
col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric")
eff, reason = effective_eligibility(col, None)
assert eff is True and reason == "numeric"