Audit findings 6.1-6.4 + the audit's remediation plan itself (docs/superpowers/plans/2026-07-20-full-audit-remediation-plan.md). - ruff: 34 → 0 (unused imports/f-strings auto-fixed; E702 semicolon lines split in test files; one unused local dropped). Suite still 819 green. - CLAUDE.md + PROJECT_STATE.md no longer claim "no database / settings in settings.json": the harness selects filesystem OR PostgreSQL session storage (repository.py, server mode), and settings flow through harness preferences with the JSON file as fallback only. - tools/replay: stub /me (SPA boot was parsing the SPA's own HTML as JSON) and /runtime/prewarm. - failSession best-effort persistence now logs its failure server-side instead of vanishing. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
109 lines
3.9 KiB
Python
109 lines
3.9 KiB
Python
"""L1: mschema/eligibility — the column-eligibility principle on fake columns.
|
|
|
|
Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
|
|
comes only from numerics, enums, temporals, booleans, and short text. Annotation
|
|
override wins over the physical classification.
|
|
"""
|
|
from datetime import datetime
|
|
|
|
from tht.config import EligibilityConfig
|
|
from tht.mschema.eligibility import classify_all, classify_column, effective_eligibility
|
|
from tht.mschema.models import (
|
|
ColumnAnnotation,
|
|
ColumnPhysical,
|
|
PhysicalSchema,
|
|
TablePhysical,
|
|
)
|
|
|
|
CFG = EligibilityConfig()
|
|
|
|
|
|
# --- classify_column (pure) ----------------------------------------------------
|
|
|
|
def test_numeric_eligible():
|
|
ok, reason = classify_column("integer", False, None, None, CFG)
|
|
assert ok and reason == "numeric"
|
|
|
|
|
|
def test_enum_eligible():
|
|
ok, reason = classify_column("tipo_ricovero", True, None, None, CFG)
|
|
assert ok and reason == "enum"
|
|
|
|
|
|
def test_temporal_eligible():
|
|
ok, reason = classify_column("timestamp without time zone", False, None, None, CFG)
|
|
assert ok and reason == "temporal"
|
|
|
|
|
|
def test_boolean_eligible():
|
|
ok, reason = classify_column("boolean", False, None, None, CFG)
|
|
assert ok and reason == "boolean"
|
|
|
|
|
|
def test_short_declared_text_eligible_without_sampling():
|
|
# varchar(100) <= max_declared_len(128) → eligible without sampling the data
|
|
ok, reason = classify_column("varchar(100)", False, None, None, CFG)
|
|
assert ok and reason == "short_text"
|
|
|
|
|
|
def test_unbounded_text_without_sampling_is_wide_text():
|
|
ok, reason = classify_column("text", False, None, None, CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
def test_long_text_sampled_within_bounds_is_short_text():
|
|
ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG)
|
|
assert ok and reason == "short_text"
|
|
|
|
|
|
def test_long_text_sampled_above_bounds_is_wide_text():
|
|
ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
def test_array_type_is_wide_text():
|
|
ok, reason = classify_column("text[]", False, None, None, CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
# --- classify_all (in-place) ---------------------------------------------------
|
|
|
|
def _schema_with(**columns) -> PhysicalSchema:
|
|
return PhysicalSchema(
|
|
database="db", schema="dw", introspected_at=datetime(2025, 1, 1),
|
|
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
|
|
)
|
|
|
|
|
|
def test_classify_all_marks_wide_text_and_clears_examples():
|
|
schema = _schema_with(
|
|
note=dict(type="text", examples=["a" * 500, "b" * 400]),
|
|
cod=dict(type="varchar(10)", examples=["X", "Y"]),
|
|
etl_last_update=dict(type="timestamp", examples=[]),
|
|
)
|
|
classify_all(schema, CFG)
|
|
cols = schema.tables["t"].columns
|
|
assert cols["note"].eligible is False
|
|
assert cols["note"].eligibility_reason == "wide_text"
|
|
assert cols["note"].examples == [] # examples cleared on ignored columns
|
|
assert cols["cod"].eligible is True
|
|
# ignore_columns default includes etl_last_update → ignored_by_name regardless of type
|
|
assert cols["etl_last_update"].eligible is False
|
|
assert cols["etl_last_update"].eligibility_reason == "ignored_by_name"
|
|
|
|
|
|
# --- effective_eligibility (annotation override wins) --------------------------
|
|
|
|
def test_annotation_override_forces_eligible():
|
|
col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text")
|
|
ann = ColumnAnnotation(eligible=True)
|
|
eff, reason = effective_eligibility(col, ann)
|
|
assert eff is True
|
|
assert reason == "override"
|
|
|
|
|
|
def test_no_annotation_falls_back_to_physical():
|
|
col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric")
|
|
eff, reason = effective_eligibility(col, None)
|
|
assert eff is True and reason == "numeric"
|