Ports the leaf data-layer modules and validates them: - mschema/ (models, eligibility, merge, render), db/ (connection, sampling, introspect, fetch_ca), rest/client.py -- renamed psdwp3->nsp, verbatim. - L0 (testcontainers, real Postgres): db connection read-only enforcement (psd_ro cannot CREATE/INSERT), introspect against a known schema (tables, columns, types, comments, FKs, enum, composite PK), sampling most-frequent values + truncation reporting. 15 tests, ~4s. - L1 (fake data): rest/client RPC contract (mocked transport -- X-API-Key header, payloads, base_url slash handling, HTTP/network error surfacing), mschema/render 3 formats (markdown, mschema-text, schema-dict) + eligibility rules (wide_text excluded, short_text/numeric/enum/temporal/ boolean eligible, annotation override wins). 25 tests. pyproject registers l0/l2 markers + addopts '-m not l2' (L2 opt-in). Deferred to their dependency-porting tasks: test_rrf.py (search needs vectorstore, B3) and the 11 CLI contract tests (need _guards/session, wired when each command lands). 'Not assumed reliable' now has real teeth for the data layer; CLI/search contracts follow.
110 lines
3.9 KiB
Python
110 lines
3.9 KiB
Python
"""L1: mschema/eligibility — the column-eligibility principle on fake columns.
|
|
|
|
Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
|
|
comes only from numerics, enums, temporals, booleans, and short text. Annotation
|
|
override wins over the physical classification.
|
|
"""
|
|
from datetime import datetime
|
|
|
|
from nsp.config import EligibilityConfig
|
|
from nsp.mschema.eligibility import classify_all, classify_column, effective_eligibility
|
|
from nsp.mschema.models import (
|
|
Annotations,
|
|
ColumnAnnotation,
|
|
ColumnPhysical,
|
|
PhysicalSchema,
|
|
TablePhysical,
|
|
)
|
|
|
|
CFG = EligibilityConfig()
|
|
|
|
|
|
# --- classify_column (pure) ----------------------------------------------------
|
|
|
|
def test_numeric_eligible():
|
|
ok, reason = classify_column("integer", False, None, None, CFG)
|
|
assert ok and reason == "numeric"
|
|
|
|
|
|
def test_enum_eligible():
|
|
ok, reason = classify_column("tipo_ricovero", True, None, None, CFG)
|
|
assert ok and reason == "enum"
|
|
|
|
|
|
def test_temporal_eligible():
|
|
ok, reason = classify_column("timestamp without time zone", False, None, None, CFG)
|
|
assert ok and reason == "temporal"
|
|
|
|
|
|
def test_boolean_eligible():
|
|
ok, reason = classify_column("boolean", False, None, None, CFG)
|
|
assert ok and reason == "boolean"
|
|
|
|
|
|
def test_short_declared_text_eligible_without_sampling():
|
|
# varchar(100) <= max_declared_len(128) → eligible without sampling the data
|
|
ok, reason = classify_column("varchar(100)", False, None, None, CFG)
|
|
assert ok and reason == "short_text"
|
|
|
|
|
|
def test_unbounded_text_without_sampling_is_wide_text():
|
|
ok, reason = classify_column("text", False, None, None, CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
def test_long_text_sampled_within_bounds_is_short_text():
|
|
ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG)
|
|
assert ok and reason == "short_text"
|
|
|
|
|
|
def test_long_text_sampled_above_bounds_is_wide_text():
|
|
ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
def test_array_type_is_wide_text():
|
|
ok, reason = classify_column("text[]", False, None, None, CFG)
|
|
assert not ok and reason == "wide_text"
|
|
|
|
|
|
# --- classify_all (in-place) ---------------------------------------------------
|
|
|
|
def _schema_with(**columns) -> PhysicalSchema:
|
|
return PhysicalSchema(
|
|
database="db", schema="dw", introspected_at=datetime(2025, 1, 1),
|
|
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
|
|
)
|
|
|
|
|
|
def test_classify_all_marks_wide_text_and_clears_examples():
|
|
schema = _schema_with(
|
|
note=dict(type="text", examples=["a" * 500, "b" * 400]),
|
|
cod=dict(type="varchar(10)", examples=["X", "Y"]),
|
|
etl_last_update=dict(type="timestamp", examples=[]),
|
|
)
|
|
classify_all(schema, CFG)
|
|
cols = schema.tables["t"].columns
|
|
assert cols["note"].eligible is False
|
|
assert cols["note"].eligibility_reason == "wide_text"
|
|
assert cols["note"].examples == [] # examples cleared on ignored columns
|
|
assert cols["cod"].eligible is True
|
|
# ignore_columns default includes etl_last_update → ignored_by_name regardless of type
|
|
assert cols["etl_last_update"].eligible is False
|
|
assert cols["etl_last_update"].eligibility_reason == "ignored_by_name"
|
|
|
|
|
|
# --- effective_eligibility (annotation override wins) --------------------------
|
|
|
|
def test_annotation_override_forces_eligible():
|
|
col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text")
|
|
ann = ColumnAnnotation(eligible=True)
|
|
eff, reason = effective_eligibility(col, ann)
|
|
assert eff is True
|
|
assert reason == "override"
|
|
|
|
|
|
def test_no_annotation_falls_back_to_physical():
|
|
col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric")
|
|
eff, reason = effective_eligibility(col, None)
|
|
assert eff is True and reason == "numeric"
|