Files
ThothII/harness/tests/test_mschema_eligibility.py
T
marcopan eb3bde90e2 test(harness): L0 testcontainers + L1 contract tests for ported db/mschema/rest (A9, spec §1)
Ports the leaf data-layer modules and validates them:
- mschema/ (models, eligibility, merge, render), db/ (connection, sampling,
  introspect, fetch_ca), rest/client.py -- renamed psdwp3->nsp, verbatim.
- L0 (testcontainers, real Postgres): db connection read-only enforcement
  (psd_ro cannot CREATE/INSERT), introspect against a known schema (tables,
  columns, types, comments, FKs, enum, composite PK), sampling most-frequent
  values + truncation reporting. 15 tests, ~4s.
- L1 (fake data): rest/client RPC contract (mocked transport -- X-API-Key
  header, payloads, base_url slash handling, HTTP/network error surfacing),
  mschema/render 3 formats (markdown, mschema-text, schema-dict) +
  eligibility rules (wide_text excluded, short_text/numeric/enum/temporal/
  boolean eligible, annotation override wins). 25 tests.

pyproject registers l0/l2 markers + addopts '-m not l2' (L2 opt-in).

Deferred to their dependency-porting tasks: test_rrf.py (search needs
vectorstore, B3) and the 11 CLI contract tests (need _guards/session, wired
when each command lands). 'Not assumed reliable' now has real teeth for the
data layer; CLI/search contracts follow.
2026-06-26 22:53:08 +02:00

110 lines
3.9 KiB
Python

"""L1: mschema/eligibility — the column-eligibility principle on fake columns.
Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
comes only from numerics, enums, temporals, booleans, and short text. Annotation
override wins over the physical classification.
"""
from datetime import datetime
from nsp.config import EligibilityConfig
from nsp.mschema.eligibility import classify_all, classify_column, effective_eligibility
from nsp.mschema.models import (
Annotations,
ColumnAnnotation,
ColumnPhysical,
PhysicalSchema,
TablePhysical,
)
CFG = EligibilityConfig()
# --- classify_column (pure) ----------------------------------------------------
def test_numeric_eligible():
ok, reason = classify_column("integer", False, None, None, CFG)
assert ok and reason == "numeric"
def test_enum_eligible():
ok, reason = classify_column("tipo_ricovero", True, None, None, CFG)
assert ok and reason == "enum"
def test_temporal_eligible():
ok, reason = classify_column("timestamp without time zone", False, None, None, CFG)
assert ok and reason == "temporal"
def test_boolean_eligible():
ok, reason = classify_column("boolean", False, None, None, CFG)
assert ok and reason == "boolean"
def test_short_declared_text_eligible_without_sampling():
# varchar(100) <= max_declared_len(128) → eligible without sampling the data
ok, reason = classify_column("varchar(100)", False, None, None, CFG)
assert ok and reason == "short_text"
def test_unbounded_text_without_sampling_is_wide_text():
ok, reason = classify_column("text", False, None, None, CFG)
assert not ok and reason == "wide_text"
def test_long_text_sampled_within_bounds_is_short_text():
ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG)
assert ok and reason == "short_text"
def test_long_text_sampled_above_bounds_is_wide_text():
ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG)
assert not ok and reason == "wide_text"
def test_array_type_is_wide_text():
ok, reason = classify_column("text[]", False, None, None, CFG)
assert not ok and reason == "wide_text"
# --- classify_all (in-place) ---------------------------------------------------
def _schema_with(**columns) -> PhysicalSchema:
return PhysicalSchema(
database="db", schema="dw", introspected_at=datetime(2025, 1, 1),
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
)
def test_classify_all_marks_wide_text_and_clears_examples():
schema = _schema_with(
note=dict(type="text", examples=["a" * 500, "b" * 400]),
cod=dict(type="varchar(10)", examples=["X", "Y"]),
etl_last_update=dict(type="timestamp", examples=[]),
)
classify_all(schema, CFG)
cols = schema.tables["t"].columns
assert cols["note"].eligible is False
assert cols["note"].eligibility_reason == "wide_text"
assert cols["note"].examples == [] # examples cleared on ignored columns
assert cols["cod"].eligible is True
# ignore_columns default includes etl_last_update → ignored_by_name regardless of type
assert cols["etl_last_update"].eligible is False
assert cols["etl_last_update"].eligibility_reason == "ignored_by_name"
# --- effective_eligibility (annotation override wins) --------------------------
def test_annotation_override_forces_eligible():
col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text")
ann = ColumnAnnotation(eligible=True)
eff, reason = effective_eligibility(col, ann)
assert eff is True
assert reason == "override"
def test_no_annotation_falls_back_to_physical():
col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric")
eff, reason = effective_eligibility(col, None)
assert eff is True and reason == "numeric"