test(harness): L0 testcontainers + L1 contract tests for ported db/mschema/rest (A9, spec §1)
Ports the leaf data-layer modules and validates them: - mschema/ (models, eligibility, merge, render), db/ (connection, sampling, introspect, fetch_ca), rest/client.py -- renamed psdwp3->nsp, verbatim. - L0 (testcontainers, real Postgres): db connection read-only enforcement (psd_ro cannot CREATE/INSERT), introspect against a known schema (tables, columns, types, comments, FKs, enum, composite PK), sampling most-frequent values + truncation reporting. 15 tests, ~4s. - L1 (fake data): rest/client RPC contract (mocked transport -- X-API-Key header, payloads, base_url slash handling, HTTP/network error surfacing), mschema/render 3 formats (markdown, mschema-text, schema-dict) + eligibility rules (wide_text excluded, short_text/numeric/enum/temporal/ boolean eligible, annotation override wins). 25 tests. pyproject registers l0/l2 markers + addopts '-m not l2' (L2 opt-in). Deferred to their dependency-porting tasks: test_rrf.py (search needs vectorstore, B3) and the 11 CLI contract tests (need _guards/session, wired when each command lands). 'Not assumed reliable' now has real teeth for the data layer; CLI/search contracts follow.
This commit is contained in:
@@ -0,0 +1,93 @@
|
||||
"""Classificazione di column eligibility (principio trasversale PsdWp3).
|
||||
|
||||
Vedi docs/superpowers/specs/2026-06-13-nsp-column-eligibility-principle.md.
|
||||
Il testo ampio (lettere di dimissione, note, anamnesi) è ignorato ovunque; i dati
|
||||
provengono solo da numerici, enum, temporali, booleani e testo breve.
|
||||
"""
|
||||
|
||||
import re
|
||||
|
||||
from nsp.config import EligibilityConfig
|
||||
from nsp.db.sampling import is_text_type
|
||||
from nsp.mschema.models import ColumnAnnotation, ColumnPhysical, PhysicalSchema
|
||||
|
||||
_NUMERIC_PREFIXES = (
|
||||
"smallint", "integer", "bigint", "numeric", "decimal", "real", "double", "money",
|
||||
)
|
||||
_TEMPORAL_PREFIXES = ("date", "time", "timestamp", "interval")
|
||||
_LEN_RE = re.compile(r"\((\d+)\)")
|
||||
|
||||
|
||||
def _declared_len(pg_type: str) -> int | None:
|
||||
m = _LEN_RE.search(pg_type)
|
||||
return int(m.group(1)) if m else None
|
||||
|
||||
|
||||
def classify_column(
|
||||
pg_type: str,
|
||||
is_enum: bool,
|
||||
sampled_avg: float | None,
|
||||
sampled_max: int | None,
|
||||
cfg: EligibilityConfig,
|
||||
) -> tuple[bool, str]:
|
||||
"""Classifica una colonna come (eligible, reason). Funzione pura, senza I/O."""
|
||||
t = pg_type.strip().lower()
|
||||
if t.endswith("[]"):
|
||||
return False, "wide_text"
|
||||
if is_enum:
|
||||
return True, "enum"
|
||||
if t.startswith(_NUMERIC_PREFIXES):
|
||||
return True, "numeric"
|
||||
if t.startswith("boolean"):
|
||||
return True, "boolean"
|
||||
if t.startswith(_TEMPORAL_PREFIXES):
|
||||
return True, "temporal"
|
||||
if t.startswith("uuid"):
|
||||
return True, "code"
|
||||
if is_text_type(t):
|
||||
declared = _declared_len(t)
|
||||
if declared is not None and declared <= cfg.max_declared_len:
|
||||
return True, "short_text"
|
||||
# bound grande o text/varchar non vincolato: decide il dato campionato
|
||||
if sampled_avg is None or sampled_max is None:
|
||||
return False, "wide_text"
|
||||
if sampled_avg <= cfg.max_avg_length and sampled_max <= cfg.max_sampled_len:
|
||||
return True, "short_text"
|
||||
return False, "wide_text"
|
||||
return False, "wide_text"
|
||||
|
||||
|
||||
def _sampled_stats(examples: list[str]) -> tuple[float | None, int | None]:
|
||||
if not examples:
|
||||
return None, None
|
||||
lengths = [len(v) for v in examples]
|
||||
return sum(lengths) / len(lengths), max(lengths)
|
||||
|
||||
|
||||
def classify_all(physical: PhysicalSchema, cfg: EligibilityConfig) -> None:
|
||||
"""Assegna eligible/eligibility_reason a ogni colonna (in-place) e azzera gli
|
||||
examples delle colonne ignored. Da chiamare DOPO add_examples."""
|
||||
ignore_by_name = {name.lower() for name in cfg.ignore_columns}
|
||||
for table in physical.tables.values():
|
||||
for column_name, column in table.columns.items():
|
||||
if column_name.lower() in ignore_by_name:
|
||||
# colonna di servizio (ETL/audit): ignorata a prescindere dal tipo
|
||||
column.eligible = False
|
||||
column.eligibility_reason = "ignored_by_name"
|
||||
column.examples = []
|
||||
continue
|
||||
avg, mx = _sampled_stats(column.examples)
|
||||
eligible, reason = classify_column(column.type, column.is_enum, avg, mx, cfg)
|
||||
column.eligible = eligible
|
||||
column.eligibility_reason = reason
|
||||
if not eligible:
|
||||
column.examples = []
|
||||
|
||||
|
||||
def effective_eligibility(
|
||||
column: ColumnPhysical, annotation: ColumnAnnotation | None
|
||||
) -> tuple[bool, str]:
|
||||
"""Eligibilità effettiva: l'override in annotations.yaml vince sul fisico."""
|
||||
if annotation is not None and annotation.eligible is not None:
|
||||
return annotation.eligible, "override"
|
||||
return column.eligible, column.eligibility_reason
|
||||
Reference in New Issue
Block a user