test(harness): L0 testcontainers + L1 contract tests for ported db/mschema/rest (A9, spec §1)
Ports the leaf data-layer modules and validates them: - mschema/ (models, eligibility, merge, render), db/ (connection, sampling, introspect, fetch_ca), rest/client.py -- renamed psdwp3->nsp, verbatim. - L0 (testcontainers, real Postgres): db connection read-only enforcement (psd_ro cannot CREATE/INSERT), introspect against a known schema (tables, columns, types, comments, FKs, enum, composite PK), sampling most-frequent values + truncation reporting. 15 tests, ~4s. - L1 (fake data): rest/client RPC contract (mocked transport -- X-API-Key header, payloads, base_url slash handling, HTTP/network error surfacing), mschema/render 3 formats (markdown, mschema-text, schema-dict) + eligibility rules (wide_text excluded, short_text/numeric/enum/temporal/ boolean eligible, annotation override wins). 25 tests. pyproject registers l0/l2 markers + addopts '-m not l2' (L2 opt-in). Deferred to their dependency-porting tasks: test_rrf.py (search needs vectorstore, B3) and the 11 CLI contract tests (need _guards/session, wired when each command lands). 'Not assumed reliable' now has real teeth for the data layer; CLI/search contracts follow.
This commit is contained in:
@@ -0,0 +1,56 @@
|
||||
"""L0: db/connection read-only enforcement against real Postgres (testcontainers).
|
||||
|
||||
The ported read-only contract: the psd_ro role can SELECT but not write, and
|
||||
can_create_in_schema / writable_tables reflect that. This is where 'ported code
|
||||
is not assumed reliable' gains real teeth for the data layer.
|
||||
"""
|
||||
import pytest
|
||||
from sqlalchemy import create_engine, text
|
||||
|
||||
from nsp.db.connection import can_create_in_schema, make_engine, ping, writable_tables
|
||||
|
||||
pytestmark = [pytest.mark.l0]
|
||||
|
||||
|
||||
def test_ping_succeeds_on_read_only_role(ro_url):
|
||||
engine = create_engine(ro_url)
|
||||
try:
|
||||
ping(engine) # SELECT 1 — must not raise
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
|
||||
def test_read_only_role_cannot_create_in_schema(ro_url):
|
||||
engine = create_engine(ro_url)
|
||||
try:
|
||||
# psd_ro has USAGE + SELECT only, not CREATE on the dw schema.
|
||||
assert can_create_in_schema(engine, "dw") is False
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
|
||||
def test_writable_tables_empty_for_read_only_role(ro_url):
|
||||
engine = create_engine(ro_url)
|
||||
try:
|
||||
tables = writable_tables(engine, "dw")
|
||||
assert tables == [] # read-only role has no INSERT/UPDATE/DELETE grants
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
|
||||
def test_read_only_role_cannot_insert(ro_url):
|
||||
"""The hard guarantee: a write attempt raises (enforced by Postgres, surfaced
|
||||
by our engine)."""
|
||||
engine = create_engine(ro_url)
|
||||
try:
|
||||
with pytest.raises(Exception):
|
||||
with engine.begin() as conn:
|
||||
conn.execute(text('INSERT INTO dw.dim_pazienti VALUES (999, %s, %s)'),
|
||||
("test", "test"))
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
|
||||
def test_admin_engine_can_create_in_schema(admin_engine):
|
||||
# Sanity: the admin (table owner) CAN create — confirms the test harness itself.
|
||||
assert can_create_in_schema(admin_engine, "dw") is True
|
||||
@@ -0,0 +1,58 @@
|
||||
"""L0: db/introspect against a known schema (testcontainers).
|
||||
Verifies the ported introspection returns the right tables, columns, types, comments,
|
||||
FKs, and indexes from a real Postgres catalog.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
from nsp.db.introspect import IntrospectionError, introspect
|
||||
|
||||
pytestmark = [pytest.mark.l0]
|
||||
|
||||
|
||||
def test_introspect_returns_known_tables(admin_engine):
|
||||
schema = introspect(admin_engine, "testdb", "dw")
|
||||
assert set(schema.tables.keys()) == {"dim_pazienti", "dim_periodi", "fct_ricoveri"}
|
||||
assert schema.db_schema == "dw"
|
||||
|
||||
|
||||
def test_introspect_columns_types_and_comments(admin_engine):
|
||||
schema = introspect(admin_engine, "testdb", "dw")
|
||||
pazienti = schema.tables["dim_pazienti"]
|
||||
assert set(pazienti.columns.keys()) == {"id_paziente", "nome", "citta"}
|
||||
assert pazienti.columns["id_paziente"].pk is True
|
||||
assert pazienti.columns["id_paziente"].nullable is False
|
||||
assert "bigint" in pazienti.columns["id_paziente"].type.lower()
|
||||
assert pazienti.comment == "Anagrafica pazienti"
|
||||
assert pazienti.columns["citta"].comment == "Comune di residenza"
|
||||
|
||||
|
||||
def test_introspect_enum_detected(admin_engine):
|
||||
schema = introspect(admin_engine, "testdb", "dw")
|
||||
tipo_col = schema.tables["fct_ricoveri"].columns["tipo"]
|
||||
assert tipo_col.is_enum is True
|
||||
|
||||
|
||||
def test_introspect_foreign_keys(admin_engine):
|
||||
schema = introspect(admin_engine, "testdb", "dw")
|
||||
ricoveri = schema.tables["fct_ricoveri"]
|
||||
fk_tables = {fk.ref_table for fk in ricoveri.foreign_keys}
|
||||
assert "dim_pazienti" in fk_tables
|
||||
assert "dim_periodi" in fk_tables
|
||||
# the composite FK to dim_periodi (anno, mese)
|
||||
periodi_fk = [fk for fk in ricoveri.foreign_keys if fk.ref_table == "dim_periodi"]
|
||||
assert periodi_fk
|
||||
assert set(periodi_fk[0].columns) == {"anno", "mese"}
|
||||
|
||||
|
||||
def test_introspect_indexes(admin_engine):
|
||||
schema = introspect(admin_engine, "testdb", "dw")
|
||||
# composite PK on dim_periodi (anno, mese)
|
||||
periodi = schema.tables["dim_periodi"]
|
||||
pk_indexes = [i for i in periodi.indexes if i.primary]
|
||||
assert pk_indexes, "dim_periodi should have a primary index"
|
||||
assert set(pk_indexes[0].columns) == {"anno", "mese"}
|
||||
|
||||
|
||||
def test_introspect_nonexistent_schema_raises(admin_engine):
|
||||
with pytest.raises(IntrospectionError):
|
||||
introspect(admin_engine, "testdb", "nonexistent_schema")
|
||||
@@ -0,0 +1,54 @@
|
||||
"""L0: db/sampling against known data (testcontainers).
|
||||
Verifies unique_values_for_lsh returns the expected most-frequent values for text
|
||||
columns, and that wide_text / non-text columns are excluded.
|
||||
"""
|
||||
import pytest
|
||||
|
||||
from nsp.config import LshConfig
|
||||
from nsp.db.introspect import introspect
|
||||
from nsp.db.sampling import is_text_type, unique_values_for_lsh
|
||||
|
||||
pytestmark = [pytest.mark.l0]
|
||||
|
||||
|
||||
def test_is_text_type():
|
||||
assert is_text_type("text")
|
||||
assert is_text_type("varchar(100)")
|
||||
assert is_text_type("character varying")
|
||||
assert not is_text_type("integer")
|
||||
assert not is_text_type("bigint")
|
||||
assert not is_text_type("timestamp without time zone")
|
||||
|
||||
|
||||
def test_unique_values_for_lsh_returns_most_frequent(admin_engine):
|
||||
schema = introspect(admin_engine, "testdb", "dw")
|
||||
# Before classify_all, all text columns are eligible=True by default. Sampling
|
||||
# only touches text types regardless.
|
||||
values, skipped, truncated = unique_values_for_lsh(
|
||||
admin_engine, schema, LshConfig(max_values_per_column=100)
|
||||
)
|
||||
# dim_pazienti.citta: Milano, Bergamo, Brescia (3 distinct, all eligible text)
|
||||
citta = values.get("dim_pazienti", {}).get("citta")
|
||||
assert citta is not None
|
||||
assert set(citta) == {"Milano", "Bergamo", "Brescia"}
|
||||
|
||||
|
||||
def test_unique_values_for_lsh_excludes_non_text(admin_engine):
|
||||
schema = introspect(admin_engine, "testdb", "dw")
|
||||
values, _, _ = unique_values_for_lsh(
|
||||
admin_engine, schema, LshConfig(max_values_per_column=100)
|
||||
)
|
||||
# id_paziente is bigint — must never appear in the LSH values.
|
||||
assert "id_paziente" not in values.get("dim_pazienti", {})
|
||||
|
||||
|
||||
def test_unique_values_for_lsh_truncation_reported(admin_engine):
|
||||
schema = introspect(admin_engine, "testdb", "dw")
|
||||
# Force a tiny cap so procedura/diagnosi columns (which have >2 distinct values)
|
||||
# are reported as truncated rather than silently cut.
|
||||
_, _, truncated = unique_values_for_lsh(
|
||||
admin_engine, schema, LshConfig(max_values_per_column=1)
|
||||
)
|
||||
truncated_cols = {(t.table, t.column) for t in truncated}
|
||||
# fct_ricoveri has several eligible text columns with distinct values
|
||||
assert any(t[0] == "fct_ricoveri" for t in truncated_cols)
|
||||
Reference in New Issue
Block a user