test(harness): L0 testcontainers + L1 contract tests for ported db/mschema/rest (A9, spec §1)

Ports the leaf data-layer modules and validates them:
- mschema/ (models, eligibility, merge, render), db/ (connection, sampling,
  introspect, fetch_ca), rest/client.py -- renamed psdwp3->nsp, verbatim.
- L0 (testcontainers, real Postgres): db connection read-only enforcement
  (psd_ro cannot CREATE/INSERT), introspect against a known schema (tables,
  columns, types, comments, FKs, enum, composite PK), sampling most-frequent
  values + truncation reporting. 15 tests, ~4s.
- L1 (fake data): rest/client RPC contract (mocked transport -- X-API-Key
  header, payloads, base_url slash handling, HTTP/network error surfacing),
  mschema/render 3 formats (markdown, mschema-text, schema-dict) +
  eligibility rules (wide_text excluded, short_text/numeric/enum/temporal/
  boolean eligible, annotation override wins). 25 tests.

pyproject registers l0/l2 markers + addopts '-m not l2' (L2 opt-in).

Deferred to their dependency-porting tasks: test_rrf.py (search needs
vectorstore, B3) and the 11 CLI contract tests (need _guards/session, wired
when each command lands). 'Not assumed reliable' now has real teeth for the
data layer; CLI/search contracts follow.
This commit is contained in:
2026-06-26 22:53:08 +02:00
parent 5f24bd1adc
commit eb3bde90e2
22 changed files with 1552 additions and 0 deletions
+32
View File
@@ -2,6 +2,10 @@ import os
import sys
from pathlib import Path
import pytest
from sqlalchemy import create_engine
from testcontainers.postgres import PostgresContainer
# Permetti `pytest` lanciato da qualsiasi directory di trovare il package `nsp`
# (installato in modalità editable nella venv, ma utile anche senza attivazione).
_ROOT = Path(__file__).resolve().parent.parent
@@ -10,3 +14,31 @@ if str(_ROOT) not in sys.path:
# Directory sessions/ risolta relativamente alla root harness (per i test che creano sessioni)
os.environ.setdefault("NSP_HARNESS_ROOT", str(_ROOT))
# --- L0 fixtures (testcontainers, real Postgres) --------------------------------
# Session-scoped: un solo container per tutta la run L0. Schema fixture caricato una
# volta (crea schema dw + dati + ruolo psd_ro read-only). Salta automaticamente se
# Docker non e' disponibile.
@pytest.fixture(scope="session")
def pg_container():
with PostgresContainer("postgres:16-alpine") as pg:
yield pg
@pytest.fixture(scope="session")
def admin_engine(pg_container):
engine = create_engine(pg_container.get_connection_url())
schema_sql = (Path(__file__).parent / "fixtures" / "schema.sql").read_text()
with engine.begin() as conn:
conn.exec_driver_sql(schema_sql)
yield engine
engine.dispose()
@pytest.fixture(scope="session")
def ro_url(pg_container, admin_engine) -> str:
host = pg_container.get_container_host_ip()
port = pg_container.get_exposed_port(5432)
return f"postgresql+psycopg2://psd_ro:psd_ro@{host}:{port}/{pg_container.dbname}"
+58
View File
@@ -0,0 +1,58 @@
CREATE SCHEMA dw;
CREATE TABLE dw.dim_pazienti (
id_paziente BIGINT PRIMARY KEY,
nome VARCHAR(100),
citta VARCHAR(100)
);
COMMENT ON TABLE dw.dim_pazienti IS 'Anagrafica pazienti';
COMMENT ON COLUMN dw.dim_pazienti.citta IS 'Comune di residenza';
CREATE TABLE dw.dim_periodi (
anno INT NOT NULL,
mese INT NOT NULL,
descrizione VARCHAR(50),
PRIMARY KEY (anno, mese)
);
CREATE TYPE dw.tipo_ricovero AS ENUM ('ordinario', 'day_hospital', 'urgenza');
CREATE TABLE dw.fct_ricoveri (
id_ricovero BIGINT PRIMARY KEY,
id_paziente BIGINT REFERENCES dw.dim_pazienti (id_paziente),
anno INT,
mese INT,
diagnosi VARCHAR(200),
procedura TEXT,
reparto VARCHAR(50),
tipo dw.tipo_ricovero,
lettera_dimissione TEXT,
etl_last_update TIMESTAMP,
FOREIGN KEY (anno, mese) REFERENCES dw.dim_periodi (anno, mese)
);
COMMENT ON TABLE dw.fct_ricoveri IS 'Fatti: ricoveri ospedalieri';
COMMENT ON COLUMN dw.fct_ricoveri.diagnosi IS 'Diagnosi principale (testo)';
INSERT INTO dw.dim_pazienti VALUES
(1, 'Mario Rossi', 'Milano'),
(2, 'Anna Bianchi', 'Bergamo'),
(3, 'Luca Verdi', 'Brescia');
INSERT INTO dw.dim_periodi VALUES
(2025, 1, 'Gennaio 2025'), (2025, 2, 'Febbraio 2025');
INSERT INTO dw.fct_ricoveri VALUES
(10, 1, 2025, 1, 'fibrillazione atriale', 'ablazione transcatetere', 'cardiologia', 'ordinario',
'Si dimette il paziente in condizioni cliniche stabili dopo ablazione transcatetere della fibrillazione atriale; si raccomanda terapia anticoagulante orale e controllo cardiologico ambulatoriale a trenta giorni dalla dimissione.',
'2025-02-01 03:00:00'),
(11, 2, 2025, 1, 'infarto miocardico acuto', 'angioplastica coronarica', 'cardiologia', 'urgenza',
'Paziente ricoverato per infarto miocardico acuto trattato con angioplastica coronarica primaria e impianto di stent medicato; decorso post-procedurale regolare, si dimette con doppia antiaggregazione e indicazione a riabilitazione cardiologica.',
'2025-02-01 03:00:00'),
(12, 3, 2025, 2, 'fibrillazione ventricolare', 'defibrillazione', 'cardiologia', 'urgenza',
'Ricovero in urgenza per fibrillazione ventricolare con arresto cardiocircolatorio rianimato; stabilizzato in terapia intensiva cardiologica, si dimette con indicazione a impianto di defibrillatore e follow-up elettrofisiologico.',
'2025-03-01 03:00:00'),
(13, 1, 2025, 2, 'scompenso cardiaco', NULL, 'pronto soccorso', 'ordinario', NULL, '2025-03-01 03:00:00');
CREATE ROLE psd_ro LOGIN PASSWORD 'psd_ro';
GRANT USAGE ON SCHEMA dw TO psd_ro;
GRANT SELECT ON ALL TABLES IN SCHEMA dw TO psd_ro;
ANALYZE;
View File
+56
View File
@@ -0,0 +1,56 @@
"""L0: db/connection read-only enforcement against real Postgres (testcontainers).
The ported read-only contract: the psd_ro role can SELECT but not write, and
can_create_in_schema / writable_tables reflect that. This is where 'ported code
is not assumed reliable' gains real teeth for the data layer.
"""
import pytest
from sqlalchemy import create_engine, text
from nsp.db.connection import can_create_in_schema, make_engine, ping, writable_tables
pytestmark = [pytest.mark.l0]
def test_ping_succeeds_on_read_only_role(ro_url):
engine = create_engine(ro_url)
try:
ping(engine) # SELECT 1 — must not raise
finally:
engine.dispose()
def test_read_only_role_cannot_create_in_schema(ro_url):
engine = create_engine(ro_url)
try:
# psd_ro has USAGE + SELECT only, not CREATE on the dw schema.
assert can_create_in_schema(engine, "dw") is False
finally:
engine.dispose()
def test_writable_tables_empty_for_read_only_role(ro_url):
engine = create_engine(ro_url)
try:
tables = writable_tables(engine, "dw")
assert tables == [] # read-only role has no INSERT/UPDATE/DELETE grants
finally:
engine.dispose()
def test_read_only_role_cannot_insert(ro_url):
"""The hard guarantee: a write attempt raises (enforced by Postgres, surfaced
by our engine)."""
engine = create_engine(ro_url)
try:
with pytest.raises(Exception):
with engine.begin() as conn:
conn.execute(text('INSERT INTO dw.dim_pazienti VALUES (999, %s, %s)'),
("test", "test"))
finally:
engine.dispose()
def test_admin_engine_can_create_in_schema(admin_engine):
# Sanity: the admin (table owner) CAN create — confirms the test harness itself.
assert can_create_in_schema(admin_engine, "dw") is True
+58
View File
@@ -0,0 +1,58 @@
"""L0: db/introspect against a known schema (testcontainers).
Verifies the ported introspection returns the right tables, columns, types, comments,
FKs, and indexes from a real Postgres catalog.
"""
import pytest
from nsp.db.introspect import IntrospectionError, introspect
pytestmark = [pytest.mark.l0]
def test_introspect_returns_known_tables(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
assert set(schema.tables.keys()) == {"dim_pazienti", "dim_periodi", "fct_ricoveri"}
assert schema.db_schema == "dw"
def test_introspect_columns_types_and_comments(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
pazienti = schema.tables["dim_pazienti"]
assert set(pazienti.columns.keys()) == {"id_paziente", "nome", "citta"}
assert pazienti.columns["id_paziente"].pk is True
assert pazienti.columns["id_paziente"].nullable is False
assert "bigint" in pazienti.columns["id_paziente"].type.lower()
assert pazienti.comment == "Anagrafica pazienti"
assert pazienti.columns["citta"].comment == "Comune di residenza"
def test_introspect_enum_detected(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
tipo_col = schema.tables["fct_ricoveri"].columns["tipo"]
assert tipo_col.is_enum is True
def test_introspect_foreign_keys(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
ricoveri = schema.tables["fct_ricoveri"]
fk_tables = {fk.ref_table for fk in ricoveri.foreign_keys}
assert "dim_pazienti" in fk_tables
assert "dim_periodi" in fk_tables
# the composite FK to dim_periodi (anno, mese)
periodi_fk = [fk for fk in ricoveri.foreign_keys if fk.ref_table == "dim_periodi"]
assert periodi_fk
assert set(periodi_fk[0].columns) == {"anno", "mese"}
def test_introspect_indexes(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
# composite PK on dim_periodi (anno, mese)
periodi = schema.tables["dim_periodi"]
pk_indexes = [i for i in periodi.indexes if i.primary]
assert pk_indexes, "dim_periodi should have a primary index"
assert set(pk_indexes[0].columns) == {"anno", "mese"}
def test_introspect_nonexistent_schema_raises(admin_engine):
with pytest.raises(IntrospectionError):
introspect(admin_engine, "testdb", "nonexistent_schema")
+54
View File
@@ -0,0 +1,54 @@
"""L0: db/sampling against known data (testcontainers).
Verifies unique_values_for_lsh returns the expected most-frequent values for text
columns, and that wide_text / non-text columns are excluded.
"""
import pytest
from nsp.config import LshConfig
from nsp.db.introspect import introspect
from nsp.db.sampling import is_text_type, unique_values_for_lsh
pytestmark = [pytest.mark.l0]
def test_is_text_type():
assert is_text_type("text")
assert is_text_type("varchar(100)")
assert is_text_type("character varying")
assert not is_text_type("integer")
assert not is_text_type("bigint")
assert not is_text_type("timestamp without time zone")
def test_unique_values_for_lsh_returns_most_frequent(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
# Before classify_all, all text columns are eligible=True by default. Sampling
# only touches text types regardless.
values, skipped, truncated = unique_values_for_lsh(
admin_engine, schema, LshConfig(max_values_per_column=100)
)
# dim_pazienti.citta: Milano, Bergamo, Brescia (3 distinct, all eligible text)
citta = values.get("dim_pazienti", {}).get("citta")
assert citta is not None
assert set(citta) == {"Milano", "Bergamo", "Brescia"}
def test_unique_values_for_lsh_excludes_non_text(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
values, _, _ = unique_values_for_lsh(
admin_engine, schema, LshConfig(max_values_per_column=100)
)
# id_paziente is bigint — must never appear in the LSH values.
assert "id_paziente" not in values.get("dim_pazienti", {})
def test_unique_values_for_lsh_truncation_reported(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
# Force a tiny cap so procedura/diagnosi columns (which have >2 distinct values)
# are reported as truncated rather than silently cut.
_, _, truncated = unique_values_for_lsh(
admin_engine, schema, LshConfig(max_values_per_column=1)
)
truncated_cols = {(t.table, t.column) for t in truncated}
# fct_ricoveri has several eligible text columns with distinct values
assert any(t[0] == "fct_ricoveri" for t in truncated_cols)
+109
View File
@@ -0,0 +1,109 @@
"""L1: mschema/eligibility — the column-eligibility principle on fake columns.
Wide text (lettere di dimissione, note, anamnesi) is excluded everywhere; data
comes only from numerics, enums, temporals, booleans, and short text. Annotation
override wins over the physical classification.
"""
from datetime import datetime
from nsp.config import EligibilityConfig
from nsp.mschema.eligibility import classify_all, classify_column, effective_eligibility
from nsp.mschema.models import (
Annotations,
ColumnAnnotation,
ColumnPhysical,
PhysicalSchema,
TablePhysical,
)
CFG = EligibilityConfig()
# --- classify_column (pure) ----------------------------------------------------
def test_numeric_eligible():
ok, reason = classify_column("integer", False, None, None, CFG)
assert ok and reason == "numeric"
def test_enum_eligible():
ok, reason = classify_column("tipo_ricovero", True, None, None, CFG)
assert ok and reason == "enum"
def test_temporal_eligible():
ok, reason = classify_column("timestamp without time zone", False, None, None, CFG)
assert ok and reason == "temporal"
def test_boolean_eligible():
ok, reason = classify_column("boolean", False, None, None, CFG)
assert ok and reason == "boolean"
def test_short_declared_text_eligible_without_sampling():
# varchar(100) <= max_declared_len(128) → eligible without sampling the data
ok, reason = classify_column("varchar(100)", False, None, None, CFG)
assert ok and reason == "short_text"
def test_unbounded_text_without_sampling_is_wide_text():
ok, reason = classify_column("text", False, None, None, CFG)
assert not ok and reason == "wide_text"
def test_long_text_sampled_within_bounds_is_short_text():
ok, reason = classify_column("varchar(500)", False, sampled_avg=10.0, sampled_max=50, cfg=CFG)
assert ok and reason == "short_text"
def test_long_text_sampled_above_bounds_is_wide_text():
ok, reason = classify_column("varchar(500)", False, sampled_avg=100.0, sampled_max=600, cfg=CFG)
assert not ok and reason == "wide_text"
def test_array_type_is_wide_text():
ok, reason = classify_column("text[]", False, None, None, CFG)
assert not ok and reason == "wide_text"
# --- classify_all (in-place) ---------------------------------------------------
def _schema_with(**columns) -> PhysicalSchema:
return PhysicalSchema(
database="db", schema="dw", introspected_at=datetime(2025, 1, 1),
tables={"t": TablePhysical(columns={k: ColumnPhysical(**v) for k, v in columns.items()})},
)
def test_classify_all_marks_wide_text_and_clears_examples():
schema = _schema_with(
note=dict(type="text", examples=["a" * 500, "b" * 400]),
cod=dict(type="varchar(10)", examples=["X", "Y"]),
etl_last_update=dict(type="timestamp", examples=[]),
)
classify_all(schema, CFG)
cols = schema.tables["t"].columns
assert cols["note"].eligible is False
assert cols["note"].eligibility_reason == "wide_text"
assert cols["note"].examples == [] # examples cleared on ignored columns
assert cols["cod"].eligible is True
# ignore_columns default includes etl_last_update → ignored_by_name regardless of type
assert cols["etl_last_update"].eligible is False
assert cols["etl_last_update"].eligibility_reason == "ignored_by_name"
# --- effective_eligibility (annotation override wins) --------------------------
def test_annotation_override_forces_eligible():
col = ColumnPhysical(type="text", eligible=False, eligibility_reason="wide_text")
ann = ColumnAnnotation(eligible=True)
eff, reason = effective_eligibility(col, ann)
assert eff is True
assert reason == "override"
def test_no_annotation_falls_back_to_physical():
col = ColumnPhysical(type="integer", eligible=True, eligibility_reason="numeric")
eff, reason = effective_eligibility(col, None)
assert eff is True and reason == "numeric"
+100
View File
@@ -0,0 +1,100 @@
"""L1: mschema/render — the 3 serialization formats on a fake PhysicalSchema.
Catches port breaks in the render layer (markdown reviewer report, mschema-text
ThothAI style, schema-dict for AV-SQL). Pure logic, no I/O.
"""
from datetime import datetime
from nsp.mschema.models import (
Annotations,
ColumnAnnotation,
ColumnPhysical,
ForeignKey,
PhysicalSchema,
TablePhysical,
)
from nsp.mschema.render import to_markdown, to_mschema_text, to_schema_dict
def _fake_schema() -> PhysicalSchema:
return PhysicalSchema(
database="testdb",
schema="dw",
introspected_at=datetime(2025, 1, 1, 0, 0, 0),
tables={
"dim_pazienti": TablePhysical(
comment="Anagrafica pazienti",
row_count=3,
columns={
"id_paziente": ColumnPhysical(type="bigint", nullable=False, pk=True),
"citta": ColumnPhysical(type="varchar(100)", examples=["Milano", "Bergamo"]),
"note": ColumnPhysical(type="text", eligible=False,
eligibility_reason="wide_text"),
},
foreign_keys=[
ForeignKey(columns=["fk_col"], ref_table="other", ref_columns=["id"]),
],
),
},
)
def test_to_markdown_includes_table_and_marks_ignored():
md = to_markdown(_fake_schema())
assert "# Schema dw (testdb)" in md
assert "## dim_pazienti" in md
assert "Anagrafica pazienti" in md
# wide_text column is struck through and labeled
assert "~~note~~" in md
assert "wide_text" in md
# foreign key reported
assert "other" in md
def test_to_mschema_text_style_and_eligibility_filter():
txt = to_mschema_text(_fake_schema())
assert "【Schema】" in txt
assert "【Foreign keys】" in txt
assert "CREATE TABLE dim_pazienti (" in txt
# eligible columns appear (column name lowercase, type uppercased)
assert "id_paziente BIGINT" in txt
assert "citta VARCHAR(100)" in txt
# wide_text column is filtered out: "note" must not appear as a CREATE TABLE column line
assert " NOTE TEXT" not in txt
assert "note" not in txt.replace("【", "").lower().split()
# PK marker
assert "PRIMARY KEY" in txt
def test_to_schema_dict_avsql_shape():
d = to_schema_dict(_fake_schema())
assert "dim_pazienti" in d
entry = d["dim_pazienti"]
assert "columns_name" in entry
assert "columns_type" in entry
assert "primary_keys" in entry
assert "foreign_keys" in entry
assert "table_to_tablefullname" in entry
assert entry["table_to_tablefullname"] == "dw.dim_pazienti"
# wide_text column filtered out of columns_name
assert "id_paziente" in entry["columns_name"]
assert "note" not in entry["columns_name"]
assert entry["primary_keys"] == ["id_paziente"]
def test_annotations_override_description_in_render():
schema = _fake_schema()
from nsp.mschema.models import TableAnnotation
ann = Annotations(tables={
"dim_pazienti": TableAnnotation(columns={
"citta": ColumnAnnotation(description="Comune di residenza"),
}),
})
md = to_markdown(schema, ann)
assert "Comune di residenza" in md
def test_render_subset_of_tables():
txt = to_mschema_text(_fake_schema(), tables=["dim_pazienti"])
# only the requested table appears
assert "dim_pazienti" in txt
+115
View File
@@ -0,0 +1,115 @@
"""L1: rest/client RPC contract (mocked transport — no network).
Verifies the ported RestClient: each rpc carries the X-API-Key header, the right
payload, base_url slash handling, and HTTP/network errors surface as RestError.
Ported (renamed psdwp3->nsp) from ChironeWp3/tests/test_rest_client.py.
"""
import pytest
import requests
from nsp.config import RestConfig
from nsp.rest.client import RestClient, RestError
class FakeResponse:
def __init__(self, payload=None, status=200, text=None):
self._payload = payload
self.status_code = status
self.text = text if text is not None else ("" if payload is None else "json")
@property
def ok(self):
return self.status_code < 400
def json(self):
if self._payload is None:
raise ValueError("no json")
return self._payload
def _client(timeout=30):
return RestClient(RestConfig(base_url="https://h/dwh/", api_key="dwh_k", timeout=timeout))
def _capture(monkeypatch, response):
calls = []
def fake_post(url, json=None, headers=None, timeout=None, verify=None):
calls.append({"url": url, "json": json, "headers": headers,
"timeout": timeout, "verify": verify})
return response
monkeypatch.setattr("nsp.rest.client.requests.post", fake_post)
return calls
def test_run_query_payload_and_rows(monkeypatch):
calls = _capture(monkeypatch, FakeResponse([{"x": 1}]))
rows = _client().run_query("SELECT 1 AS x")
assert rows == [{"x": 1}]
c = calls[0]
assert c["url"] == "https://h/dwh/rpc/run_query"
assert c["json"] == {"query_text": "SELECT 1 AS x"}
assert c["headers"]["X-API-Key"] == "dwh_k"
assert c["timeout"] == 30
def test_explain_query_maps_lines(monkeypatch):
plan = [
{"line": "Aggregate (cost=3747.24..3747.25 rows=1 width=8)"},
{"line": " -> Seq Scan on dim_patient (cost=0.00..3712.97 rows=13706 width=0)"},
]
_capture(monkeypatch, FakeResponse(plan))
lines = _client().explain_query("SELECT count(*) FROM dim_patient")
assert lines == [
"Aggregate (cost=3747.24..3747.25 rows=1 width=8)",
" -> Seq Scan on dim_patient (cost=0.00..3712.97 rows=13706 width=0)",
]
def test_ping(monkeypatch):
calls = _capture(monkeypatch, FakeResponse({"db_connected": True, "role": "postgres"}))
out = _client().ping()
assert out["db_connected"] is True
assert calls[0]["url"] == "https://h/dwh/rpc/ping"
assert calls[0]["json"] == {}
def test_top_values_payload(monkeypatch):
calls = _capture(monkeypatch, FakeResponse([{"value": "MI", "count": 13737}]))
out = _client().top_values("datawarehouse", "dim_patient", "provincia", 1000)
assert out == [{"value": "MI", "count": 13737}]
assert calls[0]["json"] == {
"schema_name": "datawarehouse",
"table_name": "dim_patient",
"column_name": "provincia",
"max_values": 1000,
}
def test_validate_select_ok_and_write(monkeypatch):
_capture(monkeypatch, FakeResponse(status=204))
assert _client().validate_select("SELECT 1") is True
_capture(monkeypatch, FakeResponse({"message": "Only SELECT / WITH statements are allowed"}, status=400))
assert _client().validate_select("DELETE FROM t") is False
def test_http_error_surfaces_message(monkeypatch):
_capture(monkeypatch, FakeResponse({"message": "boom"}, status=400))
with pytest.raises(RestError, match="boom"):
_client().run_query("SELECT bad")
def test_network_error_actionable(monkeypatch):
def boom(*a, **k):
raise requests.ConnectionError("refused")
monkeypatch.setattr("nsp.rest.client.requests.post", boom)
with pytest.raises(RestError, match="raggiungibile"):
_client().ping()
def test_base_url_without_trailing_slash(monkeypatch):
calls = _capture(monkeypatch, FakeResponse({"db_connected": True}))
RestClient(RestConfig(base_url="https://h/dwh", api_key="k")).ping()
assert calls[0]["url"] == "https://h/dwh/rpc/ping"