test(harness): L0 testcontainers + L1 contract tests for ported db/mschema/rest (A9, spec §1)

Ports the leaf data-layer modules and validates them:
- mschema/ (models, eligibility, merge, render), db/ (connection, sampling,
  introspect, fetch_ca), rest/client.py -- renamed psdwp3->nsp, verbatim.
- L0 (testcontainers, real Postgres): db connection read-only enforcement
  (psd_ro cannot CREATE/INSERT), introspect against a known schema (tables,
  columns, types, comments, FKs, enum, composite PK), sampling most-frequent
  values + truncation reporting. 15 tests, ~4s.
- L1 (fake data): rest/client RPC contract (mocked transport -- X-API-Key
  header, payloads, base_url slash handling, HTTP/network error surfacing),
  mschema/render 3 formats (markdown, mschema-text, schema-dict) +
  eligibility rules (wide_text excluded, short_text/numeric/enum/temporal/
  boolean eligible, annotation override wins). 25 tests.

pyproject registers l0/l2 markers + addopts '-m not l2' (L2 opt-in).

Deferred to their dependency-porting tasks: test_rrf.py (search needs
vectorstore, B3) and the 11 CLI contract tests (need _guards/session, wired
when each command lands). 'Not assumed reliable' now has real teeth for the
data layer; CLI/search contracts follow.
This commit is contained in:
2026-06-26 22:53:08 +02:00
parent 5f24bd1adc
commit eb3bde90e2
22 changed files with 1552 additions and 0 deletions
View File
+93
View File
@@ -0,0 +1,93 @@
"""Classificazione di column eligibility (principio trasversale PsdWp3).
Vedi docs/superpowers/specs/2026-06-13-nsp-column-eligibility-principle.md.
Il testo ampio (lettere di dimissione, note, anamnesi) è ignorato ovunque; i dati
provengono solo da numerici, enum, temporali, booleani e testo breve.
"""
import re
from nsp.config import EligibilityConfig
from nsp.db.sampling import is_text_type
from nsp.mschema.models import ColumnAnnotation, ColumnPhysical, PhysicalSchema
_NUMERIC_PREFIXES = (
"smallint", "integer", "bigint", "numeric", "decimal", "real", "double", "money",
)
_TEMPORAL_PREFIXES = ("date", "time", "timestamp", "interval")
_LEN_RE = re.compile(r"\((\d+)\)")
def _declared_len(pg_type: str) -> int | None:
m = _LEN_RE.search(pg_type)
return int(m.group(1)) if m else None
def classify_column(
pg_type: str,
is_enum: bool,
sampled_avg: float | None,
sampled_max: int | None,
cfg: EligibilityConfig,
) -> tuple[bool, str]:
"""Classifica una colonna come (eligible, reason). Funzione pura, senza I/O."""
t = pg_type.strip().lower()
if t.endswith("[]"):
return False, "wide_text"
if is_enum:
return True, "enum"
if t.startswith(_NUMERIC_PREFIXES):
return True, "numeric"
if t.startswith("boolean"):
return True, "boolean"
if t.startswith(_TEMPORAL_PREFIXES):
return True, "temporal"
if t.startswith("uuid"):
return True, "code"
if is_text_type(t):
declared = _declared_len(t)
if declared is not None and declared <= cfg.max_declared_len:
return True, "short_text"
# bound grande o text/varchar non vincolato: decide il dato campionato
if sampled_avg is None or sampled_max is None:
return False, "wide_text"
if sampled_avg <= cfg.max_avg_length and sampled_max <= cfg.max_sampled_len:
return True, "short_text"
return False, "wide_text"
return False, "wide_text"
def _sampled_stats(examples: list[str]) -> tuple[float | None, int | None]:
if not examples:
return None, None
lengths = [len(v) for v in examples]
return sum(lengths) / len(lengths), max(lengths)
def classify_all(physical: PhysicalSchema, cfg: EligibilityConfig) -> None:
"""Assegna eligible/eligibility_reason a ogni colonna (in-place) e azzera gli
examples delle colonne ignored. Da chiamare DOPO add_examples."""
ignore_by_name = {name.lower() for name in cfg.ignore_columns}
for table in physical.tables.values():
for column_name, column in table.columns.items():
if column_name.lower() in ignore_by_name:
# colonna di servizio (ETL/audit): ignorata a prescindere dal tipo
column.eligible = False
column.eligibility_reason = "ignored_by_name"
column.examples = []
continue
avg, mx = _sampled_stats(column.examples)
eligible, reason = classify_column(column.type, column.is_enum, avg, mx, cfg)
column.eligible = eligible
column.eligibility_reason = reason
if not eligible:
column.examples = []
def effective_eligibility(
column: ColumnPhysical, annotation: ColumnAnnotation | None
) -> tuple[bool, str]:
"""Eligibilità effettiva: l'override in annotations.yaml vince sul fisico."""
if annotation is not None and annotation.eligible is not None:
return annotation.eligible, "override"
return column.eligible, column.eligibility_reason
+15
View File
@@ -0,0 +1,15 @@
from nsp.mschema.models import Annotations, PhysicalSchema
def find_orphans(physical: PhysicalSchema, annotations: Annotations) -> list[str]:
"""Annotazioni che puntano a oggetti spariti dal fisico. Non le rimuove mai."""
orphans: list[str] = []
for table_name, table_ann in annotations.tables.items():
table = physical.tables.get(table_name)
if table is None:
orphans.append(table_name)
continue
for column_name in table_ann.columns:
if column_name not in table.columns:
orphans.append(f"{table_name}.{column_name}")
return orphans
+90
View File
@@ -0,0 +1,90 @@
from datetime import datetime
from pathlib import Path
from typing import Self
import yaml
from pydantic import BaseModel, Field
class _YamlModel(BaseModel):
def to_yaml(self, path: Path) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
data = self.model_dump(by_alias=True, mode="json", exclude_defaults=False)
path.write_text(
yaml.safe_dump(data, sort_keys=False, allow_unicode=True, width=120)
)
@classmethod
def from_yaml(cls, path: Path) -> Self:
raw = yaml.safe_load(path.read_text())
return cls.model_validate(raw)
class ColumnPhysical(BaseModel):
type: str
nullable: bool = True
pk: bool = False
default: str | None = None
comment: str = ""
examples: list[str] = []
is_enum: bool = False
eligible: bool = True
eligibility_reason: str = ""
class ForeignKey(BaseModel):
columns: list[str]
ref_table: str
ref_columns: list[str]
name: str = ""
class Index(BaseModel):
name: str
columns: list[str]
unique: bool = False
primary: bool = False
type: str = "btree"
class TablePhysical(BaseModel):
comment: str = ""
row_count: int = 0 # stima da pg_class.reltuples
columns: dict[str, ColumnPhysical] = {}
foreign_keys: list[ForeignKey] = []
indexes: list[Index] = []
class PhysicalSchema(_YamlModel):
database: str
db_schema: str = Field(alias="schema")
introspected_at: datetime
tables: dict[str, TablePhysical] = {}
model_config = {"populate_by_name": True}
class ColumnAnnotation(BaseModel):
description: str = ""
synonyms: list[str] = []
concepts: list[str] = []
evidence: list[str] = []
notes: str = ""
eligible: bool | None = None
class TableAnnotation(BaseModel):
description: str = ""
concepts: list[str] = []
notes: str = ""
columns: dict[str, ColumnAnnotation] = {}
class Annotations(_YamlModel):
tables: dict[str, TableAnnotation] = {}
@classmethod
def from_yaml(cls, path: Path) -> "Annotations":
if not path.exists():
return cls()
return super().from_yaml(path)
+141
View File
@@ -0,0 +1,141 @@
from typing import Any
from nsp.mschema.eligibility import effective_eligibility
from nsp.mschema.models import Annotations, ColumnAnnotation, PhysicalSchema
MAX_EXAMPLES_IN_PROMPT = 5
def _ann_col(annotations: Annotations, table: str, column: str) -> ColumnAnnotation | None:
ann = annotations.tables.get(table)
if ann is None:
return None
return ann.columns.get(column)
def _table_description(physical: PhysicalSchema, annotations: Annotations, table: str) -> str:
ann = annotations.tables.get(table)
if ann and ann.description:
return ann.description
return physical.tables[table].comment
def _column_description(
physical: PhysicalSchema, annotations: Annotations, table: str, column: str
) -> str:
ann = annotations.tables.get(table)
if ann and column in ann.columns and ann.columns[column].description:
return ann.columns[column].description
return physical.tables[table].columns[column].comment
def to_mschema_text(
physical: PhysicalSchema,
annotations: Annotations | None = None,
tables: list[str] | None = None,
) -> str:
"""Serializzazione testuale in stile ThothAI (【Schema】/【Foreign keys】)."""
annotations = annotations or Annotations()
selected = [t for t in physical.tables if tables is None or t in tables]
lines: list[str] = ["【Schema】"]
fk_lines: list[str] = []
for table_name in selected:
table = physical.tables[table_name]
desc = _table_description(physical, annotations, table_name)
if desc:
lines.append(f"-- {desc}")
lines.append(f"CREATE TABLE {table_name} (")
for column_name, column in table.columns.items():
if not effective_eligibility(column, _ann_col(annotations, table_name, column_name))[0]:
continue
line = f" {column_name} {column.type.upper()}"
if column.pk:
line += " -- PRIMARY KEY"
lines.append(line)
cdesc = _column_description(physical, annotations, table_name, column_name)
if cdesc:
lines.append(f" -- {cdesc}")
if column.examples:
shown = ", ".join(column.examples[:MAX_EXAMPLES_IN_PROMPT])
lines.append(f" -- Examples: {shown}")
lines.append(");")
for fk in table.foreign_keys:
for src, dst in zip(fk.columns, fk.ref_columns):
fk_lines.append(f"{table_name}.{src}={fk.ref_table}.{dst}")
lines.extend(["", "【Foreign keys】", *fk_lines])
return "\n".join(lines)
def to_schema_dict(
physical: PhysicalSchema, annotations: Annotations | None = None
) -> dict[str, Any]:
"""Vista compatibile con le logiche AV-SQL (schema_dict)."""
annotations = annotations or Annotations()
out: dict[str, Any] = {}
for table_name, table in physical.tables.items():
cols = [
c
for c in table.columns
if effective_eligibility(table.columns[c], _ann_col(annotations, table_name, c))[0]
]
out[table_name] = {
"columns_name": cols,
"columns_type": [table.columns[c].type for c in cols],
"columns_description": [
_column_description(physical, annotations, table_name, c) for c in cols
],
"example_values": [table.columns[c].examples for c in cols],
"table_to_tablefullname": f"{physical.db_schema}.{table_name}",
"primary_keys": [c for c in cols if table.columns[c].pk],
"foreign_keys": [
{"columns": fk.columns, "ref_table": fk.ref_table, "ref_columns": fk.ref_columns}
for fk in table.foreign_keys
],
}
return out
def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None) -> str:
"""Report leggibile per il reviewer."""
annotations = annotations or Annotations()
lines = [
f"# Schema {physical.db_schema} ({physical.database})",
"",
f"Introspezione: {physical.introspected_at.isoformat()} — "
f"{len(physical.tables)} tabelle",
]
for table_name, table in physical.tables.items():
lines += ["", f"## {table_name}", ""]
desc = _table_description(physical, annotations, table_name)
if desc:
lines += [desc, ""]
lines += [
f"Righe (stima): {table.row_count}",
"",
"| Colonna | Tipo | Null | PK | Descrizione | Esempi |",
"|---|---|---|---|---|---|",
]
for column_name, column in table.columns.items():
eligible, reason = effective_eligibility(
column, _ann_col(annotations, table_name, column_name)
)
cdesc = _column_description(physical, annotations, table_name, column_name)
if not eligible:
lines.append(
f"| ~~{column_name}~~ | {column.type} | "
f"{'sì' if column.nullable else 'no'} | "
f"{'sì' if column.pk else ''} | {cdesc} | _ignorata: {reason}_ |"
)
continue
examples = ", ".join(column.examples[:3])
lines.append(
f"| {column_name} | {column.type} | {'sì' if column.nullable else 'no'} "
f"| {'sì' if column.pk else ''} | {cdesc} | {examples} |"
)
if table.foreign_keys:
lines += ["", "Foreign keys:"]
for fk in table.foreign_keys:
lines.append(
f"- ({', '.join(fk.columns)}) → {fk.ref_table} ({', '.join(fk.ref_columns)})"
)
return "\n".join(lines)