Ports the leaf data-layer modules and validates them: - mschema/ (models, eligibility, merge, render), db/ (connection, sampling, introspect, fetch_ca), rest/client.py -- renamed psdwp3->nsp, verbatim. - L0 (testcontainers, real Postgres): db connection read-only enforcement (psd_ro cannot CREATE/INSERT), introspect against a known schema (tables, columns, types, comments, FKs, enum, composite PK), sampling most-frequent values + truncation reporting. 15 tests, ~4s. - L1 (fake data): rest/client RPC contract (mocked transport -- X-API-Key header, payloads, base_url slash handling, HTTP/network error surfacing), mschema/render 3 formats (markdown, mschema-text, schema-dict) + eligibility rules (wide_text excluded, short_text/numeric/enum/temporal/ boolean eligible, annotation override wins). 25 tests. pyproject registers l0/l2 markers + addopts '-m not l2' (L2 opt-in). Deferred to their dependency-porting tasks: test_rrf.py (search needs vectorstore, B3) and the 11 CLI contract tests (need _guards/session, wired when each command lands). 'Not assumed reliable' now has real teeth for the data layer; CLI/search contracts follow.
142 lines
5.6 KiB
Python
142 lines
5.6 KiB
Python
from typing import Any
|
|
|
|
from nsp.mschema.eligibility import effective_eligibility
|
|
from nsp.mschema.models import Annotations, ColumnAnnotation, PhysicalSchema
|
|
|
|
MAX_EXAMPLES_IN_PROMPT = 5
|
|
|
|
|
|
def _ann_col(annotations: Annotations, table: str, column: str) -> ColumnAnnotation | None:
|
|
ann = annotations.tables.get(table)
|
|
if ann is None:
|
|
return None
|
|
return ann.columns.get(column)
|
|
|
|
|
|
def _table_description(physical: PhysicalSchema, annotations: Annotations, table: str) -> str:
|
|
ann = annotations.tables.get(table)
|
|
if ann and ann.description:
|
|
return ann.description
|
|
return physical.tables[table].comment
|
|
|
|
|
|
def _column_description(
|
|
physical: PhysicalSchema, annotations: Annotations, table: str, column: str
|
|
) -> str:
|
|
ann = annotations.tables.get(table)
|
|
if ann and column in ann.columns and ann.columns[column].description:
|
|
return ann.columns[column].description
|
|
return physical.tables[table].columns[column].comment
|
|
|
|
|
|
def to_mschema_text(
|
|
physical: PhysicalSchema,
|
|
annotations: Annotations | None = None,
|
|
tables: list[str] | None = None,
|
|
) -> str:
|
|
"""Serializzazione testuale in stile ThothAI (【Schema】/【Foreign keys】)."""
|
|
annotations = annotations or Annotations()
|
|
selected = [t for t in physical.tables if tables is None or t in tables]
|
|
lines: list[str] = ["【Schema】"]
|
|
fk_lines: list[str] = []
|
|
for table_name in selected:
|
|
table = physical.tables[table_name]
|
|
desc = _table_description(physical, annotations, table_name)
|
|
if desc:
|
|
lines.append(f"-- {desc}")
|
|
lines.append(f"CREATE TABLE {table_name} (")
|
|
for column_name, column in table.columns.items():
|
|
if not effective_eligibility(column, _ann_col(annotations, table_name, column_name))[0]:
|
|
continue
|
|
line = f" {column_name} {column.type.upper()}"
|
|
if column.pk:
|
|
line += " -- PRIMARY KEY"
|
|
lines.append(line)
|
|
cdesc = _column_description(physical, annotations, table_name, column_name)
|
|
if cdesc:
|
|
lines.append(f" -- {cdesc}")
|
|
if column.examples:
|
|
shown = ", ".join(column.examples[:MAX_EXAMPLES_IN_PROMPT])
|
|
lines.append(f" -- Examples: {shown}")
|
|
lines.append(");")
|
|
for fk in table.foreign_keys:
|
|
for src, dst in zip(fk.columns, fk.ref_columns):
|
|
fk_lines.append(f"{table_name}.{src}={fk.ref_table}.{dst}")
|
|
lines.extend(["", "【Foreign keys】", *fk_lines])
|
|
return "\n".join(lines)
|
|
|
|
|
|
def to_schema_dict(
|
|
physical: PhysicalSchema, annotations: Annotations | None = None
|
|
) -> dict[str, Any]:
|
|
"""Vista compatibile con le logiche AV-SQL (schema_dict)."""
|
|
annotations = annotations or Annotations()
|
|
out: dict[str, Any] = {}
|
|
for table_name, table in physical.tables.items():
|
|
cols = [
|
|
c
|
|
for c in table.columns
|
|
if effective_eligibility(table.columns[c], _ann_col(annotations, table_name, c))[0]
|
|
]
|
|
out[table_name] = {
|
|
"columns_name": cols,
|
|
"columns_type": [table.columns[c].type for c in cols],
|
|
"columns_description": [
|
|
_column_description(physical, annotations, table_name, c) for c in cols
|
|
],
|
|
"example_values": [table.columns[c].examples for c in cols],
|
|
"table_to_tablefullname": f"{physical.db_schema}.{table_name}",
|
|
"primary_keys": [c for c in cols if table.columns[c].pk],
|
|
"foreign_keys": [
|
|
{"columns": fk.columns, "ref_table": fk.ref_table, "ref_columns": fk.ref_columns}
|
|
for fk in table.foreign_keys
|
|
],
|
|
}
|
|
return out
|
|
|
|
|
|
def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None) -> str:
|
|
"""Report leggibile per il reviewer."""
|
|
annotations = annotations or Annotations()
|
|
lines = [
|
|
f"# Schema {physical.db_schema} ({physical.database})",
|
|
"",
|
|
f"Introspezione: {physical.introspected_at.isoformat()} — "
|
|
f"{len(physical.tables)} tabelle",
|
|
]
|
|
for table_name, table in physical.tables.items():
|
|
lines += ["", f"## {table_name}", ""]
|
|
desc = _table_description(physical, annotations, table_name)
|
|
if desc:
|
|
lines += [desc, ""]
|
|
lines += [
|
|
f"Righe (stima): {table.row_count}",
|
|
"",
|
|
"| Colonna | Tipo | Null | PK | Descrizione | Esempi |",
|
|
"|---|---|---|---|---|---|",
|
|
]
|
|
for column_name, column in table.columns.items():
|
|
eligible, reason = effective_eligibility(
|
|
column, _ann_col(annotations, table_name, column_name)
|
|
)
|
|
cdesc = _column_description(physical, annotations, table_name, column_name)
|
|
if not eligible:
|
|
lines.append(
|
|
f"| ~~{column_name}~~ | {column.type} | "
|
|
f"{'sì' if column.nullable else 'no'} | "
|
|
f"{'sì' if column.pk else ''} | {cdesc} | _ignorata: {reason}_ |"
|
|
)
|
|
continue
|
|
examples = ", ".join(column.examples[:3])
|
|
lines.append(
|
|
f"| {column_name} | {column.type} | {'sì' if column.nullable else 'no'} "
|
|
f"| {'sì' if column.pk else ''} | {cdesc} | {examples} |"
|
|
)
|
|
if table.foreign_keys:
|
|
lines += ["", "Foreign keys:"]
|
|
for fk in table.foreign_keys:
|
|
lines.append(
|
|
f"- ({', '.join(fk.columns)}) → {fk.ref_table} ({', '.join(fk.ref_columns)})"
|
|
)
|
|
return "\n".join(lines)
|