test(harness): L0 testcontainers + L1 contract tests for ported db/mschema/rest (A9, spec §1)
Ports the leaf data-layer modules and validates them: - mschema/ (models, eligibility, merge, render), db/ (connection, sampling, introspect, fetch_ca), rest/client.py -- renamed psdwp3->nsp, verbatim. - L0 (testcontainers, real Postgres): db connection read-only enforcement (psd_ro cannot CREATE/INSERT), introspect against a known schema (tables, columns, types, comments, FKs, enum, composite PK), sampling most-frequent values + truncation reporting. 15 tests, ~4s. - L1 (fake data): rest/client RPC contract (mocked transport -- X-API-Key header, payloads, base_url slash handling, HTTP/network error surfacing), mschema/render 3 formats (markdown, mschema-text, schema-dict) + eligibility rules (wide_text excluded, short_text/numeric/enum/temporal/ boolean eligible, annotation override wins). 25 tests. pyproject registers l0/l2 markers + addopts '-m not l2' (L2 opt-in). Deferred to their dependency-porting tasks: test_rrf.py (search needs vectorstore, B3) and the 11 CLI contract tests (need _guards/session, wired when each command lands). 'Not assumed reliable' now has real teeth for the data layer; CLI/search contracts follow.
This commit is contained in:
@@ -0,0 +1,37 @@
|
||||
from sqlalchemy import Engine, create_engine, text
|
||||
|
||||
from nsp.config import DatabaseConfig
|
||||
|
||||
|
||||
def make_engine(cfg: DatabaseConfig) -> Engine:
|
||||
url = (
|
||||
f"postgresql+psycopg2://{cfg.user}:{cfg.password}"
|
||||
f"@{cfg.host}:{cfg.port}/{cfg.database}"
|
||||
)
|
||||
return create_engine(url, echo=False)
|
||||
|
||||
|
||||
def ping(engine: Engine) -> None:
|
||||
with engine.connect() as conn:
|
||||
conn.execute(text("SELECT 1"))
|
||||
|
||||
|
||||
def writable_tables(engine: Engine, schema: str) -> list[str]:
|
||||
"""Tabelle dello schema su cui l'utente corrente ha privilegi di scrittura."""
|
||||
q = text("""
|
||||
SELECT c.relname
|
||||
FROM pg_class c
|
||||
JOIN pg_namespace n ON n.oid = c.relnamespace
|
||||
WHERE n.nspname = :schema
|
||||
AND c.relkind IN ('r', 'p')
|
||||
AND has_table_privilege(current_user, c.oid, 'INSERT, UPDATE, DELETE')
|
||||
ORDER BY c.relname
|
||||
""")
|
||||
with engine.connect() as conn:
|
||||
return [row[0] for row in conn.execute(q, {"schema": schema})]
|
||||
|
||||
|
||||
def can_create_in_schema(engine: Engine, schema: str) -> bool:
|
||||
q = text("SELECT has_schema_privilege(current_user, :schema, 'CREATE')")
|
||||
with engine.connect() as conn:
|
||||
return bool(conn.execute(q, {"schema": schema}).scalar())
|
||||
@@ -0,0 +1,114 @@
|
||||
"""Recupero della catena di certificati presentata da un endpoint HTTPS.
|
||||
|
||||
Serve al setup di una postazione *workstation* dietro una CA interna: scarica la
|
||||
catena TLS del server REST e la salva in un bundle PEM da puntare con `PSD_SSL_CA`
|
||||
(consumato da `requests` via `verify=`). NON installa nulla nel trust store dell'OS.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import _ssl
|
||||
import socket
|
||||
import ssl
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
# Encoding atteso da Certificate.public_bytes() per la catena TLS non verificata.
|
||||
_PEM_ENCODING = getattr(_ssl, "ENCODING_PEM", 1)
|
||||
|
||||
|
||||
class CaFetchError(Exception):
|
||||
"""Errore azionabile durante il recupero della catena CA."""
|
||||
|
||||
|
||||
def parse_host_port(base_url: str) -> tuple[str, int]:
|
||||
"""Estrae (host, port) da un URL REST https. Porta di default 443."""
|
||||
parts = urlsplit(base_url)
|
||||
if parts.scheme != "https":
|
||||
raise CaFetchError(
|
||||
f"URL non https: {base_url!r}. Il recupero CA ha senso solo su HTTPS."
|
||||
)
|
||||
if not parts.hostname:
|
||||
raise CaFetchError(f"Host mancante nell'URL: {base_url!r}.")
|
||||
return parts.hostname, parts.port or 443
|
||||
|
||||
|
||||
def fetch_chain_pem(host: str, port: int = 443, timeout: int = 30) -> list[str]:
|
||||
"""Restituisce la catena di certificati presentata da host:port come lista di PEM.
|
||||
|
||||
L'handshake è volutamente *non verificato* (CERT_NONE): stiamo recuperando la catena
|
||||
per poter poi *stabilire* la fiducia, non per fidarci adesso. La verifica vera avviene
|
||||
in seguito quando `PSD_SSL_CA` punta al bundle salvato (es. `nsp db ping`).
|
||||
"""
|
||||
ctx = ssl.SSLContext(ssl.PROTOCOL_TLS_CLIENT)
|
||||
ctx.check_hostname = False
|
||||
ctx.verify_mode = ssl.CERT_NONE
|
||||
try:
|
||||
with socket.create_connection((host, port), timeout=timeout) as sock:
|
||||
with ctx.wrap_socket(sock, server_hostname=host) as tls:
|
||||
certs = _unverified_chain(tls)
|
||||
except (OSError, ssl.SSLError) as e:
|
||||
raise CaFetchError(
|
||||
f"Impossibile connettersi a {host}:{port} per recuperare i certificati: {e}"
|
||||
) from e
|
||||
|
||||
if not certs:
|
||||
raise CaFetchError(
|
||||
f"Nessun certificato presentato da {host}:{port}. "
|
||||
f"In alternativa, estrai la catena a mano con: "
|
||||
f"openssl s_client -showcerts -connect {host}:{port} -servername {host}"
|
||||
)
|
||||
return [_to_pem(c) for c in certs]
|
||||
|
||||
|
||||
def _unverified_chain(tls: ssl.SSLSocket) -> list:
|
||||
"""Catena presentata dal server. Metodo pubblico su Python >= 3.13, API interna su 3.12."""
|
||||
public = getattr(tls, "get_unverified_chain", None)
|
||||
if public is not None:
|
||||
return list(public() or [])
|
||||
sslobj = getattr(tls, "_sslobj", None)
|
||||
getter = getattr(sslobj, "get_unverified_chain", None) if sslobj is not None else None
|
||||
if getter is None:
|
||||
raise CaFetchError(
|
||||
"Questa versione di Python non espone la catena TLS. "
|
||||
"Estrai la catena a mano con `openssl s_client -showcerts`."
|
||||
)
|
||||
return list(getter() or [])
|
||||
|
||||
|
||||
def describe_pem(pem: str) -> str:
|
||||
"""Riassunto leggibile (subject / issuer) di un certificato PEM, best-effort.
|
||||
|
||||
Serve a far riconoscere all'utente la CA interna attesa (verifica out-of-band).
|
||||
Restituisce "" se il certificato non è decodificabile.
|
||||
"""
|
||||
try:
|
||||
with tempfile.NamedTemporaryFile("w", suffix=".pem", delete=False) as fh:
|
||||
fh.write(pem)
|
||||
tmp = fh.name
|
||||
try:
|
||||
info = _ssl._test_decode_cert(tmp)
|
||||
finally:
|
||||
Path(tmp).unlink(missing_ok=True)
|
||||
except (OSError, ssl.SSLError, ValueError):
|
||||
return ""
|
||||
subject = _name(info.get("subject"))
|
||||
issuer = _name(info.get("issuer"))
|
||||
return f"subject={subject} issuer={issuer}"
|
||||
|
||||
|
||||
def _name(rdns) -> str:
|
||||
"""Estrae il CN (o l'intero RDN) da una struttura subject/issuer di _test_decode_cert."""
|
||||
if not rdns:
|
||||
return "?"
|
||||
parts = {k: v for rdn in rdns for (k, v) in rdn}
|
||||
return parts.get("commonName") or ", ".join(f"{k}={v}" for k, v in parts.items())
|
||||
|
||||
|
||||
def _to_pem(cert) -> str:
|
||||
"""Converte un certificato (_ssl.Certificate o DER bytes) in PEM."""
|
||||
if isinstance(cert, (bytes, bytearray)):
|
||||
return ssl.DER_cert_to_PEM_cert(bytes(cert))
|
||||
pem = cert.public_bytes(_PEM_ENCODING)
|
||||
return pem if isinstance(pem, str) else pem.decode("ascii")
|
||||
@@ -0,0 +1,202 @@
|
||||
from datetime import UTC, datetime
|
||||
|
||||
from sqlalchemy import Engine, text
|
||||
|
||||
from nsp.mschema.models import (
|
||||
ColumnPhysical,
|
||||
ForeignKey,
|
||||
Index,
|
||||
PhysicalSchema,
|
||||
TablePhysical,
|
||||
)
|
||||
|
||||
# Query adattate da thoth_sqldb2 (Apache 2.0) — vedi src/nsp/vendor/VENDORED.md.
|
||||
|
||||
_TABLES_Q = text("""
|
||||
SELECT c.relname AS table_name,
|
||||
COALESCE(d.description, '') AS comment,
|
||||
GREATEST(c.reltuples::bigint, 0) AS row_count
|
||||
FROM pg_class c
|
||||
JOIN pg_namespace n ON n.oid = c.relnamespace
|
||||
LEFT JOIN pg_description d ON d.objoid = c.oid AND d.objsubid = 0
|
||||
WHERE c.relkind IN ('r', 'p') AND n.nspname = :schema
|
||||
ORDER BY c.relname
|
||||
""")
|
||||
|
||||
_COLUMNS_Q = text("""
|
||||
SELECT a.attname AS column_name,
|
||||
format_type(a.atttypid, a.atttypmod) AS data_type,
|
||||
(NOT a.attnotnull) AS is_nullable,
|
||||
pg_get_expr(d.adbin, d.adrelid) AS column_default,
|
||||
COALESCE(pgd.description, '') AS comment,
|
||||
(ty.typtype = 'e') AS is_enum,
|
||||
EXISTS (
|
||||
SELECT 1 FROM pg_index i
|
||||
WHERE i.indrelid = c.oid AND i.indisprimary AND a.attnum = ANY (i.indkey)
|
||||
) AS is_pk
|
||||
FROM pg_class c
|
||||
JOIN pg_namespace n ON n.oid = c.relnamespace
|
||||
JOIN pg_attribute a ON a.attrelid = c.oid
|
||||
JOIN pg_type ty ON ty.oid = a.atttypid
|
||||
LEFT JOIN pg_attrdef d ON d.adrelid = c.oid AND d.adnum = a.attnum
|
||||
LEFT JOIN pg_description pgd ON pgd.objoid = c.oid AND pgd.objsubid = a.attnum
|
||||
WHERE c.relname = :table_name AND n.nspname = :schema
|
||||
AND a.attnum > 0 AND NOT a.attisdropped
|
||||
ORDER BY a.attnum
|
||||
""")
|
||||
|
||||
_FOREIGN_KEYS_Q = text("""
|
||||
SELECT con.conname AS constraint_name,
|
||||
rel.relname AS source_table,
|
||||
a.attname AS source_column,
|
||||
frel.relname AS target_table,
|
||||
fa.attname AS target_column,
|
||||
src.ord
|
||||
FROM pg_constraint con
|
||||
JOIN pg_class rel ON rel.oid = con.conrelid
|
||||
JOIN pg_namespace ns ON ns.oid = rel.relnamespace
|
||||
JOIN pg_class frel ON frel.oid = con.confrelid
|
||||
JOIN unnest(con.conkey) WITH ORDINALITY AS src(attnum, ord) ON true
|
||||
JOIN pg_attribute a ON a.attrelid = con.conrelid AND a.attnum = src.attnum
|
||||
JOIN unnest(con.confkey) WITH ORDINALITY AS dst(attnum, ord) ON dst.ord = src.ord
|
||||
JOIN pg_attribute fa ON fa.attrelid = con.confrelid AND fa.attnum = dst.attnum
|
||||
WHERE con.contype = 'f' AND ns.nspname = :schema
|
||||
ORDER BY rel.relname, con.conname, src.ord
|
||||
""")
|
||||
|
||||
_INDEXES_Q = text("""
|
||||
SELECT i.relname AS index_name,
|
||||
t.relname AS table_name,
|
||||
ix.indisunique AS is_unique,
|
||||
ix.indisprimary AS is_primary,
|
||||
am.amname AS index_type,
|
||||
array_agg(a.attname ORDER BY a.attnum) AS columns
|
||||
FROM pg_index ix
|
||||
JOIN pg_class i ON i.oid = ix.indexrelid
|
||||
JOIN pg_class t ON t.oid = ix.indrelid
|
||||
JOIN pg_namespace n ON n.oid = t.relnamespace
|
||||
JOIN pg_am am ON am.oid = i.relam
|
||||
JOIN pg_attribute a ON a.attrelid = t.oid AND a.attnum = ANY (ix.indkey)
|
||||
WHERE n.nspname = :schema
|
||||
GROUP BY i.relname, t.relname, ix.indisunique, ix.indisprimary, am.amname
|
||||
ORDER BY t.relname, i.relname
|
||||
""")
|
||||
|
||||
|
||||
class IntrospectionError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def introspect(engine: Engine, database: str, schema: str) -> PhysicalSchema:
|
||||
with engine.connect() as conn:
|
||||
exists = conn.execute(
|
||||
text("SELECT 1 FROM pg_namespace WHERE nspname = :schema"), {"schema": schema}
|
||||
).scalar()
|
||||
if not exists:
|
||||
raise IntrospectionError(f"Schema inesistente: {schema}")
|
||||
|
||||
tables: dict[str, TablePhysical] = {}
|
||||
for trow in conn.execute(_TABLES_Q, {"schema": schema}):
|
||||
columns: dict[str, ColumnPhysical] = {}
|
||||
for crow in conn.execute(
|
||||
_COLUMNS_Q, {"table_name": trow.table_name, "schema": schema}
|
||||
):
|
||||
columns[crow.column_name] = ColumnPhysical(
|
||||
type=crow.data_type,
|
||||
nullable=bool(crow.is_nullable),
|
||||
pk=bool(crow.is_pk),
|
||||
default=crow.column_default,
|
||||
comment=crow.comment,
|
||||
is_enum=bool(crow.is_enum),
|
||||
)
|
||||
tables[trow.table_name] = TablePhysical(
|
||||
comment=trow.comment, row_count=trow.row_count, columns=columns
|
||||
)
|
||||
|
||||
# FK raggruppate per (tabella, constraint), ordinate per posizione
|
||||
grouped: dict[tuple[str, str], ForeignKey] = {}
|
||||
for row in conn.execute(_FOREIGN_KEYS_Q, {"schema": schema}):
|
||||
key = (row.source_table, row.constraint_name)
|
||||
fk = grouped.setdefault(
|
||||
key,
|
||||
ForeignKey(
|
||||
columns=[], ref_table=row.target_table, ref_columns=[],
|
||||
name=row.constraint_name,
|
||||
),
|
||||
)
|
||||
fk.columns.append(row.source_column)
|
||||
fk.ref_columns.append(row.target_column)
|
||||
for (table_name, _), fk in grouped.items():
|
||||
if table_name in tables:
|
||||
tables[table_name].foreign_keys.append(fk)
|
||||
|
||||
for row in conn.execute(_INDEXES_Q, {"schema": schema}):
|
||||
if row.table_name in tables:
|
||||
tables[row.table_name].indexes.append(
|
||||
Index(
|
||||
name=row.index_name,
|
||||
columns=list(row.columns),
|
||||
unique=bool(row.is_unique),
|
||||
primary=bool(row.is_primary),
|
||||
type=row.index_type,
|
||||
)
|
||||
)
|
||||
|
||||
return PhysicalSchema(
|
||||
database=database,
|
||||
schema=schema,
|
||||
introspected_at=datetime.now(UTC),
|
||||
tables=tables,
|
||||
)
|
||||
|
||||
|
||||
def introspect_rest(client, database: str, schema: str) -> PhysicalSchema:
|
||||
"""Introspezione via REST (rpc `list_tables`/`table_columns`/`table_comments`/
|
||||
`table_foreign_keys`). Limiti rispetto al diretto: niente indici (nessun rpc) e
|
||||
`is_enum` non disponibile (default False)."""
|
||||
tables: dict[str, TablePhysical] = {}
|
||||
for trow in client.list_tables(schema):
|
||||
if trow.get("type") != "TABLE":
|
||||
continue # le viste sono fuori scope (come l'introspezione diretta)
|
||||
table_name = trow["table"]
|
||||
|
||||
col_comments = {
|
||||
c["name"]: (c.get("comment") or "")
|
||||
for c in client.table_comments(schema, table_name)
|
||||
if c.get("object") == "COLUMN"
|
||||
}
|
||||
columns: dict[str, ColumnPhysical] = {}
|
||||
for crow in client.table_columns(schema, table_name):
|
||||
name = crow["column"]
|
||||
columns[name] = ColumnPhysical(
|
||||
type=crow["type"],
|
||||
nullable=bool(crow.get("nullable", True)),
|
||||
pk=bool(crow.get("pk", False)),
|
||||
default=crow.get("default"),
|
||||
comment=col_comments.get(name, ""),
|
||||
)
|
||||
|
||||
foreign_keys: list[ForeignKey] = []
|
||||
for fk in client.table_foreign_keys(schema, table_name):
|
||||
foreign_keys.append(
|
||||
ForeignKey(
|
||||
columns=fk.get("columns") or [fk["column"]],
|
||||
ref_table=fk.get("ref_table") or fk["target_table"],
|
||||
ref_columns=fk.get("ref_columns") or [fk["target_column"]],
|
||||
name=fk.get("name", ""),
|
||||
)
|
||||
)
|
||||
|
||||
tables[table_name] = TablePhysical(
|
||||
comment=trow.get("comment") or "",
|
||||
row_count=int(trow.get("rows") or 0),
|
||||
columns=columns,
|
||||
foreign_keys=foreign_keys,
|
||||
)
|
||||
|
||||
return PhysicalSchema(
|
||||
database=database,
|
||||
schema=schema,
|
||||
introspected_at=datetime.now(UTC),
|
||||
tables=tables,
|
||||
)
|
||||
@@ -0,0 +1,160 @@
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
|
||||
from sqlalchemy import Engine, text
|
||||
|
||||
from nsp.config import ExamplesConfig, LshConfig
|
||||
from nsp.mschema.models import Annotations, PhysicalSchema
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
TEXT_TYPE_PREFIXES = ("text", "varchar", "character", "char")
|
||||
|
||||
|
||||
def is_text_type(pg_type: str) -> bool:
|
||||
return pg_type.lower().startswith(TEXT_TYPE_PREFIXES)
|
||||
|
||||
|
||||
def add_examples(engine: Engine, physical: PhysicalSchema, cfg: ExamplesConfig) -> None:
|
||||
"""Campiona i valori distinti piu' frequenti delle colonne testuali (in-place)."""
|
||||
schema = physical.db_schema
|
||||
with engine.connect() as conn:
|
||||
for table_name, table in physical.tables.items():
|
||||
for column_name, column in table.columns.items():
|
||||
if not is_text_type(column.type):
|
||||
continue
|
||||
q = text(f'''
|
||||
SELECT "{column_name}" FROM (
|
||||
SELECT "{column_name}", count(*) AS _freq
|
||||
FROM "{schema}"."{table_name}"
|
||||
WHERE "{column_name}" IS NOT NULL AND length("{column_name}") > 0
|
||||
GROUP BY "{column_name}"
|
||||
ORDER BY _freq DESC
|
||||
LIMIT :lim
|
||||
) AS sub
|
||||
''')
|
||||
try:
|
||||
rows = conn.execute(q, {"lim": cfg.max_per_column}).fetchall()
|
||||
except Exception as e: # colonna non leggibile: si salta, non si interrompe
|
||||
logger.warning("Campionamento saltato per %s.%s: %s", table_name, column_name, e)
|
||||
continue
|
||||
column.examples = [str(r[0]) for r in rows]
|
||||
|
||||
|
||||
def add_examples_rest(client, physical: PhysicalSchema, cfg: ExamplesConfig) -> None:
|
||||
"""Variante REST di add_examples: valori più frequenti via rpc `top_values`."""
|
||||
schema = physical.db_schema
|
||||
for table_name, table in physical.tables.items():
|
||||
for column_name, column in table.columns.items():
|
||||
if not is_text_type(column.type):
|
||||
continue
|
||||
rows = client.top_values(schema, table_name, column_name, cfg.max_per_column)
|
||||
column.examples = [str(r["value"]) for r in rows if r["value"] not in (None, "")]
|
||||
|
||||
|
||||
@dataclass
|
||||
class SkippedColumn:
|
||||
table: str
|
||||
column: str
|
||||
reason: str
|
||||
|
||||
|
||||
@dataclass
|
||||
class TruncatedColumn:
|
||||
table: str
|
||||
column: str
|
||||
indexed: int # quanti valori (i più frequenti) sono stati indicizzati
|
||||
|
||||
|
||||
def unique_values_for_lsh(
|
||||
engine: Engine,
|
||||
physical: PhysicalSchema,
|
||||
cfg: LshConfig,
|
||||
annotations: Annotations | None = None,
|
||||
) -> tuple[dict[str, dict[str, list[str]]], list[SkippedColumn], list[TruncatedColumn]]:
|
||||
"""Valori delle colonne testuali *eligible* per l'indice LSH.
|
||||
|
||||
Indicizza solo colonne con eligibilità effettiva True (le `wide_text` sono escluse:
|
||||
vedi principio di column eligibility). Estrae i valori distinti *più frequenti*
|
||||
(ORDER BY frequenza); se superano `max_values_per_column` la colonna è troncata e
|
||||
segnalata (mai tagliata in silenzio).
|
||||
"""
|
||||
from nsp.mschema.eligibility import effective_eligibility
|
||||
|
||||
annotations = annotations or Annotations()
|
||||
schema = physical.db_schema
|
||||
values: dict[str, dict[str, list[str]]] = {}
|
||||
skipped: list[SkippedColumn] = []
|
||||
truncated: list[TruncatedColumn] = []
|
||||
with engine.connect() as conn:
|
||||
for table_name, table in physical.tables.items():
|
||||
table_ann = annotations.tables.get(table_name)
|
||||
for column_name, column in table.columns.items():
|
||||
if not is_text_type(column.type):
|
||||
continue
|
||||
ann_col = table_ann.columns.get(column_name) if table_ann else None
|
||||
if not effective_eligibility(column, ann_col)[0]:
|
||||
continue
|
||||
q = text(f'''
|
||||
SELECT "{column_name}" FROM (
|
||||
SELECT "{column_name}", count(*) AS _freq
|
||||
FROM "{schema}"."{table_name}"
|
||||
WHERE "{column_name}" IS NOT NULL AND length("{column_name}") > 0
|
||||
GROUP BY "{column_name}"
|
||||
ORDER BY _freq DESC, "{column_name}"
|
||||
LIMIT :lim
|
||||
) AS sub
|
||||
''')
|
||||
try:
|
||||
rows = conn.execute(q, {"lim": cfg.max_values_per_column}).fetchall()
|
||||
except Exception as e:
|
||||
skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}"))
|
||||
continue
|
||||
vals = [str(r[0]) for r in rows]
|
||||
if not vals:
|
||||
continue
|
||||
values.setdefault(table_name, {})[column_name] = vals
|
||||
if len(vals) >= cfg.max_values_per_column:
|
||||
truncated.append(TruncatedColumn(table_name, column_name, len(vals)))
|
||||
return values, skipped, truncated
|
||||
|
||||
|
||||
def unique_values_for_lsh_rest(
|
||||
client,
|
||||
physical: PhysicalSchema,
|
||||
cfg: LshConfig,
|
||||
annotations: Annotations | None = None,
|
||||
) -> tuple[dict[str, dict[str, list[str]]], list[SkippedColumn], list[TruncatedColumn]]:
|
||||
"""Variante REST di unique_values_for_lsh: valori più frequenti via rpc `top_values`.
|
||||
|
||||
Stessa logica di eligibility e di segnalazione del troncamento del transport diretto.
|
||||
"""
|
||||
from nsp.mschema.eligibility import effective_eligibility
|
||||
|
||||
annotations = annotations or Annotations()
|
||||
schema = physical.db_schema
|
||||
values: dict[str, dict[str, list[str]]] = {}
|
||||
skipped: list[SkippedColumn] = []
|
||||
truncated: list[TruncatedColumn] = []
|
||||
for table_name, table in physical.tables.items():
|
||||
table_ann = annotations.tables.get(table_name)
|
||||
for column_name, column in table.columns.items():
|
||||
if not is_text_type(column.type):
|
||||
continue
|
||||
ann_col = table_ann.columns.get(column_name) if table_ann else None
|
||||
if not effective_eligibility(column, ann_col)[0]:
|
||||
continue
|
||||
try:
|
||||
rows = client.top_values(
|
||||
schema, table_name, column_name, cfg.max_values_per_column
|
||||
)
|
||||
except Exception as e:
|
||||
skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}"))
|
||||
continue
|
||||
vals = [str(r["value"]) for r in rows if r["value"] not in (None, "")]
|
||||
if not vals:
|
||||
continue
|
||||
values.setdefault(table_name, {})[column_name] = vals
|
||||
if len(vals) >= cfg.max_values_per_column:
|
||||
truncated.append(TruncatedColumn(table_name, column_name, len(vals)))
|
||||
return values, skipped, truncated
|
||||
Reference in New Issue
Block a user