Files
ThothII/harness/tests/l0/test_db_sampling.py
T
marcopan fc5fbe6b65 refactor(harness): renaming prodotto tht (Onda -1)
Thoth (tht) è il prodotto, PSD è il cliente. Nessun riferimento al contesto
clinico nel codice.

Rinomine:
- comando+package nsp→tht (dir nsp/→tht/, 46 import, pyproject entry point)
- gate nsp-gate.js→tht-gate.js (+ rewrite token, relayIfNspFails→relayIfThtFails)
- workspace chirone.{example,test}.yaml→tht.{example,test}.yaml (generici)
- env THOTH_→THT_ (19 var) + NSP_ stragglers (NSP_HARNESS_ROOT, NSP_SESSION)
- commenti/docstring chirone/psdwp3/policlinico neutralizzati ('the reference
  implementation', 'the DWH')

Aggiunto [tool.setuptools.packages.find] include=['tht*'] (necessario: l'auto-
discovery rompeva con tht/ + workspaces/ come top-level multipli).

.env operatore aggiornato in-place (prefissi THT_, valori preservati, gitignored).

Verifica: pytest 109 passed, npm test 14 pass, tht phase meta --json OK, zero
residui nsp/THOTH_/NSP_/chirone nel package.
2026-06-27 10:33:16 +02:00

55 lines
2.2 KiB
Python

"""L0: db/sampling against known data (testcontainers).
Verifies unique_values_for_lsh returns the expected most-frequent values for text
columns, and that wide_text / non-text columns are excluded.
"""
import pytest
from tht.config import LshConfig
from tht.db.introspect import introspect
from tht.db.sampling import is_text_type, unique_values_for_lsh
pytestmark = [pytest.mark.l0]
def test_is_text_type():
assert is_text_type("text")
assert is_text_type("varchar(100)")
assert is_text_type("character varying")
assert not is_text_type("integer")
assert not is_text_type("bigint")
assert not is_text_type("timestamp without time zone")
def test_unique_values_for_lsh_returns_most_frequent(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
# Before classify_all, all text columns are eligible=True by default. Sampling
# only touches text types regardless.
values, skipped, truncated = unique_values_for_lsh(
admin_engine, schema, LshConfig(max_values_per_column=100)
)
# dim_pazienti.citta: Milano, Bergamo, Brescia (3 distinct, all eligible text)
citta = values.get("dim_pazienti", {}).get("citta")
assert citta is not None
assert set(citta) == {"Milano", "Bergamo", "Brescia"}
def test_unique_values_for_lsh_excludes_non_text(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
values, _, _ = unique_values_for_lsh(
admin_engine, schema, LshConfig(max_values_per_column=100)
)
# id_paziente is bigint — must never appear in the LSH values.
assert "id_paziente" not in values.get("dim_pazienti", {})
def test_unique_values_for_lsh_truncation_reported(admin_engine):
schema = introspect(admin_engine, "testdb", "dw")
# Force a tiny cap so procedura/diagnosi columns (which have >2 distinct values)
# are reported as truncated rather than silently cut.
_, _, truncated = unique_values_for_lsh(
admin_engine, schema, LshConfig(max_values_per_column=1)
)
truncated_cols = {(t.table, t.column) for t in truncated}
# fct_ricoveri has several eligible text columns with distinct values
assert any(t[0] == "fct_ricoveri" for t in truncated_cols)