Files
ThothII/harness/tht/cli/lsh_cmd.py
T

108 lines
4.4 KiB
Python

from pathlib import Path
import typer
from tht.cli.config_cmd import CONFIG_OPT
from tht.cli.schema_cmd import _load_config_or_exit, physical_path
lsh_app = typer.Typer(help="Indice LSH su valori dei campi (derivato, rigenerabile)")
def _lsh_dir(cfg) -> Path:
return cfg.paths.indexes / "lsh"
@lsh_app.command("build")
def build_cmd(config: Path = CONFIG_OPT) -> None:
"""Costruisce l'indice LSH dai valori del database e lo salva su pickle."""
from tht.lshindex import build_index, save_index
from tht.mschema.models import Annotations, PhysicalSchema
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
from tht.cli.schema_cmd import annotations_path
annotations = Annotations.from_yaml(annotations_path(cfg))
typer.echo("Estrazione valori (i più frequenti) dalle colonne testuali eligible...")
from tht.adapters.factory import build_dwh
from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type
from tht.mschema.eligibility import effective_eligibility
dwh = build_dwh(cfg)
values: dict[str, dict[str, list[str]]] = {}
skipped: list[SkippedColumn] = []
truncated: list[TruncatedColumn] = []
for table_name, table in physical.tables.items():
table_ann = annotations.tables.get(table_name)
for column_name, column in table.columns.items():
ann_col = table_ann.columns.get(column_name) if table_ann else None
if not is_text_type(column.type) or not effective_eligibility(column, ann_col)[0]:
continue
try:
distinct = dwh.distinct_values(table_name, column_name)
except Exception as e:
skipped.append(SkippedColumn(table_name, column_name, f"errore: {e}"))
continue
vals = [str(value) for value in distinct.values if value not in (None, "")]
if vals:
values.setdefault(table_name, {})[column_name] = vals
if distinct.truncated or len(vals) >= cfg.lsh.max_values_per_column:
truncated.append(TruncatedColumn(table_name, column_name, len(vals)))
n_values = sum(len(v) for t in values.values() for v in t.values())
typer.echo(f" {n_values} valori da {sum(len(t) for t in values.values())} colonne")
for s in skipped:
typer.secho(f" saltata {s.table}.{s.column}: {s.reason}", fg=typer.colors.YELLOW)
for t in truncated:
typer.secho(
f" troncata {t.table}.{t.column}: indicizzati i {t.indexed} valori più frequenti "
f"(limite max_values_per_column raggiunto; altri valori distinti NON indicizzati)",
fg=typer.colors.YELLOW,
)
lsh, minhashes = build_index(values, cfg.lsh, verbose=True)
save_index(lsh, minhashes, cfg.lsh, _lsh_dir(cfg), name=cfg.database.db_schema)
typer.secho(
f"OK: indice LSH ({len(minhashes)} entry) -> {_lsh_dir(cfg)}", fg=typer.colors.GREEN
)
@lsh_app.command("query")
def query_cmd(
keyword: str = typer.Argument(..., help="Termine da cercare, es. 'ablazione'."),
config: Path = CONFIG_OPT,
top: int = typer.Option(10, "--top", help="Numero massimo di risultati."),
) -> None:
"""Probe visuale: mostra i candidati LSH per un termine, con score Jaccard."""
from rich.console import Console
from rich.table import Table
from tht.lshindex import LshIndexError, load_index, query_index
cfg = _load_config_or_exit(config)
try:
lsh, minhashes, meta = load_index(_lsh_dir(cfg), name=cfg.database.db_schema)
except LshIndexError as e:
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED)
raise typer.Exit(code=1)
hits = query_index(lsh, minhashes, keyword, meta, top_n=top)
if not hits:
typer.secho(f"Nessun candidato LSH per '{keyword}'.", fg=typer.colors.YELLOW)
return
table = Table(title=f"Candidati LSH per '{keyword}' ({len(hits)})")
table.add_column("Tabella")
table.add_column("Colonna")
table.add_column("Valore")
table.add_column("Score", justify="right")
for h in hits:
table.add_row(h.table, h.column, h.value, f"{h.score:.3f}")
Console().print(table)