from pathlib import Path import typer from tht.cli.config_cmd import CONFIG_OPT from tht.cli.schema_cmd import _load_config_or_exit, physical_path lsh_app = typer.Typer(help="Indice LSH su valori dei campi (derivato, rigenerabile)") def _lsh_dir(cfg) -> Path: return cfg.paths.indexes / "lsh" def _extract_lsh_values(dwh, physical, annotations, limit): from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type from tht.mschema.eligibility import effective_eligibility values, skipped, truncated = {}, [], [] for table_name, table in physical.tables.items(): table_ann = annotations.tables.get(table_name) for column_name, column in table.columns.items(): ann_col = table_ann.columns.get(column_name) if table_ann else None if not is_text_type(column.type) or not effective_eligibility(column, ann_col)[0]: continue try: distinct = dwh.distinct_values(table_name, column_name, limit=limit) except Exception as exc: skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}")) continue vals = [str(value) for value in distinct.values if value not in (None, "")] if vals: values.setdefault(table_name, {})[column_name] = vals if distinct.truncated: truncated.append(TruncatedColumn(table_name, column_name, len(vals))) return values, skipped, truncated @lsh_app.command("build") def build_cmd(config: Path = CONFIG_OPT) -> None: """Costruisce l'indice LSH dai valori del database e lo salva su pickle.""" from tht.lshindex import build_index, save_index from tht.mschema.models import Annotations, PhysicalSchema cfg = _load_config_or_exit(config) phys_file = physical_path(cfg) if not phys_file.exists(): typer.secho( f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) physical = PhysicalSchema.from_yaml(phys_file) from tht.cli.schema_cmd import annotations_path annotations = Annotations.from_yaml(annotations_path(cfg)) typer.echo("Estrazione valori (i più frequenti) dalle colonne testuali eligible...") from tht.adapters.factory import build_dwh dwh = build_dwh(cfg) values, skipped, truncated = _extract_lsh_values( dwh, physical, annotations, cfg.lsh.max_values_per_column ) n_values = sum(len(v) for t in values.values() for v in t.values()) typer.echo(f" {n_values} valori da {sum(len(t) for t in values.values())} colonne") for s in skipped: typer.secho(f" saltata {s.table}.{s.column}: {s.reason}", fg=typer.colors.YELLOW) for t in truncated: typer.secho( f" troncata {t.table}.{t.column}: indicizzati i {t.indexed} valori più frequenti " f"(limite max_values_per_column raggiunto; altri valori distinti NON indicizzati)", fg=typer.colors.YELLOW, ) lsh, minhashes = build_index(values, cfg.lsh, verbose=True) save_index(lsh, minhashes, cfg.lsh, _lsh_dir(cfg), name=cfg.database.db_schema) typer.secho( f"OK: indice LSH ({len(minhashes)} entry) -> {_lsh_dir(cfg)}", fg=typer.colors.GREEN ) @lsh_app.command("query") def query_cmd( keyword: str = typer.Argument(..., help="Termine da cercare, es. 'ablazione'."), config: Path = CONFIG_OPT, top: int = typer.Option(10, "--top", help="Numero massimo di risultati."), ) -> None: """Probe visuale: mostra i candidati LSH per un termine, con score Jaccard.""" from rich.console import Console from rich.table import Table from tht.lshindex import LshIndexError, load_index, query_index cfg = _load_config_or_exit(config) try: lsh, minhashes, meta = load_index(_lsh_dir(cfg), name=cfg.database.db_schema) except LshIndexError as e: typer.secho(f"ERRORE: {e}", fg=typer.colors.RED) raise typer.Exit(code=1) hits = query_index(lsh, minhashes, keyword, meta, top_n=top) if not hits: typer.secho(f"Nessun candidato LSH per '{keyword}'.", fg=typer.colors.YELLOW) return table = Table(title=f"Candidati LSH per '{keyword}' ({len(hits)})") table.add_column("Tabella") table.add_column("Colonna") table.add_column("Valore") table.add_column("Score", justify="right") for h in hits: table.add_row(h.table, h.column, h.value, f"{h.score:.3f}") Console().print(table)