147 lines
6.0 KiB
Python
147 lines
6.0 KiB
Python
from pathlib import Path
|
|
|
|
import typer
|
|
|
|
from tht.cli.config_cmd import CONFIG_OPT
|
|
from tht.cli.schema_cmd import _load_config_or_exit, physical_path
|
|
|
|
lsh_app = typer.Typer(help="Indice LSH su valori dei campi (derivato, rigenerabile)")
|
|
|
|
|
|
def _lsh_dir(cfg) -> Path:
|
|
from tht.jobs.dwh_pipeline import active_generation_dir
|
|
|
|
active = active_generation_dir(cfg.paths.artifacts.parent)
|
|
if active is not None:
|
|
return active
|
|
return cfg.paths.indexes / "lsh"
|
|
|
|
|
|
def _extract_lsh_values(dwh, physical, annotations, limit):
|
|
from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type
|
|
from tht.mschema.eligibility import effective_eligibility
|
|
|
|
values, skipped, truncated = {}, [], []
|
|
for table_name, table in physical.tables.items():
|
|
table_ann = annotations.tables.get(table_name)
|
|
for column_name, column in table.columns.items():
|
|
ann_col = table_ann.columns.get(column_name) if table_ann else None
|
|
if not is_text_type(column.type) or not effective_eligibility(column, ann_col)[0]:
|
|
continue
|
|
try:
|
|
distinct = dwh.distinct_values(table_name, column_name, limit=limit)
|
|
except Exception as exc:
|
|
skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}"))
|
|
continue
|
|
vals = [str(value) for value in distinct.values if value not in (None, "")]
|
|
if vals:
|
|
values.setdefault(table_name, {})[column_name] = vals
|
|
if distinct.truncated:
|
|
truncated.append(TruncatedColumn(table_name, column_name, len(vals)))
|
|
return values, skipped, truncated
|
|
|
|
|
|
def build_lsh_artifacts(
|
|
cfg, *, dwh=None, verbose: bool = False, physical_file: Path | None = None,
|
|
output_dir: Path | None = None,
|
|
):
|
|
"""Run the existing LSH extraction/build algorithm and persist its outputs."""
|
|
from tht.adapters.factory import build_dwh
|
|
from tht.cli.schema_cmd import annotations_path
|
|
from tht.lshindex import build_index, save_index
|
|
from tht.mschema.models import Annotations, PhysicalSchema
|
|
|
|
phys_file = physical_file or physical_path(cfg)
|
|
if not phys_file.exists():
|
|
raise FileNotFoundError("physical catalog is missing; run schema introspect first")
|
|
physical = PhysicalSchema.from_yaml(phys_file)
|
|
annotations = Annotations.from_yaml(annotations_path(cfg))
|
|
target = dwh if dwh is not None else build_dwh(cfg)
|
|
values, skipped, truncated = _extract_lsh_values(
|
|
target, physical, annotations, cfg.lsh.max_values_per_column
|
|
)
|
|
lsh, minhashes = build_index(values, cfg.lsh, verbose=verbose)
|
|
save_index(
|
|
lsh, minhashes, cfg.lsh, output_dir or (cfg.paths.indexes / "lsh"),
|
|
name=cfg.database.db_schema,
|
|
)
|
|
return minhashes, skipped, truncated, values
|
|
|
|
|
|
@lsh_app.command("build")
|
|
def build_cmd(config: Path = CONFIG_OPT) -> None:
|
|
"""Costruisce l'indice LSH dai valori del database e lo salva su pickle."""
|
|
cfg = _load_config_or_exit(config)
|
|
phys_file = physical_path(cfg)
|
|
if not phys_file.exists():
|
|
typer.secho(
|
|
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
|
fg=typer.colors.RED, err=True,
|
|
)
|
|
raise typer.Exit(code=1)
|
|
typer.echo("Estrazione valori (i più frequenti) dalle colonne testuali eligible...")
|
|
from tht.jobs.dwh_pipeline import active_generation_dir
|
|
|
|
if active_generation_dir(cfg.paths.artifacts.parent) is None:
|
|
minhashes, skipped, truncated, values = build_lsh_artifacts(cfg, verbose=True)
|
|
n_values = sum(len(v) for table in values.values() for v in table.values())
|
|
n_columns = sum(len(table) for table in values.values())
|
|
else:
|
|
from tht.cli.preprocess_cmd import run_dwh_from_config
|
|
from tht.lshindex import load_index
|
|
|
|
report = run_dwh_from_config(config, steps=("lsh",))
|
|
if report.status != "succeeded":
|
|
typer.secho("ERRORE: DWH preprocessing failed", fg=typer.colors.RED, err=True)
|
|
raise typer.Exit(code=1)
|
|
_, minhashes, _ = load_index(_lsh_dir(cfg), name=cfg.database.db_schema)
|
|
skipped, truncated = [], []
|
|
n_values = len(minhashes)
|
|
n_columns = len({(entry[1], entry[2]) for entry in minhashes.values()})
|
|
typer.echo(f" {n_values} valori da {n_columns} colonne")
|
|
for s in skipped:
|
|
typer.secho(f" saltata {s.table}.{s.column}: {s.reason}", fg=typer.colors.YELLOW)
|
|
for t in truncated:
|
|
typer.secho(
|
|
f" troncata {t.table}.{t.column}: indicizzati i {t.indexed} valori più frequenti "
|
|
f"(limite max_values_per_column raggiunto; altri valori distinti NON indicizzati)",
|
|
fg=typer.colors.YELLOW,
|
|
)
|
|
|
|
typer.secho(
|
|
f"OK: indice LSH ({len(minhashes)} entry) -> {_lsh_dir(cfg)}", fg=typer.colors.GREEN
|
|
)
|
|
|
|
|
|
@lsh_app.command("query")
|
|
def query_cmd(
|
|
keyword: str = typer.Argument(..., help="Termine da cercare, es. 'ablazione'."),
|
|
config: Path = CONFIG_OPT,
|
|
top: int = typer.Option(10, "--top", help="Numero massimo di risultati."),
|
|
) -> None:
|
|
"""Probe visuale: mostra i candidati LSH per un termine, con score Jaccard."""
|
|
from rich.console import Console
|
|
from rich.table import Table
|
|
|
|
from tht.lshindex import LshIndexError, load_index, query_index
|
|
|
|
cfg = _load_config_or_exit(config)
|
|
try:
|
|
lsh, minhashes, meta = load_index(_lsh_dir(cfg), name=cfg.database.db_schema)
|
|
except LshIndexError as e:
|
|
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED)
|
|
raise typer.Exit(code=1)
|
|
|
|
hits = query_index(lsh, minhashes, keyword, meta, top_n=top)
|
|
if not hits:
|
|
typer.secho(f"Nessun candidato LSH per '{keyword}'.", fg=typer.colors.YELLOW)
|
|
return
|
|
table = Table(title=f"Candidati LSH per '{keyword}' ({len(hits)})")
|
|
table.add_column("Tabella")
|
|
table.add_column("Colonna")
|
|
table.add_column("Valore")
|
|
table.add_column("Score", justify="right")
|
|
for h in hits:
|
|
table.add_row(h.table, h.column, h.value, f"{h.score:.3f}")
|
|
Console().print(table)
|