from pathlib import Path from tht.cli.schema_cmd import physical_path def _extract_lsh_values(dwh, physical, annotations, limit, eligibility_cfg=None): from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type from tht.mschema.eligibility import effective_eligibility values, skipped, truncated = {}, [], [] for table_name, table in physical.tables.items(): table_ann = annotations.tables.get(table_name) for column_name, column in table.columns.items(): ann_col = table_ann.columns.get(column_name) if table_ann else None from_catalog = column.eligibility_reason in {"catalog", "sensitive"} if not is_text_type(column.type): continue if from_catalog and column.eligibility_reason == "sensitive": continue if from_catalog and eligibility_cfg is not None and ( column_name.lower() in {name.lower() for name in eligibility_cfg.ignore_columns} ): continue if not from_catalog and not effective_eligibility(column, ann_col)[0]: continue try: distinct = dwh.distinct_values(table_name, column_name, limit=limit) except Exception as exc: # noqa: BLE001 - an unreadable DWH column is non-fatal skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}")) continue vals = [str(value) for value in distinct.values if value not in (None, "")] if from_catalog and eligibility_cfg is not None: from tht.mschema.eligibility import classify_column lengths = [len(value) for value in vals] eligible, reason = classify_column( column.type, column.is_enum, (sum(lengths) / len(lengths)) if lengths else None, max(lengths) if lengths else None, eligibility_cfg, ) column.eligible = eligible column.eligibility_reason = reason if not eligible: continue if vals: values.setdefault(table_name, {})[column_name] = vals if distinct.truncated: truncated.append(TruncatedColumn(table_name, column_name, len(vals))) return values, skipped, truncated def build_lsh_artifacts( cfg, *, dwh=None, verbose: bool = False, physical_file: Path | None = None, output_dir: Path | None = None, ): """Run the existing LSH extraction/build algorithm and persist its outputs.""" from tht.adapters.factory import build_dwh from tht.lshindex import build_index, save_index if physical_file is None and cfg.paths.catalog_metadata_snapshot is not None: from tht.mschema.context import load_schema_context context = load_schema_context(cfg) physical, annotations = context.physical, context.annotations else: from tht.cli.schema_cmd import annotations_path from tht.mschema.models import Annotations, PhysicalSchema phys_file = physical_file or physical_path(cfg) if not phys_file.exists(): raise FileNotFoundError("physical catalog is missing; run schema introspect first") physical = PhysicalSchema.from_yaml(phys_file) annotations = Annotations.from_yaml(annotations_path(cfg)) target = dwh if dwh is not None else build_dwh(cfg) values, skipped, truncated = _extract_lsh_values( target, physical, annotations, cfg.lsh.max_values_per_column, cfg.eligibility ) lsh, minhashes = build_index(values, cfg.lsh, verbose=verbose) save_index( lsh, minhashes, cfg.lsh, output_dir or (cfg.paths.indexes / "lsh"), name=cfg.database.db_schema, ) return minhashes, skipped, truncated, values