from pathlib import Path import logging import typer from tht.adapters.factory import build_dwh from tht.cli.config_cmd import CONFIG_OPT from tht.config import ConfigError, load_config from tht.db.sampling import is_text_type from tht.mschema.eligibility import classify_all schema_app = typer.Typer(help="Gestione mschema (rappresentazione canonica dello schema)") logger = logging.getLogger(__name__) def _add_examples(dwh, phys, examples) -> None: for table_name, table in phys.tables.items(): for column_name, column in table.columns.items(): if not is_text_type(column.type): continue try: sampled = dwh.sample_column( table_name, column_name, limit=examples.max_per_column ) except Exception as exc: logger.warning("Campionamento saltato per %s.%s: %s", table_name, column_name, exc) continue column.examples = [str(value) for value in sampled if value not in (None, "")] def _load_config_or_exit(config: Path): try: return load_config(config) except ConfigError as e: typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True) raise typer.Exit(code=1) def physical_path(cfg) -> Path: return cfg.paths.artifacts / "mschema" / "physical.yaml" def annotations_path(cfg) -> Path: return cfg.paths.artifacts / "mschema" / "annotations.yaml" def refresh_catalog(cfg, *, dwh=None): """Run the existing catalog algorithm and persist its canonical output.""" target = dwh if dwh is not None else build_dwh(cfg) physical = target.introspect() _add_examples(target, physical, cfg.examples) classify_all(physical, cfg.eligibility) physical.to_yaml(physical_path(cfg)) return physical @schema_app.command("introspect") def introspect_cmd( config: Path = CONFIG_OPT, refresh: bool = typer.Option( False, "--refresh", help="Forza la re-introspezione del DWH anche se physical.yaml esiste già.", ), ) -> None: """Introspeziona lo schema target e genera artifacts/mschema/physical.yaml. Se physical.yaml esiste già, esce subito (cache); usa --refresh per rigenerarlo. """ cfg = _load_config_or_exit(config) out = physical_path(cfg) if out.exists() and not refresh: from datetime import UTC, datetime from tht.mschema.models import PhysicalSchema try: cached = PhysicalSchema.from_yaml(out) except Exception: pass # catalogo illeggibile: procedi con la re-introspezione else: ts = cached.introspected_at if ts.tzinfo is None: ts = ts.replace(tzinfo=UTC) age_days = (datetime.now(UTC) - ts).days typer.secho( f"OK (cache): {out} esistente ({len(cached.tables)} tabelle, " f"età {age_days}g). Re-introspezione solo con --refresh (manutenzione).", fg=typer.colors.GREEN, ) return try: phys = refresh_catalog(cfg) except Exception as e: typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True) raise typer.Exit(code=1) n_cols = sum(len(t.columns) for t in phys.tables.values()) n_ignored = sum( 1 for t in phys.tables.values() for c in t.columns.values() if not c.eligible ) typer.secho( f"OK: {len(phys.tables)} tabelle, {n_cols} colonne " f"({n_ignored} ignorate: testo ampio) -> {out}", fg=typer.colors.GREEN, ) @schema_app.command("check") def check_cmd(config: Path = CONFIG_OPT) -> None: """Confronta physical.yaml e annotations.yaml; segnala annotazioni orfane.""" from tht.mschema.merge import find_orphans from tht.mschema.models import Annotations, PhysicalSchema cfg = _load_config_or_exit(config) phys_file = physical_path(cfg) if not phys_file.exists(): typer.secho( f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) physical = PhysicalSchema.from_yaml(phys_file) annotations = Annotations.from_yaml(annotations_path(cfg)) ignored = [ f"{t}.{c} ({col.eligibility_reason})" for t, table in physical.tables.items() for c, col in table.columns.items() if not col.eligible ] if ignored: typer.secho( f"Colonne ignorate (testo ampio, {len(ignored)}):", fg=typer.colors.YELLOW ) for line in ignored: typer.echo(f" - {line}") orphans = find_orphans(physical, annotations) if orphans: typer.secho(f"ATTENZIONE: {len(orphans)} annotazioni orfane:", fg=typer.colors.YELLOW) for o in orphans: typer.echo(f" - {o}") raise typer.Exit(code=3) typer.secho("OK: nessuna annotazione orfana.", fg=typer.colors.GREEN) # PK con questi nomi sono identificatori generici: la regola same-name non si applica # (nel DWH reale `id` e' la PK di ~50 tabelle e produrrebbe migliaia di falsi positivi). _GENERIC_PK_NAMES = {"id", "key", "code"} @schema_app.command("suggest-fks") def suggest_fks_cmd( config: Path = CONFIG_OPT, from_sql: list[Path] = typer.Option( None, "--from-sql", help="Directory di .sql approvati da cui minare i join reali (ripetibile).", ), assume: list[str] = typer.Option( None, "--assume", help="Disambigua una PK con piu' proprietari: col=tabella_ref " "(es. cod_paz=dim_patient). Ripetibile.", ), write: bool = typer.Option( False, "--write", help="Fonde i suggerimenti in annotations.yaml (aggiunge solo FK mancanti).", ), ) -> None: """Suggerisce FK logiche per la curazione umana in annotations.yaml. Tre regole, in ordine di confidenza: (1) equi-join minati dall'SQL gia' approvato (--from-sql); (2) colonna `*time_key` verso la PK di dim_time; (3) colonna con lo stesso nome della PK di UN'ALTRA tabella, solo se quel nome ha un unico proprietario e non e' generico (id/key/code) — salvo disambiguazione esplicita con --assume. """ import yaml as _yaml from tht.mschema.fkmine import mine_join_pairs from tht.mschema.models import Annotations, ForeignKey, PhysicalSchema, TableAnnotation cfg = _load_config_or_exit(config) phys_file = physical_path(cfg) if not phys_file.exists(): typer.secho( f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) physical = PhysicalSchema.from_yaml(phys_file) ann_path = annotations_path(cfg) annotations = Annotations.from_yaml(ann_path) assumed: dict[str, str] = {} for a in assume or []: col, _, ref = a.partition("=") if not ref or ref not in physical.tables: typer.secho( f"ERRORE: --assume '{a}' non valido (atteso col=tabella nel catalogo).", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) assumed[col] = ref def _single_pk(table) -> str | None: pks = [c for c, col in table.columns.items() if col.pk] return pks[0] if len(pks) == 1 else None pk_owners: dict[str, list[str]] = {} for tname, table in physical.tables.items(): pk = _single_pk(table) if pk: pk_owners.setdefault(pk, []).append(tname) dim_time_pk = None if "dim_time" in physical.tables: dim_time_pk = _single_pk(physical.tables["dim_time"]) def _known(tname: str) -> set: keys = set() for fk in physical.tables[tname].foreign_keys: keys.add((tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns))) ann = annotations.tables.get(tname) if ann: for fk in ann.foreign_keys: keys.add((tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns))) return keys known_by_table: dict[str, set] = {t: _known(t) for t in physical.tables} suggested: dict[str, list[ForeignKey]] = {} def _add(tname: str, col: str, ref_table: str, ref_col: str) -> None: key = ((col,), ref_table, (ref_col,)) if key in known_by_table[tname]: return known_by_table[tname].add(key) suggested.setdefault(tname, []).append( ForeignKey(columns=[col], ref_table=ref_table, ref_columns=[ref_col]) ) # Regola 1: join minati dall'SQL approvato. n_sql_files = 0 mined_total = 0 for d in from_sql or []: for sql_file in sorted(d.rglob("*.sql")): n_sql_files += 1 pairs = mine_join_pairs(sql_file.read_text(), physical) mined_total += sum(pairs.values()) for (src_t, src_c, ref_t, ref_c) in pairs: _add(src_t, src_c, ref_t, ref_c) # Regole 2 e 3: convenzioni di naming. ambiguous_skipped: set[str] = set() for tname, table in physical.tables.items(): for cname in table.columns: if dim_time_pk and cname.endswith("time_key") and tname != "dim_time": _add(tname, cname, "dim_time", dim_time_pk) continue if cname in assumed: if assumed[cname] != tname: _add(tname, cname, assumed[cname], cname) continue owners = [o for o in pk_owners.get(cname, []) if o != tname] if not owners or cname in _GENERIC_PK_NAMES: continue if len(pk_owners[cname]) > 1: ambiguous_skipped.add(cname) continue _add(tname, cname, owners[0], cname) if n_sql_files: typer.secho( f"Minati {mined_total} equi-join da {n_sql_files} file SQL.", fg=typer.colors.BLUE, err=True, ) if ambiguous_skipped: typer.secho( "PK ambigue saltate dalla regola same-name (piu' tabelle proprietarie): " + ", ".join(sorted(ambiguous_skipped)) + ". Se servono, aggiungile a mano o passa --from-sql.", fg=typer.colors.YELLOW, err=True, ) n_fks = sum(len(v) for v in suggested.values()) if not suggested: typer.secho("OK: nessuna FK da suggerire.", fg=typer.colors.GREEN) return if write: for tname, fks in suggested.items(): ann = annotations.tables.setdefault(tname, TableAnnotation()) ann.foreign_keys.extend(fks) annotations.to_yaml(ann_path) typer.secho( f"OK: {n_fks} FK suggerite aggiunte a {ann_path} " f"({len(suggested)} tabelle). Rivedile a mano prima dell'uso.", fg=typer.colors.GREEN, ) return payload = { "tables": { tname: {"foreign_keys": [fk.model_dump(exclude_defaults=True) for fk in fks]} for tname, fks in suggested.items() } } typer.echo(_yaml.safe_dump(payload, sort_keys=False, allow_unicode=True)) typer.secho( f"{n_fks} FK candidate ({len(suggested)} tabelle). " f"Usa --write per fonderle in annotations.yaml, poi curale a mano.", fg=typer.colors.YELLOW, ) @schema_app.command("render") def render_cmd( config: Path = CONFIG_OPT, format: str = typer.Option( "markdown", "--format", "-f", help="Formato: markdown | mschema-text | schema-dict" ), tables: list[str] = typer.Option( None, "--table", "-t", help="Limita alle tabelle indicate (ripetibile)." ), output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."), ) -> None: """Serializza mschema (physical + annotations) nel formato richiesto.""" import json from tht.mschema.models import Annotations, PhysicalSchema from tht.mschema.render import to_markdown, to_mschema_text, to_schema_dict cfg = _load_config_or_exit(config) phys_file = physical_path(cfg) if not phys_file.exists(): typer.secho( f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) physical = PhysicalSchema.from_yaml(phys_file) annotations = Annotations.from_yaml(annotations_path(cfg)) table_filter = list(tables) if tables else None if format == "markdown": out = to_markdown(physical, annotations) elif format == "mschema-text": out = to_mschema_text(physical, annotations, tables=table_filter) elif format == "schema-dict": out = json.dumps(to_schema_dict(physical, annotations), ensure_ascii=False, indent=2) else: typer.secho(f"ERRORE: formato sconosciuto: {format}", fg=typer.colors.RED, err=True) raise typer.Exit(code=1) if output: output.parent.mkdir(parents=True, exist_ok=True) output.write_text(out) typer.secho(f"OK: scritto {output}", fg=typer.colors.GREEN) else: typer.echo(out) @schema_app.command("columns") def columns_cmd( table: str = typer.Argument(..., help="Nome tabella (chiave in physical.yaml)."), json_out: bool = typer.Option(False, "--json", help="Emetti JSON puro su stdout."), config: Path = CONFIG_OPT, ) -> None: """Elenca nome/descrizione/tipo/pk delle colonne di una tabella dal catalogo.""" import json as _json from tht.mschema.models import PhysicalSchema cfg = _load_config_or_exit(config) phys_file = physical_path(cfg) if not phys_file.exists(): typer.secho( f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", fg=typer.colors.RED, err=True, ) raise typer.Exit(code=1) physical = PhysicalSchema.from_yaml(phys_file) tbl = physical.tables.get(table) if tbl is None: typer.secho(f"ERRORE: tabella non nel catalogo: {table}", fg=typer.colors.RED, err=True) raise typer.Exit(code=1) payload = { "table": table, "description": tbl.comment, "columns": [ {"name": name, "description": col.comment, "type": col.type, "pk": col.pk} for name, col in tbl.columns.items() ], } if json_out: typer.echo(_json.dumps(payload, ensure_ascii=False)) return typer.echo(f"{table}: {tbl.comment}") for c in payload["columns"]: typer.echo(f" {'*' if c['pk'] else ' '} {c['name']} ({c['type']}) — {c['description']}")