Files
ThothII/harness/tht/cli/schema_cmd.py
T

400 lines
14 KiB
Python

from pathlib import Path
import logging
import typer
from tht.adapters.factory import build_dwh
from tht.cli.config_cmd import CONFIG_OPT
from tht.config import ConfigError, load_config
from tht.db.sampling import is_text_type
from tht.mschema.eligibility import classify_all
schema_app = typer.Typer(help="Gestione mschema (rappresentazione canonica dello schema)")
logger = logging.getLogger(__name__)
def _add_examples(dwh, phys, examples) -> None:
for table_name, table in phys.tables.items():
for column_name, column in table.columns.items():
if not is_text_type(column.type):
continue
try:
sampled = dwh.sample_column(
table_name, column_name, limit=examples.max_per_column
)
except Exception as exc:
logger.warning("Campionamento saltato per %s.%s: %s",
table_name, column_name, exc)
continue
column.examples = [str(value) for value in sampled if value not in (None, "")]
def _load_config_or_exit(config: Path):
try:
return load_config(config)
except ConfigError as e:
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
def physical_path(cfg) -> Path:
return cfg.paths.artifacts / "mschema" / "physical.yaml"
def annotations_path(cfg) -> Path:
return cfg.paths.artifacts / "mschema" / "annotations.yaml"
def refresh_catalog(cfg, *, dwh=None):
"""Run the existing catalog algorithm and persist its canonical output."""
target = dwh if dwh is not None else build_dwh(cfg)
physical = target.introspect()
_add_examples(target, physical, cfg.examples)
classify_all(physical, cfg.eligibility)
physical.to_yaml(physical_path(cfg))
return physical
@schema_app.command("introspect")
def introspect_cmd(
config: Path = CONFIG_OPT,
refresh: bool = typer.Option(
False,
"--refresh",
help="Forza la re-introspezione del DWH anche se physical.yaml esiste già.",
),
) -> None:
"""Introspeziona lo schema target e genera artifacts/mschema/physical.yaml.
Se physical.yaml esiste già, esce subito (cache); usa --refresh per rigenerarlo.
"""
cfg = _load_config_or_exit(config)
out = physical_path(cfg)
if out.exists() and not refresh:
from datetime import UTC, datetime
from tht.mschema.models import PhysicalSchema
try:
cached = PhysicalSchema.from_yaml(out)
except Exception:
pass # catalogo illeggibile: procedi con la re-introspezione
else:
ts = cached.introspected_at
if ts.tzinfo is None:
ts = ts.replace(tzinfo=UTC)
age_days = (datetime.now(UTC) - ts).days
typer.secho(
f"OK (cache): {out} esistente ({len(cached.tables)} tabelle, "
f"età {age_days}g). Re-introspezione solo con --refresh (manutenzione).",
fg=typer.colors.GREEN,
)
return
try:
phys = refresh_catalog(cfg)
except Exception as e:
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
n_cols = sum(len(t.columns) for t in phys.tables.values())
n_ignored = sum(
1 for t in phys.tables.values() for c in t.columns.values() if not c.eligible
)
typer.secho(
f"OK: {len(phys.tables)} tabelle, {n_cols} colonne "
f"({n_ignored} ignorate: testo ampio) -> {out}",
fg=typer.colors.GREEN,
)
@schema_app.command("check")
def check_cmd(config: Path = CONFIG_OPT) -> None:
"""Confronta physical.yaml e annotations.yaml; segnala annotazioni orfane."""
from tht.mschema.merge import find_orphans
from tht.mschema.models import Annotations, PhysicalSchema
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
ignored = [
f"{t}.{c} ({col.eligibility_reason})"
for t, table in physical.tables.items()
for c, col in table.columns.items()
if not col.eligible
]
if ignored:
typer.secho(
f"Colonne ignorate (testo ampio, {len(ignored)}):", fg=typer.colors.YELLOW
)
for line in ignored:
typer.echo(f" - {line}")
orphans = find_orphans(physical, annotations)
if orphans:
typer.secho(f"ATTENZIONE: {len(orphans)} annotazioni orfane:", fg=typer.colors.YELLOW)
for o in orphans:
typer.echo(f" - {o}")
raise typer.Exit(code=3)
typer.secho("OK: nessuna annotazione orfana.", fg=typer.colors.GREEN)
# PK con questi nomi sono identificatori generici: la regola same-name non si applica
# (nel DWH reale `id` e' la PK di ~50 tabelle e produrrebbe migliaia di falsi positivi).
_GENERIC_PK_NAMES = {"id", "key", "code"}
@schema_app.command("suggest-fks")
def suggest_fks_cmd(
config: Path = CONFIG_OPT,
from_sql: list[Path] = typer.Option(
None, "--from-sql",
help="Directory di .sql approvati da cui minare i join reali (ripetibile).",
),
assume: list[str] = typer.Option(
None, "--assume",
help="Disambigua una PK con piu' proprietari: col=tabella_ref "
"(es. cod_paz=dim_patient). Ripetibile.",
),
write: bool = typer.Option(
False, "--write",
help="Fonde i suggerimenti in annotations.yaml (aggiunge solo FK mancanti).",
),
) -> None:
"""Suggerisce FK logiche per la curazione umana in annotations.yaml.
Tre regole, in ordine di confidenza: (1) equi-join minati dall'SQL gia'
approvato (--from-sql); (2) colonna `*time_key` verso la PK di dim_time;
(3) colonna con lo stesso nome della PK di UN'ALTRA tabella, solo se quel
nome ha un unico proprietario e non e' generico (id/key/code) — salvo
disambiguazione esplicita con --assume.
"""
import yaml as _yaml
from tht.mschema.fkmine import mine_join_pairs
from tht.mschema.models import Annotations, ForeignKey, PhysicalSchema, TableAnnotation
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
ann_path = annotations_path(cfg)
annotations = Annotations.from_yaml(ann_path)
assumed: dict[str, str] = {}
for a in assume or []:
col, _, ref = a.partition("=")
if not ref or ref not in physical.tables:
typer.secho(
f"ERRORE: --assume '{a}' non valido (atteso col=tabella nel catalogo).",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
assumed[col] = ref
def _single_pk(table) -> str | None:
pks = [c for c, col in table.columns.items() if col.pk]
return pks[0] if len(pks) == 1 else None
pk_owners: dict[str, list[str]] = {}
for tname, table in physical.tables.items():
pk = _single_pk(table)
if pk:
pk_owners.setdefault(pk, []).append(tname)
dim_time_pk = None
if "dim_time" in physical.tables:
dim_time_pk = _single_pk(physical.tables["dim_time"])
def _known(tname: str) -> set:
keys = set()
for fk in physical.tables[tname].foreign_keys:
keys.add((tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)))
ann = annotations.tables.get(tname)
if ann:
for fk in ann.foreign_keys:
keys.add((tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)))
return keys
known_by_table: dict[str, set] = {t: _known(t) for t in physical.tables}
suggested: dict[str, list[ForeignKey]] = {}
def _add(tname: str, col: str, ref_table: str, ref_col: str) -> None:
key = ((col,), ref_table, (ref_col,))
if key in known_by_table[tname]:
return
known_by_table[tname].add(key)
suggested.setdefault(tname, []).append(
ForeignKey(columns=[col], ref_table=ref_table, ref_columns=[ref_col])
)
# Regola 1: join minati dall'SQL approvato.
n_sql_files = 0
mined_total = 0
for d in from_sql or []:
for sql_file in sorted(d.rglob("*.sql")):
n_sql_files += 1
pairs = mine_join_pairs(sql_file.read_text(), physical)
mined_total += sum(pairs.values())
for (src_t, src_c, ref_t, ref_c) in pairs:
_add(src_t, src_c, ref_t, ref_c)
# Regole 2 e 3: convenzioni di naming.
ambiguous_skipped: set[str] = set()
for tname, table in physical.tables.items():
for cname in table.columns:
if dim_time_pk and cname.endswith("time_key") and tname != "dim_time":
_add(tname, cname, "dim_time", dim_time_pk)
continue
if cname in assumed:
if assumed[cname] != tname:
_add(tname, cname, assumed[cname], cname)
continue
owners = [o for o in pk_owners.get(cname, []) if o != tname]
if not owners or cname in _GENERIC_PK_NAMES:
continue
if len(pk_owners[cname]) > 1:
ambiguous_skipped.add(cname)
continue
_add(tname, cname, owners[0], cname)
if n_sql_files:
typer.secho(
f"Minati {mined_total} equi-join da {n_sql_files} file SQL.",
fg=typer.colors.BLUE, err=True,
)
if ambiguous_skipped:
typer.secho(
"PK ambigue saltate dalla regola same-name (piu' tabelle proprietarie): "
+ ", ".join(sorted(ambiguous_skipped))
+ ". Se servono, aggiungile a mano o passa --from-sql.",
fg=typer.colors.YELLOW, err=True,
)
n_fks = sum(len(v) for v in suggested.values())
if not suggested:
typer.secho("OK: nessuna FK da suggerire.", fg=typer.colors.GREEN)
return
if write:
for tname, fks in suggested.items():
ann = annotations.tables.setdefault(tname, TableAnnotation())
ann.foreign_keys.extend(fks)
annotations.to_yaml(ann_path)
typer.secho(
f"OK: {n_fks} FK suggerite aggiunte a {ann_path} "
f"({len(suggested)} tabelle). Rivedile a mano prima dell'uso.",
fg=typer.colors.GREEN,
)
return
payload = {
"tables": {
tname: {"foreign_keys": [fk.model_dump(exclude_defaults=True) for fk in fks]}
for tname, fks in suggested.items()
}
}
typer.echo(_yaml.safe_dump(payload, sort_keys=False, allow_unicode=True))
typer.secho(
f"{n_fks} FK candidate ({len(suggested)} tabelle). "
f"Usa --write per fonderle in annotations.yaml, poi curale a mano.",
fg=typer.colors.YELLOW,
)
@schema_app.command("render")
def render_cmd(
config: Path = CONFIG_OPT,
format: str = typer.Option(
"markdown", "--format", "-f", help="Formato: markdown | mschema-text | schema-dict"
),
tables: list[str] = typer.Option(
None, "--table", "-t", help="Limita alle tabelle indicate (ripetibile)."
),
output: Path = typer.Option(None, "--output", "-o", help="File di output (default stdout)."),
) -> None:
"""Serializza mschema (physical + annotations) nel formato richiesto."""
import json
from tht.mschema.models import Annotations, PhysicalSchema
from tht.mschema.render import to_markdown, to_mschema_text, to_schema_dict
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
annotations = Annotations.from_yaml(annotations_path(cfg))
table_filter = list(tables) if tables else None
if format == "markdown":
out = to_markdown(physical, annotations)
elif format == "mschema-text":
out = to_mschema_text(physical, annotations, tables=table_filter)
elif format == "schema-dict":
out = json.dumps(to_schema_dict(physical, annotations), ensure_ascii=False, indent=2)
else:
typer.secho(f"ERRORE: formato sconosciuto: {format}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
if output:
output.parent.mkdir(parents=True, exist_ok=True)
output.write_text(out)
typer.secho(f"OK: scritto {output}", fg=typer.colors.GREEN)
else:
typer.echo(out)
@schema_app.command("columns")
def columns_cmd(
table: str = typer.Argument(..., help="Nome tabella (chiave in physical.yaml)."),
json_out: bool = typer.Option(False, "--json", help="Emetti JSON puro su stdout."),
config: Path = CONFIG_OPT,
) -> None:
"""Elenca nome/descrizione/tipo/pk delle colonne di una tabella dal catalogo."""
import json as _json
from tht.mschema.models import PhysicalSchema
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
tbl = physical.tables.get(table)
if tbl is None:
typer.secho(f"ERRORE: tabella non nel catalogo: {table}", fg=typer.colors.RED, err=True)
raise typer.Exit(code=1)
payload = {
"table": table,
"description": tbl.comment,
"columns": [
{"name": name, "description": col.comment, "type": col.type, "pk": col.pk}
for name, col in tbl.columns.items()
],
}
if json_out:
typer.echo(_json.dumps(payload, ensure_ascii=False))
return
typer.echo(f"{table}: {tbl.comment}")
for c in payload["columns"]:
typer.echo(f" {'*' if c['pk'] else ' '} {c['name']} ({c['type']}) — {c['description']}")