feat(opt): three efficiency levers for NL→SQL workflow

Lever 1: Join-graph via FK logics in annotations + suggest-fks command
  - TableAnnotation.foreign_keys field stores curated logical FKs (DWH has no FK constraints)
  - tht schema suggest-fks: mine from approved SQL, heuristics (time_key → dim_time),
    same-name discovery + explicit --assume flag for multi-owner PKs
  - mschema renders 【Foreign keys】 section populated; validation in merge.py
  - SKILL.md F4 now reads FKs from mschema-text, no custom data_time_key logic

Lever 2: Context-pack consolidation at kickoff (tht search pack)
  - Single embedding of question, reused for schema + evidence + solved searches
  - One command: tht search pack <question> --session <id> → retrieval_pack.md
  - Graceful degradation when Ollama/vector store unreachable (exit 0, empty sections)
  - SKILL.md F1 prescribes as first call; reduces model thinking turns via pre-retrieval

Lever 3: Phase-summary recap v2 auto-construction from session ledger
  - tht session show --json includes full decisions ledger
  - tht phase meta --json exports 'emits' (substantive decision types per phase)
  - Gate appends deterministic 【Decisioni registrate in questa fase】 section (appendLedgerSection)
  - Model authors only summary + checks; recap table comes from persisted state (exact by construction)
  - SKILL.md Disciplina 6: brief model output, gate fills the rest

Tests: 358 Python (including 10 FK + 3 pack + 1 session-ledger tests) + 111 JS gate tests, all pass.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-07 17:43:08 +02:00
co-authored by Claude Fable 5
parent 87e875bc81
commit e24b41b156
19 changed files with 936 additions and 29 deletions
+1
View File
@@ -71,6 +71,7 @@ def meta_cmd(
"name": p.name,
"advance": p.advance,
"artifacts_out": p.artifacts_out,
"emits": p.emits,
}
for p in wf.phases
],
+168
View File
@@ -141,6 +141,174 @@ def check_cmd(config: Path = CONFIG_OPT) -> None:
typer.secho("OK: nessuna annotazione orfana.", fg=typer.colors.GREEN)
# PK con questi nomi sono identificatori generici: la regola same-name non si applica
# (nel DWH reale `id` e' la PK di ~50 tabelle e produrrebbe migliaia di falsi positivi).
_GENERIC_PK_NAMES = {"id", "key", "code"}
@schema_app.command("suggest-fks")
def suggest_fks_cmd(
config: Path = CONFIG_OPT,
from_sql: list[Path] = typer.Option(
None, "--from-sql",
help="Directory di .sql approvati da cui minare i join reali (ripetibile).",
),
assume: list[str] = typer.Option(
None, "--assume",
help="Disambigua una PK con piu' proprietari: col=tabella_ref "
"(es. cod_paz=dim_patient). Ripetibile.",
),
write: bool = typer.Option(
False, "--write",
help="Fonde i suggerimenti in annotations.yaml (aggiunge solo FK mancanti).",
),
) -> None:
"""Suggerisce FK logiche per la curazione umana in annotations.yaml.
Tre regole, in ordine di confidenza: (1) equi-join minati dall'SQL gia'
approvato (--from-sql); (2) colonna `*time_key` verso la PK di dim_time;
(3) colonna con lo stesso nome della PK di UN'ALTRA tabella, solo se quel
nome ha un unico proprietario e non e' generico (id/key/code) — salvo
disambiguazione esplicita con --assume.
"""
import yaml as _yaml
from tht.mschema.fkmine import mine_join_pairs
from tht.mschema.models import Annotations, ForeignKey, PhysicalSchema, TableAnnotation
cfg = _load_config_or_exit(config)
phys_file = physical_path(cfg)
if not phys_file.exists():
typer.secho(
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
physical = PhysicalSchema.from_yaml(phys_file)
ann_path = annotations_path(cfg)
annotations = Annotations.from_yaml(ann_path)
assumed: dict[str, str] = {}
for a in assume or []:
col, _, ref = a.partition("=")
if not ref or ref not in physical.tables:
typer.secho(
f"ERRORE: --assume '{a}' non valido (atteso col=tabella nel catalogo).",
fg=typer.colors.RED, err=True,
)
raise typer.Exit(code=1)
assumed[col] = ref
def _single_pk(table) -> str | None:
pks = [c for c, col in table.columns.items() if col.pk]
return pks[0] if len(pks) == 1 else None
pk_owners: dict[str, list[str]] = {}
for tname, table in physical.tables.items():
pk = _single_pk(table)
if pk:
pk_owners.setdefault(pk, []).append(tname)
dim_time_pk = None
if "dim_time" in physical.tables:
dim_time_pk = _single_pk(physical.tables["dim_time"])
def _known(tname: str) -> set:
keys = set()
for fk in physical.tables[tname].foreign_keys:
keys.add((tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)))
ann = annotations.tables.get(tname)
if ann:
for fk in ann.foreign_keys:
keys.add((tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)))
return keys
known_by_table: dict[str, set] = {t: _known(t) for t in physical.tables}
suggested: dict[str, list[ForeignKey]] = {}
def _add(tname: str, col: str, ref_table: str, ref_col: str) -> None:
key = ((col,), ref_table, (ref_col,))
if key in known_by_table[tname]:
return
known_by_table[tname].add(key)
suggested.setdefault(tname, []).append(
ForeignKey(columns=[col], ref_table=ref_table, ref_columns=[ref_col])
)
# Regola 1: join minati dall'SQL approvato.
n_sql_files = 0
mined_total = 0
for d in from_sql or []:
for sql_file in sorted(d.rglob("*.sql")):
n_sql_files += 1
pairs = mine_join_pairs(sql_file.read_text(), physical)
mined_total += sum(pairs.values())
for (src_t, src_c, ref_t, ref_c) in pairs:
_add(src_t, src_c, ref_t, ref_c)
# Regole 2 e 3: convenzioni di naming.
ambiguous_skipped: set[str] = set()
for tname, table in physical.tables.items():
for cname in table.columns:
if dim_time_pk and cname.endswith("time_key") and tname != "dim_time":
_add(tname, cname, "dim_time", dim_time_pk)
continue
if cname in assumed:
if assumed[cname] != tname:
_add(tname, cname, assumed[cname], cname)
continue
owners = [o for o in pk_owners.get(cname, []) if o != tname]
if not owners or cname in _GENERIC_PK_NAMES:
continue
if len(pk_owners[cname]) > 1:
ambiguous_skipped.add(cname)
continue
_add(tname, cname, owners[0], cname)
if n_sql_files:
typer.secho(
f"Minati {mined_total} equi-join da {n_sql_files} file SQL.",
fg=typer.colors.BLUE, err=True,
)
if ambiguous_skipped:
typer.secho(
"PK ambigue saltate dalla regola same-name (piu' tabelle proprietarie): "
+ ", ".join(sorted(ambiguous_skipped))
+ ". Se servono, aggiungile a mano o passa --from-sql.",
fg=typer.colors.YELLOW, err=True,
)
n_fks = sum(len(v) for v in suggested.values())
if not suggested:
typer.secho("OK: nessuna FK da suggerire.", fg=typer.colors.GREEN)
return
if write:
for tname, fks in suggested.items():
ann = annotations.tables.setdefault(tname, TableAnnotation())
ann.foreign_keys.extend(fks)
annotations.to_yaml(ann_path)
typer.secho(
f"OK: {n_fks} FK suggerite aggiunte a {ann_path} "
f"({len(suggested)} tabelle). Rivedile a mano prima dell'uso.",
fg=typer.colors.GREEN,
)
return
payload = {
"tables": {
tname: {"foreign_keys": [fk.model_dump(exclude_defaults=True) for fk in fks]}
for tname, fks in suggested.items()
}
}
typer.echo(_yaml.safe_dump(payload, sort_keys=False, allow_unicode=True))
typer.secho(
f"{n_fks} FK candidate ({len(suggested)} tabelle). "
f"Usa --write per fonderle in annotations.yaml, poi curale a mano.",
fg=typer.colors.YELLOW,
)
@schema_app.command("render")
def render_cmd(
config: Path = CONFIG_OPT,
+153
View File
@@ -205,3 +205,156 @@ def search_cmd(
row.append((r.content[:120] + "…") if len(r.content) > 120 else r.content)
table.add_row(*row)
Console().print(table)
# Dimensioni fisse del pack (niente config: il pack deve restare piccolo perche'
# entra nel contesto del modello in un turno solo).
PACK_EVIDENCE_TOP = 5
PACK_SOLVED_TOP = 3
PACK_EXCERPT_CHARS = 400
@search_app.command("pack")
def pack_cmd(
question: str = typer.Argument(..., help="La domanda in linguaggio naturale."),
config: Path = CONFIG_OPT,
session: str = typer.Option(
None, "--session", help="Scrive il pack in sessions/<id>/retrieval_pack.md."
),
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
) -> None:
"""Context-pack F1: tabelle candidate + evidence + domande risolte in UNA chiamata.
Un solo embedding della domanda, riusato per le tre ricerche vettoriali.
Degrado gentile: se Ollama/vectordb non rispondono, le sezioni restano vuote
con un'avvertenza (exit 0) — la sessione prosegue con le ricerche live.
"""
from sqlalchemy.exc import OperationalError
from tht.cli.vector_cmd import make_embedder, open_searcher, require_vector_cfg
from tht.search import combined_search, schema_tables
from tht.solved import SOLVED_KIND
from tht.vectorstore.embeddings import EmbeddingsError
from tht.vectorstore.rest_client import VectorRestError
cfg = _load_config_or_exit(config)
require_vector_cfg(cfg)
tables: list[dict] = []
evidence: list[dict] = []
solved: list[dict] = []
warnings: list[str] = []
degrade = (VectorRestError, EmbeddingsError, OperationalError)
vec = None
searcher = embedder = None
try:
searcher = open_searcher(cfg)
embedder = make_embedder(cfg.embeddings)
vec = embedder.embed_query(question)
except degrade as e:
warnings.append(f"retrieval non disponibile ({e}): prosegui con le ricerche live")
if vec is not None:
from tht.cli.schema_cmd import physical_path
descriptions: dict[str, str] = {}
phys_file = physical_path(cfg)
if phys_file.exists():
from tht.mschema.models import PhysicalSchema
phys = PhysicalSchema.from_yaml(phys_file)
descriptions = {t: tab.comment for t, tab in phys.tables.items()}
try:
cand = combined_search(
keyword=question, lsh_hits=None, store=searcher, embedder=embedder,
top=cfg.search.schema_chunk_pool, rrf_k=cfg.search.rrf_k,
kinds=KIND_MAP["schema"], query_vec=vec,
)
tables = [
{"name": n, "rrf": round(s, 6), "description": descriptions.get(n, "")}
for n, s in schema_tables(cand, top_tables=cfg.search.top_schema_tables)
]
except degrade as e:
warnings.append(f"ricerca schema fallita ({e})")
try:
ev = combined_search(
keyword=question, lsh_hits=None, store=searcher, embedder=embedder,
top=PACK_EVIDENCE_TOP, rrf_k=cfg.search.rrf_k,
kinds=KIND_MAP["evidence"], query_vec=vec,
)
evidence = [
{"title": r.label, "status": r.status,
"excerpt": r.content[:PACK_EXCERPT_CHARS]}
for r in ev
]
except degrade as e:
warnings.append(f"ricerca evidence fallita ({e})")
try:
hits = searcher.search(vec, top_n=PACK_SOLVED_TOP, kinds=[SOLVED_KIND])
solved = [
{
"session_id": h.metadata.get("session_id", h.ref),
"question": h.metadata.get("question", h.content),
"sql": h.metadata.get("sql", ""),
"tables": h.metadata.get("tables", []),
"score": round(h.similarity, 4),
}
for h in hits
]
except degrade as e:
warnings.append(f"solved-search fallita ({e})")
for w in warnings:
typer.secho(f"ATTENZIONE: {w}", fg=typer.colors.YELLOW, err=True)
md_lines = ["# Retrieval pack", "", f"Domanda: {question}", ""]
md_lines += [f"## Tabelle candidate (top {len(tables)}, vettoriale sull'intera domanda)", ""]
if tables:
for i, t in enumerate(tables, 1):
desc = f" — {t['description']}" if t["description"] else ""
md_lines.append(f"{i}. **{t['name']}**{desc} (rrf {t['rrf']})")
else:
md_lines.append("_nessuna (retrieval non disponibile o nessun match)_")
md_lines += ["", "## Evidence rilevanti", ""]
if evidence:
for e in evidence:
status = f" [{e['status']}]" if e["status"] else ""
md_lines.append(f"- **{e['title']}**{status}: {e['excerpt']}")
else:
md_lines.append("_nessuna_")
md_lines += ["", "## Domande risolte simili (exemplar di riferimento, NON decisioni)", ""]
if solved:
for s in solved:
md_lines.append(
f"### {s['question']} \n(sessione `{s['session_id']}`; "
f"tabelle: {', '.join(s['tables']) or '-'})"
)
if s["sql"]:
md_lines += ["", "```sql", s["sql"], "```", ""]
else:
md_lines.append("_nessuna_")
if warnings:
md_lines += ["", "## Avvertenze", ""] + [f"- {w}" for w in warnings]
md = "\n".join(md_lines) + "\n"
if session:
from tht.cli.session_cmd import load_session_or_exit, session_dir
load_session_or_exit(cfg, session)
out = session_dir(cfg, session) / "retrieval_pack.md"
out.write_text(md)
if not json_out:
typer.secho(
f"OK: retrieval pack scritto in {out} "
f"({len(tables)} tabelle, {len(evidence)} evidence, {len(solved)} solved).",
fg=typer.colors.GREEN,
)
if json_out:
typer.echo(json.dumps(
{"question": question, "tables": tables, "evidence": evidence,
"solved": solved, "warnings": warnings},
ensure_ascii=False, indent=2,
))
elif not session:
typer.echo(md)
+5
View File
@@ -181,6 +181,11 @@ def show_cmd(
data = manifest.model_dump(mode="json", by_alias=True)
data["phase"] = phase
data["has_schema_linking"] = has_schema_linking
# Ledger integrale: il gate lo usa per costruire deterministicamente il
# recap delle decisioni nei riepiloghi di fase (v2).
data["decisions"] = [
d.model_dump(mode="json") for d in list_decisions(sdir)
]
typer.echo(json.dumps(data, ensure_ascii=False, indent=2))
return
+58
View File
@@ -0,0 +1,58 @@
"""Mining dei join reali dall'SQL approvato: coppie equi-join -> FK logiche candidate.
La fonte di verita' sono le query gia' validate da un umano (sql_final.sql, ctes/*.sql
delle sessioni approvate): un equi-join ricorrente tra due tabelle del catalogo, con
una delle due colonne PK della propria tabella, e' una FK logica ad alta confidenza.
"""
from collections import Counter
import sqlglot
from sqlglot import exp
from tht.mschema.models import PhysicalSchema
JoinPair = tuple[str, str, str, str] # (src_table, src_col, ref_table, ref_col)
def mine_join_pairs(sql_text: str, physical: PhysicalSchema) -> Counter:
"""Estrae le coppie equi-join tra tabelle del catalogo da un testo SQL.
Ritorna un Counter {(src_table, src_col, ref_table, ref_col): occorrenze}.
Il lato ref e' quello la cui colonna e' PK della propria tabella; coppie in cui
nessuno o entrambi i lati sono PK vengono scartate (non FK-like). Alias e CTE
vengono risolti; i riferimenti a CTE (non nel catalogo) sono ignorati.
"""
pairs: Counter = Counter()
try:
statements = sqlglot.parse(sql_text, read="postgres")
except sqlglot.errors.ParseError:
return pairs
for stmt in statements:
if stmt is None:
continue
alias_map: dict[str, str] = {}
for t in stmt.find_all(exp.Table):
alias_map[t.alias_or_name] = t.name
for eq in stmt.find_all(exp.EQ):
left, right = eq.left, eq.right
if not (isinstance(left, exp.Column) and isinstance(right, exp.Column)):
continue
if not (left.table and right.table):
continue
lt = alias_map.get(left.table, left.table)
rt = alias_map.get(right.table, right.table)
if lt == rt or lt not in physical.tables or rt not in physical.tables:
continue
lc, rc = left.name, right.name
if lc not in physical.tables[lt].columns or rc not in physical.tables[rt].columns:
continue
l_pk = physical.tables[lt].columns[lc].pk
r_pk = physical.tables[rt].columns[rc].pk
if l_pk == r_pk: # nessuna o entrambe PK: non FK-like
continue
if r_pk:
pairs[(lt, lc, rt, rc)] += 1
else:
pairs[(rt, rc, lt, lc)] += 1
return pairs
+10
View File
@@ -12,4 +12,14 @@ def find_orphans(physical: PhysicalSchema, annotations: Annotations) -> list[str
for column_name in table_ann.columns:
if column_name not in table.columns:
orphans.append(f"{table_name}.{column_name}")
for fk in table_ann.foreign_keys:
label = f"{table_name}.fk({','.join(fk.columns)})->{fk.ref_table}"
ref = physical.tables.get(fk.ref_table)
if ref is None:
orphans.append(label)
continue
missing = [c for c in fk.columns if c not in table.columns]
missing += [c for c in fk.ref_columns if c not in ref.columns]
if missing:
orphans.append(label)
return orphans
+4
View File
@@ -78,6 +78,10 @@ class TableAnnotation(BaseModel):
concepts: list[str] = []
notes: str = ""
columns: dict[str, ColumnAnnotation] = {}
# FK "logiche" curate a mano: il DWH non dichiara vincoli, quindi i join
# noti (es. data_time_key -> dim_time.day_key) vivono qui e vengono fusi
# con le FK fisiche in tutte le viste renderizzate.
foreign_keys: list[ForeignKey] = []
class Annotations(_YamlModel):
+20 -5
View File
@@ -1,11 +1,25 @@
from typing import Any
from tht.mschema.eligibility import effective_eligibility
from tht.mschema.models import Annotations, ColumnAnnotation, PhysicalSchema
from tht.mschema.models import Annotations, ColumnAnnotation, ForeignKey, PhysicalSchema
MAX_EXAMPLES_IN_PROMPT = 5
def table_foreign_keys(
physical: PhysicalSchema, annotations: Annotations, table: str
) -> list[ForeignKey]:
"""FK fisiche + FK logiche dalle annotations (dedup su columns/ref)."""
fks = list(physical.tables[table].foreign_keys)
ann = annotations.tables.get(table)
if ann:
seen = {(tuple(f.columns), f.ref_table, tuple(f.ref_columns)) for f in fks}
for fk in ann.foreign_keys:
if (tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)) not in seen:
fks.append(fk)
return fks
def _ann_col(annotations: Annotations, table: str, column: str) -> ColumnAnnotation | None:
ann = annotations.tables.get(table)
if ann is None:
@@ -59,7 +73,7 @@ def to_mschema_text(
shown = ", ".join(column.examples[:MAX_EXAMPLES_IN_PROMPT])
lines.append(f" -- Examples: {shown}")
lines.append(");")
for fk in table.foreign_keys:
for fk in table_foreign_keys(physical, annotations, table_name):
for src, dst in zip(fk.columns, fk.ref_columns):
fk_lines.append(f"{table_name}.{src}={fk.ref_table}.{dst}")
lines.extend(["", "【Foreign keys】", *fk_lines])
@@ -89,7 +103,7 @@ def to_schema_dict(
"primary_keys": [c for c in cols if table.columns[c].pk],
"foreign_keys": [
{"columns": fk.columns, "ref_table": fk.ref_table, "ref_columns": fk.ref_columns}
for fk in table.foreign_keys
for fk in table_foreign_keys(physical, annotations, table_name)
],
}
return out
@@ -132,9 +146,10 @@ def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None
f"| {column_name} | {column.type} | {'sì' if column.nullable else 'no'} "
f"| {'sì' if column.pk else ''} | {cdesc} | {examples} |"
)
if table.foreign_keys:
fks = table_foreign_keys(physical, annotations, table_name)
if fks:
lines += ["", "Foreign keys:"]
for fk in table.foreign_keys:
for fk in fks:
lines.append(
f"- ({', '.join(fk.columns)}) → {fk.ref_table} ({', '.join(fk.ref_columns)})"
)
+8 -2
View File
@@ -97,8 +97,12 @@ def combined_search(
top: int,
rrf_k: int,
kinds: list[str] | None,
query_vec: list[float] | None = None,
) -> list[SearchResult]:
"""Fonde LSH (valori di campo) e pgvector con Reciprocal Rank Fusion."""
"""Fonde LSH (valori di campo) e pgvector con Reciprocal Rank Fusion.
`query_vec` permette di riusare un embedding gia' calcolato della stessa
keyword (es. `tht search pack`, che fa piu' ricerche sulla stessa domanda)."""
rankings: dict[str, list[tuple[str, float]]] = {}
lsh_values: dict[str, str] = {}
if lsh_hits:
@@ -106,7 +110,9 @@ def combined_search(
rankings["lsh"] = [(key, score) for key, score, _ in aggregated]
lsh_values = {key: value for key, _, value in aggregated}
vector_hits = store.search(embedder.embed_query(keyword), top_n=top * 2, kinds=kinds)
if query_vec is None:
query_vec = embedder.embed_query(keyword)
vector_hits = store.search(query_vec, top_n=top * 2, kinds=kinds)
rankings["vector"] = [(_vector_key(h), h.similarity) for h in vector_hits]
by_key = {_vector_key(h): h for h in vector_hits}