feat(opt): three efficiency levers for NL→SQL workflow
Lever 1: Join-graph via FK logics in annotations + suggest-fks command
- TableAnnotation.foreign_keys field stores curated logical FKs (DWH has no FK constraints)
- tht schema suggest-fks: mine from approved SQL, heuristics (time_key → dim_time),
same-name discovery + explicit --assume flag for multi-owner PKs
- mschema renders 【Foreign keys】 section populated; validation in merge.py
- SKILL.md F4 now reads FKs from mschema-text, no custom data_time_key logic
Lever 2: Context-pack consolidation at kickoff (tht search pack)
- Single embedding of question, reused for schema + evidence + solved searches
- One command: tht search pack <question> --session <id> → retrieval_pack.md
- Graceful degradation when Ollama/vector store unreachable (exit 0, empty sections)
- SKILL.md F1 prescribes as first call; reduces model thinking turns via pre-retrieval
Lever 3: Phase-summary recap v2 auto-construction from session ledger
- tht session show --json includes full decisions ledger
- tht phase meta --json exports 'emits' (substantive decision types per phase)
- Gate appends deterministic 【Decisioni registrate in questa fase】 section (appendLedgerSection)
- Model authors only summary + checks; recap table comes from persisted state (exact by construction)
- SKILL.md Disciplina 6: brief model output, gate fills the rest
Tests: 358 Python (including 10 FK + 3 pack + 1 session-ledger tests) + 111 JS gate tests, all pass.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -71,6 +71,7 @@ def meta_cmd(
|
||||
"name": p.name,
|
||||
"advance": p.advance,
|
||||
"artifacts_out": p.artifacts_out,
|
||||
"emits": p.emits,
|
||||
}
|
||||
for p in wf.phases
|
||||
],
|
||||
|
||||
@@ -141,6 +141,174 @@ def check_cmd(config: Path = CONFIG_OPT) -> None:
|
||||
typer.secho("OK: nessuna annotazione orfana.", fg=typer.colors.GREEN)
|
||||
|
||||
|
||||
# PK con questi nomi sono identificatori generici: la regola same-name non si applica
|
||||
# (nel DWH reale `id` e' la PK di ~50 tabelle e produrrebbe migliaia di falsi positivi).
|
||||
_GENERIC_PK_NAMES = {"id", "key", "code"}
|
||||
|
||||
|
||||
@schema_app.command("suggest-fks")
|
||||
def suggest_fks_cmd(
|
||||
config: Path = CONFIG_OPT,
|
||||
from_sql: list[Path] = typer.Option(
|
||||
None, "--from-sql",
|
||||
help="Directory di .sql approvati da cui minare i join reali (ripetibile).",
|
||||
),
|
||||
assume: list[str] = typer.Option(
|
||||
None, "--assume",
|
||||
help="Disambigua una PK con piu' proprietari: col=tabella_ref "
|
||||
"(es. cod_paz=dim_patient). Ripetibile.",
|
||||
),
|
||||
write: bool = typer.Option(
|
||||
False, "--write",
|
||||
help="Fonde i suggerimenti in annotations.yaml (aggiunge solo FK mancanti).",
|
||||
),
|
||||
) -> None:
|
||||
"""Suggerisce FK logiche per la curazione umana in annotations.yaml.
|
||||
|
||||
Tre regole, in ordine di confidenza: (1) equi-join minati dall'SQL gia'
|
||||
approvato (--from-sql); (2) colonna `*time_key` verso la PK di dim_time;
|
||||
(3) colonna con lo stesso nome della PK di UN'ALTRA tabella, solo se quel
|
||||
nome ha un unico proprietario e non e' generico (id/key/code) — salvo
|
||||
disambiguazione esplicita con --assume.
|
||||
"""
|
||||
import yaml as _yaml
|
||||
|
||||
from tht.mschema.fkmine import mine_join_pairs
|
||||
from tht.mschema.models import Annotations, ForeignKey, PhysicalSchema, TableAnnotation
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
phys_file = physical_path(cfg)
|
||||
if not phys_file.exists():
|
||||
typer.secho(
|
||||
f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
physical = PhysicalSchema.from_yaml(phys_file)
|
||||
ann_path = annotations_path(cfg)
|
||||
annotations = Annotations.from_yaml(ann_path)
|
||||
|
||||
assumed: dict[str, str] = {}
|
||||
for a in assume or []:
|
||||
col, _, ref = a.partition("=")
|
||||
if not ref or ref not in physical.tables:
|
||||
typer.secho(
|
||||
f"ERRORE: --assume '{a}' non valido (atteso col=tabella nel catalogo).",
|
||||
fg=typer.colors.RED, err=True,
|
||||
)
|
||||
raise typer.Exit(code=1)
|
||||
assumed[col] = ref
|
||||
|
||||
def _single_pk(table) -> str | None:
|
||||
pks = [c for c, col in table.columns.items() if col.pk]
|
||||
return pks[0] if len(pks) == 1 else None
|
||||
|
||||
pk_owners: dict[str, list[str]] = {}
|
||||
for tname, table in physical.tables.items():
|
||||
pk = _single_pk(table)
|
||||
if pk:
|
||||
pk_owners.setdefault(pk, []).append(tname)
|
||||
|
||||
dim_time_pk = None
|
||||
if "dim_time" in physical.tables:
|
||||
dim_time_pk = _single_pk(physical.tables["dim_time"])
|
||||
|
||||
def _known(tname: str) -> set:
|
||||
keys = set()
|
||||
for fk in physical.tables[tname].foreign_keys:
|
||||
keys.add((tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)))
|
||||
ann = annotations.tables.get(tname)
|
||||
if ann:
|
||||
for fk in ann.foreign_keys:
|
||||
keys.add((tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)))
|
||||
return keys
|
||||
|
||||
known_by_table: dict[str, set] = {t: _known(t) for t in physical.tables}
|
||||
suggested: dict[str, list[ForeignKey]] = {}
|
||||
|
||||
def _add(tname: str, col: str, ref_table: str, ref_col: str) -> None:
|
||||
key = ((col,), ref_table, (ref_col,))
|
||||
if key in known_by_table[tname]:
|
||||
return
|
||||
known_by_table[tname].add(key)
|
||||
suggested.setdefault(tname, []).append(
|
||||
ForeignKey(columns=[col], ref_table=ref_table, ref_columns=[ref_col])
|
||||
)
|
||||
|
||||
# Regola 1: join minati dall'SQL approvato.
|
||||
n_sql_files = 0
|
||||
mined_total = 0
|
||||
for d in from_sql or []:
|
||||
for sql_file in sorted(d.rglob("*.sql")):
|
||||
n_sql_files += 1
|
||||
pairs = mine_join_pairs(sql_file.read_text(), physical)
|
||||
mined_total += sum(pairs.values())
|
||||
for (src_t, src_c, ref_t, ref_c) in pairs:
|
||||
_add(src_t, src_c, ref_t, ref_c)
|
||||
|
||||
# Regole 2 e 3: convenzioni di naming.
|
||||
ambiguous_skipped: set[str] = set()
|
||||
for tname, table in physical.tables.items():
|
||||
for cname in table.columns:
|
||||
if dim_time_pk and cname.endswith("time_key") and tname != "dim_time":
|
||||
_add(tname, cname, "dim_time", dim_time_pk)
|
||||
continue
|
||||
if cname in assumed:
|
||||
if assumed[cname] != tname:
|
||||
_add(tname, cname, assumed[cname], cname)
|
||||
continue
|
||||
owners = [o for o in pk_owners.get(cname, []) if o != tname]
|
||||
if not owners or cname in _GENERIC_PK_NAMES:
|
||||
continue
|
||||
if len(pk_owners[cname]) > 1:
|
||||
ambiguous_skipped.add(cname)
|
||||
continue
|
||||
_add(tname, cname, owners[0], cname)
|
||||
|
||||
if n_sql_files:
|
||||
typer.secho(
|
||||
f"Minati {mined_total} equi-join da {n_sql_files} file SQL.",
|
||||
fg=typer.colors.BLUE, err=True,
|
||||
)
|
||||
if ambiguous_skipped:
|
||||
typer.secho(
|
||||
"PK ambigue saltate dalla regola same-name (piu' tabelle proprietarie): "
|
||||
+ ", ".join(sorted(ambiguous_skipped))
|
||||
+ ". Se servono, aggiungile a mano o passa --from-sql.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
|
||||
n_fks = sum(len(v) for v in suggested.values())
|
||||
if not suggested:
|
||||
typer.secho("OK: nessuna FK da suggerire.", fg=typer.colors.GREEN)
|
||||
return
|
||||
|
||||
if write:
|
||||
for tname, fks in suggested.items():
|
||||
ann = annotations.tables.setdefault(tname, TableAnnotation())
|
||||
ann.foreign_keys.extend(fks)
|
||||
annotations.to_yaml(ann_path)
|
||||
typer.secho(
|
||||
f"OK: {n_fks} FK suggerite aggiunte a {ann_path} "
|
||||
f"({len(suggested)} tabelle). Rivedile a mano prima dell'uso.",
|
||||
fg=typer.colors.GREEN,
|
||||
)
|
||||
return
|
||||
|
||||
payload = {
|
||||
"tables": {
|
||||
tname: {"foreign_keys": [fk.model_dump(exclude_defaults=True) for fk in fks]}
|
||||
for tname, fks in suggested.items()
|
||||
}
|
||||
}
|
||||
typer.echo(_yaml.safe_dump(payload, sort_keys=False, allow_unicode=True))
|
||||
typer.secho(
|
||||
f"{n_fks} FK candidate ({len(suggested)} tabelle). "
|
||||
f"Usa --write per fonderle in annotations.yaml, poi curale a mano.",
|
||||
fg=typer.colors.YELLOW,
|
||||
)
|
||||
|
||||
|
||||
@schema_app.command("render")
|
||||
def render_cmd(
|
||||
config: Path = CONFIG_OPT,
|
||||
|
||||
@@ -205,3 +205,156 @@ def search_cmd(
|
||||
row.append((r.content[:120] + "…") if len(r.content) > 120 else r.content)
|
||||
table.add_row(*row)
|
||||
Console().print(table)
|
||||
|
||||
|
||||
# Dimensioni fisse del pack (niente config: il pack deve restare piccolo perche'
|
||||
# entra nel contesto del modello in un turno solo).
|
||||
PACK_EVIDENCE_TOP = 5
|
||||
PACK_SOLVED_TOP = 3
|
||||
PACK_EXCERPT_CHARS = 400
|
||||
|
||||
|
||||
@search_app.command("pack")
|
||||
def pack_cmd(
|
||||
question: str = typer.Argument(..., help="La domanda in linguaggio naturale."),
|
||||
config: Path = CONFIG_OPT,
|
||||
session: str = typer.Option(
|
||||
None, "--session", help="Scrive il pack in sessions/<id>/retrieval_pack.md."
|
||||
),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
) -> None:
|
||||
"""Context-pack F1: tabelle candidate + evidence + domande risolte in UNA chiamata.
|
||||
|
||||
Un solo embedding della domanda, riusato per le tre ricerche vettoriali.
|
||||
Degrado gentile: se Ollama/vectordb non rispondono, le sezioni restano vuote
|
||||
con un'avvertenza (exit 0) — la sessione prosegue con le ricerche live.
|
||||
"""
|
||||
from sqlalchemy.exc import OperationalError
|
||||
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher, require_vector_cfg
|
||||
from tht.search import combined_search, schema_tables
|
||||
from tht.solved import SOLVED_KIND
|
||||
from tht.vectorstore.embeddings import EmbeddingsError
|
||||
from tht.vectorstore.rest_client import VectorRestError
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_cfg(cfg)
|
||||
|
||||
tables: list[dict] = []
|
||||
evidence: list[dict] = []
|
||||
solved: list[dict] = []
|
||||
warnings: list[str] = []
|
||||
degrade = (VectorRestError, EmbeddingsError, OperationalError)
|
||||
|
||||
vec = None
|
||||
searcher = embedder = None
|
||||
try:
|
||||
searcher = open_searcher(cfg)
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
vec = embedder.embed_query(question)
|
||||
except degrade as e:
|
||||
warnings.append(f"retrieval non disponibile ({e}): prosegui con le ricerche live")
|
||||
|
||||
if vec is not None:
|
||||
from tht.cli.schema_cmd import physical_path
|
||||
|
||||
descriptions: dict[str, str] = {}
|
||||
phys_file = physical_path(cfg)
|
||||
if phys_file.exists():
|
||||
from tht.mschema.models import PhysicalSchema
|
||||
|
||||
phys = PhysicalSchema.from_yaml(phys_file)
|
||||
descriptions = {t: tab.comment for t, tab in phys.tables.items()}
|
||||
try:
|
||||
cand = combined_search(
|
||||
keyword=question, lsh_hits=None, store=searcher, embedder=embedder,
|
||||
top=cfg.search.schema_chunk_pool, rrf_k=cfg.search.rrf_k,
|
||||
kinds=KIND_MAP["schema"], query_vec=vec,
|
||||
)
|
||||
tables = [
|
||||
{"name": n, "rrf": round(s, 6), "description": descriptions.get(n, "")}
|
||||
for n, s in schema_tables(cand, top_tables=cfg.search.top_schema_tables)
|
||||
]
|
||||
except degrade as e:
|
||||
warnings.append(f"ricerca schema fallita ({e})")
|
||||
try:
|
||||
ev = combined_search(
|
||||
keyword=question, lsh_hits=None, store=searcher, embedder=embedder,
|
||||
top=PACK_EVIDENCE_TOP, rrf_k=cfg.search.rrf_k,
|
||||
kinds=KIND_MAP["evidence"], query_vec=vec,
|
||||
)
|
||||
evidence = [
|
||||
{"title": r.label, "status": r.status,
|
||||
"excerpt": r.content[:PACK_EXCERPT_CHARS]}
|
||||
for r in ev
|
||||
]
|
||||
except degrade as e:
|
||||
warnings.append(f"ricerca evidence fallita ({e})")
|
||||
try:
|
||||
hits = searcher.search(vec, top_n=PACK_SOLVED_TOP, kinds=[SOLVED_KIND])
|
||||
solved = [
|
||||
{
|
||||
"session_id": h.metadata.get("session_id", h.ref),
|
||||
"question": h.metadata.get("question", h.content),
|
||||
"sql": h.metadata.get("sql", ""),
|
||||
"tables": h.metadata.get("tables", []),
|
||||
"score": round(h.similarity, 4),
|
||||
}
|
||||
for h in hits
|
||||
]
|
||||
except degrade as e:
|
||||
warnings.append(f"solved-search fallita ({e})")
|
||||
|
||||
for w in warnings:
|
||||
typer.secho(f"ATTENZIONE: {w}", fg=typer.colors.YELLOW, err=True)
|
||||
|
||||
md_lines = ["# Retrieval pack", "", f"Domanda: {question}", ""]
|
||||
md_lines += [f"## Tabelle candidate (top {len(tables)}, vettoriale sull'intera domanda)", ""]
|
||||
if tables:
|
||||
for i, t in enumerate(tables, 1):
|
||||
desc = f" — {t['description']}" if t["description"] else ""
|
||||
md_lines.append(f"{i}. **{t['name']}**{desc} (rrf {t['rrf']})")
|
||||
else:
|
||||
md_lines.append("_nessuna (retrieval non disponibile o nessun match)_")
|
||||
md_lines += ["", "## Evidence rilevanti", ""]
|
||||
if evidence:
|
||||
for e in evidence:
|
||||
status = f" [{e['status']}]" if e["status"] else ""
|
||||
md_lines.append(f"- **{e['title']}**{status}: {e['excerpt']}")
|
||||
else:
|
||||
md_lines.append("_nessuna_")
|
||||
md_lines += ["", "## Domande risolte simili (exemplar di riferimento, NON decisioni)", ""]
|
||||
if solved:
|
||||
for s in solved:
|
||||
md_lines.append(
|
||||
f"### {s['question']} \n(sessione `{s['session_id']}`; "
|
||||
f"tabelle: {', '.join(s['tables']) or '-'})"
|
||||
)
|
||||
if s["sql"]:
|
||||
md_lines += ["", "```sql", s["sql"], "```", ""]
|
||||
else:
|
||||
md_lines.append("_nessuna_")
|
||||
if warnings:
|
||||
md_lines += ["", "## Avvertenze", ""] + [f"- {w}" for w in warnings]
|
||||
md = "\n".join(md_lines) + "\n"
|
||||
|
||||
if session:
|
||||
from tht.cli.session_cmd import load_session_or_exit, session_dir
|
||||
|
||||
load_session_or_exit(cfg, session)
|
||||
out = session_dir(cfg, session) / "retrieval_pack.md"
|
||||
out.write_text(md)
|
||||
if not json_out:
|
||||
typer.secho(
|
||||
f"OK: retrieval pack scritto in {out} "
|
||||
f"({len(tables)} tabelle, {len(evidence)} evidence, {len(solved)} solved).",
|
||||
fg=typer.colors.GREEN,
|
||||
)
|
||||
if json_out:
|
||||
typer.echo(json.dumps(
|
||||
{"question": question, "tables": tables, "evidence": evidence,
|
||||
"solved": solved, "warnings": warnings},
|
||||
ensure_ascii=False, indent=2,
|
||||
))
|
||||
elif not session:
|
||||
typer.echo(md)
|
||||
|
||||
@@ -181,6 +181,11 @@ def show_cmd(
|
||||
data = manifest.model_dump(mode="json", by_alias=True)
|
||||
data["phase"] = phase
|
||||
data["has_schema_linking"] = has_schema_linking
|
||||
# Ledger integrale: il gate lo usa per costruire deterministicamente il
|
||||
# recap delle decisioni nei riepiloghi di fase (v2).
|
||||
data["decisions"] = [
|
||||
d.model_dump(mode="json") for d in list_decisions(sdir)
|
||||
]
|
||||
typer.echo(json.dumps(data, ensure_ascii=False, indent=2))
|
||||
return
|
||||
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Mining dei join reali dall'SQL approvato: coppie equi-join -> FK logiche candidate.
|
||||
|
||||
La fonte di verita' sono le query gia' validate da un umano (sql_final.sql, ctes/*.sql
|
||||
delle sessioni approvate): un equi-join ricorrente tra due tabelle del catalogo, con
|
||||
una delle due colonne PK della propria tabella, e' una FK logica ad alta confidenza.
|
||||
"""
|
||||
|
||||
from collections import Counter
|
||||
|
||||
import sqlglot
|
||||
from sqlglot import exp
|
||||
|
||||
from tht.mschema.models import PhysicalSchema
|
||||
|
||||
JoinPair = tuple[str, str, str, str] # (src_table, src_col, ref_table, ref_col)
|
||||
|
||||
|
||||
def mine_join_pairs(sql_text: str, physical: PhysicalSchema) -> Counter:
|
||||
"""Estrae le coppie equi-join tra tabelle del catalogo da un testo SQL.
|
||||
|
||||
Ritorna un Counter {(src_table, src_col, ref_table, ref_col): occorrenze}.
|
||||
Il lato ref e' quello la cui colonna e' PK della propria tabella; coppie in cui
|
||||
nessuno o entrambi i lati sono PK vengono scartate (non FK-like). Alias e CTE
|
||||
vengono risolti; i riferimenti a CTE (non nel catalogo) sono ignorati.
|
||||
"""
|
||||
pairs: Counter = Counter()
|
||||
try:
|
||||
statements = sqlglot.parse(sql_text, read="postgres")
|
||||
except sqlglot.errors.ParseError:
|
||||
return pairs
|
||||
for stmt in statements:
|
||||
if stmt is None:
|
||||
continue
|
||||
alias_map: dict[str, str] = {}
|
||||
for t in stmt.find_all(exp.Table):
|
||||
alias_map[t.alias_or_name] = t.name
|
||||
for eq in stmt.find_all(exp.EQ):
|
||||
left, right = eq.left, eq.right
|
||||
if not (isinstance(left, exp.Column) and isinstance(right, exp.Column)):
|
||||
continue
|
||||
if not (left.table and right.table):
|
||||
continue
|
||||
lt = alias_map.get(left.table, left.table)
|
||||
rt = alias_map.get(right.table, right.table)
|
||||
if lt == rt or lt not in physical.tables or rt not in physical.tables:
|
||||
continue
|
||||
lc, rc = left.name, right.name
|
||||
if lc not in physical.tables[lt].columns or rc not in physical.tables[rt].columns:
|
||||
continue
|
||||
l_pk = physical.tables[lt].columns[lc].pk
|
||||
r_pk = physical.tables[rt].columns[rc].pk
|
||||
if l_pk == r_pk: # nessuna o entrambe PK: non FK-like
|
||||
continue
|
||||
if r_pk:
|
||||
pairs[(lt, lc, rt, rc)] += 1
|
||||
else:
|
||||
pairs[(rt, rc, lt, lc)] += 1
|
||||
return pairs
|
||||
@@ -12,4 +12,14 @@ def find_orphans(physical: PhysicalSchema, annotations: Annotations) -> list[str
|
||||
for column_name in table_ann.columns:
|
||||
if column_name not in table.columns:
|
||||
orphans.append(f"{table_name}.{column_name}")
|
||||
for fk in table_ann.foreign_keys:
|
||||
label = f"{table_name}.fk({','.join(fk.columns)})->{fk.ref_table}"
|
||||
ref = physical.tables.get(fk.ref_table)
|
||||
if ref is None:
|
||||
orphans.append(label)
|
||||
continue
|
||||
missing = [c for c in fk.columns if c not in table.columns]
|
||||
missing += [c for c in fk.ref_columns if c not in ref.columns]
|
||||
if missing:
|
||||
orphans.append(label)
|
||||
return orphans
|
||||
|
||||
@@ -78,6 +78,10 @@ class TableAnnotation(BaseModel):
|
||||
concepts: list[str] = []
|
||||
notes: str = ""
|
||||
columns: dict[str, ColumnAnnotation] = {}
|
||||
# FK "logiche" curate a mano: il DWH non dichiara vincoli, quindi i join
|
||||
# noti (es. data_time_key -> dim_time.day_key) vivono qui e vengono fusi
|
||||
# con le FK fisiche in tutte le viste renderizzate.
|
||||
foreign_keys: list[ForeignKey] = []
|
||||
|
||||
|
||||
class Annotations(_YamlModel):
|
||||
|
||||
@@ -1,11 +1,25 @@
|
||||
from typing import Any
|
||||
|
||||
from tht.mschema.eligibility import effective_eligibility
|
||||
from tht.mschema.models import Annotations, ColumnAnnotation, PhysicalSchema
|
||||
from tht.mschema.models import Annotations, ColumnAnnotation, ForeignKey, PhysicalSchema
|
||||
|
||||
MAX_EXAMPLES_IN_PROMPT = 5
|
||||
|
||||
|
||||
def table_foreign_keys(
|
||||
physical: PhysicalSchema, annotations: Annotations, table: str
|
||||
) -> list[ForeignKey]:
|
||||
"""FK fisiche + FK logiche dalle annotations (dedup su columns/ref)."""
|
||||
fks = list(physical.tables[table].foreign_keys)
|
||||
ann = annotations.tables.get(table)
|
||||
if ann:
|
||||
seen = {(tuple(f.columns), f.ref_table, tuple(f.ref_columns)) for f in fks}
|
||||
for fk in ann.foreign_keys:
|
||||
if (tuple(fk.columns), fk.ref_table, tuple(fk.ref_columns)) not in seen:
|
||||
fks.append(fk)
|
||||
return fks
|
||||
|
||||
|
||||
def _ann_col(annotations: Annotations, table: str, column: str) -> ColumnAnnotation | None:
|
||||
ann = annotations.tables.get(table)
|
||||
if ann is None:
|
||||
@@ -59,7 +73,7 @@ def to_mschema_text(
|
||||
shown = ", ".join(column.examples[:MAX_EXAMPLES_IN_PROMPT])
|
||||
lines.append(f" -- Examples: {shown}")
|
||||
lines.append(");")
|
||||
for fk in table.foreign_keys:
|
||||
for fk in table_foreign_keys(physical, annotations, table_name):
|
||||
for src, dst in zip(fk.columns, fk.ref_columns):
|
||||
fk_lines.append(f"{table_name}.{src}={fk.ref_table}.{dst}")
|
||||
lines.extend(["", "【Foreign keys】", *fk_lines])
|
||||
@@ -89,7 +103,7 @@ def to_schema_dict(
|
||||
"primary_keys": [c for c in cols if table.columns[c].pk],
|
||||
"foreign_keys": [
|
||||
{"columns": fk.columns, "ref_table": fk.ref_table, "ref_columns": fk.ref_columns}
|
||||
for fk in table.foreign_keys
|
||||
for fk in table_foreign_keys(physical, annotations, table_name)
|
||||
],
|
||||
}
|
||||
return out
|
||||
@@ -132,9 +146,10 @@ def to_markdown(physical: PhysicalSchema, annotations: Annotations | None = None
|
||||
f"| {column_name} | {column.type} | {'sì' if column.nullable else 'no'} "
|
||||
f"| {'sì' if column.pk else ''} | {cdesc} | {examples} |"
|
||||
)
|
||||
if table.foreign_keys:
|
||||
fks = table_foreign_keys(physical, annotations, table_name)
|
||||
if fks:
|
||||
lines += ["", "Foreign keys:"]
|
||||
for fk in table.foreign_keys:
|
||||
for fk in fks:
|
||||
lines.append(
|
||||
f"- ({', '.join(fk.columns)}) → {fk.ref_table} ({', '.join(fk.ref_columns)})"
|
||||
)
|
||||
|
||||
@@ -97,8 +97,12 @@ def combined_search(
|
||||
top: int,
|
||||
rrf_k: int,
|
||||
kinds: list[str] | None,
|
||||
query_vec: list[float] | None = None,
|
||||
) -> list[SearchResult]:
|
||||
"""Fonde LSH (valori di campo) e pgvector con Reciprocal Rank Fusion."""
|
||||
"""Fonde LSH (valori di campo) e pgvector con Reciprocal Rank Fusion.
|
||||
|
||||
`query_vec` permette di riusare un embedding gia' calcolato della stessa
|
||||
keyword (es. `tht search pack`, che fa piu' ricerche sulla stessa domanda)."""
|
||||
rankings: dict[str, list[tuple[str, float]]] = {}
|
||||
lsh_values: dict[str, str] = {}
|
||||
if lsh_hits:
|
||||
@@ -106,7 +110,9 @@ def combined_search(
|
||||
rankings["lsh"] = [(key, score) for key, score, _ in aggregated]
|
||||
lsh_values = {key: value for key, _, value in aggregated}
|
||||
|
||||
vector_hits = store.search(embedder.embed_query(keyword), top_n=top * 2, kinds=kinds)
|
||||
if query_vec is None:
|
||||
query_vec = embedder.embed_query(keyword)
|
||||
vector_hits = store.search(query_vec, top_n=top * 2, kinds=kinds)
|
||||
rankings["vector"] = [(_vector_key(h), h.similarity) for h in vector_hits]
|
||||
by_key = {_vector_key(h): h for h in vector_hits}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user