feat(opt): three efficiency levers for NL→SQL workflow
Lever 1: Join-graph via FK logics in annotations + suggest-fks command
- TableAnnotation.foreign_keys field stores curated logical FKs (DWH has no FK constraints)
- tht schema suggest-fks: mine from approved SQL, heuristics (time_key → dim_time),
same-name discovery + explicit --assume flag for multi-owner PKs
- mschema renders 【Foreign keys】 section populated; validation in merge.py
- SKILL.md F4 now reads FKs from mschema-text, no custom data_time_key logic
Lever 2: Context-pack consolidation at kickoff (tht search pack)
- Single embedding of question, reused for schema + evidence + solved searches
- One command: tht search pack <question> --session <id> → retrieval_pack.md
- Graceful degradation when Ollama/vector store unreachable (exit 0, empty sections)
- SKILL.md F1 prescribes as first call; reduces model thinking turns via pre-retrieval
Lever 3: Phase-summary recap v2 auto-construction from session ledger
- tht session show --json includes full decisions ledger
- tht phase meta --json exports 'emits' (substantive decision types per phase)
- Gate appends deterministic 【Decisioni registrate in questa fase】 section (appendLedgerSection)
- Model authors only summary + checks; recap table comes from persisted state (exact by construction)
- SKILL.md Disciplina 6: brief model output, gate fills the rest
Tests: 358 Python (including 10 FK + 3 pack + 1 session-ledger tests) + 111 JS gate tests, all pass.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,115 @@
|
||||
import json
|
||||
from datetime import datetime
|
||||
from types import SimpleNamespace
|
||||
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from tht.cli import app
|
||||
from tht.mschema.models import ColumnPhysical, PhysicalSchema, TablePhysical
|
||||
from tht.vectorstore.embeddings import EmbeddingsError
|
||||
|
||||
|
||||
class _FakeEmbedder:
|
||||
def __init__(self):
|
||||
self.calls = 0
|
||||
|
||||
def embed_query(self, text):
|
||||
self.calls += 1
|
||||
return [0.1, 0.2, 0.3]
|
||||
|
||||
|
||||
class _FakeSearcher:
|
||||
def search(self, vec, top_n, kinds=None):
|
||||
if kinds == ["solved_question"]:
|
||||
return [SimpleNamespace(
|
||||
kind="memory", ref="s-1", id="m1", title="q solved",
|
||||
similarity=0.91, content="quanti pazienti nel 2024?",
|
||||
metadata={"session_id": "2026-01-01-000000-x", "sql": "SELECT 1",
|
||||
"tables": ["fact_ablazione"], "question": "quanti pazienti nel 2024?"},
|
||||
)]
|
||||
if kinds == ["schema_table", "schema_column"]:
|
||||
return [SimpleNamespace(
|
||||
kind="schema_table", ref="fact_ablazione", id="t1",
|
||||
title="Tabella fact_ablazione", similarity=0.88,
|
||||
content="Tabella fact_ablazione", metadata={},
|
||||
)]
|
||||
if kinds == ["evidence"]:
|
||||
return [SimpleNamespace(
|
||||
kind="evidence", ref="ev1", id="ev1", title="Dominio ablazione",
|
||||
similarity=0.8, content="L'ablazione e' una procedura...",
|
||||
metadata={"status": "approved"},
|
||||
)]
|
||||
return []
|
||||
|
||||
|
||||
def _workspace(tmp_path, with_session=None):
|
||||
PhysicalSchema(
|
||||
database="d", schema="s", introspected_at=datetime(2026, 1, 1),
|
||||
tables={"fact_ablazione": TablePhysical(
|
||||
comment="Ablazioni", columns={"cod_paz": ColumnPhysical(type="bigint")})},
|
||||
).to_yaml(tmp_path / "artifacts" / "mschema" / "physical.yaml")
|
||||
cfg = tmp_path / "workspace.yaml"
|
||||
cfg.write_text(
|
||||
"database: {database: d, schema: s, user: u, password: p, transport: direct}\n"
|
||||
"vector_db: {database: v, schema: public, user: u, password: p}\n"
|
||||
"embeddings: {base_url: 'http://localhost:11434', model: nomic-embed-text, dim: 8}\n"
|
||||
f"paths: {{artifacts: {tmp_path/'artifacts'}, indexes: {tmp_path/'i'}, "
|
||||
f"sessions: {tmp_path/'sessions'}}}\n"
|
||||
)
|
||||
if with_session:
|
||||
sdir = tmp_path / "sessions" / with_session
|
||||
sdir.mkdir(parents=True)
|
||||
(sdir / "session_manifest.yaml").write_text(
|
||||
f"id: {with_session}\nquestion: q\ndatabase: d\nschema: s\n"
|
||||
"created_at: 2026-01-01T00:00:00+00:00\nstatus: open\n"
|
||||
)
|
||||
return cfg
|
||||
|
||||
|
||||
def _patch(monkeypatch, embedder, searcher):
|
||||
import tht.cli.vector_cmd as vc
|
||||
|
||||
monkeypatch.setattr(vc, "make_embedder", lambda _cfg: embedder)
|
||||
monkeypatch.setattr(vc, "open_searcher", lambda _cfg: searcher)
|
||||
|
||||
|
||||
def test_pack_single_embed_and_sections(tmp_path, monkeypatch):
|
||||
cfg = _workspace(tmp_path)
|
||||
emb = _FakeEmbedder()
|
||||
_patch(monkeypatch, emb, _FakeSearcher())
|
||||
res = CliRunner().invoke(app, ["search", "pack", "quanti pazienti", "-c", str(cfg)])
|
||||
assert res.exit_code == 0, res.output
|
||||
assert emb.calls == 1 # UN solo embedding per le tre ricerche
|
||||
assert "fact_ablazione" in res.output and "Ablazioni" in res.output
|
||||
assert "Dominio ablazione" in res.output
|
||||
assert "SELECT 1" in res.output
|
||||
|
||||
|
||||
def test_pack_json_and_session_file(tmp_path, monkeypatch):
|
||||
sid = "2026-01-01-000000-test"
|
||||
cfg = _workspace(tmp_path, with_session=sid)
|
||||
_patch(monkeypatch, _FakeEmbedder(), _FakeSearcher())
|
||||
res = CliRunner().invoke(
|
||||
app, ["search", "pack", "q", "-c", str(cfg), "--session", sid, "--json"]
|
||||
)
|
||||
assert res.exit_code == 0, res.output
|
||||
data = json.loads(res.output)
|
||||
assert data["tables"][0]["name"] == "fact_ablazione"
|
||||
pack = tmp_path / "sessions" / sid / "retrieval_pack.md"
|
||||
assert pack.exists()
|
||||
assert "Retrieval pack" in pack.read_text()
|
||||
|
||||
|
||||
def test_pack_degrades_gracefully(tmp_path, monkeypatch):
|
||||
cfg = _workspace(tmp_path)
|
||||
|
||||
class _Broken:
|
||||
def embed_query(self, text):
|
||||
raise EmbeddingsError("ollama down")
|
||||
|
||||
_patch(monkeypatch, _Broken(), _FakeSearcher())
|
||||
res = CliRunner().invoke(app, ["search", "pack", "q", "-c", str(cfg), "--json"])
|
||||
assert res.exit_code == 0, res.output
|
||||
data = json.loads(res.output[res.output.index("{"):])
|
||||
assert data["tables"] == [] and data["evidence"] == [] and data["solved"] == []
|
||||
assert any("retrieval non disponibile" in w for w in data["warnings"])
|
||||
Reference in New Issue
Block a user