feat(opt): three efficiency levers for NL→SQL workflow

Lever 1: Join-graph via FK logics in annotations + suggest-fks command
  - TableAnnotation.foreign_keys field stores curated logical FKs (DWH has no FK constraints)
  - tht schema suggest-fks: mine from approved SQL, heuristics (time_key → dim_time),
    same-name discovery + explicit --assume flag for multi-owner PKs
  - mschema renders 【Foreign keys】 section populated; validation in merge.py
  - SKILL.md F4 now reads FKs from mschema-text, no custom data_time_key logic

Lever 2: Context-pack consolidation at kickoff (tht search pack)
  - Single embedding of question, reused for schema + evidence + solved searches
  - One command: tht search pack <question> --session <id> → retrieval_pack.md
  - Graceful degradation when Ollama/vector store unreachable (exit 0, empty sections)
  - SKILL.md F1 prescribes as first call; reduces model thinking turns via pre-retrieval

Lever 3: Phase-summary recap v2 auto-construction from session ledger
  - tht session show --json includes full decisions ledger
  - tht phase meta --json exports 'emits' (substantive decision types per phase)
  - Gate appends deterministic 【Decisioni registrate in questa fase】 section (appendLedgerSection)
  - Model authors only summary + checks; recap table comes from persisted state (exact by construction)
  - SKILL.md Disciplina 6: brief model output, gate fills the rest

Tests: 358 Python (including 10 FK + 3 pack + 1 session-ledger tests) + 111 JS gate tests, all pass.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-07 17:43:08 +02:00
co-authored by Claude Fable 5
parent 87e875bc81
commit e24b41b156
19 changed files with 936 additions and 29 deletions
@@ -199,3 +199,32 @@ test("enrichPhaseSummaryV2 preserves open_questions/checks untouched", () => {
assert.deepEqual(out.checks, [{ label: "x", status: "ok" }]);
assert.deepEqual(out.open_questions, ["domanda aperta"]);
});
// --- appendLedgerSection -----------------------------------------------------------
const { appendLedgerSection } = require("../enrich.js");
test("appendLedgerSection appends only the phase's substantive decisions", () => {
const data = { schema_version: 2, summary: "s", sections: [{ title: "Criteri", items: [] }] };
const decisions = [
{ seq: 1, type: "concept_clarified", subject: "ablazione", detail: "solo transcatetere", rationale: "r1" },
{ seq: 2, type: "phase_approved", subject: "phase:1", detail: "", rationale: "" }, // meta: non in emits
{ seq: 3, type: "table_promoted", subject: "dim_patient", detail: "", rationale: "" }, // altra fase
];
const out = appendLedgerSection(data, decisions, ["concept_clarified", "ambiguity_open"]);
assert.equal(out.sections.length, 2);
const ledger = out.sections[1];
assert.equal(ledger.title, "Decisioni registrate in questa fase (dal ledger)");
assert.equal(ledger.items.length, 1);
assert.equal(ledger.items[0].label, "concept_clarified: ablazione");
assert.equal(ledger.items[0].value, "solo transcatetere");
// input non mutato
assert.equal(data.sections.length, 1);
});
test("appendLedgerSection is a no-op without matching decisions or with empty ledger", () => {
const data = { schema_version: 2, summary: "s" };
assert.equal(appendLedgerSection(data, [], ["concept_clarified"]), data);
assert.equal(appendLedgerSection(data, [{ type: "sql_approved", subject: "x" }], ["concept_clarified"]), data);
assert.equal(appendLedgerSection(data, undefined, undefined), data);
});
+22
View File
@@ -151,10 +151,32 @@ function enrichCteResultColumns(payload, planTables, getColumns) {
return { ...payload, columns };
}
// Contract C bis. Appends a deterministic "decisions of this phase" section built
// from the session ledger, so the model's recap can stay thin (Discipline 6 exactness
// comes from persisted state, not model prose). `decisions` is session-show's ledger
// dump; `emits` is the phase's substantive decision-type list from workflow.yaml.
// No matching decisions -> data returned unchanged. Pure: returns a NEW object.
function appendLedgerSection(data, decisions, emits) {
const emitSet = new Set(emits || []);
const rows = (decisions || []).filter((d) => d && emitSet.has(d.type));
if (rows.length === 0) return data;
const section = {
title: "Decisioni registrate in questa fase (dal ledger)",
items: rows.map((d) => ({
label: `${d.type}: ${d.subject}`,
value: d.detail || "",
kind: "decision",
rationale: d.rationale || "",
})),
};
return { ...data, sections: [...(data.sections || []), section] };
}
module.exports = {
memoizeGetColumns,
enrichCtePlanV2,
buildCteResultV2,
enrichCteResultColumns,
enrichPhaseSummaryV2,
appendLedgerSection,
};
+17 -1
View File
@@ -42,6 +42,7 @@ import {
buildCteResultV2,
enrichCteResultColumns,
enrichPhaseSummaryV2,
appendLedgerSection,
} from "./gate/enrich.js";
import { isReserved } from "./reserved-labels.mjs";
@@ -898,11 +899,26 @@ export default function (pi) {
"Riepilogo di fase v2 non valido:\n- " + v.errors.join("\n- "),
);
}
const enriched = enrichPhaseSummaryV2(
let enriched = enrichPhaseSummaryV2(
data,
phaseMetaForNum(ctx, curNum),
makeGetColumns(ctx),
);
// Recap deterministico: appende la sezione "decisioni di questa fase"
// dal ledger (best-effort: un errore di lettura non blocca il gate).
try {
const show = JSON.parse(
tht(ctx, ["session", "show", session, "--json"]),
);
const meta = phaseMeta(ctx).phases.find((p) => p.num === curNum);
enriched = appendLedgerSection(
enriched,
show.decisions,
meta ? meta.emits : [],
);
} catch {
// ledger non leggibile: il riepilogo resta quello del modello
}
artifact = { ...artifact, data: enriched };
}
+38 -16
View File
@@ -81,7 +81,10 @@ substantive decisions.
For a phase-closing gate (`reviewer_confirm kind:"phase"`) prefer the **structured
v2 recap** `artifact:{kind:"phase", data:{schema_version:2, …}}`: you author
`summary` (1-3 sentence markdown), `checks[]`, `sections[]` and `tables[]`; the gate
fills `phase` (from workflow meta) and every `description` from the catalog. Every
fills `phase` (from workflow meta) and every `description` from the catalog, and
APPENDS a deterministic section "Decisioni registrate in questa fase (dal ledger)"
— do NOT re-enumerate the phase's recorded decisions yourself: author only the
summary, the checks and the context the ledger cannot express. Every
`sections[].items[]` MUST cite the concrete **table**, **column** and the **value**
that motivates the choice (booleans, time windows, thresholds) — not just prose.
Compact example:
@@ -124,6 +127,13 @@ substantive decisions.
11. **Rollback (D15).** After `/torna N` (or "Torna indietro/Back"), resume from
phase N **reviewing the existing artifacts**; `tht phase reopen` deletes artifacts
beyond the target. Do NOT re-run `tht` commands for artifacts that are still valid.
12. **This skill is the complete contract.** Every command, flag and behavior you
need is named in this skill and its reference docs (`rewriting.md`, `cte.md`,
`sql-generation.md`). Do NOT run `--help`, do NOT read the harness source
(`tht/`, `.pi/extensions/`, tests) to figure out how a command works, and do NOT
explore the filesystem with `find`/`grep`/`cat` for that purpose. If something
genuinely seems missing or a command behaves unexpectedly, say so to the reviewer
instead of reverse-engineering the tooling.
## Phase 0 — Resume (cold start)
@@ -148,21 +158,25 @@ it is complete. (The backend already refuses resume for finalized/archived sessi
Prerequisite: you must already be in Phase 1.
**F1 toolbox.** The only commands you need here are `tht search find` and `tht schema
render` — both fast, read-only lookups over workspace artifacts already on disk.
Evidence lives in `<workspace>/evidence/**` and is what `tht search find --kind evidence`
returns — do not browse it with `find`/`cat`. Do NOT run `tht schema introspect`: it is
a maintenance command that re-reads the remote DWH (~3 minutes); the catalog
`artifacts/mschema/physical.yaml` is already in the workspace. Do NOT explore with
`--help` or ad-hoc shell commands — every command you need is named in this skill.
**F1 toolbox.** The only commands you need here are `tht search pack`, `tht search
find` and `tht schema render` — all fast, read-only lookups over workspace artifacts
already on disk. Evidence lives in `<workspace>/evidence/**` and is what `tht search
find --kind evidence` returns — do not browse it with `find`/`cat`. Do NOT run `tht
schema introspect`: it is a maintenance command that re-reads the remote DWH (~3
minutes); the catalog `artifacts/mschema/physical.yaml` is already in the workspace.
Do NOT explore with `--help` or ad-hoc shell commands — every command you need is
named in this skill.
1. Explore the DWH and knowledge base: `tht search find "<term>"` (evidence + schema, LSH
over real values) and `tht search find --kind evidence "<term>"`. The LSH exposes
EVERY column where a value appears — it does not collapse to a single best match,
so a value like "ablazione" may anchor on multiple columns.
First list ALL the ambiguous terms in the question, then run the `tht search find`
calls for every term in ONE batch (a single message with multiple shell invocations)
— not one lookup per turn.
1. **First call, one shot:** `tht search pack "<original question>" --session <id>` —
it bundles candidate tables, relevant evidence and similar solved questions for the
WHOLE question in a single command (one embedding, three searches) and persists
`retrieval_pack.md` in the session. Read it before anything else; it usually
answers "which tables/evidence matter here" without further exploration.
Then ground the individual ambiguous terms: list ALL of them and run the
`tht search find "<term>"` / `tht search find --kind evidence "<term>"` calls in
ONE batch (a single message with multiple shell invocations) — not one lookup per
turn. The LSH exposes EVERY column where a value appears — it does not collapse
to a single best match, so a value like "ablazione" may anchor on multiple columns.
2. For each ambiguity (clinical term, population, time window, outcome), present the
candidate interpretations (`recommended:true` on the best) + "Altro". Pick the widget
by the question's shape:
@@ -244,6 +258,10 @@ Prerequisite: Phase 3 closed.
1. `tht schema render --format mschema-text` for the schema context (the catalog
`artifacts/mschema/physical.yaml` is already in the workspace; only if render fails
with `physical.yaml non trovato`, run `tht schema introspect` once, then render).
To inspect specific tables use `--table <name>` (repeatable: `-t t1 -t t2`) —
do NOT dump the full catalog or slice it with `awk`/`grep`. The session's
`retrieval_pack.md` (built in F1) already lists the candidate tables for the
question — start from those.
Copy table/column names EXACTLY from it — never invent objects.
Also run `tht memory solved-search "<question>" --json`: similar already-solved
questions show which tables comparable questions used. Cite relevant precedents
@@ -262,7 +280,11 @@ Prerequisite: Phase 3 closed.
to reference other columns as join keys or filter predicates when the query
requires them.
Propose joins separately in `reviewer_decide(advance:false)`, registering
`join_modified`.
`join_modified`. Ground them in the `【Foreign keys】` section of the mschema-text
render: it lists the curated logical FKs of the workspace (e.g.
`fact_x.cod_paz=dim_patient.cod_paz`, `*_time_key=dim_time.day_key`) — prefer
those to joins you derive yourself, and flag to the reviewer any join you need
that is NOT in the list.
3. **Value grounding (D14a).** If a cited value (e.g. "ablazione") matches MULTIPLE
columns (a boolean flag + a free-text patologia field), present a `reviewer_decide`
with a `value_grounded` option for each candidate column (the LSH exposes all of
+20 -2
View File
@@ -13,8 +13,8 @@ Rules (from the AV-SQL discipline, hold verbatim):
3. Better one column too many than one too few: if unsure, include it.
4. One CTE = one informative subset with a clear purpose (e.g. "ricoveri with
ablazione in 2025"), named in a speaking snake_case.
5. CTEs can chain-reference each other; the last one in the file is the one that
`tht cte test` will query.
5. CTEs can chain-reference each other **within the same file**; the last one in the
file is the one that `tht cte test` will query (see the execution contract below).
6. Each file in `sessions/<id>/ctes/<name>.sql` contains ONLY the `WITH ... AS (...)`
block (multi-CTE allowed), WITHOUT a trailing SELECT. A `SELECT ...` line after
the WITH block causes an error in `tht cte test`: never add it. CTEs are tested
@@ -24,6 +24,24 @@ Rules (from the AV-SQL discipline, hold verbatim):
7. Filters: use field values verified with `tht search` (LSH match on real values),
not imagined values.
## How `tht cte test` executes (complete contract — do not read the harness source)
- Each CTE file is **standalone**: the test reads ONLY `ctes/<name>.sql`, appends
`SELECT * FROM <last CTE defined in that file>` and runs it read-only against the
DWH with an injected LIMIT and statement timeout. Cross-file references are NOT
resolved: to build on an earlier CTE, repeat its definition in the same `WITH`
chain (that is why multi-CTE files are allowed). A test normally completes in
well under a second.
- Before execution the SQL is validated statically: parsable, a single statement,
read-only by structure, no blacklisted functions, and every referenced table must
exist in the catalog (names defined in the `WITH` chain are exempt). Tables outside
the promoted perimeter produce warnings. Unqualified columns are not statically
checked — the DWH will catch them at run time.
- Test order is enforced by the CLI from the approved plan (exit 5 names the CTE
whose turn it is). Every outcome (ok or error) is appended to `cte_tests.json`;
the gate reads the persisted file + last test via `tht cte info`, so never paste
SQL, columns or preview rows into the gate text.
Presentation to the reviewer, for each CTE in the plan:
> **<name>** — purpose: <one line>
@@ -21,9 +21,9 @@ copy-pasteable: no rationale comments (that lives in the audit artifacts).
## Time dimension (analysis by year/month/quarter)
Fact tables have `data_time_key` (`integer`, format `YYYYMMDD`): it is the FK to
`dim_time.day_key`. **This FK is NOT declared** in the DWH (facts have
`foreign_keys: []`), so it will NOT appear in `schema_linking.json`: you must add it
by hand to the join.
`dim_time.day_key`. The DWH does not declare it, but the workspace annotations do:
it appears in the `【Foreign keys】` section of the mschema-text render (every
`*_time_key` column maps to `dim_time.day_key`) — take it from there for the join.
- To extract year, month, quarter, semester etc. do
`JOIN dim_time dt ON dt.day_key = <fact>.data_time_key` and use the dimension's