test(workflow): complete observable baseline (#21)

This commit is contained in:
2026-08-24 00:40:11 +02:00
parent dc9726cb35
commit d15bb59c3d
4 changed files with 626 additions and 173 deletions
+41 -14
View File
@@ -12,20 +12,26 @@ baseline green without weakening its assertions.
### Pi gate ### Pi gate
Run `npm test` from the harness package. From `harness/`, run `npm test` with the repository's supported Node 24 runtime.
The gate suite fixes: The gate suite fixes:
- the registered Pi tool names and their required and optional parameters; - the complete registered Pi tool schemas, including nested types and enum-like constraints;
- the exact workflow definition and injected session skill bytes; - the semantic workflow definition and the exact injected session skill bytes;
- widget descriptors and reviewer response semantics; - widget descriptors and reviewer response semantics;
- F1 clarification and explicitly accepted open ambiguity; - F1 clarification and explicitly accepted open ambiguity;
- F2 Memory applied, deselected, and absent; - F2 Memory applied, deselected, and absent;
- F3 rewritten question and assumptions, including mutation failure ordering; - F3 rewritten question and assumptions, including mutation failure ordering;
- F4 Evidence acceptance and rejection; - F4 Evidence used, accepted, rejected, and legacy-without-corpus projections;
- F8 Memory promotion accepted, declined, and absent, including mutation failure ordering; - F8 Memory promotion accepted, declined, and absent, including mutation failure ordering;
- resume reconstruction for the touched F1, F2, F3, F4, and F8 states;
- artifact payload compatibility, anti-bypass behavior, and final phase closing. - artifact payload compatibility, anti-bypass behavior, and final phase closing.
The baseline intentionally checks widget structure and domain content without freezing the
pre-existing Italian chrome emitted by the gate. Repository policy requires UI chrome and labels
to migrate to English in their owning workstream; this contract must not turn that mismatch into a
new compatibility requirement.
### Harness CLI and persistence ### Harness CLI and persistence
Run the default pytest suite from the harness package. The suite fixes: Run the default pytest suite from the harness package. The suite fixes:
@@ -42,7 +48,18 @@ working local Docker daemon and remain part of the default suite when Docker is
### Backend bridge ### Backend bridge
Run the backend test suite followed by TypeScript typechecking. The suite fixes: From `backend/`, the passing automated baseline is:
```sh
npx vitest run test/tht-runner.test.ts test/pi-process-manager.test.ts \
test/session-bridge.test.ts test/sse-hub.test.ts test/sse-route.test.ts \
test/routes-sessions.test.ts test/e2e-f1.test.ts \
test/workspace-preprocessing-service.test.ts test/evidence-materialization.test.ts
npx tsc --noEmit -p .
npm run build
```
These suites fix:
- CLI argument ordering and JSON/error propagation across the runner boundary; - CLI argument ordering and JSON/error propagation across the runner boundary;
- new-session versus resume Pi prompts; - new-session versus resume Pi prompts;
@@ -52,7 +69,18 @@ Run the backend test suite followed by TypeScript typechecking. The suite fixes:
### Frontend client ### Frontend client
Run the frontend test suite followed by TypeScript typechecking. The suite fixes: From `frontend/`, the passing automated baseline is:
```sh
npx vitest run src/store/sessionStore.test.ts src/stream/useSessionStream.test.tsx \
src/widgets/registry.test.tsx src/widgets/SelectWidget.test.tsx \
src/widgets/MultiselectWidget.test.tsx src/widgets/ArtifactWidget.test.tsx \
src/shell/f1-loop.test.tsx src/shell/SessionDocumentsPanel.test.tsx
npx tsc -b
npm run build
```
These suites fix:
- widget registry and gate response payloads; - widget registry and gate response payloads;
- `ui_request`, `text_delta`, activity, usage, and lifecycle event reduction; - `ui_request`, `text_delta`, activity, usage, and lifecycle event reduction;
@@ -78,18 +106,17 @@ question to finalization, followed by resume verification, belongs to the final
ticket. If its environment or credentials are unavailable, it must remain recorded as a pending ticket. If its environment or credentials are unavailable, it must remain recorded as a pending
manual gate rather than being reported as passed. manual gate rather than being reported as passed.
## Pre-existing full-suite exceptions ## Full-suite diagnostic exceptions
The workflow baseline and every focused seam above pass on the source commit from which this Every command defined above as part of the automated baseline exits successfully. Running the
branch was created. Two unrelated full-suite failures also reproduce unchanged on that base broader backend and frontend suites is still useful as a diagnostic, but those full suites are not
checkout and are therefore recorded rather than hidden or repaired in this refactoring ticket: the executable acceptance gate for this ticket because two unrelated failures reproduce unchanged
on the source commit from which this branch was created:
- the backend authentication runtime-projection suite currently rejects ten positive fixtures - the backend authentication runtime-projection suite currently rejects ten positive fixtures
with its fail-closed public error; with its fail-closed public error;
- one frontend application-shell authentication test does not render the expected trusted-upstream - one frontend application-shell authentication test does not render the expected trusted-upstream
display name. display name.
The focused backend workflow suite, backend typecheck and build, focused frontend workflow suite, These two exceptions must remain visible until their owning workstream resolves them; they must not
frontend typecheck and build, complete harness pytest suite, Ruff, and complete Pi gate suite all be used to relax any workflow assertion or to describe a nonzero command as a passing baseline.
pass. These two exceptions must remain visible until their owning workstream resolves them; they
must not be used to relax any workflow assertion.
@@ -42,23 +42,45 @@ const PHASE_META = JSON.stringify({
function useShell({ phase, preview = [], fail = () => null }) { function useShell({ phase, preview = [], fail = () => null }) {
const calls = []; const calls = [];
shell.current = (_file, args, options = {}) => { shell.current = (_file, args, options = {}) => {
const command = args.join(" "); calls.push({ args: [...args], input: options.input });
calls.push({ command, input: options.input }); const failure = fail(args);
const failure = fail(command);
if (failure) { if (failure) {
const error = new Error(failure.stderr); const error = new Error(failure.stderr);
error.status = failure.status; error.status = failure.status;
error.stderr = failure.stderr; error.stderr = failure.stderr;
throw error; throw error;
} }
if (command === "phase meta --json") return PHASE_META; if (sameArgs(args, ["phase", "meta", "--json"])) return PHASE_META;
if (command.startsWith("phase show --session ")) return `Fase corrente: ${phase}\n`; if (startsWithArgs(args, ["phase", "show", "--session"])) return `Fase corrente: ${phase}\n`;
if (command.startsWith("memory promote --session ")) return JSON.stringify(preview); if (startsWithArgs(args, ["memory", "promote", "--session"])) return JSON.stringify(preview);
return ""; return "";
}; };
return calls; return calls;
} }
function sameArgs(actual, expected) {
return actual.length === expected.length && actual.every((value, index) => value === expected[index]);
}
function startsWithArgs(actual, prefix) {
return prefix.every((value, index) => actual[index] === value);
}
function mutationArgs(calls, prefixes) {
return calls
.filter(({ args }) => prefixes.some((prefix) => startsWithArgs(args, prefix)))
.map(({ args }) => args);
}
function memoryPromotionMutationArgs(calls) {
return mutationArgs(calls, [
["memory", "save-one"],
["decision", "add"],
["phase", "advance"],
["session", "finalize"],
]);
}
async function setupGate(shellOptions) { async function setupGate(shellOptions) {
const calls = useShell(shellOptions); const calls = useShell(shellOptions);
const runtime = createFakePi(); const runtime = createFakePi();
@@ -83,65 +105,27 @@ test("the Pi gate exposes the approved tool names and parameter boundaries", ()
const publicContract = [...tools].map(([name, { def }]) => ({ const publicContract = [...tools].map(([name, { def }]) => ({
name, name,
required: def.parameters.required ?? [], parameters: stripDescriptions(def.parameters),
properties: Object.keys(def.parameters.properties),
})); }));
assert.deepEqual(publicContract, [ const expected = JSON.parse(fs.readFileSync(
{ path.join(__dirname, "golden", "pi_tool_schemas.json"),
name: "reviewer_select", "utf8",
required: ["session", "title", "options"], ));
properties: ["session", "title", "options", "intro", "advance"], assert.deepEqual(publicContract, expected);
},
{
name: "reviewer_datamart",
required: ["session"],
properties: ["session"],
},
{
name: "reviewer_decide",
required: ["session", "title", "options"],
properties: ["session", "title", "options", "allow_empty", "advance"],
},
{
name: "reviewer_schema_linking",
required: ["session", "title", "tables"],
properties: ["session", "title", "tables", "advance"],
},
{
name: "reviewer_confirm",
required: ["session", "kind", "title", "artifact"],
properties: ["session", "kind", "title", "artifact", "names"],
},
{
name: "reviewer_memory_promote",
required: ["session"],
properties: ["session"],
},
{
name: "rewrite_question",
required: ["session", "question"],
properties: ["session", "question", "assumptions"],
},
{
name: "write_schema_linking",
required: ["session", "schema_linking"],
properties: ["session", "schema_linking"],
},
{
name: "write_cte_sql",
required: ["session", "name", "sql"],
properties: ["session", "name", "sql"],
},
{
name: "write_final_sql",
required: ["session", "sql"],
properties: ["session", "sql"],
},
]);
}); });
test("the injected skill and workflow definition stay byte-identical during extraction", () => { function stripDescriptions(value, insideProperties = false) {
if (Array.isArray(value)) return value.map((item) => stripDescriptions(item));
if (value && typeof value === "object") {
return Object.fromEntries(Object.entries(value)
.filter(([key]) => insideProperties || key !== "description")
.map(([key, item]) => [key, stripDescriptions(item, key === "properties")]));
}
return value;
}
test("the injected skill stays byte-identical during extraction", () => {
const harnessRoot = path.resolve(__dirname, "..", "..", "..", ".."); const harnessRoot = path.resolve(__dirname, "..", "..", "..", "..");
const digest = (relativePath) => crypto const digest = (relativePath) => crypto
.createHash("sha256") .createHash("sha256")
@@ -152,10 +136,6 @@ test("the injected skill and workflow definition stay byte-identical during extr
digest(path.join(".pi", "skills", "tht-sessione", "SKILL.md")), digest(path.join(".pi", "skills", "tht-sessione", "SKILL.md")),
"626a794071c095a4f20fffabb3bab901f05c101590adbdc58e45adfae56f3219", "626a794071c095a4f20fffabb3bab901f05c101590adbdc58e45adfae56f3219",
); );
assert.equal(
digest("workflow.yaml"),
"a0604c3dfac7960dcfbf4a584383a602ee953a9fffac97a469281ceef1284d38",
);
}); });
test("F1 persists a concrete clarification directly from the select widget", async () => { test("F1 persists a concrete clarification directly from the select widget", async () => {
@@ -214,13 +194,13 @@ test("F1 persists a concrete clarification directly from the select widget", asy
reserved: ["back", "exit", "other"], reserved: ["back", "exit", "other"],
}, },
); );
assert.ok(calls.some(({ command }) => command === [ assert.ok(calls.some(({ args }) => sameArgs(args, [
"decision add --session s1", "decision", "add", "--session", "s1",
"--type concept_clarified", "--type", "concept_clarified",
"--subject ablazione", "--subject", "ablazione",
"--detail procedura clinica", "--detail", "procedura clinica",
"--rationale scelta dal reviewer", "--rationale", "scelta dal reviewer",
].join(" "))); ])));
assert.match(result.content[0].text, /Decisione registrata \(concept_clarified\)/); assert.match(result.content[0].text, /Decisione registrata \(concept_clarified\)/);
}); });
@@ -240,7 +220,7 @@ test("F1 ask-only and control choices do not mutate the ledger", async () => {
ctx, ctx,
); );
assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false); assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false);
assert.match(result.content[0].text, /Scelta del reviewer: Chiedi un dettaglio/); assert.match(result.content[0].text, /Scelta del reviewer: Chiedi un dettaglio/);
}); });
@@ -269,13 +249,13 @@ test("F1 persists an explicitly accepted open ambiguity", async () => {
ctx, ctx,
); );
assert.equal(calls.some(({ command }) => command === [ assert.equal(calls.some(({ args }) => sameArgs(args, [
"decision add --session s1", "decision", "add", "--session", "s1",
"--type ambiguity_open", "--type", "ambiguity_open",
"--subject termine clinico", "--subject", "termine clinico",
"--detail nessuna definizione conclusiva", "--detail", "nessuna definizione conclusiva",
"--rationale rischio accettato dal reviewer", "--rationale", "rischio accettato dal reviewer",
].join(" ")), true); ])), true);
assert.match(result.content[0].text, /Decisione registrata \(ambiguity_open\)/); assert.match(result.content[0].text, /Decisione registrata \(ambiguity_open\)/);
}); });
@@ -297,7 +277,7 @@ const MEMORY_OPTIONS = [
test("F2 applies selected Memory content and keeps the substantive phase open", async () => { test("F2 applies selected Memory content and keeps the substantive phase open", async () => {
const { ctx, tools, calls } = await setupGate({ const { ctx, tools, calls } = await setupGate({
phase: 2, phase: 2,
fail: (command) => command.startsWith("phase advance --auto") fail: (args) => startsWithArgs(args, ["phase", "advance", "--auto"])
? { status: 6, stderr: "phase not auto-eligible" } ? { status: 6, stderr: "phase not auto-eligible" }
: null, : null,
}); });
@@ -322,21 +302,15 @@ test("F2 applies selected Memory content and keeps the substantive phase open",
{ {
phase: descriptor.phase, phase: descriptor.phase,
widget: descriptor.widget, widget: descriptor.widget,
title: descriptor.title,
allowEmpty: descriptor.allow_empty, allowEmpty: descriptor.allow_empty,
selected: descriptor.selected, selected: descriptor.selected,
selectionLabel: descriptor.selection_label,
confirmLabel: descriptor.confirm_label,
options: descriptor.options, options: descriptor.options,
}, },
{ {
phase: "F2", phase: "F2",
widget: "multiselect", widget: "multiselect",
title: "Seleziona le memory da applicare alla domanda",
allowEmpty: true, allowEmpty: true,
selected: ["mem-0042"], selected: ["mem-0042"],
selectionLabel: "memory da applicare",
confirmLabel: "Applica le memory selezionate",
options: [ options: [
{ {
id: "mem-0042", id: "mem-0042",
@@ -349,13 +323,17 @@ test("F2 applies selected Memory content and keeps the substantive phase open",
], ],
}, },
); );
const mutations = calls assert.equal(typeof descriptor.title, "string");
.map(({ command }) => command) assert.equal(typeof descriptor.selection_label, "string");
.filter((command) => command.startsWith("decision ") || command.startsWith("phase advance")); assert.equal(typeof descriptor.confirm_label, "string");
const mutations = mutationArgs(calls, [["decision"], ["phase", "advance"]]);
assert.deepEqual(mutations, [ assert.deepEqual(mutations, [
"decision add --session s1 --type concept_clarified --subject paziente attivo " + [
"--detail flag_attivo = TRUE --rationale Riusa mem-0042 per la stessa definizione", "decision", "add", "--session", "s1", "--type", "concept_clarified",
"phase advance --auto --session s1", "--subject", "paziente attivo", "--detail", "flag_attivo = TRUE",
"--rationale", "Riusa mem-0042 per la stessa definizione",
],
["phase", "advance", "--auto", "--session", "s1"],
]); ]);
assert.match(result.content[0].text, /La fase resta aperta/); assert.match(result.content[0].text, /La fase resta aperta/);
}); });
@@ -378,8 +356,10 @@ test("F2 accepts an empty selection without recording a rejection", async () =>
ctx, ctx,
); );
assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false); assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false);
assert.equal(calls.some(({ command }) => command === "phase advance --auto --session s1"), true); assert.equal(calls.some(({ args }) => sameArgs(
args, ["phase", "advance", "--auto", "--session", "s1"],
)), true);
assert.match(result.content[0].text, /Nessuna decisione registrata.*Fase avanzata/s); assert.match(result.content[0].text, /Nessuna decisione registrata.*Fase avanzata/s);
}); });
@@ -401,19 +381,20 @@ test("F2 with no Memory candidates notifies once and advances without an empty w
ctx, ctx,
); );
assert.deepEqual(ctx.notifications, [{ assert.equal(ctx.notifications.length, 1);
message: "Nessuna memory riutilizzabile per questa domanda — passo alla fase successiva.", assert.equal(ctx.notifications[0].level, "info");
level: "info", assert.equal(typeof ctx.notifications[0].message, "string");
}]); assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false);
assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false); assert.equal(calls.some(({ args }) => sameArgs(
assert.equal(calls.some(({ command }) => command === "phase advance --auto --session s1"), true); args, ["phase", "advance", "--auto", "--session", "s1"],
)), true);
assert.match(result.content[0].text, /Fase memoria vuota/); assert.match(result.content[0].text, /Fase memoria vuota/);
}); });
test("F3 stops before ledger and advance when writing the rewritten question fails", async () => { test("F3 stops before ledger and advance when writing the rewritten question fails", async () => {
const { ctx, tools, calls } = await setupGate({ const { ctx, tools, calls } = await setupGate({
phase: 3, phase: 3,
fail: (command) => command.startsWith("session set-question ") fail: (args) => startsWithArgs(args, ["session", "set-question"])
? { status: 1, stderr: "question write failed" } ? { status: 1, stderr: "question write failed" }
: null, : null,
}); });
@@ -426,12 +407,14 @@ test("F3 stops before ledger and advance when writing the rewritten question fai
ctx, ctx,
); );
const mutations = calls.map(({ command }) => command).filter((command) => const mutations = mutationArgs(calls, [
command.startsWith("session set-question ") || ["session", "set-question"], ["decision", "add"], ["phase", "advance"],
command.startsWith("decision add ") || ]);
command.startsWith("phase advance"));
assert.deepEqual(mutations, [ assert.deepEqual(mutations, [
"session set-question s1 --question Domanda riscritta --assumption Assunzione A", [
"session", "set-question", "s1", "--question", "Domanda riscritta",
"--assumption", "Assunzione A",
],
]); ]);
assert.match(result.content[0].text, /question write failed/); assert.match(result.content[0].text, /question write failed/);
}); });
@@ -439,7 +422,7 @@ test("F3 stops before ledger and advance when writing the rewritten question fai
test("F3 keeps the rewritten question but does not advance when the ledger write fails", async () => { test("F3 keeps the rewritten question but does not advance when the ledger write fails", async () => {
const { ctx, tools, calls } = await setupGate({ const { ctx, tools, calls } = await setupGate({
phase: 3, phase: 3,
fail: (command) => command.startsWith("decision add ") fail: (args) => startsWithArgs(args, ["decision", "add"])
? { status: 1, stderr: "ledger write failed" } ? { status: 1, stderr: "ledger write failed" }
: null, : null,
}); });
@@ -452,14 +435,18 @@ test("F3 keeps the rewritten question but does not advance when the ledger write
ctx, ctx,
); );
const mutations = calls.map(({ command }) => command).filter((command) => const mutations = mutationArgs(calls, [
command.startsWith("session set-question ") || ["session", "set-question"], ["decision", "add"], ["phase", "advance"],
command.startsWith("decision add ") || ]);
command.startsWith("phase advance"));
assert.deepEqual(mutations, [ assert.deepEqual(mutations, [
"session set-question s1 --question Domanda riscritta --assumption Assunzione A", [
"decision add --session s1 --type question_rewritten --subject domanda " + "session", "set-question", "s1", "--question", "Domanda riscritta",
"--detail Domanda riscritta", "--assumption", "Assunzione A",
],
[
"decision", "add", "--session", "s1", "--type", "question_rewritten",
"--subject", "domanda", "--detail", "Domanda riscritta",
],
]); ]);
assert.match(result.content[0].text, /ledger write failed/); assert.match(result.content[0].text, /ledger write failed/);
}); });
@@ -507,12 +494,10 @@ test("F4 persists exactly the reviewer-selected Evidence disposition", async ()
{ id: "accept", label: "Usa la fonte", detail: undefined }, { id: "accept", label: "Usa la fonte", detail: undefined },
{ id: "reject", label: "Scarta la fonte", detail: undefined }, { id: "reject", label: "Scarta la fonte", detail: undefined },
]); ]);
const decisions = calls const decisions = mutationArgs(calls, [["decision", "add"]]);
.map(({ command }) => command)
.filter((command) => command.startsWith("decision add "));
const expectedType = selected === "accept" ? "evidence_accepted" : "evidence_rejected"; const expectedType = selected === "accept" ? "evidence_accepted" : "evidence_rejected";
assert.equal(decisions.length, 1); assert.equal(decisions.length, 1);
assert.match(decisions[0], new RegExp(`--type ${expectedType} `)); assert.equal(decisions[0][decisions[0].indexOf("--type") + 1], expectedType);
assert.match(result.content[0].text, new RegExp(expectedType)); assert.match(result.content[0].text, new RegExp(expectedType));
} }
}); });
@@ -545,17 +530,16 @@ test("F8 saves an accepted Memory before its ledger marker and finalizes", async
assert.equal(descriptor.widget, "multiselect"); assert.equal(descriptor.widget, "multiselect");
assert.deepEqual(descriptor.selected, ["seq-5"]); assert.deepEqual(descriptor.selected, ["seq-5"]);
assert.match(descriptor.content, /flag_attivo = TRUE/); assert.match(descriptor.content, /flag_attivo = TRUE/);
const mutations = calls.map(({ command }) => command).filter((command) => const mutations = memoryPromotionMutationArgs(calls);
command.startsWith("memory save-one ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance") ||
command.startsWith("session finalize"));
assert.deepEqual(mutations, [ assert.deepEqual(mutations, [
"memory save-one --session s1 --decision 5 --json", ["memory", "save-one", "--session", "s1", "--decision", "5", "--json"],
"decision add --session s1 --type memory_promoted --subject paziente attivo " + [
"--detail seq:5 --rationale scelta dal reviewer", "decision", "add", "--session", "s1", "--type", "memory_promoted",
"phase advance --session s1", "--subject", "paziente attivo", "--detail", "seq:5",
"session finalize s1", "--rationale", "scelta dal reviewer",
],
["phase", "advance", "--session", "s1"],
["session", "finalize", "s1"],
]); ]);
assert.match(result.content[0].text, /1 memorie salvate.*sessione finalizzata/s); assert.match(result.content[0].text, /1 memorie salvate.*sessione finalizzata/s);
}); });
@@ -572,16 +556,14 @@ test("F8 records a declined candidate without saving it and finalizes", async ()
ctx, ctx,
); );
const mutations = calls.map(({ command }) => command).filter((command) => const mutations = memoryPromotionMutationArgs(calls);
command.startsWith("memory save-one ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance") ||
command.startsWith("session finalize"));
assert.deepEqual(mutations, [ assert.deepEqual(mutations, [
"decision add --session s1 --type memory_promotion_declined " + [
"--subject paziente attivo --detail seq:5", "decision", "add", "--session", "s1", "--type", "memory_promotion_declined",
"phase advance --session s1", "--subject", "paziente attivo", "--detail", "seq:5",
"session finalize s1", ],
["phase", "advance", "--session", "s1"],
["session", "finalize", "s1"],
]); ]);
assert.match(result.content[0].text, /1 candidati scartati.*sessione finalizzata/s); assert.match(result.content[0].text, /1 candidati scartati.*sessione finalizzata/s);
}); });
@@ -598,12 +580,11 @@ test("F8 with no promotion candidates finalizes without showing a widget", async
ctx, ctx,
); );
assert.equal(calls.some(({ command }) => command.startsWith("memory save-one ")), false); assert.equal(calls.some(({ args }) => startsWithArgs(args, ["memory", "save-one"])), false);
assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false); assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false);
assert.deepEqual( assert.deepEqual(
calls.map(({ command }) => command).filter((command) => mutationArgs(calls, [["phase", "advance"], ["session", "finalize"]]),
command.startsWith("phase advance") || command.startsWith("session finalize")), [["phase", "advance", "--session", "s1"], ["session", "finalize", "s1"]],
["phase advance --session s1", "session finalize s1"],
); );
assert.equal(ctx.notifications.length, 1); assert.equal(ctx.notifications.length, 1);
assert.match(result.content[0].text, /Nessun candidato.*sessione finalizzata/s); assert.match(result.content[0].text, /Nessun candidato.*sessione finalizzata/s);
@@ -613,7 +594,7 @@ test("F8 does not write the ledger or finalize when the vector save fails", asyn
const { ctx, tools, calls } = await setupGate({ const { ctx, tools, calls } = await setupGate({
phase: 8, phase: 8,
preview: [PROMOTION_CANDIDATE], preview: [PROMOTION_CANDIDATE],
fail: (command) => command.startsWith("memory save-one ") fail: (args) => startsWithArgs(args, ["memory", "save-one"])
? { status: 1, stderr: "vector save failed" } ? { status: 1, stderr: "vector save failed" }
: null, : null,
}); });
@@ -627,12 +608,10 @@ test("F8 does not write the ledger or finalize when the vector save fails", asyn
ctx, ctx,
); );
const mutations = calls.map(({ command }) => command).filter((command) => const mutations = memoryPromotionMutationArgs(calls);
command.startsWith("memory save-one ") || assert.deepEqual(mutations, [
command.startsWith("decision add ") || ["memory", "save-one", "--session", "s1", "--decision", "5", "--json"],
command.startsWith("phase advance") || ]);
command.startsWith("session finalize"));
assert.deepEqual(mutations, ["memory save-one --session s1 --decision 5 --json"]);
assert.match(result.content[0].text, /vector save failed.*Recupero manuale/s); assert.match(result.content[0].text, /vector save failed.*Recupero manuale/s);
}); });
@@ -640,7 +619,7 @@ test("F8 reports manual recovery and does not finalize after save succeeds but l
const { ctx, tools, calls } = await setupGate({ const { ctx, tools, calls } = await setupGate({
phase: 8, phase: 8,
preview: [PROMOTION_CANDIDATE], preview: [PROMOTION_CANDIDATE],
fail: (command) => command.includes("--type memory_promoted") fail: (args) => args.includes("--type") && args.includes("memory_promoted")
? { status: 1, stderr: "promotion ledger failed" } ? { status: 1, stderr: "promotion ledger failed" }
: null, : null,
}); });
@@ -654,15 +633,14 @@ test("F8 reports manual recovery and does not finalize after save succeeds but l
ctx, ctx,
); );
const mutations = calls.map(({ command }) => command).filter((command) => const mutations = memoryPromotionMutationArgs(calls);
command.startsWith("memory save-one ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance") ||
command.startsWith("session finalize"));
assert.deepEqual(mutations, [ assert.deepEqual(mutations, [
"memory save-one --session s1 --decision 5 --json", ["memory", "save-one", "--session", "s1", "--decision", "5", "--json"],
"decision add --session s1 --type memory_promoted --subject paziente attivo " + [
"--detail seq:5 --rationale scelta dal reviewer", "decision", "add", "--session", "s1", "--type", "memory_promoted",
"--subject", "paziente attivo", "--detail", "seq:5",
"--rationale", "scelta dal reviewer",
],
]); ]);
assert.match(result.content[0].text, /vectordb.*NON registrata.*recupero manuale/is); assert.match(result.content[0].text, /vectordb.*NON registrata.*recupero manuale/is);
}); });
@@ -0,0 +1,199 @@
[
{
"name": "reviewer_select",
"parameters": {
"type": "object",
"required": ["session", "title", "options"],
"properties": {
"session": { "type": "string" },
"title": { "type": "string" },
"options": {
"type": "array",
"items": {
"type": "object",
"required": ["id", "label"],
"properties": {
"id": { "type": "string" },
"label": { "type": "string" },
"decision": {
"type": "object",
"required": ["type", "subject"],
"properties": {
"type": { "type": "string" },
"subject": { "type": "string" },
"detail": { "type": "string" },
"rationale": { "type": "string" }
}
},
"recommended": { "type": "boolean" }
}
}
},
"intro": { "type": "string" },
"advance": { "type": "boolean" }
}
}
},
{
"name": "reviewer_datamart",
"parameters": {
"type": "object",
"required": ["session"],
"properties": { "session": { "type": "string" } }
}
},
{
"name": "reviewer_decide",
"parameters": {
"type": "object",
"required": ["session", "title", "options"],
"properties": {
"session": { "type": "string" },
"title": { "type": "string" },
"options": {
"type": "array",
"items": {
"type": "object",
"required": ["id", "label", "decision"],
"properties": {
"id": { "type": "string" },
"label": { "type": "string" },
"description": { "type": "string" },
"decision": {
"type": "object",
"required": ["type", "subject"],
"properties": {
"type": { "type": "string" },
"subject": { "type": "string" },
"detail": { "type": "string" },
"rationale": { "type": "string" }
}
},
"recommended": { "type": "boolean" }
}
}
},
"allow_empty": { "type": "boolean" },
"advance": { "type": "boolean" }
}
}
},
{
"name": "reviewer_schema_linking",
"parameters": {
"type": "object",
"required": ["session", "title", "tables"],
"properties": {
"session": { "type": "string" },
"title": { "type": "string" },
"tables": {
"type": "array",
"items": {
"type": "object",
"required": ["id", "name", "kind"],
"properties": {
"id": { "type": "string" },
"name": { "type": "string" },
"kind": {
"anyOf": [
{ "type": "string", "const": "promote" },
{ "type": "string", "const": "exclude" }
]
},
"rationale": { "type": "string" },
"suggested_columns": {
"type": "array",
"items": { "type": "string" }
},
"recommended": { "type": "boolean" }
}
}
},
"advance": { "type": "boolean" }
}
}
},
{
"name": "reviewer_confirm",
"parameters": {
"type": "object",
"required": ["session", "kind", "title", "artifact"],
"properties": {
"session": { "type": "string" },
"kind": {
"anyOf": [
{ "type": "string", "const": "phase" },
{ "type": "string", "const": "cte_plan" },
{ "type": "string", "const": "cte_result" },
{ "type": "string", "const": "sql" }
]
},
"title": { "type": "string" },
"artifact": {
"type": "object",
"required": ["kind", "data"],
"properties": {
"kind": { "type": "string" },
"data": {},
"version": { "type": "number" }
}
},
"names": { "type": "array", "items": { "type": "string" } }
}
}
},
{
"name": "reviewer_memory_promote",
"parameters": {
"type": "object",
"required": ["session"],
"properties": { "session": { "type": "string" } }
}
},
{
"name": "rewrite_question",
"parameters": {
"type": "object",
"required": ["session", "question"],
"properties": {
"session": { "type": "string" },
"question": { "type": "string" },
"assumptions": {}
}
}
},
{
"name": "write_schema_linking",
"parameters": {
"type": "object",
"required": ["session", "schema_linking"],
"properties": {
"session": { "type": "string" },
"schema_linking": {}
}
}
},
{
"name": "write_cte_sql",
"parameters": {
"type": "object",
"required": ["session", "name", "sql"],
"properties": {
"session": { "type": "string" },
"name": { "type": "string" },
"sql": { "type": "string" }
}
}
},
{
"name": "write_final_sql",
"parameters": {
"type": "object",
"required": ["session", "sql"],
"properties": {
"session": { "type": "string" },
"sql": { "type": "string" }
}
}
}
]
@@ -0,0 +1,249 @@
"""Executable baseline for persisted workflow behavior touched by the refactor."""
from dataclasses import asdict
from datetime import UTC, datetime
import hashlib
import pytest
from tht.corpus.models import CanonicalDocument, CorpusManifest
from tht.corpus.store import CorpusStore
from tht.decisions import DecisionRecord, append_decision
from tht.phase import current_phase, effective_decisions
from tht.session.artifacts import build_evidence_entries
from tht.session.models import Candidate, SchemaLinking
from tht.workflow import load_workflow
def test_workflow_definition_has_the_approved_semantic_contract():
workflow = load_workflow()
assert workflow.schema_version == 1
assert workflow.max_phase == 8
assert [asdict(phase) for phase in workflow.phases] == [
{
"id": "F1",
"num": 1,
"name": "chiarimento",
"advance": "kind:phase",
"prerequisites": [],
"artifacts_out": [],
"emits": ["concept_clarified", "ambiguity_open"],
},
{
"id": "F2",
"num": 2,
"name": "memoria",
"advance": "auto_if_empty",
"prerequisites": [],
"artifacts_out": [],
"emits": ["memory_rejected", "concept_clarified"],
},
{
"id": "F3",
"num": 3,
"name": "riscrittura",
"advance": "kind:phase",
"prerequisites": [{"decision_exists": "question_rewritten"}],
"artifacts_out": ["question.md"],
"emits": ["question_rewritten"],
},
{
"id": "F4",
"num": 4,
"name": "schema_linking",
"advance": "reviewer_decide",
"prerequisites": [],
"artifacts_out": ["schema_linking.json"],
"emits": [
"table_promoted",
"table_excluded",
"column_promoted",
"column_excluded",
"column_corrected",
"join_modified",
"evidence_accepted",
"evidence_rejected",
"value_grounded",
"concept_formula_approved",
"concept_formula_rejected",
],
},
{
"id": "F5",
"num": 5,
"name": "sintesi",
"advance": "kind:phase",
"prerequisites": [
{"file_validates": ["schema_linking.json", "SchemaLinking"]}
],
"artifacts_out": [],
"emits": [],
},
{
"id": "F6",
"num": 6,
"name": "cte",
"advance": "auto_if_empty_or_skipped",
"prerequisites": [
{
"any": [
{"decision_subject_exists": ["phase_skipped", "phase:6"]},
{"all_ctes_approved": True},
]
}
],
"artifacts_out": ["cte_plan.json", "ctes/", "cte_tests.json"],
"emits": ["cte_approved", "cte_corrected", "cte_rejected"],
},
{
"id": "F7",
"num": 7,
"name": "sql_finale",
"advance": "kind:phase",
"prerequisites": [{"decision_exists": "sql_approved"}],
"artifacts_out": ["sql_final.sql"],
"emits": ["sql_revised", "sql_approved", "sql_rejected"],
},
{
"id": "F8",
"num": 8,
"name": "datamart",
"advance": "reviewer_decide",
"prerequisites": [
{
"any": [
{"decision_exists": "datamart_requested"},
{"decision_exists": "datamart_declined"},
]
}
],
"artifacts_out": [],
"emits": [
"datamart_requested",
"datamart_declined",
"memory_promoted",
"memory_promotion_declined",
],
},
]
def _record(seq: int, type_: str, subject: str) -> DecisionRecord:
return DecisionRecord(
seq=seq,
ts=datetime(2026, 8, 24, tzinfo=UTC),
type=type_,
subject=subject,
)
def _linking(*evidence_ids: str, decision_seq: int = 17) -> SchemaLinking:
return SchemaLinking(
question="q",
candidates=[
Candidate(
kind="table",
name="fact_procedure",
evidence=list(evidence_ids),
decision="promoted",
decision_seq=decision_seq,
)
],
)
def test_schema_linking_evidence_used_resolves_from_the_active_canonical_corpus(tmp_path):
content = "# Curated definition\n"
digest = hashlib.sha256(content.encode()).hexdigest()
document = CanonicalDocument(
document_id=f"doc:{digest}",
source_id="fs:evi-used",
source_uri="file:///curated/evi-used.md",
source_fingerprint=f"sha256:{'a' * 64}",
content_hash=f"sha256:{digest}",
content=content,
pipeline_version="evidence-v1",
metadata={"frontmatter": {"id": "evi-used"}},
)
store = CorpusStore(tmp_path / "corpus")
generation = store.stage(
CorpusManifest(documents=(document,)),
{document.document_id: content},
)
store.publish(generation)
entries = build_evidence_entries(
[],
_linking("evi-used"),
tmp_path / "artifacts" / "evidence",
)
assert entries == [{
"id": "evi-used",
"file": str(
tmp_path / "artifacts" / ".materialized-evidence" / f"{digest}.md"
),
"esito": "usata",
"decision_seq": 17,
}]
def test_legacy_evidence_without_a_canonical_corpus_keeps_used_and_reviewed_outcomes(tmp_path):
evidence_root = tmp_path / "artifacts" / "evidence"
evidence_root.mkdir(parents=True)
for evidence_id in ("evi-used", "evi-accepted", "evi-rejected"):
(evidence_root / f"{evidence_id}.md").write_text(f"# {evidence_id}\n")
entries = build_evidence_entries(
[
_record(21, "evidence_accepted", "evi-accepted"),
_record(22, "evidence_rejected", "evi-rejected"),
],
_linking("evi-used", "evi-accepted"),
evidence_root,
)
assert [(entry["id"], entry["esito"], entry["decision_seq"]) for entry in entries] == [
("evi-used", "usata", 17),
("evi-accepted", "accettata", 21),
("evi-rejected", "scartata", 22),
]
assert all(entry["file"].endswith(f"{entry['id']}.md") for entry in entries)
@pytest.mark.parametrize(
("approved_through", "phase_decisions", "expected_phase"),
[
(0, [("concept_clarified", "ablazione")], 1),
(1, [("concept_clarified", "paziente attivo")], 2),
(2, [("question_rewritten", "domanda")], 3),
(3, [("evidence_accepted", "evi-7")], 4),
(
7,
[
("datamart_declined", "phase:8"),
("memory_promotion_declined", "paziente attivo"),
],
8,
),
],
ids=["F1", "F2", "F3", "F4-Evidence", "F8"],
)
def test_resume_reconstructs_each_touched_open_phase(
tmp_path, approved_through, phase_decisions, expected_phase
):
session_dir = tmp_path / f"resume-f{expected_phase}"
session_dir.mkdir()
for phase in range(1, approved_through + 1):
append_decision(
session_dir,
type="phase_approved",
subject=f"phase:{phase}",
)
for decision_type, subject in phase_decisions:
append_decision(session_dir, type=decision_type, subject=subject)
assert current_phase(session_dir) == expected_phase
effective = {(decision.type, decision.subject) for decision in effective_decisions(session_dir)}
assert set(phase_decisions) <= effective