test(workflow): freeze observable contracts (#21)

This commit is contained in:
2026-08-24 00:25:17 +02:00
parent b5db0cd3c1
commit dc9726cb35
2 changed files with 763 additions and 0 deletions
@@ -0,0 +1,95 @@
# Workflow observable baseline
This contract freezes the externally observable behavior that the conservative modular
refactoring must preserve. It describes what callers and reviewers can observe; it does not
prescribe the internal location of the implementation.
Changing an expectation in this baseline is a behavior change and requires an explicit product
decision. Moving code between Workflow core, Disambiguation, Memory, and Evidence must keep the
baseline green without weakening its assertions.
## Automated seams
### Pi gate
Run `npm test` from the harness package.
The gate suite fixes:
- the registered Pi tool names and their required and optional parameters;
- the exact workflow definition and injected session skill bytes;
- widget descriptors and reviewer response semantics;
- F1 clarification and explicitly accepted open ambiguity;
- F2 Memory applied, deselected, and absent;
- F3 rewritten question and assumptions, including mutation failure ordering;
- F4 Evidence acceptance and rejection;
- F8 Memory promotion accepted, declined, and absent, including mutation failure ordering;
- artifact payload compatibility, anti-bypass behavior, and final phase closing.
### Harness CLI and persistence
Run the default pytest suite from the harness package. The suite fixes:
- pristine JSON output, human output separation, exit codes, and CLI error behavior;
- decision ledger folding, retraction, reopen ordering, and current-phase reconstruction;
- question, schema-linking, CTE, SQL, validation, and session-document projections;
- Evidence source, corpus, search, citation, and legacy-without-active-corpus behavior;
- Memory search, promotion, solved-question, and vector-write behavior;
- filesystem session persistence and PostgreSQL repository parity.
The default pytest configuration excludes only tests marked `l2`. Tests marked `l0` require a
working local Docker daemon and remain part of the default suite when Docker is available.
### Backend bridge
Run the backend test suite followed by TypeScript typechecking. The suite fixes:
- CLI argument ordering and JSON/error propagation across the runner boundary;
- new-session versus resume Pi prompts;
- refusal to resume finalized, archived, foreign, unavailable, or read-only sessions;
- Pi RPC to client event mapping, SSE replay/reset behavior, and runtime replacement ordering;
- failure persistence and sanitization before a client-visible response.
### Frontend client
Run the frontend test suite followed by TypeScript typechecking. The suite fixes:
- widget registry and gate response payloads;
- `ui_request`, `text_delta`, activity, usage, and lifecycle event reduction;
- stream replacement, cursor reset, reconnection, and pending-text flush behavior;
- session document projections shown to the reviewer.
## Mutation ordering
The following sequences are part of the observable failure contract:
1. F3 writes the rewritten question, appends `question_rewritten` to the ledger, then advances.
A failure stops the remaining operations.
2. F8 saves one reusable Memory vector, appends its `memory_promoted` marker, advances F8, then
finalizes. A failed vector write leaves no marker; a failed marker after a successful vector
write returns the manual recovery instruction and does not finalize.
3. A declined F8 candidate writes only `memory_promotion_declined`; an absent candidate writes no
Memory decision and still closes F8.
## Environment-dependent acceptance
Real-model and remote-DWH tests remain opt-in through the `l2` marker. The live journey from a new
question to finalization, followed by resume verification, belongs to the final live-acceptance
ticket. If its environment or credentials are unavailable, it must remain recorded as a pending
manual gate rather than being reported as passed.
## Pre-existing full-suite exceptions
The workflow baseline and every focused seam above pass on the source commit from which this
branch was created. Two unrelated full-suite failures also reproduce unchanged on that base
checkout and are therefore recorded rather than hidden or repaired in this refactoring ticket:
- the backend authentication runtime-projection suite currently rejects ten positive fixtures
with its fail-closed public error;
- one frontend application-shell authentication test does not render the expected trusted-upstream
display name.
The focused backend workflow suite, backend typecheck and build, focused frontend workflow suite,
frontend typecheck and build, complete harness pytest suite, Ruff, and complete Pi gate suite all
pass. These two exceptions must remain visible until their owning workstream resolves them; they
must not be used to relax any workflow assertion.
@@ -0,0 +1,668 @@
const test = require("node:test");
const assert = require("node:assert");
const crypto = require("node:crypto");
const cp = require("node:child_process");
const fs = require("node:fs");
const { createRequire } = require("node:module");
const path = require("node:path");
const { createFakePi } = require("./fake_pi_runtime.js");
const GATE = path.join(__dirname, "..", "..", "tht-gate.js");
globalThis.require = createRequire(GATE);
const shell = { current: () => "" };
cp.execFileSync = (file, args, options) => shell.current(file, args, options);
const installGate = require(GATE).default ?? require(GATE);
const PHASE_META = JSON.stringify({
max_phase: 8,
phases: [
{ num: 1, id: "F1", name: "chiarimento", emits: ["concept_clarified", "ambiguity_open"] },
{ num: 2, id: "F2", name: "memoria", emits: ["memory_rejected", "concept_clarified"] },
{ num: 3, id: "F3", name: "riscrittura", emits: ["question_rewritten"] },
{
num: 4,
id: "F4",
name: "schema_linking",
emits: ["evidence_accepted", "evidence_rejected"],
},
{ num: 5, id: "F5", name: "sintesi", emits: [] },
{ num: 6, id: "F6", name: "cte", emits: [] },
{ num: 7, id: "F7", name: "sql_finale", emits: ["sql_approved"] },
{
num: 8,
id: "F8",
name: "datamart",
emits: ["datamart_declined", "memory_promoted", "memory_promotion_declined"],
},
],
});
function useShell({ phase, preview = [], fail = () => null }) {
const calls = [];
shell.current = (_file, args, options = {}) => {
const command = args.join(" ");
calls.push({ command, input: options.input });
const failure = fail(command);
if (failure) {
const error = new Error(failure.stderr);
error.status = failure.status;
error.stderr = failure.stderr;
throw error;
}
if (command === "phase meta --json") return PHASE_META;
if (command.startsWith("phase show --session ")) return `Fase corrente: ${phase}\n`;
if (command.startsWith("memory promote --session ")) return JSON.stringify(preview);
return "";
};
return calls;
}
async function setupGate(shellOptions) {
const calls = useShell(shellOptions);
const runtime = createFakePi();
runtime.ctx.cwd = "/nonexistent-thothii-contract-cwd";
runtime.ctx.mode = "rpc";
installGate(runtime.pi);
await runtime.pi.emit("session_start", {});
return { ...runtime, calls };
}
function answerNextWidget(ctx, choices, capture) {
ctx.ui.input = async (title) => {
const descriptor = JSON.parse(title);
capture?.(descriptor);
return JSON.stringify({ id: descriptor.id, choices });
};
}
test("the Pi gate exposes the approved tool names and parameter boundaries", () => {
const { pi, tools } = createFakePi();
installGate(pi);
const publicContract = [...tools].map(([name, { def }]) => ({
name,
required: def.parameters.required ?? [],
properties: Object.keys(def.parameters.properties),
}));
assert.deepEqual(publicContract, [
{
name: "reviewer_select",
required: ["session", "title", "options"],
properties: ["session", "title", "options", "intro", "advance"],
},
{
name: "reviewer_datamart",
required: ["session"],
properties: ["session"],
},
{
name: "reviewer_decide",
required: ["session", "title", "options"],
properties: ["session", "title", "options", "allow_empty", "advance"],
},
{
name: "reviewer_schema_linking",
required: ["session", "title", "tables"],
properties: ["session", "title", "tables", "advance"],
},
{
name: "reviewer_confirm",
required: ["session", "kind", "title", "artifact"],
properties: ["session", "kind", "title", "artifact", "names"],
},
{
name: "reviewer_memory_promote",
required: ["session"],
properties: ["session"],
},
{
name: "rewrite_question",
required: ["session", "question"],
properties: ["session", "question", "assumptions"],
},
{
name: "write_schema_linking",
required: ["session", "schema_linking"],
properties: ["session", "schema_linking"],
},
{
name: "write_cte_sql",
required: ["session", "name", "sql"],
properties: ["session", "name", "sql"],
},
{
name: "write_final_sql",
required: ["session", "sql"],
properties: ["session", "sql"],
},
]);
});
test("the injected skill and workflow definition stay byte-identical during extraction", () => {
const harnessRoot = path.resolve(__dirname, "..", "..", "..", "..");
const digest = (relativePath) => crypto
.createHash("sha256")
.update(fs.readFileSync(path.join(harnessRoot, relativePath)))
.digest("hex");
assert.equal(
digest(path.join(".pi", "skills", "tht-sessione", "SKILL.md")),
"626a794071c095a4f20fffabb3bab901f05c101590adbdc58e45adfae56f3219",
);
assert.equal(
digest("workflow.yaml"),
"a0604c3dfac7960dcfbf4a584383a602ee953a9fffac97a469281ceef1284d38",
);
});
test("F1 persists a concrete clarification directly from the select widget", async () => {
const { ctx, tools, calls } = await setupGate({ phase: 1 });
let descriptor;
answerNextWidget(ctx, ["procedure"], (value) => { descriptor = value; });
const result = await tools.get("reviewer_select").def.execute(
"select-f1",
{
session: "s1",
title: "Che cosa significa ablazione?",
intro: "Una sola interpretazione può essere corretta.",
options: [
{
id: "procedure",
label: "Procedura clinica",
recommended: true,
decision: {
type: "concept_clarified",
subject: "ablazione",
detail: "procedura clinica",
rationale: "scelta dal reviewer",
},
},
{ id: "ask", label: "Serve un altro chiarimento" },
],
},
null,
null,
ctx,
);
assert.deepEqual(
{
type: descriptor.type,
widget: descriptor.widget,
phase: descriptor.phase,
title: descriptor.title,
intro: descriptor.intro,
recommended: descriptor.recommended,
options: descriptor.options,
reserved: descriptor.reserved,
},
{
type: "ui_request",
widget: "select",
phase: "F1",
title: "Che cosa significa ablazione?",
intro: "Una sola interpretazione può essere corretta.",
recommended: "procedure",
options: [
{ id: "procedure", label: "Procedura clinica" },
{ id: "ask", label: "Serve un altro chiarimento" },
],
reserved: ["back", "exit", "other"],
},
);
assert.ok(calls.some(({ command }) => command === [
"decision add --session s1",
"--type concept_clarified",
"--subject ablazione",
"--detail procedura clinica",
"--rationale scelta dal reviewer",
].join(" ")));
assert.match(result.content[0].text, /Decisione registrata \(concept_clarified\)/);
});
test("F1 ask-only and control choices do not mutate the ledger", async () => {
const { ctx, tools, calls } = await setupGate({ phase: 1 });
answerNextWidget(ctx, ["ask"]);
const result = await tools.get("reviewer_select").def.execute(
"select-f1-ask",
{
session: "s1",
title: "Serve un altro chiarimento?",
options: [{ id: "ask", label: "Chiedi un dettaglio" }],
},
null,
null,
ctx,
);
assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false);
assert.match(result.content[0].text, /Scelta del reviewer: Chiedi un dettaglio/);
});
test("F1 persists an explicitly accepted open ambiguity", async () => {
const { ctx, tools, calls } = await setupGate({ phase: 1 });
answerNextWidget(ctx, ["leave-open"]);
const result = await tools.get("reviewer_select").def.execute(
"select-f1-open",
{
session: "s1",
title: "Come trattare il termine non risolto?",
options: [{
id: "leave-open",
label: "Lascia aperta l'ambiguità",
decision: {
type: "ambiguity_open",
subject: "termine clinico",
detail: "nessuna definizione conclusiva",
rationale: "rischio accettato dal reviewer",
},
}],
},
null,
null,
ctx,
);
assert.equal(calls.some(({ command }) => command === [
"decision add --session s1",
"--type ambiguity_open",
"--subject termine clinico",
"--detail nessuna definizione conclusiva",
"--rationale rischio accettato dal reviewer",
].join(" ")), true);
assert.match(result.content[0].text, /Decisione registrata \(ambiguity_open\)/);
});
const MEMORY_OPTIONS = [
{
id: "mem-0042",
label: "Paziente attivo",
description: "Decisione concept_clarified: paziente attivo\nflag_attivo = TRUE",
recommended: true,
decision: {
type: "concept_clarified",
subject: "paziente attivo",
detail: "flag_attivo = TRUE",
rationale: "Riusa mem-0042 per la stessa definizione",
},
},
];
test("F2 applies selected Memory content and keeps the substantive phase open", async () => {
const { ctx, tools, calls } = await setupGate({
phase: 2,
fail: (command) => command.startsWith("phase advance --auto")
? { status: 6, stderr: "phase not auto-eligible" }
: null,
});
let descriptor;
answerNextWidget(ctx, ["mem-0042"], (value) => { descriptor = value; });
const result = await tools.get("reviewer_decide").def.execute(
"memory-selected",
{
session: "s1",
title: "Memorie candidate",
options: MEMORY_OPTIONS,
allow_empty: true,
advance: true,
},
null,
null,
ctx,
);
assert.deepEqual(
{
phase: descriptor.phase,
widget: descriptor.widget,
title: descriptor.title,
allowEmpty: descriptor.allow_empty,
selected: descriptor.selected,
selectionLabel: descriptor.selection_label,
confirmLabel: descriptor.confirm_label,
options: descriptor.options,
},
{
phase: "F2",
widget: "multiselect",
title: "Seleziona le memory da applicare alla domanda",
allowEmpty: true,
selected: ["mem-0042"],
selectionLabel: "memory da applicare",
confirmLabel: "Applica le memory selezionate",
options: [
{
id: "mem-0042",
label: "Paziente attivo",
detail: "Decisione concept_clarified: paziente attivo\nflag_attivo = TRUE",
rationale: "Riusa mem-0042 per la stessa definizione",
meta: { memory_id: "mem-0042" },
selected: true,
},
],
},
);
const mutations = calls
.map(({ command }) => command)
.filter((command) => command.startsWith("decision ") || command.startsWith("phase advance"));
assert.deepEqual(mutations, [
"decision add --session s1 --type concept_clarified --subject paziente attivo " +
"--detail flag_attivo = TRUE --rationale Riusa mem-0042 per la stessa definizione",
"phase advance --auto --session s1",
]);
assert.match(result.content[0].text, /La fase resta aperta/);
});
test("F2 accepts an empty selection without recording a rejection", async () => {
const { ctx, tools, calls } = await setupGate({ phase: 2 });
answerNextWidget(ctx, []);
const result = await tools.get("reviewer_decide").def.execute(
"memory-deselected",
{
session: "s1",
title: "Memorie candidate",
options: MEMORY_OPTIONS,
allow_empty: true,
advance: true,
},
null,
null,
ctx,
);
assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false);
assert.equal(calls.some(({ command }) => command === "phase advance --auto --session s1"), true);
assert.match(result.content[0].text, /Nessuna decisione registrata.*Fase avanzata/s);
});
test("F2 with no Memory candidates notifies once and advances without an empty widget", async () => {
const { ctx, tools, calls } = await setupGate({ phase: 2 });
ctx.ui.input = async () => { throw new Error("an empty Memory widget must not be shown"); };
const result = await tools.get("reviewer_decide").def.execute(
"memory-absent",
{
session: "s1",
title: "Memorie candidate",
options: [],
allow_empty: true,
advance: true,
},
null,
null,
ctx,
);
assert.deepEqual(ctx.notifications, [{
message: "Nessuna memory riutilizzabile per questa domanda — passo alla fase successiva.",
level: "info",
}]);
assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false);
assert.equal(calls.some(({ command }) => command === "phase advance --auto --session s1"), true);
assert.match(result.content[0].text, /Fase memoria vuota/);
});
test("F3 stops before ledger and advance when writing the rewritten question fails", async () => {
const { ctx, tools, calls } = await setupGate({
phase: 3,
fail: (command) => command.startsWith("session set-question ")
? { status: 1, stderr: "question write failed" }
: null,
});
const result = await tools.get("rewrite_question").def.execute(
"rewrite-write-failure",
{ session: "s1", question: "Domanda riscritta", assumptions: ["Assunzione A"] },
null,
null,
ctx,
);
const mutations = calls.map(({ command }) => command).filter((command) =>
command.startsWith("session set-question ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance"));
assert.deepEqual(mutations, [
"session set-question s1 --question Domanda riscritta --assumption Assunzione A",
]);
assert.match(result.content[0].text, /question write failed/);
});
test("F3 keeps the rewritten question but does not advance when the ledger write fails", async () => {
const { ctx, tools, calls } = await setupGate({
phase: 3,
fail: (command) => command.startsWith("decision add ")
? { status: 1, stderr: "ledger write failed" }
: null,
});
const result = await tools.get("rewrite_question").def.execute(
"rewrite-ledger-failure",
{ session: "s1", question: "Domanda riscritta", assumptions: ["Assunzione A"] },
null,
null,
ctx,
);
const mutations = calls.map(({ command }) => command).filter((command) =>
command.startsWith("session set-question ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance"));
assert.deepEqual(mutations, [
"session set-question s1 --question Domanda riscritta --assumption Assunzione A",
"decision add --session s1 --type question_rewritten --subject domanda " +
"--detail Domanda riscritta",
]);
assert.match(result.content[0].text, /ledger write failed/);
});
test("F4 persists exactly the reviewer-selected Evidence disposition", async () => {
for (const selected of ["accept", "reject"]) {
const { ctx, tools, calls } = await setupGate({ phase: 4 });
let descriptor;
answerNextWidget(ctx, [selected], (value) => { descriptor = value; });
const result = await tools.get("reviewer_decide").def.execute(
`evidence-${selected}`,
{
session: "s1",
title: "Valuta la fonte Evidence",
options: [
{
id: "accept",
label: "Usa la fonte",
description: "Evidence evi-7: definizione clinica",
decision: {
type: "evidence_accepted",
subject: "evi-7",
detail: "definizione clinica",
},
},
{
id: "reject",
label: "Scarta la fonte",
description: "Evidence evi-7: definizione clinica",
decision: {
type: "evidence_rejected",
subject: "evi-7",
detail: "non pertinente",
},
},
],
},
null,
null,
ctx,
);
assert.deepEqual(descriptor.options.map(({ id, label, detail }) => ({ id, label, detail })), [
{ id: "accept", label: "Usa la fonte", detail: undefined },
{ id: "reject", label: "Scarta la fonte", detail: undefined },
]);
const decisions = calls
.map(({ command }) => command)
.filter((command) => command.startsWith("decision add "));
const expectedType = selected === "accept" ? "evidence_accepted" : "evidence_rejected";
assert.equal(decisions.length, 1);
assert.match(decisions[0], new RegExp(`--type ${expectedType} `));
assert.match(result.content[0].text, new RegExp(expectedType));
}
});
const PROMOTION_CANDIDATE = {
decision_seq: 5,
type: "concept_clarified",
subject: "paziente attivo",
detail: "flag_attivo = TRUE",
rationale: "scelta dal reviewer",
question_context: "quanti pazienti attivi",
tables: [],
concepts: ["paziente attivo"],
};
test("F8 saves an accepted Memory before its ledger marker and finalizes", async () => {
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE] });
let descriptor;
answerNextWidget(ctx, ["seq-5"], (value) => { descriptor = value; });
const result = await tools.get("reviewer_memory_promote").def.execute(
"promote-accepted",
{ session: "s1" },
null,
null,
ctx,
);
assert.equal(descriptor.phase, "F8");
assert.equal(descriptor.widget, "multiselect");
assert.deepEqual(descriptor.selected, ["seq-5"]);
assert.match(descriptor.content, /flag_attivo = TRUE/);
const mutations = calls.map(({ command }) => command).filter((command) =>
command.startsWith("memory save-one ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance") ||
command.startsWith("session finalize"));
assert.deepEqual(mutations, [
"memory save-one --session s1 --decision 5 --json",
"decision add --session s1 --type memory_promoted --subject paziente attivo " +
"--detail seq:5 --rationale scelta dal reviewer",
"phase advance --session s1",
"session finalize s1",
]);
assert.match(result.content[0].text, /1 memorie salvate.*sessione finalizzata/s);
});
test("F8 records a declined candidate without saving it and finalizes", async () => {
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE] });
answerNextWidget(ctx, []);
const result = await tools.get("reviewer_memory_promote").def.execute(
"promote-declined",
{ session: "s1" },
null,
null,
ctx,
);
const mutations = calls.map(({ command }) => command).filter((command) =>
command.startsWith("memory save-one ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance") ||
command.startsWith("session finalize"));
assert.deepEqual(mutations, [
"decision add --session s1 --type memory_promotion_declined " +
"--subject paziente attivo --detail seq:5",
"phase advance --session s1",
"session finalize s1",
]);
assert.match(result.content[0].text, /1 candidati scartati.*sessione finalizzata/s);
});
test("F8 with no promotion candidates finalizes without showing a widget", async () => {
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [] });
ctx.ui.input = async () => { throw new Error("a promotion widget must not be shown"); };
const result = await tools.get("reviewer_memory_promote").def.execute(
"promote-absent",
{ session: "s1" },
null,
null,
ctx,
);
assert.equal(calls.some(({ command }) => command.startsWith("memory save-one ")), false);
assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false);
assert.deepEqual(
calls.map(({ command }) => command).filter((command) =>
command.startsWith("phase advance") || command.startsWith("session finalize")),
["phase advance --session s1", "session finalize s1"],
);
assert.equal(ctx.notifications.length, 1);
assert.match(result.content[0].text, /Nessun candidato.*sessione finalizzata/s);
});
test("F8 does not write the ledger or finalize when the vector save fails", async () => {
const { ctx, tools, calls } = await setupGate({
phase: 8,
preview: [PROMOTION_CANDIDATE],
fail: (command) => command.startsWith("memory save-one ")
? { status: 1, stderr: "vector save failed" }
: null,
});
answerNextWidget(ctx, ["seq-5"]);
const result = await tools.get("reviewer_memory_promote").def.execute(
"promote-save-failure",
{ session: "s1" },
null,
null,
ctx,
);
const mutations = calls.map(({ command }) => command).filter((command) =>
command.startsWith("memory save-one ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance") ||
command.startsWith("session finalize"));
assert.deepEqual(mutations, ["memory save-one --session s1 --decision 5 --json"]);
assert.match(result.content[0].text, /vector save failed.*Recupero manuale/s);
});
test("F8 reports manual recovery and does not finalize after save succeeds but ledger fails", async () => {
const { ctx, tools, calls } = await setupGate({
phase: 8,
preview: [PROMOTION_CANDIDATE],
fail: (command) => command.includes("--type memory_promoted")
? { status: 1, stderr: "promotion ledger failed" }
: null,
});
answerNextWidget(ctx, ["seq-5"]);
const result = await tools.get("reviewer_memory_promote").def.execute(
"promote-ledger-failure",
{ session: "s1" },
null,
null,
ctx,
);
const mutations = calls.map(({ command }) => command).filter((command) =>
command.startsWith("memory save-one ") ||
command.startsWith("decision add ") ||
command.startsWith("phase advance") ||
command.startsWith("session finalize"));
assert.deepEqual(mutations, [
"memory save-one --session s1 --decision 5 --json",
"decision add --session s1 --type memory_promoted --subject paziente attivo " +
"--detail seq:5 --rationale scelta dal reviewer",
]);
assert.match(result.content[0].text, /vectordb.*NON registrata.*recupero manuale/is);
});