From d15bb59c3dfff8f9cedbe50d434ef1da8ed5ff55 Mon Sep 17 00:00:00 2001 From: mptyl Date: Mon, 24 Aug 2026 00:40:11 +0200 Subject: [PATCH] test(workflow): complete observable baseline (#21) --- .../contracts/workflow-observable-baseline.md | 55 +++- .../gate_workflow_observable_contract.test.js | 296 ++++++++---------- .../__tests__/golden/pi_tool_schemas.json | 199 ++++++++++++ .../test_workflow_observable_contract.py | 249 +++++++++++++++ 4 files changed, 626 insertions(+), 173 deletions(-) create mode 100644 harness/.pi/extensions/gate/__tests__/golden/pi_tool_schemas.json create mode 100644 harness/tests/test_workflow_observable_contract.py diff --git a/docs/contracts/workflow-observable-baseline.md b/docs/contracts/workflow-observable-baseline.md index 15b16b60..e9fe2919 100644 --- a/docs/contracts/workflow-observable-baseline.md +++ b/docs/contracts/workflow-observable-baseline.md @@ -12,20 +12,26 @@ baseline green without weakening its assertions. ### Pi gate -Run `npm test` from the harness package. +From `harness/`, run `npm test` with the repository's supported Node 24 runtime. The gate suite fixes: -- the registered Pi tool names and their required and optional parameters; -- the exact workflow definition and injected session skill bytes; +- the complete registered Pi tool schemas, including nested types and enum-like constraints; +- the semantic workflow definition and the exact injected session skill bytes; - widget descriptors and reviewer response semantics; - F1 clarification and explicitly accepted open ambiguity; - F2 Memory applied, deselected, and absent; - F3 rewritten question and assumptions, including mutation failure ordering; -- F4 Evidence acceptance and rejection; +- F4 Evidence used, accepted, rejected, and legacy-without-corpus projections; - F8 Memory promotion accepted, declined, and absent, including mutation failure ordering; +- resume reconstruction for the touched F1, F2, F3, F4, and F8 states; - artifact payload compatibility, anti-bypass behavior, and final phase closing. +The baseline intentionally checks widget structure and domain content without freezing the +pre-existing Italian chrome emitted by the gate. Repository policy requires UI chrome and labels +to migrate to English in their owning workstream; this contract must not turn that mismatch into a +new compatibility requirement. + ### Harness CLI and persistence Run the default pytest suite from the harness package. The suite fixes: @@ -42,7 +48,18 @@ working local Docker daemon and remain part of the default suite when Docker is ### Backend bridge -Run the backend test suite followed by TypeScript typechecking. The suite fixes: +From `backend/`, the passing automated baseline is: + +```sh +npx vitest run test/tht-runner.test.ts test/pi-process-manager.test.ts \ + test/session-bridge.test.ts test/sse-hub.test.ts test/sse-route.test.ts \ + test/routes-sessions.test.ts test/e2e-f1.test.ts \ + test/workspace-preprocessing-service.test.ts test/evidence-materialization.test.ts +npx tsc --noEmit -p . +npm run build +``` + +These suites fix: - CLI argument ordering and JSON/error propagation across the runner boundary; - new-session versus resume Pi prompts; @@ -52,7 +69,18 @@ Run the backend test suite followed by TypeScript typechecking. The suite fixes: ### Frontend client -Run the frontend test suite followed by TypeScript typechecking. The suite fixes: +From `frontend/`, the passing automated baseline is: + +```sh +npx vitest run src/store/sessionStore.test.ts src/stream/useSessionStream.test.tsx \ + src/widgets/registry.test.tsx src/widgets/SelectWidget.test.tsx \ + src/widgets/MultiselectWidget.test.tsx src/widgets/ArtifactWidget.test.tsx \ + src/shell/f1-loop.test.tsx src/shell/SessionDocumentsPanel.test.tsx +npx tsc -b +npm run build +``` + +These suites fix: - widget registry and gate response payloads; - `ui_request`, `text_delta`, activity, usage, and lifecycle event reduction; @@ -78,18 +106,17 @@ question to finalization, followed by resume verification, belongs to the final ticket. If its environment or credentials are unavailable, it must remain recorded as a pending manual gate rather than being reported as passed. -## Pre-existing full-suite exceptions +## Full-suite diagnostic exceptions -The workflow baseline and every focused seam above pass on the source commit from which this -branch was created. Two unrelated full-suite failures also reproduce unchanged on that base -checkout and are therefore recorded rather than hidden or repaired in this refactoring ticket: +Every command defined above as part of the automated baseline exits successfully. Running the +broader backend and frontend suites is still useful as a diagnostic, but those full suites are not +the executable acceptance gate for this ticket because two unrelated failures reproduce unchanged +on the source commit from which this branch was created: - the backend authentication runtime-projection suite currently rejects ten positive fixtures with its fail-closed public error; - one frontend application-shell authentication test does not render the expected trusted-upstream display name. -The focused backend workflow suite, backend typecheck and build, focused frontend workflow suite, -frontend typecheck and build, complete harness pytest suite, Ruff, and complete Pi gate suite all -pass. These two exceptions must remain visible until their owning workstream resolves them; they -must not be used to relax any workflow assertion. +These two exceptions must remain visible until their owning workstream resolves them; they must not +be used to relax any workflow assertion or to describe a nonzero command as a passing baseline. diff --git a/harness/.pi/extensions/gate/__tests__/gate_workflow_observable_contract.test.js b/harness/.pi/extensions/gate/__tests__/gate_workflow_observable_contract.test.js index 5680fff3..bdd67048 100644 --- a/harness/.pi/extensions/gate/__tests__/gate_workflow_observable_contract.test.js +++ b/harness/.pi/extensions/gate/__tests__/gate_workflow_observable_contract.test.js @@ -42,23 +42,45 @@ const PHASE_META = JSON.stringify({ function useShell({ phase, preview = [], fail = () => null }) { const calls = []; shell.current = (_file, args, options = {}) => { - const command = args.join(" "); - calls.push({ command, input: options.input }); - const failure = fail(command); + calls.push({ args: [...args], input: options.input }); + const failure = fail(args); if (failure) { const error = new Error(failure.stderr); error.status = failure.status; error.stderr = failure.stderr; throw error; } - if (command === "phase meta --json") return PHASE_META; - if (command.startsWith("phase show --session ")) return `Fase corrente: ${phase}\n`; - if (command.startsWith("memory promote --session ")) return JSON.stringify(preview); + if (sameArgs(args, ["phase", "meta", "--json"])) return PHASE_META; + if (startsWithArgs(args, ["phase", "show", "--session"])) return `Fase corrente: ${phase}\n`; + if (startsWithArgs(args, ["memory", "promote", "--session"])) return JSON.stringify(preview); return ""; }; return calls; } +function sameArgs(actual, expected) { + return actual.length === expected.length && actual.every((value, index) => value === expected[index]); +} + +function startsWithArgs(actual, prefix) { + return prefix.every((value, index) => actual[index] === value); +} + +function mutationArgs(calls, prefixes) { + return calls + .filter(({ args }) => prefixes.some((prefix) => startsWithArgs(args, prefix))) + .map(({ args }) => args); +} + +function memoryPromotionMutationArgs(calls) { + return mutationArgs(calls, [ + ["memory", "save-one"], + ["decision", "add"], + ["phase", "advance"], + ["session", "finalize"], + ]); +} + async function setupGate(shellOptions) { const calls = useShell(shellOptions); const runtime = createFakePi(); @@ -83,65 +105,27 @@ test("the Pi gate exposes the approved tool names and parameter boundaries", () const publicContract = [...tools].map(([name, { def }]) => ({ name, - required: def.parameters.required ?? [], - properties: Object.keys(def.parameters.properties), + parameters: stripDescriptions(def.parameters), })); - assert.deepEqual(publicContract, [ - { - name: "reviewer_select", - required: ["session", "title", "options"], - properties: ["session", "title", "options", "intro", "advance"], - }, - { - name: "reviewer_datamart", - required: ["session"], - properties: ["session"], - }, - { - name: "reviewer_decide", - required: ["session", "title", "options"], - properties: ["session", "title", "options", "allow_empty", "advance"], - }, - { - name: "reviewer_schema_linking", - required: ["session", "title", "tables"], - properties: ["session", "title", "tables", "advance"], - }, - { - name: "reviewer_confirm", - required: ["session", "kind", "title", "artifact"], - properties: ["session", "kind", "title", "artifact", "names"], - }, - { - name: "reviewer_memory_promote", - required: ["session"], - properties: ["session"], - }, - { - name: "rewrite_question", - required: ["session", "question"], - properties: ["session", "question", "assumptions"], - }, - { - name: "write_schema_linking", - required: ["session", "schema_linking"], - properties: ["session", "schema_linking"], - }, - { - name: "write_cte_sql", - required: ["session", "name", "sql"], - properties: ["session", "name", "sql"], - }, - { - name: "write_final_sql", - required: ["session", "sql"], - properties: ["session", "sql"], - }, - ]); + const expected = JSON.parse(fs.readFileSync( + path.join(__dirname, "golden", "pi_tool_schemas.json"), + "utf8", + )); + assert.deepEqual(publicContract, expected); }); -test("the injected skill and workflow definition stay byte-identical during extraction", () => { +function stripDescriptions(value, insideProperties = false) { + if (Array.isArray(value)) return value.map((item) => stripDescriptions(item)); + if (value && typeof value === "object") { + return Object.fromEntries(Object.entries(value) + .filter(([key]) => insideProperties || key !== "description") + .map(([key, item]) => [key, stripDescriptions(item, key === "properties")])); + } + return value; +} + +test("the injected skill stays byte-identical during extraction", () => { const harnessRoot = path.resolve(__dirname, "..", "..", "..", ".."); const digest = (relativePath) => crypto .createHash("sha256") @@ -152,10 +136,6 @@ test("the injected skill and workflow definition stay byte-identical during extr digest(path.join(".pi", "skills", "tht-sessione", "SKILL.md")), "626a794071c095a4f20fffabb3bab901f05c101590adbdc58e45adfae56f3219", ); - assert.equal( - digest("workflow.yaml"), - "a0604c3dfac7960dcfbf4a584383a602ee953a9fffac97a469281ceef1284d38", - ); }); test("F1 persists a concrete clarification directly from the select widget", async () => { @@ -214,13 +194,13 @@ test("F1 persists a concrete clarification directly from the select widget", asy reserved: ["back", "exit", "other"], }, ); - assert.ok(calls.some(({ command }) => command === [ - "decision add --session s1", - "--type concept_clarified", - "--subject ablazione", - "--detail procedura clinica", - "--rationale scelta dal reviewer", - ].join(" "))); + assert.ok(calls.some(({ args }) => sameArgs(args, [ + "decision", "add", "--session", "s1", + "--type", "concept_clarified", + "--subject", "ablazione", + "--detail", "procedura clinica", + "--rationale", "scelta dal reviewer", + ]))); assert.match(result.content[0].text, /Decisione registrata \(concept_clarified\)/); }); @@ -240,7 +220,7 @@ test("F1 ask-only and control choices do not mutate the ledger", async () => { ctx, ); - assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false); + assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false); assert.match(result.content[0].text, /Scelta del reviewer: Chiedi un dettaglio/); }); @@ -269,13 +249,13 @@ test("F1 persists an explicitly accepted open ambiguity", async () => { ctx, ); - assert.equal(calls.some(({ command }) => command === [ - "decision add --session s1", - "--type ambiguity_open", - "--subject termine clinico", - "--detail nessuna definizione conclusiva", - "--rationale rischio accettato dal reviewer", - ].join(" ")), true); + assert.equal(calls.some(({ args }) => sameArgs(args, [ + "decision", "add", "--session", "s1", + "--type", "ambiguity_open", + "--subject", "termine clinico", + "--detail", "nessuna definizione conclusiva", + "--rationale", "rischio accettato dal reviewer", + ])), true); assert.match(result.content[0].text, /Decisione registrata \(ambiguity_open\)/); }); @@ -297,7 +277,7 @@ const MEMORY_OPTIONS = [ test("F2 applies selected Memory content and keeps the substantive phase open", async () => { const { ctx, tools, calls } = await setupGate({ phase: 2, - fail: (command) => command.startsWith("phase advance --auto") + fail: (args) => startsWithArgs(args, ["phase", "advance", "--auto"]) ? { status: 6, stderr: "phase not auto-eligible" } : null, }); @@ -322,21 +302,15 @@ test("F2 applies selected Memory content and keeps the substantive phase open", { phase: descriptor.phase, widget: descriptor.widget, - title: descriptor.title, allowEmpty: descriptor.allow_empty, selected: descriptor.selected, - selectionLabel: descriptor.selection_label, - confirmLabel: descriptor.confirm_label, options: descriptor.options, }, { phase: "F2", widget: "multiselect", - title: "Seleziona le memory da applicare alla domanda", allowEmpty: true, selected: ["mem-0042"], - selectionLabel: "memory da applicare", - confirmLabel: "Applica le memory selezionate", options: [ { id: "mem-0042", @@ -349,13 +323,17 @@ test("F2 applies selected Memory content and keeps the substantive phase open", ], }, ); - const mutations = calls - .map(({ command }) => command) - .filter((command) => command.startsWith("decision ") || command.startsWith("phase advance")); + assert.equal(typeof descriptor.title, "string"); + assert.equal(typeof descriptor.selection_label, "string"); + assert.equal(typeof descriptor.confirm_label, "string"); + const mutations = mutationArgs(calls, [["decision"], ["phase", "advance"]]); assert.deepEqual(mutations, [ - "decision add --session s1 --type concept_clarified --subject paziente attivo " + - "--detail flag_attivo = TRUE --rationale Riusa mem-0042 per la stessa definizione", - "phase advance --auto --session s1", + [ + "decision", "add", "--session", "s1", "--type", "concept_clarified", + "--subject", "paziente attivo", "--detail", "flag_attivo = TRUE", + "--rationale", "Riusa mem-0042 per la stessa definizione", + ], + ["phase", "advance", "--auto", "--session", "s1"], ]); assert.match(result.content[0].text, /La fase resta aperta/); }); @@ -378,8 +356,10 @@ test("F2 accepts an empty selection without recording a rejection", async () => ctx, ); - assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false); - assert.equal(calls.some(({ command }) => command === "phase advance --auto --session s1"), true); + assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false); + assert.equal(calls.some(({ args }) => sameArgs( + args, ["phase", "advance", "--auto", "--session", "s1"], + )), true); assert.match(result.content[0].text, /Nessuna decisione registrata.*Fase avanzata/s); }); @@ -401,19 +381,20 @@ test("F2 with no Memory candidates notifies once and advances without an empty w ctx, ); - assert.deepEqual(ctx.notifications, [{ - message: "Nessuna memory riutilizzabile per questa domanda — passo alla fase successiva.", - level: "info", - }]); - assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false); - assert.equal(calls.some(({ command }) => command === "phase advance --auto --session s1"), true); + assert.equal(ctx.notifications.length, 1); + assert.equal(ctx.notifications[0].level, "info"); + assert.equal(typeof ctx.notifications[0].message, "string"); + assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false); + assert.equal(calls.some(({ args }) => sameArgs( + args, ["phase", "advance", "--auto", "--session", "s1"], + )), true); assert.match(result.content[0].text, /Fase memoria vuota/); }); test("F3 stops before ledger and advance when writing the rewritten question fails", async () => { const { ctx, tools, calls } = await setupGate({ phase: 3, - fail: (command) => command.startsWith("session set-question ") + fail: (args) => startsWithArgs(args, ["session", "set-question"]) ? { status: 1, stderr: "question write failed" } : null, }); @@ -426,12 +407,14 @@ test("F3 stops before ledger and advance when writing the rewritten question fai ctx, ); - const mutations = calls.map(({ command }) => command).filter((command) => - command.startsWith("session set-question ") || - command.startsWith("decision add ") || - command.startsWith("phase advance")); + const mutations = mutationArgs(calls, [ + ["session", "set-question"], ["decision", "add"], ["phase", "advance"], + ]); assert.deepEqual(mutations, [ - "session set-question s1 --question Domanda riscritta --assumption Assunzione A", + [ + "session", "set-question", "s1", "--question", "Domanda riscritta", + "--assumption", "Assunzione A", + ], ]); assert.match(result.content[0].text, /question write failed/); }); @@ -439,7 +422,7 @@ test("F3 stops before ledger and advance when writing the rewritten question fai test("F3 keeps the rewritten question but does not advance when the ledger write fails", async () => { const { ctx, tools, calls } = await setupGate({ phase: 3, - fail: (command) => command.startsWith("decision add ") + fail: (args) => startsWithArgs(args, ["decision", "add"]) ? { status: 1, stderr: "ledger write failed" } : null, }); @@ -452,14 +435,18 @@ test("F3 keeps the rewritten question but does not advance when the ledger write ctx, ); - const mutations = calls.map(({ command }) => command).filter((command) => - command.startsWith("session set-question ") || - command.startsWith("decision add ") || - command.startsWith("phase advance")); + const mutations = mutationArgs(calls, [ + ["session", "set-question"], ["decision", "add"], ["phase", "advance"], + ]); assert.deepEqual(mutations, [ - "session set-question s1 --question Domanda riscritta --assumption Assunzione A", - "decision add --session s1 --type question_rewritten --subject domanda " + - "--detail Domanda riscritta", + [ + "session", "set-question", "s1", "--question", "Domanda riscritta", + "--assumption", "Assunzione A", + ], + [ + "decision", "add", "--session", "s1", "--type", "question_rewritten", + "--subject", "domanda", "--detail", "Domanda riscritta", + ], ]); assert.match(result.content[0].text, /ledger write failed/); }); @@ -507,12 +494,10 @@ test("F4 persists exactly the reviewer-selected Evidence disposition", async () { id: "accept", label: "Usa la fonte", detail: undefined }, { id: "reject", label: "Scarta la fonte", detail: undefined }, ]); - const decisions = calls - .map(({ command }) => command) - .filter((command) => command.startsWith("decision add ")); + const decisions = mutationArgs(calls, [["decision", "add"]]); const expectedType = selected === "accept" ? "evidence_accepted" : "evidence_rejected"; assert.equal(decisions.length, 1); - assert.match(decisions[0], new RegExp(`--type ${expectedType} `)); + assert.equal(decisions[0][decisions[0].indexOf("--type") + 1], expectedType); assert.match(result.content[0].text, new RegExp(expectedType)); } }); @@ -545,17 +530,16 @@ test("F8 saves an accepted Memory before its ledger marker and finalizes", async assert.equal(descriptor.widget, "multiselect"); assert.deepEqual(descriptor.selected, ["seq-5"]); assert.match(descriptor.content, /flag_attivo = TRUE/); - const mutations = calls.map(({ command }) => command).filter((command) => - command.startsWith("memory save-one ") || - command.startsWith("decision add ") || - command.startsWith("phase advance") || - command.startsWith("session finalize")); + const mutations = memoryPromotionMutationArgs(calls); assert.deepEqual(mutations, [ - "memory save-one --session s1 --decision 5 --json", - "decision add --session s1 --type memory_promoted --subject paziente attivo " + - "--detail seq:5 --rationale scelta dal reviewer", - "phase advance --session s1", - "session finalize s1", + ["memory", "save-one", "--session", "s1", "--decision", "5", "--json"], + [ + "decision", "add", "--session", "s1", "--type", "memory_promoted", + "--subject", "paziente attivo", "--detail", "seq:5", + "--rationale", "scelta dal reviewer", + ], + ["phase", "advance", "--session", "s1"], + ["session", "finalize", "s1"], ]); assert.match(result.content[0].text, /1 memorie salvate.*sessione finalizzata/s); }); @@ -572,16 +556,14 @@ test("F8 records a declined candidate without saving it and finalizes", async () ctx, ); - const mutations = calls.map(({ command }) => command).filter((command) => - command.startsWith("memory save-one ") || - command.startsWith("decision add ") || - command.startsWith("phase advance") || - command.startsWith("session finalize")); + const mutations = memoryPromotionMutationArgs(calls); assert.deepEqual(mutations, [ - "decision add --session s1 --type memory_promotion_declined " + - "--subject paziente attivo --detail seq:5", - "phase advance --session s1", - "session finalize s1", + [ + "decision", "add", "--session", "s1", "--type", "memory_promotion_declined", + "--subject", "paziente attivo", "--detail", "seq:5", + ], + ["phase", "advance", "--session", "s1"], + ["session", "finalize", "s1"], ]); assert.match(result.content[0].text, /1 candidati scartati.*sessione finalizzata/s); }); @@ -598,12 +580,11 @@ test("F8 with no promotion candidates finalizes without showing a widget", async ctx, ); - assert.equal(calls.some(({ command }) => command.startsWith("memory save-one ")), false); - assert.equal(calls.some(({ command }) => command.startsWith("decision add ")), false); + assert.equal(calls.some(({ args }) => startsWithArgs(args, ["memory", "save-one"])), false); + assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false); assert.deepEqual( - calls.map(({ command }) => command).filter((command) => - command.startsWith("phase advance") || command.startsWith("session finalize")), - ["phase advance --session s1", "session finalize s1"], + mutationArgs(calls, [["phase", "advance"], ["session", "finalize"]]), + [["phase", "advance", "--session", "s1"], ["session", "finalize", "s1"]], ); assert.equal(ctx.notifications.length, 1); assert.match(result.content[0].text, /Nessun candidato.*sessione finalizzata/s); @@ -613,7 +594,7 @@ test("F8 does not write the ledger or finalize when the vector save fails", asyn const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE], - fail: (command) => command.startsWith("memory save-one ") + fail: (args) => startsWithArgs(args, ["memory", "save-one"]) ? { status: 1, stderr: "vector save failed" } : null, }); @@ -627,12 +608,10 @@ test("F8 does not write the ledger or finalize when the vector save fails", asyn ctx, ); - const mutations = calls.map(({ command }) => command).filter((command) => - command.startsWith("memory save-one ") || - command.startsWith("decision add ") || - command.startsWith("phase advance") || - command.startsWith("session finalize")); - assert.deepEqual(mutations, ["memory save-one --session s1 --decision 5 --json"]); + const mutations = memoryPromotionMutationArgs(calls); + assert.deepEqual(mutations, [ + ["memory", "save-one", "--session", "s1", "--decision", "5", "--json"], + ]); assert.match(result.content[0].text, /vector save failed.*Recupero manuale/s); }); @@ -640,7 +619,7 @@ test("F8 reports manual recovery and does not finalize after save succeeds but l const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE], - fail: (command) => command.includes("--type memory_promoted") + fail: (args) => args.includes("--type") && args.includes("memory_promoted") ? { status: 1, stderr: "promotion ledger failed" } : null, }); @@ -654,15 +633,14 @@ test("F8 reports manual recovery and does not finalize after save succeeds but l ctx, ); - const mutations = calls.map(({ command }) => command).filter((command) => - command.startsWith("memory save-one ") || - command.startsWith("decision add ") || - command.startsWith("phase advance") || - command.startsWith("session finalize")); + const mutations = memoryPromotionMutationArgs(calls); assert.deepEqual(mutations, [ - "memory save-one --session s1 --decision 5 --json", - "decision add --session s1 --type memory_promoted --subject paziente attivo " + - "--detail seq:5 --rationale scelta dal reviewer", + ["memory", "save-one", "--session", "s1", "--decision", "5", "--json"], + [ + "decision", "add", "--session", "s1", "--type", "memory_promoted", + "--subject", "paziente attivo", "--detail", "seq:5", + "--rationale", "scelta dal reviewer", + ], ]); assert.match(result.content[0].text, /vectordb.*NON registrata.*recupero manuale/is); }); diff --git a/harness/.pi/extensions/gate/__tests__/golden/pi_tool_schemas.json b/harness/.pi/extensions/gate/__tests__/golden/pi_tool_schemas.json new file mode 100644 index 00000000..b7587cf4 --- /dev/null +++ b/harness/.pi/extensions/gate/__tests__/golden/pi_tool_schemas.json @@ -0,0 +1,199 @@ +[ + { + "name": "reviewer_select", + "parameters": { + "type": "object", + "required": ["session", "title", "options"], + "properties": { + "session": { "type": "string" }, + "title": { "type": "string" }, + "options": { + "type": "array", + "items": { + "type": "object", + "required": ["id", "label"], + "properties": { + "id": { "type": "string" }, + "label": { "type": "string" }, + "decision": { + "type": "object", + "required": ["type", "subject"], + "properties": { + "type": { "type": "string" }, + "subject": { "type": "string" }, + "detail": { "type": "string" }, + "rationale": { "type": "string" } + } + }, + "recommended": { "type": "boolean" } + } + } + }, + "intro": { "type": "string" }, + "advance": { "type": "boolean" } + } + } + }, + { + "name": "reviewer_datamart", + "parameters": { + "type": "object", + "required": ["session"], + "properties": { "session": { "type": "string" } } + } + }, + { + "name": "reviewer_decide", + "parameters": { + "type": "object", + "required": ["session", "title", "options"], + "properties": { + "session": { "type": "string" }, + "title": { "type": "string" }, + "options": { + "type": "array", + "items": { + "type": "object", + "required": ["id", "label", "decision"], + "properties": { + "id": { "type": "string" }, + "label": { "type": "string" }, + "description": { "type": "string" }, + "decision": { + "type": "object", + "required": ["type", "subject"], + "properties": { + "type": { "type": "string" }, + "subject": { "type": "string" }, + "detail": { "type": "string" }, + "rationale": { "type": "string" } + } + }, + "recommended": { "type": "boolean" } + } + } + }, + "allow_empty": { "type": "boolean" }, + "advance": { "type": "boolean" } + } + } + }, + { + "name": "reviewer_schema_linking", + "parameters": { + "type": "object", + "required": ["session", "title", "tables"], + "properties": { + "session": { "type": "string" }, + "title": { "type": "string" }, + "tables": { + "type": "array", + "items": { + "type": "object", + "required": ["id", "name", "kind"], + "properties": { + "id": { "type": "string" }, + "name": { "type": "string" }, + "kind": { + "anyOf": [ + { "type": "string", "const": "promote" }, + { "type": "string", "const": "exclude" } + ] + }, + "rationale": { "type": "string" }, + "suggested_columns": { + "type": "array", + "items": { "type": "string" } + }, + "recommended": { "type": "boolean" } + } + } + }, + "advance": { "type": "boolean" } + } + } + }, + { + "name": "reviewer_confirm", + "parameters": { + "type": "object", + "required": ["session", "kind", "title", "artifact"], + "properties": { + "session": { "type": "string" }, + "kind": { + "anyOf": [ + { "type": "string", "const": "phase" }, + { "type": "string", "const": "cte_plan" }, + { "type": "string", "const": "cte_result" }, + { "type": "string", "const": "sql" } + ] + }, + "title": { "type": "string" }, + "artifact": { + "type": "object", + "required": ["kind", "data"], + "properties": { + "kind": { "type": "string" }, + "data": {}, + "version": { "type": "number" } + } + }, + "names": { "type": "array", "items": { "type": "string" } } + } + } + }, + { + "name": "reviewer_memory_promote", + "parameters": { + "type": "object", + "required": ["session"], + "properties": { "session": { "type": "string" } } + } + }, + { + "name": "rewrite_question", + "parameters": { + "type": "object", + "required": ["session", "question"], + "properties": { + "session": { "type": "string" }, + "question": { "type": "string" }, + "assumptions": {} + } + } + }, + { + "name": "write_schema_linking", + "parameters": { + "type": "object", + "required": ["session", "schema_linking"], + "properties": { + "session": { "type": "string" }, + "schema_linking": {} + } + } + }, + { + "name": "write_cte_sql", + "parameters": { + "type": "object", + "required": ["session", "name", "sql"], + "properties": { + "session": { "type": "string" }, + "name": { "type": "string" }, + "sql": { "type": "string" } + } + } + }, + { + "name": "write_final_sql", + "parameters": { + "type": "object", + "required": ["session", "sql"], + "properties": { + "session": { "type": "string" }, + "sql": { "type": "string" } + } + } + } +] diff --git a/harness/tests/test_workflow_observable_contract.py b/harness/tests/test_workflow_observable_contract.py new file mode 100644 index 00000000..be672024 --- /dev/null +++ b/harness/tests/test_workflow_observable_contract.py @@ -0,0 +1,249 @@ +"""Executable baseline for persisted workflow behavior touched by the refactor.""" + +from dataclasses import asdict +from datetime import UTC, datetime +import hashlib + +import pytest + +from tht.corpus.models import CanonicalDocument, CorpusManifest +from tht.corpus.store import CorpusStore +from tht.decisions import DecisionRecord, append_decision +from tht.phase import current_phase, effective_decisions +from tht.session.artifacts import build_evidence_entries +from tht.session.models import Candidate, SchemaLinking +from tht.workflow import load_workflow + + +def test_workflow_definition_has_the_approved_semantic_contract(): + workflow = load_workflow() + + assert workflow.schema_version == 1 + assert workflow.max_phase == 8 + assert [asdict(phase) for phase in workflow.phases] == [ + { + "id": "F1", + "num": 1, + "name": "chiarimento", + "advance": "kind:phase", + "prerequisites": [], + "artifacts_out": [], + "emits": ["concept_clarified", "ambiguity_open"], + }, + { + "id": "F2", + "num": 2, + "name": "memoria", + "advance": "auto_if_empty", + "prerequisites": [], + "artifacts_out": [], + "emits": ["memory_rejected", "concept_clarified"], + }, + { + "id": "F3", + "num": 3, + "name": "riscrittura", + "advance": "kind:phase", + "prerequisites": [{"decision_exists": "question_rewritten"}], + "artifacts_out": ["question.md"], + "emits": ["question_rewritten"], + }, + { + "id": "F4", + "num": 4, + "name": "schema_linking", + "advance": "reviewer_decide", + "prerequisites": [], + "artifacts_out": ["schema_linking.json"], + "emits": [ + "table_promoted", + "table_excluded", + "column_promoted", + "column_excluded", + "column_corrected", + "join_modified", + "evidence_accepted", + "evidence_rejected", + "value_grounded", + "concept_formula_approved", + "concept_formula_rejected", + ], + }, + { + "id": "F5", + "num": 5, + "name": "sintesi", + "advance": "kind:phase", + "prerequisites": [ + {"file_validates": ["schema_linking.json", "SchemaLinking"]} + ], + "artifacts_out": [], + "emits": [], + }, + { + "id": "F6", + "num": 6, + "name": "cte", + "advance": "auto_if_empty_or_skipped", + "prerequisites": [ + { + "any": [ + {"decision_subject_exists": ["phase_skipped", "phase:6"]}, + {"all_ctes_approved": True}, + ] + } + ], + "artifacts_out": ["cte_plan.json", "ctes/", "cte_tests.json"], + "emits": ["cte_approved", "cte_corrected", "cte_rejected"], + }, + { + "id": "F7", + "num": 7, + "name": "sql_finale", + "advance": "kind:phase", + "prerequisites": [{"decision_exists": "sql_approved"}], + "artifacts_out": ["sql_final.sql"], + "emits": ["sql_revised", "sql_approved", "sql_rejected"], + }, + { + "id": "F8", + "num": 8, + "name": "datamart", + "advance": "reviewer_decide", + "prerequisites": [ + { + "any": [ + {"decision_exists": "datamart_requested"}, + {"decision_exists": "datamart_declined"}, + ] + } + ], + "artifacts_out": [], + "emits": [ + "datamart_requested", + "datamart_declined", + "memory_promoted", + "memory_promotion_declined", + ], + }, + ] + + +def _record(seq: int, type_: str, subject: str) -> DecisionRecord: + return DecisionRecord( + seq=seq, + ts=datetime(2026, 8, 24, tzinfo=UTC), + type=type_, + subject=subject, + ) + + +def _linking(*evidence_ids: str, decision_seq: int = 17) -> SchemaLinking: + return SchemaLinking( + question="q", + candidates=[ + Candidate( + kind="table", + name="fact_procedure", + evidence=list(evidence_ids), + decision="promoted", + decision_seq=decision_seq, + ) + ], + ) + + +def test_schema_linking_evidence_used_resolves_from_the_active_canonical_corpus(tmp_path): + content = "# Curated definition\n" + digest = hashlib.sha256(content.encode()).hexdigest() + document = CanonicalDocument( + document_id=f"doc:{digest}", + source_id="fs:evi-used", + source_uri="file:///curated/evi-used.md", + source_fingerprint=f"sha256:{'a' * 64}", + content_hash=f"sha256:{digest}", + content=content, + pipeline_version="evidence-v1", + metadata={"frontmatter": {"id": "evi-used"}}, + ) + store = CorpusStore(tmp_path / "corpus") + generation = store.stage( + CorpusManifest(documents=(document,)), + {document.document_id: content}, + ) + store.publish(generation) + + entries = build_evidence_entries( + [], + _linking("evi-used"), + tmp_path / "artifacts" / "evidence", + ) + + assert entries == [{ + "id": "evi-used", + "file": str( + tmp_path / "artifacts" / ".materialized-evidence" / f"{digest}.md" + ), + "esito": "usata", + "decision_seq": 17, + }] + + +def test_legacy_evidence_without_a_canonical_corpus_keeps_used_and_reviewed_outcomes(tmp_path): + evidence_root = tmp_path / "artifacts" / "evidence" + evidence_root.mkdir(parents=True) + for evidence_id in ("evi-used", "evi-accepted", "evi-rejected"): + (evidence_root / f"{evidence_id}.md").write_text(f"# {evidence_id}\n") + + entries = build_evidence_entries( + [ + _record(21, "evidence_accepted", "evi-accepted"), + _record(22, "evidence_rejected", "evi-rejected"), + ], + _linking("evi-used", "evi-accepted"), + evidence_root, + ) + + assert [(entry["id"], entry["esito"], entry["decision_seq"]) for entry in entries] == [ + ("evi-used", "usata", 17), + ("evi-accepted", "accettata", 21), + ("evi-rejected", "scartata", 22), + ] + assert all(entry["file"].endswith(f"{entry['id']}.md") for entry in entries) + + +@pytest.mark.parametrize( + ("approved_through", "phase_decisions", "expected_phase"), + [ + (0, [("concept_clarified", "ablazione")], 1), + (1, [("concept_clarified", "paziente attivo")], 2), + (2, [("question_rewritten", "domanda")], 3), + (3, [("evidence_accepted", "evi-7")], 4), + ( + 7, + [ + ("datamart_declined", "phase:8"), + ("memory_promotion_declined", "paziente attivo"), + ], + 8, + ), + ], + ids=["F1", "F2", "F3", "F4-Evidence", "F8"], +) +def test_resume_reconstructs_each_touched_open_phase( + tmp_path, approved_through, phase_decisions, expected_phase +): + session_dir = tmp_path / f"resume-f{expected_phase}" + session_dir.mkdir() + for phase in range(1, approved_through + 1): + append_decision( + session_dir, + type="phase_approved", + subject=f"phase:{phase}", + ) + for decision_type, subject in phase_decisions: + append_decision(session_dir, type=decision_type, subject=subject) + + assert current_phase(session_dir) == expected_phase + effective = {(decision.type, decision.subject) for decision in effective_decisions(session_dir)} + assert set(phase_decisions) <= effective