feat: implement memory and evidence administration with guided repairs
Publish documentation / publish (push) Successful in 1m27s
Publish documentation / publish (push) Successful in 1m27s
Add PostgreSQL-backed memory, editable evidence with source review and activation, and human-approved archive repairs across the harness, API, and UI. Include migrations, deployment support, regression coverage, and validation documentation. Refresh permissions from validated session roles so existing administrator logins can access newly deployed archive management features.
This commit is contained in:
@@ -0,0 +1,61 @@
|
||||
const test = require("node:test");
|
||||
const assert = require("node:assert/strict");
|
||||
const { installRepairGate } = require("../memory/repair.js");
|
||||
|
||||
function gate({ answers, fail = false, resume = false }) {
|
||||
let tool;
|
||||
const calls = [], widgets = [];
|
||||
const repair = { repair_id: "receipt", choice: null, status: "proposed", can_apply: true,
|
||||
saved: false, indexed: false, options: [{ id: "fix", content: { detail: "Correction" } }] };
|
||||
installRepairGate({ registerTool: def => { tool = def; } }, {
|
||||
workflow: { activate() {}, phase: () => ({ id: "F4" }) },
|
||||
memory: { execute: (_ctx, args) => {
|
||||
calls.push(args);
|
||||
if (args[0] === "repair-apply") {
|
||||
const choice = args[args.indexOf("--choice") + 1];
|
||||
if (choice === "reject") Object.assign(repair, { choice, status: "rejected" });
|
||||
else Object.assign(repair, { choice, saved: true, indexed: !fail,
|
||||
status: fail ? "pending_activation" : "active" });
|
||||
fail = false;
|
||||
}
|
||||
return JSON.stringify(repair);
|
||||
} },
|
||||
waitForReviewer: async (_ctx, descriptor) => {
|
||||
widgets.push(descriptor);
|
||||
assert.ok(answers.length, "gate must not ask an unbounded extra question");
|
||||
return answers.shift();
|
||||
},
|
||||
toTextResult: text => text,
|
||||
});
|
||||
return { calls, widgets, run: () => tool.execute("id", {
|
||||
session: "s", ...(resume ? { repair_id: "receipt" } : { proposal: { reason: "conflict", options: [] } }),
|
||||
}, null, null, {}) };
|
||||
}
|
||||
|
||||
test("human choice applies once and the next widget reports activation", async () => {
|
||||
const g = gate({ answers: [{ choices: ["fix"] }, { choices: ["continue"] }] });
|
||||
assert.equal(JSON.parse(await g.run()).indexed, true);
|
||||
assert.deepEqual(g.calls.map(args => args[0]), ["repair-prepare", "repair-apply"]);
|
||||
assert.equal(g.widgets[0].repair.saved, false);
|
||||
assert.equal(g.widgets[1].repair.status, "active");
|
||||
});
|
||||
test("rejection asks for reformulation without a normal workflow decision", async () => {
|
||||
const g = gate({ answers: [{ choices: ["reject"] }] });
|
||||
assert.match(await g.run(), /rejected as inadequate/);
|
||||
assert.equal(g.calls.length, 2);
|
||||
});
|
||||
test("retry keeps the receipt and selected choice, then reports true activation", async () => {
|
||||
const g = gate({ resume: true, fail: true,
|
||||
answers: [{ choices: ["fix"] }, { choices: ["fix"] }, { choices: ["continue"] }] });
|
||||
assert.equal(JSON.parse(await g.run()).indexed, true);
|
||||
assert.equal(g.widgets[1].repair.status, "pending_activation");
|
||||
assert.equal(g.calls[0][0], "repair-show");
|
||||
assert.deepEqual(g.calls[1], g.calls[2]);
|
||||
});
|
||||
test("free text and forged choices never apply a correction", async () => {
|
||||
for (const response of [{ control: "freetext", text: "Explain the conflict" }, { choices: ["forged"] }]) {
|
||||
const g = gate({ answers: [response] });
|
||||
await g.run();
|
||||
assert.equal(g.calls.length, 1);
|
||||
}
|
||||
});
|
||||
@@ -1,163 +1,44 @@
|
||||
const test = require("node:test");
|
||||
const assert = require("node:assert");
|
||||
const assert = require("node:assert/strict");
|
||||
const { createMemoryGate } = require("../memory/index.js");
|
||||
const { buildMultiselectRequest } = require("../core/builders.js");
|
||||
const { isReserved } = require("../core/reserved-labels.mjs");
|
||||
const { createFakePi } = require("./fake_pi_runtime.js");
|
||||
|
||||
|
||||
const TABLE_CANDIDATE = {
|
||||
decision_seq: 3,
|
||||
type: "table_promoted",
|
||||
subject: "fact_seeablazione",
|
||||
detail: "tabella principale ablazioni",
|
||||
rationale: "scelta dal reviewer",
|
||||
question_context: "quante ablazioni nel 2023",
|
||||
tables: ["fact_seeablazione"],
|
||||
concepts: [],
|
||||
};
|
||||
|
||||
const MEMORY_CANDIDATE = {
|
||||
decision_seq: 5,
|
||||
type: "concept_clarified",
|
||||
subject: "paziente attivo",
|
||||
detail: "flag_attivo = TRUE",
|
||||
rationale: "scelta dal reviewer",
|
||||
question_context: "quante ablazioni nel 2023",
|
||||
tables: [],
|
||||
concepts: ["paziente attivo"],
|
||||
};
|
||||
|
||||
|
||||
test("the public Memory facade owns F8 policy and mutation ordering", async () => {
|
||||
const { pi, tools, ctx } = createFakePi();
|
||||
function setup({ phase = 8, response, failure = null } = {}) {
|
||||
const runtime = createFakePi();
|
||||
const calls = [];
|
||||
let descriptor;
|
||||
const duplicateMemory = { ...MEMORY_CANDIDATE, decision_seq: 6 };
|
||||
const declinedMemory = {
|
||||
...MEMORY_CANDIDATE,
|
||||
decision_seq: 7,
|
||||
subject: "ricovero indice",
|
||||
detail: "first_event",
|
||||
};
|
||||
ctx.ui.input = async (title) => {
|
||||
descriptor = JSON.parse(title);
|
||||
calls.push(["review", descriptor.options.map((option) => option.id)]);
|
||||
return JSON.stringify({ id: descriptor.id, choices: ["seq-5"] });
|
||||
};
|
||||
|
||||
const memoryGate = createMemoryGate({
|
||||
workflow: {
|
||||
activate: () => calls.push(["activate"]),
|
||||
phase: () => ({ number: 8, id: "F8" }),
|
||||
close: (_ctx, _session, phaseNumber, summary) => {
|
||||
calls.push(["close", phaseNumber, summary]);
|
||||
return null;
|
||||
},
|
||||
},
|
||||
memory: {
|
||||
execute: (_ctx, args) => {
|
||||
calls.push(["memory-execute", args]);
|
||||
return JSON.stringify([
|
||||
TABLE_CANDIDATE,
|
||||
MEMORY_CANDIDATE,
|
||||
duplicateMemory,
|
||||
declinedMemory,
|
||||
]);
|
||||
},
|
||||
mutate: (_ctx, args, recovery) => {
|
||||
calls.push(["memory-mutate", args, recovery]);
|
||||
return null;
|
||||
},
|
||||
},
|
||||
ledger: {
|
||||
record: (_ctx, session, decision, recovery) => {
|
||||
calls.push(["ledger", session, decision, recovery]);
|
||||
return null;
|
||||
},
|
||||
},
|
||||
reviewer: {
|
||||
buildMultiselect: buildMultiselectRequest,
|
||||
isReserved,
|
||||
},
|
||||
waitForReviewer: async (runtimeContext, widget) => {
|
||||
const response = await runtimeContext.ui.input(JSON.stringify(widget), "");
|
||||
return JSON.parse(response);
|
||||
},
|
||||
toTextResult: (text) => ({ content: [{ type: "text", text }] }),
|
||||
const summary = { summary_id: "summary", items: [{ id: "rule", card: { subject: "Rule" } }] };
|
||||
createMemoryGate({
|
||||
workflow: { activate() {}, phase: () => ({ number: phase, id: "F" + phase }),
|
||||
close: () => { calls.push("close"); return "closed"; } },
|
||||
memory: { execute: () => JSON.stringify(summary),
|
||||
mutate: async (_ctx, args) => { calls.push(["save", args]); return failure; } },
|
||||
ledger: { record: (_ctx, _session, decision) => { calls.push(["ledger", decision]); } },
|
||||
waitForReviewer: async (_ctx, widget) => { calls.push(["widget", widget]); return response; },
|
||||
toTextResult: text => ({ content: [{ type: "text", text }] }),
|
||||
}).install(runtime.pi);
|
||||
return { calls, run: () => runtime.tools.get("reviewer_memory_promote").def.execute(
|
||||
"id", { session: "s1" }, null, null, runtime.ctx) };
|
||||
}
|
||||
for (const control of ["back", "exit", "freetext"]) {
|
||||
test("review control " + control + " never saves or closes", async () => {
|
||||
const { calls, run } = setup({ response: { control, text: "Revise scope" } });
|
||||
await run();
|
||||
assert.deepEqual(calls.map(call => call[0]), ["widget"]);
|
||||
});
|
||||
memoryGate.install(pi);
|
||||
|
||||
const result = await tools.get("reviewer_memory_promote").def.execute(
|
||||
"promote-via-facade",
|
||||
{ session: "s1" },
|
||||
null,
|
||||
null,
|
||||
ctx,
|
||||
);
|
||||
|
||||
assert.deepEqual(descriptor.options, [
|
||||
{
|
||||
id: "seq-5",
|
||||
label: "concept_clarified: paziente attivo",
|
||||
detail: "flag_attivo = TRUE",
|
||||
rationale: "scelta dal reviewer",
|
||||
meta: { question_context: "quante ablazioni nel 2023" },
|
||||
selected: true,
|
||||
},
|
||||
{
|
||||
id: "seq-7",
|
||||
label: "concept_clarified: ricovero indice",
|
||||
detail: "first_event",
|
||||
rationale: "scelta dal reviewer",
|
||||
meta: { question_context: "quante ablazioni nel 2023" },
|
||||
selected: true,
|
||||
},
|
||||
]);
|
||||
assert.deepEqual(descriptor.selected, ["seq-5", "seq-7"]);
|
||||
assert.doesNotMatch(descriptor.content, /fact_seeablazione/);
|
||||
assert.match(descriptor.content, /flag_attivo = TRUE/);
|
||||
assert.deepEqual(calls.slice(0, 7), [
|
||||
["activate"],
|
||||
[
|
||||
"memory-execute",
|
||||
["promote", "--session", "s1", "--preview", "--json"],
|
||||
],
|
||||
["review", ["seq-5", "seq-7"]],
|
||||
[
|
||||
"memory-mutate",
|
||||
["save-one", "--session", "s1", "--decision", "5", "--json"],
|
||||
"Recupero manuale (umano): tht memory save-one --session s1 " +
|
||||
"--decision 5. Finora salvate: 0.",
|
||||
],
|
||||
[
|
||||
"ledger",
|
||||
"s1",
|
||||
{
|
||||
type: "memory_promoted",
|
||||
subject: "paziente attivo",
|
||||
detail: "seq:5",
|
||||
rationale: "scelta dal reviewer",
|
||||
},
|
||||
"Memoria salvata nel vectordb ma decisione memory_promoted NON registrata: " +
|
||||
"recupero manuale (umano) con tht decision add --session s1 " +
|
||||
"--type memory_promoted --subject \"paziente attivo\" --detail seq:5.",
|
||||
],
|
||||
[
|
||||
"ledger",
|
||||
"s1",
|
||||
{
|
||||
type: "memory_promotion_declined",
|
||||
subject: "ricovero indice",
|
||||
detail: "seq:7",
|
||||
},
|
||||
"",
|
||||
],
|
||||
[
|
||||
"close",
|
||||
8,
|
||||
"Promozione registrata: 1 memorie salvate nel vectordb, 1 candidati scartati.",
|
||||
],
|
||||
]);
|
||||
assert.match(result.content[0].text, /1 memorie salvate.*1 candidati scartati/);
|
||||
}
|
||||
test("a mismatched review identity cannot mutate Memory", async () => {
|
||||
const { calls, run } = setup({ response: { text: JSON.stringify({ summary_id: "old", items: [] }) } });
|
||||
assert.match((await run()).content[0].text, /Invalid/);
|
||||
assert.equal(calls.length, 1);
|
||||
});
|
||||
test("review is only presented at F8", async () => {
|
||||
const { calls, run } = setup({ phase: 7 });
|
||||
assert.match((await run()).content[0].text, /end of F8/);
|
||||
assert.equal(calls.length, 0);
|
||||
});
|
||||
test("a persistence failure prevents the review marker and closing", async () => {
|
||||
const { calls, run } = setup({ failure: "database unavailable",
|
||||
response: { text: JSON.stringify({ summary_id: "summary", items: [] }) } });
|
||||
assert.equal(await run(), "database unavailable");
|
||||
assert.deepEqual(calls.map(call => call[0]), ["widget", "save"]);
|
||||
});
|
||||
|
||||
@@ -117,6 +117,18 @@ test("the Memory facade normalizes search hits and applies only the selected F2
|
||||
assert.match(result.content[0].text, /1 decisioni.*La fase resta aperta/s);
|
||||
});
|
||||
|
||||
test("authoritative UUID card identities survive normalization into the reviewer widget", async () => {
|
||||
const { ctx, memoryGate, descriptor } = setupRecall(["first"]);
|
||||
const id = "mem-11111111-1111-4111-8111-111111111111";
|
||||
await memoryGate.reviewRecall(ctx, {
|
||||
session: "s1", title: "Memory", allow_empty: true, advance: false,
|
||||
options: [{ ...MEMORY_OPTIONS[0], decision: {
|
||||
...MEMORY_OPTIONS[0].decision, rationale: `Riusa ${id}`,
|
||||
} }],
|
||||
}, "F2");
|
||||
assert.equal(descriptor().options[0].meta.memory_id, id);
|
||||
});
|
||||
|
||||
|
||||
test("the Memory facade accepts a deselected F2 result without a rejection", async () => {
|
||||
const { ctx, calls, memoryGate } = setupRecall([]);
|
||||
|
||||
@@ -25,7 +25,7 @@ echo "$@" >> "${log}"
|
||||
case "$1 $2" in
|
||||
"phase show") echo "Fase corrente: ${phase}";;
|
||||
"phase meta") echo '{"max_phase":8,"phases":[{"num":2,"id":"F2","emits":[]},{"num":8,"id":"F8","emits":[]}]}';;
|
||||
"memory promote") echo "[]";;
|
||||
"memory summary") echo '{"summary_id":"saved-review","reviewed":true}';;
|
||||
"session show") echo '{"status":"${status}"}';;
|
||||
*) echo "OK";;
|
||||
esac
|
||||
@@ -52,7 +52,7 @@ async function runPromote(t, { phase }) {
|
||||
return { res, calls: fake.calls() };
|
||||
}
|
||||
|
||||
test("F8 + zero candidati: il gate avanza la fase e finalizza da solo", async (t) => {
|
||||
test("F8 recupera un riepilogo già salvato e finalizza da solo", async (t) => {
|
||||
const { res, calls } = await runPromote(t, { phase: 8 });
|
||||
const text = res.content[0].text;
|
||||
assert.match(text, /sessione finalizzata \(s1\)/);
|
||||
@@ -63,7 +63,7 @@ test("F8 + zero candidati: il gate avanza la fase e finalizza da solo", async (t
|
||||
|
||||
test("fuori dall'ultima fase non chiude nulla (comportamento precedente)", async (t) => {
|
||||
const { res, calls } = await runPromote(t, { phase: 2 });
|
||||
assert.match(res.content[0].text, /prosegui con la chiusura della sessione/);
|
||||
assert.match(res.content[0].text, /end of F8/);
|
||||
assert.doesNotMatch(calls, /phase advance/);
|
||||
assert.doesNotMatch(calls, /session finalize/);
|
||||
});
|
||||
|
||||
@@ -33,12 +33,12 @@ const PHASE_META = JSON.stringify({
|
||||
num: 8,
|
||||
id: "F8",
|
||||
name: "datamart",
|
||||
emits: ["datamart_declined", "memory_promoted", "memory_promotion_declined"],
|
||||
emits: ["datamart_declined", "memory_summary_reviewed"],
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
function useShell({ phase, preview = [], fail = () => null }) {
|
||||
function useShell({ phase, preview = [], fail = () => null, indexed = true, reviewed = false }) {
|
||||
const calls = [];
|
||||
shell.current = (_file, args, options = {}) => {
|
||||
calls.push({ args: [...args], input: options.input });
|
||||
@@ -51,7 +51,8 @@ function useShell({ phase, preview = [], fail = () => null }) {
|
||||
}
|
||||
if (sameArgs(args, ["phase", "meta", "--json"])) return PHASE_META;
|
||||
if (startsWithArgs(args, ["phase", "show", "--session"])) return `Fase corrente: ${phase}\n`;
|
||||
if (startsWithArgs(args, ["memory", "promote", "--session"])) return JSON.stringify(preview);
|
||||
if (startsWithArgs(args, ["memory", "summary", "--session"])) return JSON.stringify({ summary_id: "summary-1", items: preview, reviewed });
|
||||
if (startsWithArgs(args, ["memory", "review-apply"])) return JSON.stringify({ indexed, saved: true });
|
||||
return "";
|
||||
};
|
||||
return calls;
|
||||
@@ -73,7 +74,7 @@ function mutationArgs(calls, prefixes) {
|
||||
|
||||
function memoryPromotionMutationArgs(calls) {
|
||||
return mutationArgs(calls, [
|
||||
["memory", "save-one"],
|
||||
["memory", "review-apply"],
|
||||
["decision", "add"],
|
||||
["phase", "advance"],
|
||||
["session", "finalize"],
|
||||
@@ -511,144 +512,55 @@ test("F4 persists exactly the reviewer-selected Evidence disposition", async ()
|
||||
});
|
||||
|
||||
const PROMOTION_CANDIDATE = {
|
||||
decision_seq: 5,
|
||||
type: "concept_clarified",
|
||||
subject: "paziente attivo",
|
||||
detail: "flag_attivo = TRUE",
|
||||
rationale: "scelta dal reviewer",
|
||||
question_context: "quanti pazienti attivi",
|
||||
tables: [],
|
||||
concepts: ["paziente attivo"],
|
||||
id: "decision-5", card: { family: "domain_clarification", subject: "paziente attivo", detail: "flag_attivo = TRUE" },
|
||||
};
|
||||
|
||||
test("F8 saves an accepted Memory before its ledger marker and finalizes", async () => {
|
||||
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE] });
|
||||
let descriptor;
|
||||
answerNextWidget(ctx, ["seq-5"], (value) => { descriptor = value; });
|
||||
|
||||
const result = await tools.get("reviewer_memory_promote").def.execute(
|
||||
"promote-accepted",
|
||||
{ session: "s1" },
|
||||
null,
|
||||
null,
|
||||
ctx,
|
||||
);
|
||||
|
||||
assert.equal(descriptor.phase, "F8");
|
||||
assert.equal(descriptor.widget, "multiselect");
|
||||
assert.deepEqual(descriptor.selected, ["seq-5"]);
|
||||
assert.match(descriptor.content, /flag_attivo = TRUE/);
|
||||
const mutations = memoryPromotionMutationArgs(calls);
|
||||
assert.deepEqual(mutations, [
|
||||
["memory", "save-one", "--session", "s1", "--decision", "5", "--json"],
|
||||
[
|
||||
"decision", "add", "--session", "s1", "--type", "memory_promoted",
|
||||
"--subject", "paziente attivo", "--detail", "seq:5",
|
||||
"--rationale", "scelta dal reviewer",
|
||||
],
|
||||
["phase", "advance", "--session", "s1"],
|
||||
["session", "finalize", "s1"],
|
||||
]);
|
||||
assert.match(result.content[0].text, /1 memorie salvate.*sessione finalizzata/s);
|
||||
});
|
||||
|
||||
test("F8 records a declined candidate without saving it and finalizes", async () => {
|
||||
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE] });
|
||||
answerNextWidget(ctx, []);
|
||||
|
||||
const result = await tools.get("reviewer_memory_promote").def.execute(
|
||||
"promote-declined",
|
||||
{ session: "s1" },
|
||||
null,
|
||||
null,
|
||||
ctx,
|
||||
);
|
||||
|
||||
const mutations = memoryPromotionMutationArgs(calls);
|
||||
assert.deepEqual(mutations, [
|
||||
[
|
||||
"decision", "add", "--session", "s1", "--type", "memory_promotion_declined",
|
||||
"--subject", "paziente attivo", "--detail", "seq:5",
|
||||
],
|
||||
["phase", "advance", "--session", "s1"],
|
||||
["session", "finalize", "s1"],
|
||||
]);
|
||||
assert.match(result.content[0].text, /1 candidati scartati.*sessione finalizzata/s);
|
||||
});
|
||||
|
||||
test("F8 with no promotion candidates finalizes without showing a widget", async () => {
|
||||
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [] });
|
||||
ctx.ui.input = async () => { throw new Error("a promotion widget must not be shown"); };
|
||||
|
||||
const result = await tools.get("reviewer_memory_promote").def.execute(
|
||||
"promote-absent",
|
||||
{ session: "s1" },
|
||||
null,
|
||||
null,
|
||||
ctx,
|
||||
);
|
||||
|
||||
assert.equal(calls.some(({ args }) => startsWithArgs(args, ["memory", "save-one"])), false);
|
||||
assert.equal(calls.some(({ args }) => startsWithArgs(args, ["decision", "add"])), false);
|
||||
assert.deepEqual(
|
||||
mutationArgs(calls, [["phase", "advance"], ["session", "finalize"]]),
|
||||
[["phase", "advance", "--session", "s1"], ["session", "finalize", "s1"]],
|
||||
);
|
||||
assert.equal(ctx.notifications.length, 1);
|
||||
assert.match(result.content[0].text, /Nessun candidato.*sessione finalizzata/s);
|
||||
});
|
||||
|
||||
test("F8 does not write the ledger or finalize when the vector save fails", async () => {
|
||||
const { ctx, tools, calls } = await setupGate({
|
||||
phase: 8,
|
||||
preview: [PROMOTION_CANDIDATE],
|
||||
fail: (args) => startsWithArgs(args, ["memory", "save-one"])
|
||||
? { status: 1, stderr: "vector save failed" }
|
||||
: null,
|
||||
function answerReview(ctx, items) {
|
||||
ctx.ui.input = async title => {
|
||||
const descriptor = JSON.parse(title);
|
||||
assert.equal(descriptor.widget, "memory-review");
|
||||
return JSON.stringify({ id: descriptor.id, text: JSON.stringify({
|
||||
summary_id: descriptor.summary.summary_id, items,
|
||||
}) });
|
||||
};
|
||||
}
|
||||
for (const selected of [true, false]) {
|
||||
test(`F8 persists the edited review (selected=${selected}) before its marker and finalization`, async () => {
|
||||
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE] });
|
||||
const items = selected ? [{ ...PROMOTION_CANDIDATE, card: { ...PROMOTION_CANDIDATE.card, detail: "Reviewer correction" } }] : [];
|
||||
answerReview(ctx, items);
|
||||
const result = await tools.get("reviewer_memory_promote").def.execute("review", { session: "s1" }, null, null, ctx);
|
||||
const mutations = memoryPromotionMutationArgs(calls);
|
||||
assert.deepEqual(mutations[0], ["memory", "review-apply", "--session", "s1", "--review-json",
|
||||
JSON.stringify({ summary_id: "summary-1", items }), "--json"]);
|
||||
assert(mutations[1].includes("memory_summary_reviewed"));
|
||||
assert.deepEqual(mutations.slice(2), [["phase", "advance", "--session", "s1"], ["session", "finalize", "s1"]]);
|
||||
assert.match(result.content[0].text, /sessione finalizzata/s);
|
||||
});
|
||||
answerNextWidget(ctx, ["seq-5"]);
|
||||
|
||||
const result = await tools.get("reviewer_memory_promote").def.execute(
|
||||
"promote-save-failure",
|
||||
{ session: "s1" },
|
||||
null,
|
||||
null,
|
||||
ctx,
|
||||
);
|
||||
|
||||
const mutations = memoryPromotionMutationArgs(calls);
|
||||
assert.deepEqual(mutations, [
|
||||
["memory", "save-one", "--session", "s1", "--decision", "5", "--json"],
|
||||
]);
|
||||
assert.match(result.content[0].text, /vector save failed.*Recupero manuale/s);
|
||||
}
|
||||
test("F8 warns about pending indexing but preserves the saved review and finalizes", async () => {
|
||||
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE], indexed: false });
|
||||
answerReview(ctx, [PROMOTION_CANDIDATE]);
|
||||
await tools.get("reviewer_memory_promote").def.execute("pending", { session: "s1" }, null, null, ctx);
|
||||
assert(ctx.notifications.some(({ message, level }) => level === "warning" && message.includes("Memory saved")));
|
||||
assert(memoryPromotionMutationArgs(calls).some(args => args.includes("memory_summary_reviewed")));
|
||||
});
|
||||
|
||||
test("F8 reports manual recovery and does not finalize after save succeeds but ledger fails", async () => {
|
||||
const { ctx, tools, calls } = await setupGate({
|
||||
phase: 8,
|
||||
preview: [PROMOTION_CANDIDATE],
|
||||
fail: (args) => args.includes("--type") && args.includes("memory_promoted")
|
||||
? { status: 1, stderr: "promotion ledger failed" }
|
||||
: null,
|
||||
test("F8 recovers a saved review without asking again or saving cards again", async () => {
|
||||
const { ctx, tools, calls } = await setupGate({ phase: 8, reviewed: true });
|
||||
ctx.ui.input = async () => { throw new Error("review must not be shown twice"); };
|
||||
await tools.get("reviewer_memory_promote").def.execute("recover", { session: "s1" }, null, null, ctx);
|
||||
const mutations = memoryPromotionMutationArgs(calls);
|
||||
assert.equal(mutations.length, 3);
|
||||
assert(mutations[0].includes("memory_summary_reviewed"));
|
||||
assert.deepEqual(mutations.at(-1), ["session", "finalize", "s1"]);
|
||||
});
|
||||
for (const failingStep of ["review-apply", "memory_summary_reviewed"]) {
|
||||
test(`F8 stops after failure at ${failingStep} without finalizing`, async () => {
|
||||
const { ctx, tools, calls } = await setupGate({ phase: 8, preview: [PROMOTION_CANDIDATE],
|
||||
fail: args => args.includes(failingStep) ? { status: 1, stderr: "persistence unavailable" } : null });
|
||||
answerReview(ctx, [PROMOTION_CANDIDATE]);
|
||||
const result = await tools.get("reviewer_memory_promote").def.execute("fail", { session: "s1" }, null, null, ctx);
|
||||
const mutations = memoryPromotionMutationArgs(calls);
|
||||
assert.equal(mutations.length, failingStep === "review-apply" ? 1 : 2);
|
||||
assert.match(result.content[0].text, /persistence unavailable/s);
|
||||
});
|
||||
answerNextWidget(ctx, ["seq-5"]);
|
||||
|
||||
const result = await tools.get("reviewer_memory_promote").def.execute(
|
||||
"promote-ledger-failure",
|
||||
{ session: "s1" },
|
||||
null,
|
||||
null,
|
||||
ctx,
|
||||
);
|
||||
|
||||
const mutations = memoryPromotionMutationArgs(calls);
|
||||
assert.deepEqual(mutations, [
|
||||
["memory", "save-one", "--session", "s1", "--decision", "5", "--json"],
|
||||
[
|
||||
"decision", "add", "--session", "s1", "--type", "memory_promoted",
|
||||
"--subject", "paziente attivo", "--detail", "seq:5",
|
||||
"--rationale", "scelta dal reviewer",
|
||||
],
|
||||
]);
|
||||
assert.match(result.content[0].text, /vectordb.*NON registrata.*recupero manuale/is);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -150,6 +150,36 @@
|
||||
"properties": { "session": { "type": "string" } }
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "reviewer_archive_repair",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"required": ["session"],
|
||||
"properties": {
|
||||
"session": { "type": "string" },
|
||||
"repair_id": { "type": "string" },
|
||||
"proposal": {
|
||||
"type": "object", "required": ["reason", "options"],
|
||||
"properties": {
|
||||
"reason": { "type": "string" },
|
||||
"options": {
|
||||
"type": "array", "minItems": 1, "maxItems": 5,
|
||||
"items": {
|
||||
"type": "object",
|
||||
"required": ["id", "label", "archive", "target_id", "revision", "content"],
|
||||
"properties": {
|
||||
"id": { "type": "string" }, "label": { "type": "string" },
|
||||
"archive": { "anyOf": [{ "const": "memory", "type": "string" }, { "const": "evidence", "type": "string" }] },
|
||||
"target_id": { "type": "string" }, "revision": { "type": "string" },
|
||||
"content": { "type": "object", "patternProperties": { "^.*$": {} } }
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"name": "rewrite_question",
|
||||
"parameters": {
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { Type } from "typebox";
|
||||
import { installRepairGate } from "./repair.js";
|
||||
|
||||
// Compatibility note: reviewer labels move verbatim from the composition root.
|
||||
// Tickets #22 and #23 require observable parity; translating existing chrome is a
|
||||
@@ -9,7 +10,7 @@ function normalizedText(value) {
|
||||
}
|
||||
|
||||
|
||||
const MEMORY_ID_RE = /\bmem-\d{4,}\b/i;
|
||||
const MEMORY_ID_RE = /\bmem-(?:[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}|\d{4,})\b/i;
|
||||
|
||||
|
||||
function memoryOptionKey(option) {
|
||||
@@ -160,208 +161,69 @@ async function reviewRecall(ctx, params, phase, dependencies) {
|
||||
}
|
||||
|
||||
|
||||
// The candidates come from `tht memory promote --preview --json` (deterministic,
|
||||
// reviewer-approved decisions only); the model never authors them.
|
||||
function dedupePromotionCandidates(candidates) {
|
||||
const seen = new Set();
|
||||
return candidates.filter((candidate) => {
|
||||
if (candidate.type !== "concept_clarified") return false;
|
||||
const key = [
|
||||
candidate.type,
|
||||
candidate.subject,
|
||||
candidate.detail,
|
||||
candidate.rationale,
|
||||
candidate.question_context,
|
||||
].map(normalizedText).join("\u0000");
|
||||
if (seen.has(key)) return false;
|
||||
seen.add(key);
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
function promotionOptions(candidates) {
|
||||
return candidates.map((candidate) => ({
|
||||
id: `seq-${candidate.decision_seq}`,
|
||||
label: `${candidate.type}: ${candidate.subject}`,
|
||||
detail: candidate.detail || "",
|
||||
rationale: candidate.rationale || "",
|
||||
meta: { question_context: candidate.question_context || "" },
|
||||
}));
|
||||
}
|
||||
|
||||
|
||||
function promotionContent(candidates) {
|
||||
return candidates
|
||||
.map(
|
||||
(candidate) =>
|
||||
`- **${candidate.type}: ${candidate.subject}** (decisione #${candidate.decision_seq})\n` +
|
||||
` ${candidate.detail || ""}\n` +
|
||||
` Motivo: ${candidate.rationale || "—"}\n` +
|
||||
` Domanda di contesto: ${candidate.question_context || "—"}`,
|
||||
)
|
||||
.join("\n");
|
||||
}
|
||||
|
||||
|
||||
function splitPromotionChoices(candidates, choices) {
|
||||
const chosen = new Set(choices ?? []);
|
||||
const promote = [];
|
||||
const decline = [];
|
||||
for (const candidate of candidates) {
|
||||
(chosen.has(`seq-${candidate.decision_seq}`) ? promote : decline).push(candidate);
|
||||
}
|
||||
return { promote, decline };
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Install the public F8 Memory tool against the narrow capability supplied by the
|
||||
* workflow composition root. Memory owns candidate policy, presentation and mutation
|
||||
* order; the capability keeps shared phase, ledger and persistence mechanisms in core.
|
||||
*/
|
||||
function installMemoryGate(
|
||||
pi,
|
||||
{ workflow, memory, ledger, reviewer, waitForReviewer, toTextResult },
|
||||
) {
|
||||
pi.registerTool({
|
||||
name: "reviewer_memory_promote",
|
||||
label: "Promozione memorie riusabili (reviewer)",
|
||||
description:
|
||||
"F8 (prima della chiusura di fase): propone al reviewer i candidati di promozione " +
|
||||
"calcolati dalla CLI (tht memory promote --preview: solo concept_clarified, " +
|
||||
"max 5, esclusi i gia' promossi/rifiutati). Le selezioni " +
|
||||
"vengono salvate nel vectordb (tht memory save-one) e registrate come memory_promoted; " +
|
||||
"le deselezioni come memory_promotion_declined (non riproposte). Nessun parametro oltre " +
|
||||
"alla sessione: i candidati sono deterministici, NON li scrivi tu. Registrata la " +
|
||||
"promozione, in F8 il gate chiude la fase e finalizza la sessione da solo: NON " +
|
||||
"presentare un reviewer_confirm dopo.",
|
||||
parameters: Type.Object({
|
||||
session: Type.String(),
|
||||
}),
|
||||
async execute(_id, params, _signal, _onUpdate, ctx) {
|
||||
workflow.activate();
|
||||
try {
|
||||
const { session } = params;
|
||||
const phase = workflow.phase(ctx, session);
|
||||
let candidates;
|
||||
try {
|
||||
candidates = JSON.parse(memory.execute(
|
||||
ctx,
|
||||
["promote", "--session", session, "--preview", "--json"],
|
||||
));
|
||||
} catch (error) {
|
||||
const message = (error.stderr || error.message || String(error)).toString().trim();
|
||||
return toTextResult(`Preview di promozione non disponibile: ${message}`);
|
||||
}
|
||||
if (!Array.isArray(candidates) || candidates.length === 0) {
|
||||
await ctx.ui.notify(
|
||||
"Nessuna decisione riusabile da promuovere in memoria per questa sessione.",
|
||||
"info",
|
||||
);
|
||||
const closed = workflow.close(
|
||||
ctx, session, phase.number, "Nessun candidato di promozione.",
|
||||
);
|
||||
if (closed) return closed;
|
||||
return toTextResult(
|
||||
"Nessun candidato di promozione: prosegui con la chiusura della sessione.",
|
||||
);
|
||||
}
|
||||
candidates = dedupePromotionCandidates(candidates);
|
||||
if (candidates.length === 0) {
|
||||
await ctx.ui.notify(
|
||||
"Nessun concetto chiarito da salvare come memory per questa sessione.",
|
||||
"info",
|
||||
);
|
||||
const closed = workflow.close(
|
||||
ctx, session, phase.number, "Nessun candidato di promozione.",
|
||||
);
|
||||
if (closed) return closed;
|
||||
return toTextResult(
|
||||
"Nessun candidato di promozione: prosegui con la chiusura della sessione.",
|
||||
);
|
||||
}
|
||||
const options = promotionOptions(candidates);
|
||||
const widget = reviewer.buildMultiselect({
|
||||
id: `u${Date.now()}`,
|
||||
phase: phase.id,
|
||||
title: "Quali concetti chiariti salvare nella memoria riutilizzabile?",
|
||||
allowEmpty: true,
|
||||
options,
|
||||
selected: options.map((option) => option.id),
|
||||
content: promotionContent(candidates),
|
||||
});
|
||||
const response = await waitForReviewer(ctx, widget);
|
||||
if (response.control === "freetext") {
|
||||
return toTextResult(
|
||||
`Altro (reviewer): ${response.text}. Valuta e ripresenta il gate.`,
|
||||
);
|
||||
}
|
||||
if (response.control === "back") {
|
||||
return toTextResult("Il reviewer vuole tornare indietro.");
|
||||
}
|
||||
if (response.control === "exit") {
|
||||
return toTextResult("Il reviewer vuole uscire.");
|
||||
}
|
||||
const { promote, decline } = splitPromotionChoices(candidates, response.choices);
|
||||
let saved = 0;
|
||||
for (const candidate of promote) {
|
||||
const saveError = memory.mutate(
|
||||
ctx,
|
||||
[
|
||||
"save-one", "--session", session,
|
||||
"--decision", String(candidate.decision_seq), "--json",
|
||||
],
|
||||
`Recupero manuale (umano): tht memory save-one --session ${session} ` +
|
||||
`--decision ${candidate.decision_seq}. Finora salvate: ${saved}.`,
|
||||
);
|
||||
if (saveError) return saveError;
|
||||
const ledgerError = ledger.record(
|
||||
ctx,
|
||||
session,
|
||||
{
|
||||
type: "memory_promoted",
|
||||
subject: candidate.subject,
|
||||
detail: `seq:${candidate.decision_seq}`,
|
||||
rationale: candidate.rationale || candidate.detail || "",
|
||||
},
|
||||
"Memoria salvata nel vectordb ma decisione memory_promoted NON registrata: " +
|
||||
`recupero manuale (umano) con tht decision add --session ${session} ` +
|
||||
`--type memory_promoted --subject "${candidate.subject}" ` +
|
||||
`--detail seq:${candidate.decision_seq}.`,
|
||||
);
|
||||
if (ledgerError) return ledgerError;
|
||||
saved++;
|
||||
}
|
||||
for (const candidate of decline) {
|
||||
const ledgerError = ledger.record(ctx, session, {
|
||||
type: "memory_promotion_declined",
|
||||
subject: candidate.subject,
|
||||
detail: `seq:${candidate.decision_seq}`,
|
||||
}, "");
|
||||
if (ledgerError) return ledgerError;
|
||||
}
|
||||
const summary =
|
||||
`Promozione registrata: ${saved} memorie salvate nel vectordb, ` +
|
||||
`${decline.length} candidati scartati.`;
|
||||
const closed = workflow.close(ctx, session, phase.number, summary);
|
||||
if (closed) return closed;
|
||||
return toTextResult(summary);
|
||||
} catch (fatal) {
|
||||
const message = (fatal.stderr || fatal.message || String(fatal)).toString().trim();
|
||||
return toTextResult(
|
||||
`[reviewer_memory_promote ERRORE INTERNO] ${message}. ` +
|
||||
"Riprova o usa un approccio diverso.",
|
||||
);
|
||||
}
|
||||
},
|
||||
});
|
||||
function installMemoryGate(pi, { workflow, memory, ledger, waitForReviewer, toTextResult }) {
|
||||
pi.registerTool({
|
||||
name: "reviewer_memory_promote",
|
||||
label: "Review Memory summary",
|
||||
description: "F8: show the editable Memory summary from persisted proposals and approved " +
|
||||
"decisions. The reviewer selects additions, updates and links. Only those choices " +
|
||||
"are saved. Then close F8 and finalize. Prepare reusable rules during the workflow " +
|
||||
"with tht memory propose; do not call review-apply yourself.",
|
||||
parameters: Type.Object({ session: Type.String() }),
|
||||
async execute(_id, params, _signal, _onUpdate, ctx) {
|
||||
workflow.activate();
|
||||
try {
|
||||
const { session } = params;
|
||||
const phase = workflow.phase(ctx, session);
|
||||
if (phase.number > 8) return workflow.close(ctx, session, phase.number, "Memory reviewed.");
|
||||
if (phase.number !== 8) return toTextResult("Review the Memory summary at the end of F8.");
|
||||
const summary = JSON.parse(memory.execute(ctx, ["summary", "--session", session, "--json"]));
|
||||
if (summary.reviewed) {
|
||||
const failure = ledger.record(ctx, session, {
|
||||
type: "memory_summary_reviewed", subject: summary.summary_id,
|
||||
detail: "Previously saved Memory review recovered.",
|
||||
}, "Retry to record the recovered Memory review.");
|
||||
if (failure) return failure;
|
||||
return workflow.close(ctx, session, phase.number, "Memory review recovered.");
|
||||
}
|
||||
const response = await waitForReviewer(ctx, {
|
||||
id: `u${Date.now()}`, schema_version: 1, phase: phase.id,
|
||||
widget: "memory-review", title: "Memory for future questions",
|
||||
summary, reserved: ["other", "back", "exit"],
|
||||
});
|
||||
if (response.control) return toTextResult(
|
||||
response.control === "freetext"
|
||||
? `Reviewer feedback: ${response.text}. Revise the proposals and present the summary again.`
|
||||
: `Reviewer requested: ${response.control}.`);
|
||||
const selection = JSON.parse(response.text || "{}");
|
||||
if (selection.summary_id !== summary.summary_id || !Array.isArray(selection.items))
|
||||
return toTextResult("Invalid Memory review response. Present the summary again.");
|
||||
const failure = await memory.mutate(ctx, [
|
||||
"review-apply", "--session", session, "--review-json", JSON.stringify(selection), "--json",
|
||||
], "The Memory review is recoverable. Present the summary again if its content changed.");
|
||||
if (failure) return failure;
|
||||
const selected = new Set(selection.items.map(item => item.id));
|
||||
const ledgerFailure = ledger.record(ctx, session, {
|
||||
type: "memory_summary_reviewed", subject: summary.summary_id,
|
||||
detail: `Saved ${selected.size} cards; declined ${summary.items.length-selected.size}.`,
|
||||
}, "Memory is saved. Retry to record the review and finalize.");
|
||||
if (ledgerFailure) return ledgerFailure;
|
||||
const message = `Memory reviewed: ${selected.size} cards saved, ${summary.items.length-selected.size} declined.`;
|
||||
return workflow.close(ctx, session, phase.number, message) || toTextResult(message);
|
||||
} catch (error) {
|
||||
return toTextResult(`[reviewer_memory_promote] ${String(error.stderr || error.message || error).trim()}`);
|
||||
}
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
export function createMemoryGate(dependencies) {
|
||||
return {
|
||||
install: (pi) => installMemoryGate(pi, dependencies),
|
||||
install: (pi) => {
|
||||
installMemoryGate(pi, dependencies);
|
||||
installRepairGate(pi, dependencies);
|
||||
},
|
||||
reviewRecall: (ctx, params, phase) => reviewRecall(ctx, params, phase, dependencies),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
import { Type } from "typebox";
|
||||
|
||||
export function installRepairGate(pi, { workflow, memory, waitForReviewer, toTextResult }) {
|
||||
pi.registerTool({
|
||||
name: "reviewer_archive_repair",
|
||||
label: "Resolve Memory / Evidence conflict",
|
||||
description: "Present closed archive correction choices with current and resulting content. " +
|
||||
"Pass a reason and options (id, label, archive memory|evidence, target_id, revision, " +
|
||||
"complete content). Only the human selects the persistent correction. " +
|
||||
"Resume with repair_id from tht memory repairs --session. Never run repair-apply yourself. " +
|
||||
"This gate does not approve or advance the current workflow phase.",
|
||||
parameters: Type.Object({
|
||||
session: Type.String(),
|
||||
repair_id: Type.Optional(Type.String()),
|
||||
proposal: Type.Optional(Type.Object({
|
||||
reason: Type.String(), options: Type.Array(Type.Object({
|
||||
id: Type.String(), label: Type.String(),
|
||||
archive: Type.Union([Type.Literal("memory"), Type.Literal("evidence")]),
|
||||
target_id: Type.String(), revision: Type.String(),
|
||||
content: Type.Record(Type.String(), Type.Unknown()),
|
||||
}), { minItems: 1, maxItems: 5 }),
|
||||
})),
|
||||
}),
|
||||
async execute(_id, params, _signal, _update, ctx) {
|
||||
workflow.activate();
|
||||
try {
|
||||
if (!!params.repair_id === !!params.proposal)
|
||||
return toTextResult("Supply either a new proposal or an existing repair_id.");
|
||||
const run = args => JSON.parse(memory.execute(ctx,
|
||||
[...args, "--session", params.session, "--json"]));
|
||||
let repair = params.repair_id
|
||||
? run(["repair-show", "--repair-id", params.repair_id])
|
||||
: run(["repair-prepare", "--proposal-json", JSON.stringify(params.proposal)]);
|
||||
let error;
|
||||
while (true) {
|
||||
const response = await waitForReviewer(ctx, {
|
||||
id: `u${Date.now()}`, schema_version: 1,
|
||||
phase: workflow.phase(ctx, params.session).id,
|
||||
widget: "archive-repair", title: "Resolve archive conflict",
|
||||
repair, error, reserved: ["other", "back", "exit"],
|
||||
});
|
||||
if (response.control) return toTextResult(response.control === "freetext"
|
||||
? `Reviewer feedback: ${response.text}. Reformulate the repair choices. ` +
|
||||
`Existing receipt ${repair.repair_id}: ${repair.status}.`
|
||||
: `Reviewer requested ${response.control}. Repair ${repair.repair_id}: ${repair.status}.`);
|
||||
const selected = response.choices;
|
||||
if (!Array.isArray(selected) || selected.length !== 1)
|
||||
return toTextResult("Invalid archive repair selection. Present the gate again.");
|
||||
const choice = selected[0];
|
||||
if (choice === "continue" && (repair.choice || !repair.can_apply))
|
||||
return toTextResult(JSON.stringify({ repair_id: repair.repair_id,
|
||||
status: repair.status, saved: repair.saved, indexed: repair.indexed,
|
||||
choice: repair.choice,
|
||||
correction: repair.options.find(o => o.id === repair.choice)?.content,
|
||||
next: "Continue the normal question review gates. Archive status above is authoritative." }));
|
||||
if (choice !== "reject" && !repair.options.some(o => o.id === choice))
|
||||
return toTextResult("Select only a displayed repair choice.");
|
||||
try {
|
||||
repair = run(["repair-apply", "--repair-id", repair.repair_id, "--choice", choice]);
|
||||
error = undefined;
|
||||
if (repair.status === "rejected")
|
||||
return toTextResult("All repair proposals were rejected as inadequate. " +
|
||||
"No archive was changed. Reformulate the options for the reviewer.");
|
||||
} catch (failure) {
|
||||
error = "The correction could not complete. Review the current archive state before retrying.";
|
||||
repair = run(["repair-show", "--repair-id", repair.repair_id]);
|
||||
}
|
||||
}
|
||||
} catch (error) {
|
||||
return toTextResult(`[reviewer_archive_repair] ${String(error.stderr || error.message || error).trim()}`);
|
||||
}
|
||||
},
|
||||
});
|
||||
}
|
||||
@@ -152,7 +152,7 @@ const FORBIDDEN = [
|
||||
/\btht\s+decision\s+add\b/,
|
||||
/\btht\s+cte\s+plan\b/,
|
||||
// La promozione in memoria passa dal reviewer (reviewer_memory_promote), mai da shell.
|
||||
/\btht\s+memory\s+(promote|save-one)\b/,
|
||||
/\btht\s+memory\s+(promote|save-one|review-apply|repair-apply|create|update|delete|admin|index)\b/,
|
||||
// La re-introspezione del DWH (~3 min) è manutenzione fuori sessione, mai in workflow.
|
||||
/\btht\s+schema\s+introspect\b[^\n]*--refresh/,
|
||||
];
|
||||
@@ -627,9 +627,17 @@ export default function (pi) {
|
||||
},
|
||||
memory: {
|
||||
execute: (ctx, args) => tht(ctx, ["memory", ...args]),
|
||||
mutate: (ctx, args, recovery) => relayIfThtFails(
|
||||
ctx, ["memory", ...args], recovery,
|
||||
),
|
||||
mutate: async (ctx, args, recovery) => {
|
||||
try {
|
||||
const result = JSON.parse(tht(ctx, ["memory", ...args]));
|
||||
if (result.indexed === false) {
|
||||
await ctx.ui.notify("Memory saved. Index update is incomplete; an administrator can retry it in Memory management.", "warning");
|
||||
}
|
||||
return null;
|
||||
} catch (error) {
|
||||
return textResult(`${(error.stderr || error.message || String(error)).toString().trim()} ${recovery}`);
|
||||
}
|
||||
},
|
||||
},
|
||||
ledger: {
|
||||
validate: validateDecisionTypes,
|
||||
|
||||
@@ -9,6 +9,10 @@ You receive exactly one normalized Source Evidence request. Return one JSON obje
|
||||
only a `candidates` array, without Markdown fences, comments, or explanatory text. The
|
||||
host strictly rejects unknown or missing fields.
|
||||
|
||||
Candidate JSON uses response protocol version 1. The host renders accepted candidates
|
||||
as editable Curated Evidence v4; supply the typed payload and exact excerpts in this
|
||||
response, leaving file rendering and provenance metadata to the host.
|
||||
|
||||
Use only facts present in `normalized_text`. Never merge, cite, or infer facts from
|
||||
another source. Reuse an `existing_id` only when it was supplied in `previous_units`;
|
||||
otherwise omit it. Preserve prior reviewed wording when it is still supported. When a
|
||||
|
||||
@@ -81,6 +81,9 @@ kind:"cte_result"` of the plan (F6), `reviewer_confirm kind:"sql"` (F7), and
|
||||
Back" (rollback, see discipline 11), or "Esci/Exit" (session abort). If the
|
||||
reviewer closes without choosing, the gate re-presents the same widget — there is
|
||||
no silent skip.
|
||||
When Memory and Evidence conflict and a shared archive needs correction, read
|
||||
`archive-repair.md` and use `reviewer_archive_repair`. Its receipt reports the
|
||||
persistent correction; the normal phase gate still approves the current question.
|
||||
6. **Self-contained messages.** When you call any `reviewer_*` tool, ALWAYS include
|
||||
in the `message` (or in the `options`' labels/descriptions) a concise recap of the
|
||||
context the reviewer needs to decide: what was asked, what you found, what each
|
||||
@@ -310,6 +313,10 @@ Prerequisite: Phase 3 closed.
|
||||
questions show which tables comparable questions used. Cite relevant precedents
|
||||
(session id + tables) to the reviewer as CONTEXT — they are reference material,
|
||||
NOT decisions to apply; their filters/periods may not transfer.
|
||||
Run `tht memory rules "<question>" --session <id> --json` for reusable join rules
|
||||
and explained errors. Read `memory-review.md` when a rule is relevant or a reviewer
|
||||
approves a reusable correction. Present the applicable rule and its Memory ID in
|
||||
the existing table/join proposal; that gate decides its use for this question.
|
||||
2. Propose tables to promote/exclude with **`reviewer_schema_linking`**: pass
|
||||
`tables[]` as `{id, name, kind: "promote"|"exclude", rationale, suggested_columns}`.
|
||||
Do NOT list every column yourself — the gate loads the full column set (with
|
||||
@@ -383,6 +390,8 @@ Prerequisite: Phase 5 closed.
|
||||
|
||||
1. Read `cte.md`. Decompose the rewritten question into CTEs (Agent View Generation):
|
||||
each CTE captures an informative subset with a clear purpose, named in snake_case.
|
||||
Consult `tht memory rules "<question>" --session <id> --json` and follow
|
||||
`memory-review.md` for reusable calculation rules and explained errors.
|
||||
`tht memory solved-search "<question>" --json` shows how similar solved questions
|
||||
were structured — use as reference only.
|
||||
2. Present the full CTE plan to the reviewer with `reviewer_confirm kind:"cte_plan"`,
|
||||
@@ -428,6 +437,8 @@ Prerequisite: Phase 6 closed.
|
||||
|
||||
1. Read `sql-generation.md`. Recursive divide-and-conquer: the CTEs approved in
|
||||
Phase 6 are the preferred building blocks (reuse them by name).
|
||||
Consult `tht memory rules "<question>" --session <id> --json`; show applicable
|
||||
rules and Memory IDs in the SQL explanation approved by the existing SQL gate.
|
||||
`tht memory solved-search "<question>" --json` gives the final SQL of similar
|
||||
solved questions: reference exemplars — never copy filters, periods or
|
||||
populations without checking them against the current rewritten question.
|
||||
@@ -460,15 +471,12 @@ Prerequisite: Phase 7 closed.
|
||||
2. On a server, if the reviewer chose yes: `tht datamart generate` (stub — raises
|
||||
NotImplementedError for now). Tell
|
||||
the reviewer that dbt generation is not implemented yet.
|
||||
3. **Memory promotion closes the session.** Call `reviewer_memory_promote` with ONLY
|
||||
the session id: the gate computes the candidates itself (`tht memory promote
|
||||
--preview` — the 3 reusable types, already excluding promoted/declined ones) and
|
||||
shows the reviewer a pre-selected checklist. Selected → saved to the vectordb +
|
||||
`memory_promoted`; deselected → `memory_promotion_declined` (never re-proposed).
|
||||
After recording the promotion (even with zero candidates) the gate advances F8 and
|
||||
finalizes the session itself — do NOT present a `reviewer_confirm kind:"phase"`
|
||||
afterwards: there is nothing left to approve. When the gate answers "sessione
|
||||
finalizzata", give the reviewer the final summary and end the turn.
|
||||
3. **Memory review closes the session.** Follow `memory-review.md` to finish any
|
||||
persisted proposals, then call `reviewer_memory_promote` with the session ID.
|
||||
The reviewer edits and selects additions, explicit updates and links in one
|
||||
summary, including the consultative solved question. Only the selected cards
|
||||
are saved. The gate records `memory_summary_reviewed`, advances F8 and finalizes.
|
||||
Once it reports finalization, give the session summary and end the turn.
|
||||
4. If the gate reports an error instead (e.g. the datamart decision is missing),
|
||||
fix the prerequisite and call `reviewer_memory_promote` again. Only if the gate
|
||||
says the session is still open, close with `reviewer_confirm kind:"phase"` as a
|
||||
@@ -478,7 +486,8 @@ Prerequisite: Phase 7 closed.
|
||||
|
||||
When the promotion gate (or, as fallback, the F8 phase gate) closes Phase 8, the
|
||||
gate calls `tht session finalize` automatically.
|
||||
Finalize also indexes the question→SQL pair in the vectordb (kind `solved_question`,
|
||||
best-effort — on failure recover with `tht memory solved-index <id>`). The persisted
|
||||
Exemplars are saved only when selected in the Memory summary. Finalize performs
|
||||
no additional automatic Memory writes. Pending indexing is recovered through
|
||||
Memory management. The persisted
|
||||
state (ledger `review_decisions.jsonl` + artifacts) is the truth: what is not
|
||||
recorded did not happen.
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
# Resolve a conflict in the shared archives
|
||||
|
||||
Use this gate when the retrieved Memory and Evidence disagree and the reviewer
|
||||
must decide which archive to correct. State the conflict, its effect on the current
|
||||
question, and the concrete alternative corrections. Each choice replaces one existing
|
||||
Memory Card or one Evidence unit; offer only changes that resolve the stated conflict.
|
||||
|
||||
1. Read the complete current target and revision through
|
||||
`tht memory repair-target --session <id> --archive memory|evidence --target-id <id> --json`.
|
||||
Keep all content fields that the proposed correction does not change. An Evidence
|
||||
correction retains its identity, kind and original source history. Evidence must
|
||||
already be consolidated; external file edits must be reconciled first.
|
||||
2. Call `reviewer_archive_repair` with `session` and `proposal`:
|
||||
`{reason, options:[{id,label,archive,target_id,revision,content}]}`.
|
||||
`content` is the complete resulting card or Evidence unit. Use one to five distinct
|
||||
choice IDs, excluding the reserved `reject` and `continue`. The gate loads the current
|
||||
content itself and shows both versions. The human selects the archive correction.
|
||||
3. A rejection means every proposal was inadequate. Reformulate the choices using the
|
||||
reviewer's feedback and present a new proposal. No archive mutation follows rejection.
|
||||
A non-administrator can reject or continue the question; shared corrections require
|
||||
an administrator. Do not disguise a shared correction as a final-summary promotion.
|
||||
4. The result distinguishes `active`, `pending_activation`, `applying`, and `superseded`.
|
||||
`saved` alone does not establish retrieval availability. The gate offers retry of
|
||||
the same approved correction after an index failure. A newer archive edit requires
|
||||
reconciliation and a new proposal; replay never overwrites that edit.
|
||||
5. After interruption, run `tht memory repairs --session <id> --json`, then call
|
||||
`reviewer_archive_repair` with `session` and `repair_id`. This refreshes current
|
||||
archive status. The list's `recorded_status` is historical. Resume the existing
|
||||
receipt for recovery instead of proposing the already-saved correction again.
|
||||
6. Continue the ordinary clarification/schema/SQL gate for the current question,
|
||||
carrying the chosen meaning and the reported archive status. Archive repair does
|
||||
not advance a phase. If activation remains pending, report that explicitly.
|
||||
|
||||
The extension alone invokes `repair-apply` after the human response. Model-authored
|
||||
shell commands may prepare or inspect proposals, but cannot apply them. Correction
|
||||
receipts, like phase artifacts, persist independently of the live chat. Git remains
|
||||
an operator action after reviewing the Evidence file diff.
|
||||
@@ -0,0 +1,74 @@
|
||||
# Reusable Memory during the workflow
|
||||
|
||||
Consult rules at schema linking and SQL construction with
|
||||
`tht memory rules "<current question>" --session <id> --json`.
|
||||
An optional `--filters '{"table":"orders","column":"id"}'` narrows the physical
|
||||
context. The runtime fixes database and schema. A returned card is a candidate:
|
||||
explain its scope and Memory ID in the existing join, CTE or SQL proposal. That
|
||||
gate approves the concrete use; a retrieved link is not an approval. Exemplars
|
||||
remain references to previous questions.
|
||||
|
||||
## Prepare additions and updates
|
||||
|
||||
After a reviewer approves a reusable definition, join/calculation rule or explained
|
||||
correction, prepare a card citing the effective decision sequence(s) from
|
||||
`tht session show <id>`. Preserve the explanation and physical dependencies.
|
||||
Keep query-specific choices (for example only using 2024) in the solved question.
|
||||
An unselected option, unexplained rejection or timeout is not a reusable source.
|
||||
When an error and correction state the same rule, prepare one card.
|
||||
|
||||
Write the complete current proposal list to a temporary JSON file and run
|
||||
`tht memory propose --session <id> --data <file.json>`. This persists
|
||||
`memory_proposals.json` in the session without changing the shared archive.
|
||||
Reuse proposal IDs when refining their wording. The accepted shape is:
|
||||
|
||||
```json
|
||||
[
|
||||
{
|
||||
"id": "order-grain",
|
||||
"source_seqs": [11],
|
||||
"reason": "The reviewer corrected repeated header totals; this also applies to future questions.",
|
||||
"card": {
|
||||
"family": "sql_rule",
|
||||
"subject": "Order totals after joining lines",
|
||||
"detail": "Aggregate each order once before combining it with line totals.",
|
||||
"scope": "Sales orders and order lines",
|
||||
"rationale": "A line join repeats the header amount once per line.",
|
||||
"concepts": ["order grain"],
|
||||
"dependencies": [{"database": "warehouse", "schema_name": "sales", "table": "orders", "column": "total"}],
|
||||
"links": []
|
||||
}
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
Use the actual decision numbers and schema identifiers from this session. Families
|
||||
are `domain_clarification`, `sql_rule`, `explained_error` and `solved_question`.
|
||||
Explained errors require both the corrected behavior and an approved explanation.
|
||||
The summary adds domain clarifications and the current approved exemplar when they
|
||||
are not covered by authored proposals, so author only the additions/changes needed.
|
||||
|
||||
For an update, include `target_id` and `target_revision` from the current recalled
|
||||
card, explain the change in `reason`, and provide its complete resulting content.
|
||||
The reviewer sees the current card alongside the proposal. A conflicting manual
|
||||
edit requires a refreshed proposal and another review; it is not overwritten.
|
||||
Keep different rules with different scopes separate. Exact duplicates do not need
|
||||
another card. Semantic deletion belongs to Memory management.
|
||||
|
||||
Links contain `target_id` and `meaning`. Use a current Memory ID for an existing
|
||||
destination or `proposal:<proposal-id>` for another new card in the summary. The
|
||||
reviewer may edit/remove links; a link to a new card requires selecting that card.
|
||||
|
||||
## Final review
|
||||
|
||||
At the end of F8, call `reviewer_memory_promote`. The widget permits content, scope,
|
||||
concept, dependency and link edits and accepts an empty selection. Approved exemplar
|
||||
SQL remains the session's solution; changing it requires returning to SQL review.
|
||||
The gate applies selected cards and links atomically in PostgreSQL and then updates
|
||||
Qdrant. An incomplete index update is reported and retained for explicit retry.
|
||||
An interrupted review is recovered from its persisted receipt rather than applied
|
||||
again. Once the gate reports finalization, end the session without another gate.
|
||||
|
||||
For a Memory/Evidence conflict requiring a persistent correction, follow
|
||||
`archive-repair.md`. A correction already saved by that gate does not need a
|
||||
duplicate update in the final summary.
|
||||
@@ -1,12 +1,9 @@
|
||||
3. **Memory promotion closes the session.** Call `reviewer_memory_promote` with ONLY
|
||||
the session id: the gate computes the candidates itself (`tht memory promote
|
||||
--preview` — the 3 reusable types, already excluding promoted/declined ones) and
|
||||
shows the reviewer a pre-selected checklist. Selected → saved to the vectordb +
|
||||
`memory_promoted`; deselected → `memory_promotion_declined` (never re-proposed).
|
||||
After recording the promotion (even with zero candidates) the gate advances F8 and
|
||||
finalizes the session itself — do NOT present a `reviewer_confirm kind:"phase"`
|
||||
afterwards: there is nothing left to approve. When the gate answers "sessione
|
||||
finalizzata", give the reviewer the final summary and end the turn.
|
||||
3. **Memory review closes the session.** Follow `memory-review.md` to finish any
|
||||
persisted proposals, then call `reviewer_memory_promote` with the session ID.
|
||||
The reviewer edits and selects additions, explicit updates and links in one
|
||||
summary, including the consultative solved question. Only the selected cards
|
||||
are saved. The gate records `memory_summary_reviewed`, advances F8 and finalizes.
|
||||
Once it reports finalization, give the session summary and end the turn.
|
||||
4. If the gate reports an error instead (e.g. the datamart decision is missing),
|
||||
fix the prerequisite and call `reviewer_memory_promote` again. Only if the gate
|
||||
says the session is still open, close with `reviewer_confirm kind:"phase"` as a
|
||||
|
||||
@@ -1,2 +1,3 @@
|
||||
Finalize also indexes the question→SQL pair in the vectordb (kind `solved_question`,
|
||||
best-effort — on failure recover with `tht memory solved-index <id>`). The persisted
|
||||
Exemplars are saved only when selected in the Memory summary. Finalize performs
|
||||
no additional automatic Memory writes. Pending indexing is recovered through
|
||||
Memory management. The persisted
|
||||
|
||||
@@ -2,3 +2,7 @@
|
||||
questions show which tables comparable questions used. Cite relevant precedents
|
||||
(session id + tables) to the reviewer as CONTEXT — they are reference material,
|
||||
NOT decisions to apply; their filters/periods may not transfer.
|
||||
Run `tht memory rules "<question>" --session <id> --json` for reusable join rules
|
||||
and explained errors. Read `memory-review.md` when a rule is relevant or a reviewer
|
||||
approves a reusable correction. Present the applicable rule and its Memory ID in
|
||||
the existing table/join proposal; that gate decides its use for this question.
|
||||
|
||||
@@ -1,2 +1,4 @@
|
||||
Consult `tht memory rules "<question>" --session <id> --json` and follow
|
||||
`memory-review.md` for reusable calculation rules and explained errors.
|
||||
`tht memory solved-search "<question>" --json` shows how similar solved questions
|
||||
were structured — use as reference only.
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
Consult `tht memory rules "<question>" --session <id> --json`; show applicable
|
||||
rules and Memory IDs in the SQL explanation approved by the existing SQL gate.
|
||||
`tht memory solved-search "<question>" --json` gives the final SQL of similar
|
||||
solved questions: reference exemplars — never copy filters, periods or
|
||||
populations without checking them against the current rewritten question.
|
||||
|
||||
@@ -81,6 +81,9 @@ kind:"cte_result"` of the plan (F6), `reviewer_confirm kind:"sql"` (F7), and
|
||||
Back" (rollback, see discipline 11), or "Esci/Exit" (session abort). If the
|
||||
reviewer closes without choosing, the gate re-presents the same widget — there is
|
||||
no silent skip.
|
||||
When Memory and Evidence conflict and a shared archive needs correction, read
|
||||
`archive-repair.md` and use `reviewer_archive_repair`. Its receipt reports the
|
||||
persistent correction; the normal phase gate still approves the current question.
|
||||
6. **Self-contained messages.** When you call any `reviewer_*` tool, ALWAYS include
|
||||
in the `message` (or in the `options`' labels/descriptions) a concise recap of the
|
||||
context the reviewer needs to decide: what was asked, what you found, what each
|
||||
|
||||
@@ -35,7 +35,7 @@ dev = [
|
||||
include = ["tht*"]
|
||||
|
||||
[tool.setuptools.package-data]
|
||||
tht = ["migrations/sessions/*.sql"]
|
||||
tht = ["migrations/sessions/*.sql", "migrations/memory/*.sql"]
|
||||
|
||||
[tool.ruff]
|
||||
line-length = 100
|
||||
|
||||
@@ -66,9 +66,22 @@
|
||||
"config check",
|
||||
"db fetch-ca",
|
||||
"doctor",
|
||||
"memory admin",
|
||||
"memory create",
|
||||
"memory delete",
|
||||
"memory index",
|
||||
"memory list",
|
||||
"memory pending",
|
||||
"memory propose",
|
||||
"memory repair-target",
|
||||
"memory repair-prepare",
|
||||
"memory repair-show",
|
||||
"memory repair-apply",
|
||||
"memory repairs",
|
||||
"memory summary",
|
||||
"memory review-apply",
|
||||
"memory rules",
|
||||
"memory retry",
|
||||
"memory show",
|
||||
"memory update"
|
||||
],
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,5 +1,6 @@
|
||||
import json
|
||||
import uuid
|
||||
from contextlib import nullcontext
|
||||
from datetime import UTC, datetime
|
||||
from types import SimpleNamespace
|
||||
|
||||
@@ -7,7 +8,9 @@ from typer.testing import CliRunner
|
||||
|
||||
from tht.cli import app
|
||||
from tht.decisions import DecisionInput
|
||||
from tht.memory import MemoryRecord, recall_memories, save_registry
|
||||
from tht.memory import MemoryRecord, recall_memories
|
||||
from tht.memory.models import Card
|
||||
from tht.memory.service import MemoryService
|
||||
from tht.phase import current_phase
|
||||
from tht.session.filesystem_repository import FilesystemSessionRepository
|
||||
from tht.session.models import PrincipalContext, SessionManifest
|
||||
@@ -42,7 +45,7 @@ class Searcher:
|
||||
self.hits = hits
|
||||
self.calls = []
|
||||
|
||||
def search(self, embedding, *, top_n, kinds):
|
||||
def search(self, embedding, *, top_n, kinds, **kwargs):
|
||||
self.calls.append((embedding, top_n, kinds))
|
||||
return self.hits
|
||||
|
||||
@@ -145,11 +148,21 @@ def test_recall_cli_reconstructs_applied_and_rejected_memory_from_persisted_f2_s
|
||||
|
||||
records = [_memory("mem-0001"), _memory("mem-0003")]
|
||||
searcher = Searcher([
|
||||
SimpleNamespace(ref="mem-0003", similarity=0.9),
|
||||
SimpleNamespace(ref="mem-0001", similarity=0.8),
|
||||
SimpleNamespace(ref="mem-0003", similarity=0.9,
|
||||
metadata={"memory_revision": "r", "memory_format": 2}),
|
||||
SimpleNamespace(ref="mem-0001", similarity=0.8,
|
||||
metadata={"memory_revision": "r", "memory_format": 2}),
|
||||
])
|
||||
embedder = Embedder()
|
||||
save_registry(records, tmp_path / "artifacts" / "memory" / "registry.jsonl")
|
||||
cards = {r.id: Card(id=r.id, family="domain_clarification", subject=r.subject,
|
||||
detail=r.detail, scope="psd-clinical", workspace_id="psd-clinical", origin="workflow",
|
||||
created_at=r.ts, updated_at=r.ts, revision="r", indexed=True) for r in records}
|
||||
archive = SimpleNamespace(list=lambda query: {}, get=lambda identity: cards[identity],
|
||||
close=lambda: None)
|
||||
archive.operation = lambda: nullcontext(archive)
|
||||
service = MemoryService(archive, PrincipalContext(issuer="local", subject="reviewer"),
|
||||
store_factory=lambda: None, embedder_factory=Embedder)
|
||||
monkeypatch.setattr("tht.cli.memory_cmd.memory_service", lambda cfg: service)
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.open_searcher", lambda cfg: searcher)
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: embedder)
|
||||
|
||||
@@ -165,4 +178,4 @@ def test_recall_cli_reconstructs_applied_and_rejected_memory_from_persisted_f2_s
|
||||
assert json.loads(response.stdout) == []
|
||||
assert current_phase(repository.get(session_id)) == 2
|
||||
assert embedder.questions == ["active patients"]
|
||||
assert searcher.calls == [([0.1, 0.2], 5, ["memory"])]
|
||||
assert searcher.calls == [([0.1, 0.2], 20, ["memory"])]
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
"""Deterministic retrieval policy tests, separate from actual embedding/Qdrant recovery."""
|
||||
|
||||
from contextlib import nullcontext
|
||||
from datetime import UTC, datetime
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from tht.memory.models import Card, MemoryNotFound
|
||||
from tht.memory.retrieval import RecallScope, expand_and_rank
|
||||
|
||||
|
||||
def card(identity, **values):
|
||||
return Card(id=identity, workspace_id="sales", family="domain_clarification",
|
||||
subject=identity, scope="Sales", origin="manual", revision="r", indexed=True,
|
||||
created_at=datetime.now(UTC), updated_at=datetime.now(UTC), **values)
|
||||
|
||||
|
||||
class Archive:
|
||||
def __init__(self, *cards):
|
||||
self.cards = {c.id: c for c in cards}
|
||||
self.reads = []
|
||||
|
||||
def get(self, identity):
|
||||
self.reads.append(identity)
|
||||
if identity not in self.cards:
|
||||
raise MemoryNotFound(identity)
|
||||
return self.cards[identity]
|
||||
|
||||
def operation(self):
|
||||
return nullcontext(self)
|
||||
|
||||
|
||||
def hit(identity, **metadata):
|
||||
return SimpleNamespace(ref=identity, metadata={"memory_revision": "r", "memory_format": 2,
|
||||
**metadata})
|
||||
|
||||
|
||||
def link(identity):
|
||||
return {"target_id": identity, "meaning": "Requires the grain clarification"}
|
||||
|
||||
|
||||
def rank(repo, hits, **kwargs):
|
||||
return expand_and_rank(repo, hits, scope=kwargs.pop("scope", RecallScope()),
|
||||
family="domain_clarification", excluded=kwargs.pop("excluded", set()), top=100, **kwargs)
|
||||
|
||||
|
||||
def test_link_only_candidates_cycles_duplicates_depth_and_current_content():
|
||||
repo = Archive(card("a", links=[link("b")]),
|
||||
card("b", detail="Current correction", links=[link("a"), link("c")]),
|
||||
card("c", links=[link("d")]), card("d"))
|
||||
results = rank(repo, [hit("a"), hit("a")])
|
||||
assert [r.card.id for r in results] == ["a", "b", "c"]
|
||||
assert results[1].card.detail == "Current correction"
|
||||
assert results[2].path == ("a", "b", "c")
|
||||
assert len(repo.reads) == len(set(repo.reads)) == 3
|
||||
assert rank(repo, [hit("a")]) == results # duplicate seeds do not boost a score
|
||||
|
||||
|
||||
def test_direct_and_graph_candidates_are_reranked_together_without_cycle_boost():
|
||||
repo = Archive(card("a", links=[link("c")]), card("b"), card("c", links=[link("a")]))
|
||||
results = rank(repo, [hit("a"), hit("b"), hit("c")])
|
||||
assert [r.card.id for r in results] == ["a", "c", "b"]
|
||||
assert len(results) == 3
|
||||
|
||||
|
||||
@pytest.mark.parametrize("invalid", ["missing", "pending", "excluded", "wrong_scope", "family"])
|
||||
def test_ineligible_link_targets_cannot_be_returned_or_used_as_bridges(invalid):
|
||||
target = card("b", links=[link("c")])
|
||||
if invalid == "pending":
|
||||
target.indexed = False
|
||||
if invalid == "wrong_scope":
|
||||
target.scope = "Purchases"
|
||||
if invalid == "family":
|
||||
target.family = "sql_rule"
|
||||
repo = Archive(card("a", links=[link("b")]), card("c"),
|
||||
*([] if invalid == "missing" else [target]))
|
||||
results = rank(repo, [hit("a")], scope=RecallScope(scope="Sales"),
|
||||
excluded={"b"} if invalid == "excluded" else set())
|
||||
assert [r.card.id for r in results] == ["a"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("metadata", [{"memory_revision": "old"}, {"memory_format": 1}])
|
||||
def test_stale_seeds_do_not_expand(metadata):
|
||||
repo = Archive(card("a", links=[link("b")]), card("b"))
|
||||
assert rank(repo, [hit("a", **metadata)]) == []
|
||||
|
||||
|
||||
def test_physical_scope_matches_one_dependency_and_includes_workspace_and_parent_rules():
|
||||
scope = RecallScope(database="sales", schema_name="public", table="orders", column="id")
|
||||
assert scope.matches(card("global"))
|
||||
assert scope.matches(card("database", dependencies=[{"database": "sales"}]))
|
||||
assert scope.matches(card("table", dependencies=[{"database": "sales",
|
||||
"schema_name": "public", "table": "orders"}]))
|
||||
assert not scope.matches(card("split", dependencies=[
|
||||
{"database": "sales", "schema_name": "public", "table": "orders", "column": "amount"},
|
||||
{"database": "purchases", "schema_name": "public", "table": "orders", "column": "id"},
|
||||
]))
|
||||
assert not scope.matches(card("schema", dependencies=[
|
||||
{"database": "sales", "schema_name": "audit", "table": "orders", "column": "id"},
|
||||
]))
|
||||
|
||||
|
||||
def test_scope_and_concepts_are_explicit_and_combined():
|
||||
scope = RecallScope(scope="Sales", concepts=["orders", "grain"])
|
||||
assert scope.matches(card("a", concepts=["orders", "grain"]))
|
||||
assert not scope.matches(card("b", concepts=["orders"]))
|
||||
with pytest.raises(ValueError):
|
||||
RecallScope(column="id")
|
||||
|
||||
|
||||
def test_fanout_and_total_visits_are_bounded():
|
||||
cards = [card(f"c{i:04}", links=[link(f"c{j:04}") for j in range(i+1, i+101)])
|
||||
for i in range(500)]
|
||||
repo = Archive(*cards)
|
||||
results = rank(repo, [hit("c0000")])
|
||||
assert "c0100" not in {r.card.id for r in results}
|
||||
assert len(repo.reads) <= 200
|
||||
repo.reads.clear()
|
||||
rank(repo, [hit(c.id) for c in cards[:100]])
|
||||
assert len(repo.reads) <= 200
|
||||
@@ -9,7 +9,7 @@ from tht.session.filesystem_repository import FilesystemSessionRepository
|
||||
from tht.session.models import PrincipalContext, SessionManifest
|
||||
|
||||
|
||||
def test_finalize_commits_session_before_best_effort_post_commit_read_failure(
|
||||
def test_finalize_commits_session_without_implicitly_reading_or_saving_memory(
|
||||
tmp_path, monkeypatch, capsys
|
||||
):
|
||||
repository = FilesystemSessionRepository(
|
||||
@@ -83,5 +83,5 @@ def test_finalize_commits_session_before_best_effort_post_commit_read_failure(
|
||||
|
||||
captured = capsys.readouterr()
|
||||
assert repository.get(session_id).manifest.status == "finalized"
|
||||
assert "finalized snapshot unavailable" in captured.err
|
||||
assert "finalized snapshot unavailable" not in captured.err
|
||||
assert f"OK: sessione {session_id} finalizzata" in captured.out
|
||||
|
||||
@@ -6,7 +6,6 @@ import typer
|
||||
|
||||
from tht.cli import db_cmd, memory_cmd
|
||||
from tht.cli.lsh_cmd import _extract_lsh_values
|
||||
from tht.memory import MemoryRecord
|
||||
from tht.mschema.models import Annotations, ColumnPhysical, PhysicalSchema, TablePhysical
|
||||
from tht.ports.dwh import DistinctValues, DwhHealth
|
||||
|
||||
@@ -57,58 +56,25 @@ def test_lsh_extraction_honors_configured_limit(limit, truncated):
|
||||
assert [report.indexed for report in reports] == ([limit] if truncated else [])
|
||||
|
||||
|
||||
def test_memory_command_writes_through_factory_vector_store(monkeypatch):
|
||||
store = SimpleNamespace(existing_hashes=lambda *args: {}, upsert=lambda table, rows: 1)
|
||||
captured = []
|
||||
original_upsert = store.upsert
|
||||
store.upsert = lambda table, rows: captured.extend(rows) or original_upsert(table, rows)
|
||||
# Legacy server deployments wrote directly through the factory and intentionally
|
||||
# did not configure the workstation-only REST writer key.
|
||||
cfg = SimpleNamespace(profile="server", embeddings=object(), vector_write_rest=None)
|
||||
manifest = SimpleNamespace(id="s1")
|
||||
snapshot = SimpleNamespace(manifest=manifest, decisions=[], artifacts={})
|
||||
record = MemoryRecord(
|
||||
id="m1", ts=datetime(2026, 1, 1, tzinfo=UTC), session_id="s1",
|
||||
decision_seq=7, type="concept_clarified", subject="paziente attivo",
|
||||
detail="flag_attivo = TRUE", question_context="q",
|
||||
)
|
||||
monkeypatch.setattr(memory_cmd, "_load_config_or_exit", lambda path: cfg)
|
||||
monkeypatch.setattr(memory_cmd, "load_snapshot_or_exit", lambda cfg, session: snapshot)
|
||||
monkeypatch.setattr(memory_cmd, "registry_path", lambda cfg: None)
|
||||
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store)
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder",
|
||||
lambda cfg: SimpleNamespace(embed_documents=lambda texts: [[0.1]]))
|
||||
monkeypatch.setattr("tht.memory.promote_snapshot", lambda *args, **kwargs: None)
|
||||
monkeypatch.setattr("tht.memory.load_registry", lambda path: [record])
|
||||
memory_cmd.save_one_cmd(session="s1", decision=7, json_out=True)
|
||||
from tht.ports.vector import VectorWriteRecord
|
||||
assert len(captured) == 1 and isinstance(captured[0], VectorWriteRecord)
|
||||
|
||||
|
||||
def test_solved_index_writes_through_writer_only_factory_store(monkeypatch):
|
||||
writer_only_store = SimpleNamespace(
|
||||
capabilities=SimpleNamespace(search=False, upsert=True),
|
||||
existing_hashes=lambda *args: {},
|
||||
upsert=lambda table, rows: 1,
|
||||
)
|
||||
cfg = SimpleNamespace(embeddings=object(), vector_write_rest=object())
|
||||
manifest = SimpleNamespace(id="s1")
|
||||
snapshot = SimpleNamespace(manifest=manifest, decisions=[], artifacts={})
|
||||
def test_save_one_passes_the_selected_source_to_authoritative_service(monkeypatch, capsys):
|
||||
snapshot = object()
|
||||
calls = []
|
||||
|
||||
service = SimpleNamespace(close=lambda: None, promote=lambda source, seqs:
|
||||
calls.append((source, seqs)) or [{"indexed": False, "saved": True}])
|
||||
monkeypatch.setattr(memory_cmd, "_load_config_or_exit", lambda path: object())
|
||||
monkeypatch.setattr(memory_cmd, "load_snapshot_or_exit", lambda cfg, session: snapshot)
|
||||
monkeypatch.setattr(
|
||||
"tht.adapters.factory.build_vector_store",
|
||||
lambda cfg, require_write: calls.append(require_write) or writer_only_store,
|
||||
)
|
||||
monkeypatch.setattr("tht.cli.sql_cmd.promoted_tables_for", lambda *args: [])
|
||||
monkeypatch.setattr(
|
||||
"tht.memory.index_solved_question",
|
||||
lambda loaded, tables, *, store, embedder: int(
|
||||
loaded is snapshot and tables == [] and store is writer_only_store
|
||||
),
|
||||
)
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object())
|
||||
monkeypatch.setattr(memory_cmd, "memory_service", lambda cfg: service)
|
||||
memory_cmd.save_one_cmd(session="s1", decision=7, json_out=True)
|
||||
assert calls == [(snapshot, [7])]
|
||||
assert '"indexed": false' in capsys.readouterr().out
|
||||
|
||||
assert memory_cmd.index_solved_session(cfg, "s1") == 1
|
||||
assert calls == [True]
|
||||
|
||||
def test_solved_recovery_uses_authority_without_recreating_historical_content(monkeypatch):
|
||||
snapshot = object()
|
||||
calls = []
|
||||
service = SimpleNamespace(close=lambda: None, retry_solved=lambda source:
|
||||
calls.append(source) or {"indexed": True, "action": "upsert"})
|
||||
monkeypatch.setattr(memory_cmd, "load_snapshot_or_exit", lambda cfg, session: snapshot)
|
||||
monkeypatch.setattr(memory_cmd, "memory_service", lambda cfg: service)
|
||||
assert memory_cmd.index_solved_session(object(), "s1") == 1
|
||||
assert calls == [snapshot]
|
||||
|
||||
@@ -35,7 +35,7 @@ def test_typer_tree_matches_the_approved_command_surface():
|
||||
expected = set(approved["maintained"]) | set(approved["enhanced"])
|
||||
|
||||
assert len(approved["maintained"]) == 61
|
||||
assert len(approved["enhanced"]) == 8
|
||||
assert len(approved["enhanced"]) == 21
|
||||
assert len(approved["erased"]) == 14
|
||||
assert not (expected & set(approved["erased"]))
|
||||
assert _leaf_paths(get_command(app)) == expected
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
import json
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
from test_evidence_local_archive import write_unit
|
||||
from test_preprocess_cli import _runtime_config
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from tht.cli import app
|
||||
from tht.evidence.administration import ConsolidationError, browse, consolidate_from_config
|
||||
from tht.evidence.local_archive import LocalEvidenceArchive
|
||||
|
||||
|
||||
def test_browse_filters_working_files_without_index_or_dwh_and_reports_invalid_files(tmp_path):
|
||||
path = write_unit(tmp_path)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
path.write_text(path.read_text().replace("Use order ID and year.", "Use all three parts of the key."))
|
||||
invalid = path.parent / "broken.md"
|
||||
invalid.write_text("not an Evidence document")
|
||||
result = browse(tmp_path, {"kind": "domain", "q": "three"})
|
||||
assert result["total"] == 1
|
||||
assert result["items"][0]["status"] == "modified"
|
||||
assert result["items"][0]["payload"]["rule"] == "Use all three parts of the key."
|
||||
assert result["errors"][0]["file"] == "curated/domain/broken.md"
|
||||
path.unlink()
|
||||
assert browse(tmp_path, {"status": "removed"})["total"] == 1
|
||||
|
||||
|
||||
def test_public_consolidation_retries_saved_content_and_does_not_use_git(monkeypatch, tmp_path):
|
||||
import tht.cli.preprocess_cmd as command
|
||||
import tht.config as config_module
|
||||
path = write_unit(tmp_path)
|
||||
cfg = SimpleNamespace(evidence=SimpleNamespace(local_archive_root=tmp_path))
|
||||
monkeypatch.setattr(config_module, "load_config", lambda _: cfg)
|
||||
calls = []
|
||||
def stage(_, *, local_snapshot):
|
||||
calls.append(local_snapshot)
|
||||
if len(calls) == 1:
|
||||
raise RuntimeError("private endpoint failure")
|
||||
return SimpleNamespace(status="succeeded")
|
||||
monkeypatch.setattr(command, "run_from_config", stage)
|
||||
monkeypatch.setattr("subprocess.run", lambda *a, **kw: pytest.fail("Consolidation must not run Git"))
|
||||
with pytest.raises(ConsolidationError, match="Retry consolidation") as failure:
|
||||
consolidate_from_config(tmp_path / "config.yaml")
|
||||
assert failure.value.saved
|
||||
assert path.exists()
|
||||
assert LocalEvidenceArchive(tmp_path).active_snapshot() is None
|
||||
assert consolidate_from_config(tmp_path / "config.yaml").status == "succeeded"
|
||||
assert calls[0] == calls[1]
|
||||
|
||||
|
||||
def test_cli_failure_is_pristine_json_with_a_correctable_error(monkeypatch, tmp_path):
|
||||
config = _runtime_config(tmp_path)
|
||||
def fail(_):
|
||||
raise ConsolidationError("curated/domain/rule.md: Missing Rule section")
|
||||
monkeypatch.setattr("tht.evidence.administration.consolidate_from_config", fail)
|
||||
result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(config)])
|
||||
assert result.exit_code == 1
|
||||
assert "Missing Rule section" in json.loads(result.stdout)["error"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("flags", [["gc"], ["--dry-run"], ["--resume", "a" * 32]])
|
||||
def test_cli_rejects_conflicting_consolidation_options_before_any_work(monkeypatch, flags):
|
||||
monkeypatch.setattr("tht.cli.preprocess_cmd.gc_from_config", lambda *a, **kw: pytest.fail("No cleanup"))
|
||||
result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", *flags])
|
||||
assert result.exit_code == 2
|
||||
assert json.loads(result.stdout)["code"] == "invalid_consolidation"
|
||||
|
||||
|
||||
def test_first_consolidation_reports_legacy_conversion_error(monkeypatch, tmp_path):
|
||||
from tht.evidence.authoring import EvidencePreparationError
|
||||
|
||||
write_unit(tmp_path)
|
||||
(tmp_path / "evidence/manifest.yaml").write_text("invalid")
|
||||
monkeypatch.setattr("tht.config.load_config", lambda _: SimpleNamespace(
|
||||
evidence=SimpleNamespace(local_archive_root=tmp_path)))
|
||||
def fail(_):
|
||||
raise EvidencePreparationError("source_invalid", "source/guide.md")
|
||||
monkeypatch.setattr("tht.evidence.authoring.migrate_workspace_evidence", fail)
|
||||
with pytest.raises(ConsolidationError, match="source/guide.md"):
|
||||
consolidate_from_config(tmp_path / "config.yaml")
|
||||
|
||||
|
||||
def test_preprocessing_uses_only_the_active_snapshot_and_clear_preserves_primary_files(monkeypatch, tmp_path):
|
||||
import yaml
|
||||
|
||||
from tht.cli.preprocess_cmd import clear_from_config
|
||||
from tht.config import load_config
|
||||
from tht.evidence.sources import build_sources
|
||||
path = write_unit(tmp_path / "local")
|
||||
archive = LocalEvidenceArchive(tmp_path / "local")
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
config = _runtime_config(tmp_path)
|
||||
raw = yaml.safe_load(config.read_text())
|
||||
raw["evidence"]["local_archive_root"] = str(tmp_path / "local")
|
||||
config.write_text(yaml.safe_dump(raw))
|
||||
path.write_text(path.read_text().replace("Use order ID and year.", "Unconsolidated text."))
|
||||
cfg = load_config(config)
|
||||
sources = build_sources(cfg.evidence)
|
||||
assert len(sources) == 1
|
||||
assert sources[0].root == archive.active_snapshot()
|
||||
saved = path.read_bytes()
|
||||
monkeypatch.setenv("THT_PROFILE", "server")
|
||||
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda *a, **kw: SimpleNamespace(clear_reference=lambda: True))
|
||||
clear_from_config(config)
|
||||
assert path.read_bytes() == saved
|
||||
assert archive.active_snapshot().is_dir()
|
||||
|
||||
|
||||
def test_manual_git_sequence_preserves_edits_additions_deletions_and_archive_metadata(tmp_path):
|
||||
import subprocess
|
||||
|
||||
def git(root, *args):
|
||||
return subprocess.run(["git", "-C", str(root), *args], check=True, capture_output=True)
|
||||
|
||||
remote = tmp_path / "remote.git"
|
||||
repo = tmp_path / "repo"
|
||||
repo.mkdir()
|
||||
git(tmp_path, "init", "--bare", str(remote))
|
||||
git(repo, "init", "--initial-branch=main")
|
||||
git(repo, "config", "user.name", "Evidence test")
|
||||
git(repo, "config", "user.email", "evidence@example.invalid")
|
||||
git(repo, "remote", "add", "origin", str(remote))
|
||||
root = repo / "workspace"
|
||||
path = write_unit(root)
|
||||
original = path.read_text()
|
||||
from tht.evidence.canonical import parse_curated_markdown
|
||||
identity = parse_curated_markdown(original).id
|
||||
removed = path.with_name("removed.md")
|
||||
removed.write_text(original.replace(identity, "evidence:removed"))
|
||||
archive = LocalEvidenceArchive(root)
|
||||
archive.consolidate(actor="Curator", activate=lambda _: None)
|
||||
git(repo, "add", "-A", "--", "workspace/evidence")
|
||||
git(repo, "commit", "--only", "-m", "Baseline", "--", "workspace/evidence")
|
||||
|
||||
path.write_text(path.read_text().replace("Use order ID and year.", "Use the approved compound key."))
|
||||
added = path.with_name("added.md")
|
||||
added.write_text(original.replace(identity, "evidence:added"))
|
||||
removed.unlink()
|
||||
archive.consolidate(actor="Curator", activate=lambda _: None)
|
||||
git(repo, "status", "--short")
|
||||
git(repo, "diff", "HEAD", "--", "workspace/evidence")
|
||||
git(repo, "add", "-A", "--", "workspace/evidence")
|
||||
git(repo, "commit", "--only", "-m", "Curate Evidence", "--", "workspace/evidence")
|
||||
git(repo, "push", "origin", "main")
|
||||
clone = tmp_path / "clone"
|
||||
git(tmp_path, "clone", "--branch", "main", str(remote), str(clone))
|
||||
copy = clone / "workspace"
|
||||
assert (copy / path.relative_to(root)).read_bytes() == path.read_bytes()
|
||||
assert (copy / added.relative_to(root)).exists()
|
||||
assert not (copy / removed.relative_to(root)).exists()
|
||||
assert (copy / "evidence/local-manifest.yaml").read_bytes() == (root / "evidence/local-manifest.yaml").read_bytes()
|
||||
assert LocalEvidenceArchive(copy).active_snapshot().is_dir()
|
||||
assert {unit["status"] for unit in browse(copy, {})["items"]} == {"active"}
|
||||
@@ -350,12 +350,12 @@ def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tm
|
||||
assert restructurer.requests[0].previous_units[0].id == "evidence:fascia-pediatrica"
|
||||
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
|
||||
curated = load_curated_tree(tmp_path / "evidence" / "curated")[0]
|
||||
assert curated.schema_version == 3
|
||||
assert curated.schema_version == 4
|
||||
assert "# Fascia pediatrica\n" in curated_path.read_text(encoding="utf-8")
|
||||
assert validate_workspace_evidence(tmp_path).publishable is True
|
||||
|
||||
|
||||
def test_migrate_workspace_evidence_rewrites_v1_units_as_v3_without_a_model_call(tmp_path):
|
||||
def test_migrate_workspace_evidence_rewrites_v1_units_as_v4_without_a_model_call(tmp_path):
|
||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||
|
||||
@@ -365,15 +365,16 @@ def test_migrate_workspace_evidence_rewrites_v1_units_as_v3_without_a_model_call
|
||||
migrated = load_curated_tree(tmp_path / "evidence" / "curated")[0]
|
||||
assert report.migrated == ("evidence:fascia-pediatrica",)
|
||||
assert report.unchanged == ()
|
||||
assert migrated.schema_version == 3
|
||||
assert migrated.schema_version == 4
|
||||
assert migrated.payload.rule == "La fascia pediatrica comprende i minori."
|
||||
text = curated_path.read_text(encoding="utf-8")
|
||||
assert "## Regola\n\n<!-- tht:raw-rule:" in text
|
||||
assert "## Regola\n\n" in text
|
||||
assert "<!-- tht:" not in text
|
||||
assert "La fascia pediatrica comprende i minori." in text
|
||||
assert report.findings == ()
|
||||
|
||||
|
||||
def test_migrate_workspace_evidence_rewrites_v2_units_as_table_free_v3(tmp_path):
|
||||
def test_migrate_workspace_evidence_rewrites_v2_units_as_editable_v4(tmp_path):
|
||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||
evidence = _evidence(source_text).model_copy(update={"schema_version": 2})
|
||||
_write_workspace(tmp_path, evidence, source_text)
|
||||
@@ -388,8 +389,8 @@ def test_migrate_workspace_evidence_rewrites_v2_units_as_table_free_v3(tmp_path)
|
||||
assert first.unchanged == ()
|
||||
assert second.migrated == ()
|
||||
assert second.unchanged == ("evidence:fascia-pediatrica",)
|
||||
assert migrated.schema_version == 3
|
||||
assert text.startswith("<!-- tht:metadata:")
|
||||
assert migrated.schema_version == 4
|
||||
assert text.startswith("---\nschema_version: 4")
|
||||
assert not any(line.startswith("|") for line in text.splitlines())
|
||||
|
||||
|
||||
@@ -422,20 +423,24 @@ def test_migrate_workspace_evidence_rewrites_legacy_v3_rule_presentation(tmp_pat
|
||||
migrated_text = curated_path.read_text(encoding="utf-8")
|
||||
assert report.migrated == ("evidence:fascia-pediatrica",)
|
||||
assert report.unchanged == ()
|
||||
assert "## Regola\n\n<!-- tht:raw-rule:" in migrated_text
|
||||
assert "## Regola\n\n" in migrated_text
|
||||
assert "<!-- tht:" not in migrated_text
|
||||
assert load_curated_tree(tmp_path / "evidence" / "curated")[0].payload.rule == (
|
||||
evidence.payload.rule
|
||||
)
|
||||
|
||||
|
||||
def test_migrate_workspace_evidence_rejects_dirty_curated_files_in_a_nested_workspace(tmp_path):
|
||||
def test_migrate_workspace_evidence_preserves_uncommitted_content_without_git_operations(tmp_path):
|
||||
subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True)
|
||||
workspace_root = tmp_path / "psd-clinical"
|
||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||
_write_workspace(workspace_root, _evidence(source_text), source_text)
|
||||
|
||||
with pytest.raises(EvidencePreparationError, match="authoring_worktree_dirty"):
|
||||
migrate_workspace_evidence(workspace_root)
|
||||
result = migrate_workspace_evidence(workspace_root)
|
||||
assert result.migrated == ("evidence:fascia-pediatrica",)
|
||||
assert load_curated_tree(workspace_root / "evidence/curated")[0].payload.rule == _evidence(source_text).payload.rule
|
||||
assert subprocess.run(["git", "-C", str(tmp_path), "rev-parse", "HEAD"],
|
||||
capture_output=True, check=False).returncode != 0
|
||||
|
||||
|
||||
def test_prepare_can_issue_independent_source_calls_concurrently(tmp_path):
|
||||
@@ -601,7 +606,7 @@ def test_prepare_marks_an_omitted_prior_unit_for_human_review(tmp_path):
|
||||
"supporting_excerpt_missing", "unresolved_review_item",
|
||||
]
|
||||
retained = load_curated_tree(tmp_path / "evidence" / "curated")[0]
|
||||
assert retained.schema_version == 3
|
||||
assert retained.schema_version == 4
|
||||
assert retained.review_items[0].code == "source_no_longer_supports_unit"
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
import pytest
|
||||
|
||||
from tht.evidence.canonical import (
|
||||
CuratedEvidence,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
parse_curated_markdown,
|
||||
)
|
||||
|
||||
|
||||
def unit(kind="domain", payload=None):
|
||||
return CuratedEvidence.model_validate(
|
||||
{
|
||||
"schema_version": 4,
|
||||
"id": "evidence:example",
|
||||
"title": "Example",
|
||||
"kind": kind,
|
||||
"purposes": ["sql_generation"],
|
||||
"language": "en",
|
||||
"provenance": {"kind": "manual", "declared_by": "curator"},
|
||||
"payload": payload or {"rule": "Count orders once, using their complete key."},
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("kind", "payload"),
|
||||
[
|
||||
("domain", {"rule": "A rule.\n\n### Explanation\n\n- First line\n- Second line"}),
|
||||
(
|
||||
"glossary",
|
||||
{
|
||||
"definition": "A patient.",
|
||||
"synonyms": ["person", "a\nmultiline synonym"],
|
||||
"variants": [],
|
||||
},
|
||||
),
|
||||
(
|
||||
"enum",
|
||||
{
|
||||
"column": "sales.orders.status",
|
||||
"values": {"": "Unknown", "A": "Active\n\nwith details"},
|
||||
},
|
||||
),
|
||||
("example", {"question": "How many?", "interpretation": "Count distinct orders."}),
|
||||
(
|
||||
"mapping",
|
||||
{"concept": "Orders", "tables": ["sales.orders"], "columns": ["sales.orders.id"]},
|
||||
),
|
||||
("normalization", {"input": "A", "output": "active", "rule": "Expand the abbreviation."}),
|
||||
("formula", {"concept": "Total", "columns": ["sales.lines.amount"], "sql": "sum(amount)"}),
|
||||
(
|
||||
"reference",
|
||||
{"url": "https://example.test/rule", "label": "Policy", "description": "Rule source."},
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_editable_round_trip_preserves_every_typed_payload(kind, payload):
|
||||
original = unit(kind, payload)
|
||||
rendered = dump_curated_markdown(original)
|
||||
assert parse_curated_markdown(rendered) == original
|
||||
assert "<!-- tht:" not in rendered
|
||||
assert "payload:" not in rendered
|
||||
|
||||
|
||||
def test_visible_edit_is_the_only_authoritative_text():
|
||||
original = unit()
|
||||
edited = dump_curated_markdown(original).replace(
|
||||
"Count orders once, using their complete key.", "Use order ID and year."
|
||||
)
|
||||
parsed = parse_curated_markdown(edited)
|
||||
assert parsed.payload.rule == "Use order ID and year."
|
||||
assert parsed.id == original.id
|
||||
|
||||
|
||||
def test_manual_file_needs_no_source_document_hash_or_encoded_metadata():
|
||||
text = """---
|
||||
schema_version: 4
|
||||
id: evidence:manual
|
||||
kind: domain
|
||||
language: it
|
||||
purposes: [sql_generation]
|
||||
applies_to:
|
||||
tables: [sales.orders]
|
||||
---
|
||||
|
||||
# Una regola manuale
|
||||
|
||||
## Regola
|
||||
|
||||
Le righe vengono collegate tramite numero ordine ed esercizio.
|
||||
"""
|
||||
parsed = parse_curated_markdown(text)
|
||||
assert isinstance(parsed.provenance, ManualEvidenceProvenance)
|
||||
assert parsed.provenance.original is None
|
||||
assert parsed.payload.rule.startswith("Le righe")
|
||||
|
||||
|
||||
def test_code_fences_cannot_inject_structural_sections():
|
||||
original = unit(
|
||||
"normalization",
|
||||
{
|
||||
"input": "A",
|
||||
"output": "active",
|
||||
"rule": "Example:\n\n```markdown\n## Input\nDo not parse this heading.\n```",
|
||||
},
|
||||
)
|
||||
assert parse_curated_markdown(dump_curated_markdown(original)) == original
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"mutation",
|
||||
[
|
||||
lambda text: text.replace("## Rule", "## Missing"),
|
||||
lambda text: text + "\n## Rule\n\nAnother rule.\n",
|
||||
lambda text: text.replace("kind: domain", "kind: domain\npayload: {rule: hidden}"),
|
||||
lambda text: text.replace("kind: domain", "kind: glossary\nkind: domain"),
|
||||
lambda text: text.replace("kind: domain", "kind: ["),
|
||||
lambda text: text.replace("- sql_generation", ""),
|
||||
lambda text: text.replace("language: en", "language: ''"),
|
||||
],
|
||||
)
|
||||
def test_invalid_edit_is_rejected_with_a_correctable_error(mutation):
|
||||
with pytest.raises(ValueError):
|
||||
parse_curated_markdown(mutation(dump_curated_markdown(unit())))
|
||||
|
||||
|
||||
def test_unrepresentable_legacy_content_is_reported_instead_of_silently_trimmed():
|
||||
with pytest.raises(ValueError, match="losslessly"):
|
||||
dump_curated_markdown(unit(payload={"rule": " significant indentation"}))
|
||||
@@ -0,0 +1,249 @@
|
||||
import os
|
||||
import shutil
|
||||
from pathlib import Path
|
||||
from uuid import uuid4
|
||||
|
||||
import pytest
|
||||
import requests
|
||||
from testcontainers.core.container import DockerContainer
|
||||
from testcontainers.core.waiting_utils import wait_for_logs
|
||||
|
||||
from tht.adapters.vector.qdrant import QdrantVectorStore
|
||||
from tht.evidence.adapters import FilesystemEvidenceSource
|
||||
from tht.evidence.canonical import CuratedEvidence, dump_curated_markdown
|
||||
from tht.evidence.corpus.chunk import ChunkPolicy
|
||||
from tht.evidence.corpus.pipeline import CorpusPipeline
|
||||
from tht.evidence.corpus.store import CorpusStore
|
||||
from tht.evidence.local_archive import LocalEvidenceArchive
|
||||
from tht.evidence.search import ActiveEvidenceSearcher, EvidenceSearchContext, search_evidence
|
||||
from tht.ports.vector import VectorWriteRecord
|
||||
from tht.vectorstore.records import VectorRecord
|
||||
|
||||
pytestmark = pytest.mark.l0
|
||||
|
||||
|
||||
class Embeddings:
|
||||
def embed_documents(self, texts):
|
||||
return [[1.0, 0.0, 0.0] for _ in texts]
|
||||
|
||||
def embed_query(self, query):
|
||||
return [1.0, 0.0, 0.0]
|
||||
|
||||
|
||||
def test_editable_archive_reindexes_visible_edits_for_core_and_preserves_other_indexes(tmp_path, monkeypatch):
|
||||
with DockerContainer("qdrant/qdrant:v1.18.2").with_exposed_ports(6333) as container:
|
||||
wait_for_logs(container, "Qdrant HTTP listening on 6333")
|
||||
url = f"http://{container.get_container_host_ip()}:{container.get_exposed_port(6333)}"
|
||||
workspace = "editable-" + uuid4().hex
|
||||
collections = {key: workspace + "-" + key for key in ("reference", "memory")}
|
||||
for collection in collections.values():
|
||||
requests.put(
|
||||
f"{url}/collections/{collection}",
|
||||
json={
|
||||
"vectors": {"size": 3, "distance": "Cosine"},
|
||||
"sparse_vectors": {"bm25": {"modifier": "idf"}},
|
||||
},
|
||||
timeout=10,
|
||||
).raise_for_status()
|
||||
vectors = QdrantVectorStore(
|
||||
base_url=url, collections=collections, workspace_id=workspace, expected_dimension=3
|
||||
)
|
||||
for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]:
|
||||
vectors.upsert(
|
||||
collection,
|
||||
[
|
||||
VectorWriteRecord(
|
||||
record=VectorRecord(
|
||||
id=kind,
|
||||
ref=kind,
|
||||
kind=kind,
|
||||
title="Preserve",
|
||||
content="Unrelated content",
|
||||
),
|
||||
embedding=[1, 0, 0],
|
||||
content_hash="sha256:" + "a" * 64,
|
||||
)
|
||||
],
|
||||
)
|
||||
corpus = CorpusStore(tmp_path / "corpus")
|
||||
archive = LocalEvidenceArchive(tmp_path / "workspace")
|
||||
path = archive.evidence / "curated/domain/order-key.md"
|
||||
path.parent.mkdir(parents=True)
|
||||
unit = CuratedEvidence.model_validate(
|
||||
{
|
||||
"schema_version": 4,
|
||||
"id": "evidence:order-key",
|
||||
"title": "Order key",
|
||||
"kind": "domain",
|
||||
"purposes": ["sql_generation"],
|
||||
"language": "en",
|
||||
"provenance": {"kind": "manual", "declared_by": "Alice"},
|
||||
"payload": {"rule": "Join orders using the complete business key."},
|
||||
}
|
||||
)
|
||||
path.write_text(dump_curated_markdown(unit))
|
||||
|
||||
def activate(snapshot):
|
||||
result = CorpusPipeline(
|
||||
store=corpus,
|
||||
sources=[FilesystemEvidenceSource(snapshot, patterns=("curated/**/*.md",))],
|
||||
embedder=Embeddings(),
|
||||
vector_store=vectors,
|
||||
embedding_model="fixture",
|
||||
embedding_dimensions=3,
|
||||
chunk_policy=ChunkPolicy(version="semantic:v1", max_chars=5000),
|
||||
pipeline_version="editable-v4",
|
||||
workspace_id=workspace,
|
||||
sparse_language="english",
|
||||
).run()
|
||||
assert result.status == "succeeded", (result, result.review_items)
|
||||
|
||||
class Delegate:
|
||||
def search(self, embedding, top_n=10, kinds=None, **kwargs):
|
||||
return vectors.search(["evidence"], embedding, limit=top_n, kinds=kinds, **kwargs)
|
||||
|
||||
searcher = ActiveEvidenceSearcher(corpus, Delegate(), workspace, "english")
|
||||
|
||||
def lookup():
|
||||
result = search_evidence(
|
||||
"order key",
|
||||
"sql_generation",
|
||||
EvidenceSearchContext(),
|
||||
searcher=searcher,
|
||||
embedder=Embeddings(),
|
||||
)
|
||||
assert result.status == "available"
|
||||
return result.results
|
||||
|
||||
archive.consolidate(actor="Alice", activate=activate)
|
||||
assert lookup()[0].evidence_id == unit.id
|
||||
assert lookup()[0].provenance["kind"] == "manual"
|
||||
path.write_text(
|
||||
path.read_text().replace(
|
||||
"complete business key", "order number, financial year and company"
|
||||
)
|
||||
)
|
||||
assert "financial year" not in " ".join(lookup()[0].excerpts)
|
||||
archive.consolidate(actor="Bob", activate=activate)
|
||||
assert "financial year" in " ".join(lookup()[0].excerpts)
|
||||
assert lookup()[0].provenance["declared_by"] == "Bob"
|
||||
previous_snapshot = archive.active_snapshot()
|
||||
valid = path.read_text()
|
||||
path.write_text(
|
||||
valid.replace(
|
||||
"Join orders using the order number, financial year and company.", "x" * 6000
|
||||
)
|
||||
)
|
||||
with pytest.raises(AssertionError, match="blocked"):
|
||||
archive.consolidate(actor="Bob", activate=activate)
|
||||
assert archive.active_snapshot() == previous_snapshot
|
||||
assert "financial year" in " ".join(lookup()[0].excerpts)
|
||||
path.write_text(valid)
|
||||
archive.consolidate(actor="Bob", activate=activate)
|
||||
path.unlink()
|
||||
archive.consolidate(actor="Bob", activate=activate)
|
||||
assert lookup() == ()
|
||||
# Optional real corpus probe, always copied into the test's private archive.
|
||||
if supplied := os.environ.get("THT_E1_PSD_COPY"):
|
||||
from tht.evidence.canonical import load_curated_tree
|
||||
|
||||
source_root = Path(supplied) / "evidence"
|
||||
expected = {u.id for u in load_curated_tree(source_root / "curated")}
|
||||
assert len(expected) == 35
|
||||
shutil.copytree(
|
||||
source_root / "curated", archive.evidence / "curated", dirs_exist_ok=True
|
||||
)
|
||||
shutil.copytree(source_root / "source", archive.evidence / "source", dirs_exist_ok=True)
|
||||
archive.consolidate(actor="Validation", activate=activate)
|
||||
actual = {
|
||||
d.metadata["curated_evidence"]["id"] for d in corpus.active_manifest().documents
|
||||
}
|
||||
assert actual == expected
|
||||
print("PSD v4: all 35 converted Evidence units indexed with unchanged identities")
|
||||
# Exercise the installed harness entry point against the same real Qdrant.
|
||||
import json
|
||||
|
||||
import yaml
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from tht.cli import app
|
||||
from tht.cli.preprocess_cmd import clear_from_config, run_from_config
|
||||
runtime = tmp_path / "runtime.yaml"
|
||||
runtime.write_text(yaml.safe_dump({
|
||||
"runtime_identity": {"workspace_id": workspace, "workspace_revision": "a" * 40},
|
||||
"dwh": {"type": "postgres_direct", "connection": {"database": "unused", "schema": "public", "user": "unused", "password": "unused"}},
|
||||
"vectors": {"type": "qdrant", "base_url": "http://qdrant:6333", "collection": workspace},
|
||||
"embeddings": {"provider": "ollama_internal", "base_url": "http://embedding:11434", "model": "fixture", "dim": 3},
|
||||
"evidence": {"schema_version": 2, "local_archive_root": str(archive.root), "sources": [{"type": "http", "urls": ["https://must-not-be-fetched.invalid/source.md"]}]},
|
||||
"vector": {"max_chunk_chars": 5000},
|
||||
"roots": {"sessions": str(tmp_path / "sessions"), "artifacts": str(tmp_path / "artifacts"), "indexes": str(tmp_path / "indexes")},
|
||||
}))
|
||||
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda *a, **kw: vectors)
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: Embeddings())
|
||||
monkeypatch.setenv("THT_PRINCIPAL_SUBJECT", "Installed curator")
|
||||
path.write_text(dump_curated_markdown(unit))
|
||||
result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)])
|
||||
assert result.exit_code == 0, result.output
|
||||
assert json.loads(result.stdout)["status"] == "succeeded"
|
||||
assert any(hit.provenance.get("declared_by") == "Installed curator" for hit in lookup())
|
||||
for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]:
|
||||
assert vectors.existing_hashes(collection, [kind])
|
||||
primary = path.read_bytes()
|
||||
monkeypatch.setenv("THT_PROFILE", "server")
|
||||
clear_from_config(runtime)
|
||||
assert path.read_bytes() == primary
|
||||
assert archive.active_snapshot().is_dir()
|
||||
# Full preprocessing must recreate Reference after Clear; Evidence alone
|
||||
# cannot claim the missing Schema derivatives are ready.
|
||||
requests.put(f"{url}/collections/{collections['reference']}", json={
|
||||
"vectors": {"size": 3, "distance": "Cosine"},
|
||||
"sparse_vectors": {"bm25": {"modifier": "idf"}},
|
||||
}, timeout=10).raise_for_status()
|
||||
assert run_from_config(runtime).status == "succeeded"
|
||||
assert any(hit.evidence_id == unit.id for hit in lookup())
|
||||
|
||||
# E3 traverses the public source commands and the same real index activation.
|
||||
from test_evidence_imports import Refiner, Remote
|
||||
|
||||
from tht.evidence.imports import reviews
|
||||
vectors.upsert("schema_records", [VectorWriteRecord(record=VectorRecord(
|
||||
id="schema_table", ref="schema_table", kind="schema_table", title="Preserve",
|
||||
content="Unrelated schema"), embedding=[1, 0, 0], content_hash="sha256:" + "a" * 64)])
|
||||
remote, refiner = Remote(), Refiner()
|
||||
monkeypatch.setattr("tht.evidence.imports.acquisition_sources", lambda cfg: [remote])
|
||||
monkeypatch.setattr("tht.evidence.authoring.PiEvidenceRestructurer", lambda *a, **kw: refiner)
|
||||
|
||||
def source_command(*args):
|
||||
outcome = CliRunner().invoke(app, ["evidence", "sources", *args, "--json", "-c", str(runtime)])
|
||||
assert outcome.exit_code == 0, outcome.output
|
||||
return json.loads(outcome.stdout)
|
||||
|
||||
source_command("refresh")
|
||||
row = reviews(archive)[0]
|
||||
imported_id = row["proposed"][0]["id"]
|
||||
assert not any(hit.evidence_id == imported_id for hit in lookup())
|
||||
source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace")
|
||||
assert any(hit.evidence_id == imported_id for hit in lookup())
|
||||
imported = archive.evidence / f"curated/domain/{imported_id[9:]}.md"
|
||||
imported.write_text(imported.read_text().replace("## Rule\n\nUse order ID and year.", "## Rule\n\nKeep the curator's company key."))
|
||||
result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)])
|
||||
assert result.exit_code == 0, result.output
|
||||
remote.text = "The refreshed source has another key."
|
||||
source_command("refresh")
|
||||
assert any("company key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id)
|
||||
row = reviews(archive)[0]
|
||||
source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace")
|
||||
assert any("another key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id)
|
||||
assert not any("company key" in " ".join(hit.excerpts) for hit in lookup() if hit.evidence_id == imported_id)
|
||||
imported.unlink()
|
||||
result = CliRunner().invoke(app, ["preprocess", "evidence", "--consolidate", "--json", "-c", str(runtime)])
|
||||
assert result.exit_code == 0, result.output
|
||||
remote.text = "Do not regenerate the retired source rule."
|
||||
source_command("refresh")
|
||||
row = reviews(archive)[0]
|
||||
assert not row["proposed"]
|
||||
source_command("decide", "--source-id", row["id"], "--revision", row["revision"], "--decision", "replace")
|
||||
assert not any(hit.evidence_id == imported_id for hit in lookup())
|
||||
for collection, kind in [("schema_records", "schema_table"), ("memory", "memory")]:
|
||||
assert vectors.existing_hashes(collection, [kind])
|
||||
assert vectors.existing_hashes("memory", ["memory"])
|
||||
@@ -97,6 +97,7 @@ def test_source_factory_preserves_legacy_first_order_and_filesystem_configuratio
|
||||
(legacy_root / "evidence").mkdir(parents=True)
|
||||
configured_root.mkdir()
|
||||
cfg = SimpleNamespace(evidence=SimpleNamespace(
|
||||
local_archive_root=None,
|
||||
source_root=legacy_root,
|
||||
evidence_dir="evidence",
|
||||
sources=[SimpleNamespace(
|
||||
@@ -124,6 +125,7 @@ def test_legacy_source_discovers_only_curated_evidence_units(tmp_path):
|
||||
(legacy_root / "evidence" / "curated" / "domain").mkdir(parents=True)
|
||||
(legacy_root / "evidence" / "curated" / "domain" / "patient.md").write_text("curated")
|
||||
cfg = SimpleNamespace(evidence=SimpleNamespace(
|
||||
local_archive_root=None,
|
||||
source_root=legacy_root,
|
||||
evidence_dir="evidence",
|
||||
sources=[],
|
||||
|
||||
@@ -0,0 +1,239 @@
|
||||
import json
|
||||
|
||||
import pytest
|
||||
from test_evidence_local_archive import read_active, write_unit
|
||||
|
||||
from tht.evidence.adapters import FilesystemEvidenceSource
|
||||
from tht.evidence.authoring import RestructureCandidate
|
||||
from tht.evidence.canonical import EvidenceProvenance, ManualEvidenceProvenance
|
||||
from tht.evidence.contracts import AcquiredDocument, SourceObject
|
||||
from tht.evidence.imports import decide, refresh, reviews
|
||||
from tht.evidence.local_archive import ArchiveConflict, LocalEvidenceArchive, _digest
|
||||
|
||||
|
||||
class Refiner:
|
||||
def __init__(self):
|
||||
self.calls = []
|
||||
|
||||
def restructure(self, request):
|
||||
self.calls.append(request)
|
||||
return (RestructureCandidate(schema_version=1,
|
||||
existing_id=request.previous_units[0].id if request.previous_units else None,
|
||||
title="Order key", kind="domain", language="en", purposes=("sql_generation",),
|
||||
payload={"rule": request.normalized_text.strip()}, supporting_excerpts=(request.normalized_text.strip(),)),)
|
||||
|
||||
|
||||
class Remote:
|
||||
def __init__(self, uri="https://docs.example.test/rule.md"):
|
||||
self.uri = uri
|
||||
self.text = "Use order ID and year."
|
||||
self.calls = 0
|
||||
self.fail = False
|
||||
self.absent = False
|
||||
|
||||
def discover(self):
|
||||
self.calls += 1
|
||||
if self.fail:
|
||||
raise RuntimeError("private access credential must not leak")
|
||||
if not self.absent:
|
||||
yield SourceObject(source_id="test:rule", uri=self.uri,
|
||||
fingerprint="sha256:" + _digest(self.text.encode()))
|
||||
|
||||
def acquire(self, item):
|
||||
self.calls += 1
|
||||
return AcquiredDocument(source=item, content=self.text.encode(), media_type="text/markdown")
|
||||
|
||||
|
||||
def empty_archive(tmp_path):
|
||||
(tmp_path / "evidence/curated").mkdir(parents=True)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.initialize()
|
||||
return archive
|
||||
|
||||
|
||||
def choose(archive, choice="replace", activate=lambda _: None):
|
||||
row = next(r for r in reviews(archive) if r["status"] in {"review", "applying"})
|
||||
return decide(archive, source_id=row["id"], revision=row["revision"], decision=choice,
|
||||
actor="Curator", activate=activate)
|
||||
|
||||
|
||||
def test_local_source_refresh_preserves_manual_correction_and_explicit_choices(tmp_path):
|
||||
path = write_unit(tmp_path, source=True)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.consolidate(actor="Curator", activate=lambda _: None)
|
||||
source = FilesystemEvidenceSource(archive.evidence, patterns=("source/*.md",))
|
||||
refiner = Refiner()
|
||||
assert refresh(archive, [source], refiner)["counts"]["unchanged"] == 1
|
||||
assert not refiner.calls
|
||||
path.write_text(path.read_text().replace("## Rule\n\nUse order ID and year.", "## Rule\n\nUse company as well."))
|
||||
archive.consolidate(actor="Curator", activate=lambda _: None)
|
||||
original = archive.active_snapshot()
|
||||
(archive.evidence / "source/orders.md").write_text("The new source says use ID only.")
|
||||
refresh(archive, [source], refiner)
|
||||
assert archive.active_snapshot() == original
|
||||
assert reviews(archive)[0]["current"][0]["payload"]["rule"] == "Use company as well."
|
||||
choose(archive, "keep")
|
||||
unit = read_active(archive)
|
||||
assert unit.payload.rule == "Use company as well."
|
||||
assert isinstance(unit.provenance, ManualEvidenceProvenance)
|
||||
assert unit.provenance.original.supporting_excerpts == ("Use order ID and year.",)
|
||||
assert refresh(archive, [source], refiner)["counts"]["unchanged"] == 1
|
||||
(archive.evidence / "source/orders.md").write_text("Now use ID, year and company.")
|
||||
refresh(archive, [source], refiner)
|
||||
choose(archive)
|
||||
unit = read_active(archive)
|
||||
assert unit.payload.rule == "Now use ID, year and company."
|
||||
assert isinstance(unit.provenance, EvidenceProvenance)
|
||||
assert (archive.active_snapshot() / unit.provenance.source_file).read_text().strip() == unit.payload.rule
|
||||
assert (original / "curated/domain/order-key.md").exists()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("uri", ["https://docs.example.test/rule.md", "s3://documents/rule.md"])
|
||||
def test_read_only_remote_import_preserves_bytes_and_never_refreshes_implicitly(tmp_path, uri):
|
||||
archive = empty_archive(tmp_path)
|
||||
remote, refiner = Remote(uri), Refiner()
|
||||
refresh(archive, [remote], refiner)
|
||||
assert archive.active_snapshot() is None
|
||||
choose(archive)
|
||||
assert read_active(archive).payload.rule == remote.text
|
||||
assert remote.calls == 2
|
||||
archive.consolidate(actor="Curator", activate=lambda _: None)
|
||||
assert remote.calls == 2
|
||||
assert refresh(archive, [remote], refiner)["counts"]["unchanged"] == 1
|
||||
assert len(refiner.calls) == 1
|
||||
saved = next((archive.metadata / "acquisitions").rglob("*.json"))
|
||||
assert json.loads(saved.read_text())["source"]["uri"] == uri
|
||||
assert json.loads(saved.read_text())["raw_base64"]
|
||||
|
||||
|
||||
def test_access_failure_and_missing_source_do_not_remove_or_replace_curated_content(tmp_path):
|
||||
archive, remote, refiner = empty_archive(tmp_path), Remote(), Refiner()
|
||||
refresh(archive, [remote], refiner)
|
||||
choose(archive)
|
||||
before = (archive.metadata / "sources.json").read_bytes()
|
||||
active = archive.active_snapshot()
|
||||
remote.fail = True
|
||||
with pytest.raises(RuntimeError):
|
||||
refresh(archive, [remote], refiner)
|
||||
assert (archive.metadata / "sources.json").read_bytes() == before
|
||||
remote.fail, remote.absent = False, True
|
||||
refresh(archive, [remote], refiner)
|
||||
assert reviews(archive)[0]["availability"] == "missing"
|
||||
assert archive.active_snapshot() == active
|
||||
|
||||
|
||||
def test_refresh_does_not_resurrect_deleted_units_with_new_model_ids(tmp_path):
|
||||
archive, remote, refiner = empty_archive(tmp_path), Remote(), Refiner()
|
||||
refresh(archive, [remote], refiner)
|
||||
choose(archive)
|
||||
(archive.evidence / "curated/domain/order-key.md").unlink()
|
||||
archive.consolidate(actor="Curator", activate=lambda _: None)
|
||||
remote.text = "Changed source could regenerate the deleted rule."
|
||||
refresh(archive, [remote], refiner)
|
||||
row = reviews(archive)[0]
|
||||
assert row["suppressed"] and row["proposed"] == []
|
||||
choose(archive)
|
||||
assert archive._files(archive.active_snapshot()) == {}
|
||||
|
||||
|
||||
def test_failed_activation_retries_saved_source_decision_without_reacquisition(tmp_path):
|
||||
archive, remote, refiner = empty_archive(tmp_path), Remote(), Refiner()
|
||||
refresh(archive, [remote], refiner)
|
||||
def fail(_):
|
||||
raise RuntimeError("offline index")
|
||||
with pytest.raises(RuntimeError, match="offline index"):
|
||||
choose(archive, activate=fail)
|
||||
assert reviews(archive)[0]["status"] == "applying"
|
||||
assert archive.active_snapshot() is None
|
||||
choose(archive)
|
||||
assert read_active(archive).payload.rule == remote.text
|
||||
assert remote.calls == 2
|
||||
|
||||
|
||||
def test_source_retirement_decision_also_suppresses_future_regeneration(tmp_path):
|
||||
archive, remote, refiner = empty_archive(tmp_path), Remote(), Refiner()
|
||||
refresh(archive, [remote], refiner)
|
||||
choose(archive)
|
||||
remote.text = "The source no longer supports the old rule."
|
||||
class EmptyRefiner:
|
||||
def restructure(self, request):
|
||||
return ()
|
||||
refresh(archive, [remote], EmptyRefiner())
|
||||
assert reviews(archive)[0]["removed_ids"] == ["evidence:order-key"]
|
||||
choose(archive)
|
||||
remote.text = "A newly worded source could restore the same rule."
|
||||
refresh(archive, [remote], refiner)
|
||||
assert reviews(archive)[0]["proposed"] == []
|
||||
|
||||
|
||||
def test_retry_never_overwrites_an_intervening_external_edit(tmp_path):
|
||||
archive, remote, refiner = empty_archive(tmp_path), Remote(), Refiner()
|
||||
refresh(archive, [remote], refiner)
|
||||
def fail(_):
|
||||
raise RuntimeError("offline index")
|
||||
with pytest.raises(RuntimeError):
|
||||
choose(archive, activate=fail)
|
||||
path = archive.evidence / "curated/domain/order-key.md"
|
||||
edited = path.read_text().replace("Use order ID and year.", "External correction.")
|
||||
path.write_text(edited)
|
||||
with pytest.raises(ArchiveConflict, match="changed during"):
|
||||
choose(archive)
|
||||
assert path.read_text() == edited
|
||||
assert archive.active_snapshot() is None
|
||||
|
||||
|
||||
def test_external_edit_invalidates_comparison_and_same_source_can_be_reviewed_again(tmp_path):
|
||||
archive, remote, refiner = empty_archive(tmp_path), Remote(), Refiner()
|
||||
refresh(archive, [remote], refiner)
|
||||
choose(archive)
|
||||
remote.text = "New source."
|
||||
refresh(archive, [remote], refiner)
|
||||
old_revision = reviews(archive)[0]["revision"]
|
||||
path = archive.evidence / "curated/domain/order-key.md"
|
||||
path.write_text(path.read_text().replace("## Rule\n\nUse order ID and year.", "## Rule\n\nManual correction."))
|
||||
with pytest.raises(ArchiveConflict, match="changed since"):
|
||||
choose(archive)
|
||||
refresh(archive, [remote], refiner)
|
||||
assert reviews(archive)[0]["revision"] != old_revision
|
||||
choose(archive, "keep")
|
||||
assert read_active(archive).payload.rule == "Manual correction."
|
||||
|
||||
|
||||
def test_decision_recovers_interruption_between_journal_and_archive_write(monkeypatch, tmp_path):
|
||||
archive, remote, refiner = empty_archive(tmp_path), Remote(), Refiner()
|
||||
refresh(archive, [remote], refiner)
|
||||
write = archive._write_state
|
||||
monkeypatch.setattr(archive, "_write_state", lambda _: (_ for _ in ()).throw(OSError("interrupted")))
|
||||
with pytest.raises(OSError):
|
||||
choose(archive)
|
||||
monkeypatch.setattr(archive, "_write_state", write)
|
||||
choose(archive)
|
||||
assert read_active(archive).payload.rule == remote.text
|
||||
|
||||
|
||||
def test_local_incoming_draft_needs_no_database_or_source_server(tmp_path):
|
||||
archive = empty_archive(tmp_path)
|
||||
(archive.evidence / "incoming").mkdir()
|
||||
(archive.evidence / "incoming/draft.md").write_text("A new domain rule.")
|
||||
refresh(archive, [FilesystemEvidenceSource(archive.evidence, patterns=("incoming/*.md",))], Refiner())
|
||||
choose(archive)
|
||||
assert read_active(archive).payload.rule == "A new domain rule."
|
||||
|
||||
|
||||
def test_source_cli_access_error_is_sanitized_json(monkeypatch, tmp_path):
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from tht.cli import app
|
||||
def fail(*a, **kw):
|
||||
raise RuntimeError("private credential")
|
||||
monkeypatch.setattr("tht.evidence.administration.source_action", fail)
|
||||
result = CliRunner().invoke(app, ["evidence", "sources", "refresh", "--json", "-c", str(tmp_path / "cfg")])
|
||||
assert result.exit_code == 1
|
||||
assert json.loads(result.stdout)["status"] == "failed"
|
||||
assert "private credential" not in result.stdout
|
||||
|
||||
|
||||
def test_installed_refiner_uses_deployment_resources_outside_the_python_wheel(monkeypatch, tmp_path):
|
||||
from tht.evidence.authoring import authoring_skill_path
|
||||
monkeypatch.setenv("THT_HARNESS_DIR", str(tmp_path / "deployment"))
|
||||
assert authoring_skill_path() == tmp_path / "deployment/.pi/skills/tht-evidence-authoring/SKILL.md"
|
||||
@@ -0,0 +1,235 @@
|
||||
import pytest
|
||||
import yaml
|
||||
|
||||
from tht.evidence.authoring import (
|
||||
EvidencePreparationError,
|
||||
normalize_source_text,
|
||||
prepare_workspace_evidence,
|
||||
)
|
||||
from tht.evidence.canonical import (
|
||||
CuratedEvidence,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
parse_curated_markdown,
|
||||
)
|
||||
from tht.evidence.local_archive import ArchiveConflict, LocalEvidenceArchive, _digest
|
||||
|
||||
|
||||
def write_unit(root, *, rule="Use order ID and year.", source=False):
|
||||
provenance = {"kind": "manual", "declared_by": "local curator"}
|
||||
if source:
|
||||
path = root / "evidence/source/orders.md"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(rule)
|
||||
provenance = {
|
||||
"source_file": "source/orders.md",
|
||||
"source_sha256": "sha256:" + _digest(normalize_source_text(rule).encode()),
|
||||
"supporting_excerpts": [rule],
|
||||
}
|
||||
unit = CuratedEvidence.model_validate(
|
||||
{
|
||||
"schema_version": 4,
|
||||
"id": "evidence:order-key",
|
||||
"title": "Order key",
|
||||
"kind": "domain",
|
||||
"language": "en",
|
||||
"purposes": ["sql_generation"],
|
||||
"payload": {"rule": rule},
|
||||
"provenance": provenance,
|
||||
}
|
||||
)
|
||||
path = root / "evidence/curated/domain/order-key.md"
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(dump_curated_markdown(unit))
|
||||
return path
|
||||
|
||||
|
||||
def read_active(archive):
|
||||
return parse_curated_markdown(
|
||||
(archive.active_snapshot() / "curated/domain/order-key.md").read_text()
|
||||
)
|
||||
|
||||
|
||||
def test_manual_creation_and_visible_edit_reach_only_the_consolidated_snapshot(tmp_path):
|
||||
path = write_unit(tmp_path)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
first = archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
assert first["status"] == "active"
|
||||
assert read_active(archive).provenance.declared_by == "Alice"
|
||||
path.write_text(path.read_text().replace("Use order ID and year.", "Use ID, year and company."))
|
||||
assert read_active(archive).payload.rule == "Use order ID and year."
|
||||
candidate = archive.consolidate(actor="Bob")
|
||||
assert candidate["status"] == "pending_activation"
|
||||
assert read_active(archive).payload.rule == "Use order ID and year."
|
||||
archive.consolidate(actor="Bob", activate=lambda _: None)
|
||||
assert read_active(archive).payload.rule == "Use ID, year and company."
|
||||
assert read_active(archive).provenance.declared_by == "Bob"
|
||||
|
||||
|
||||
def test_document_correction_becomes_manual_and_retains_original_source_separately(tmp_path):
|
||||
path = write_unit(tmp_path, source=True)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.initialize()
|
||||
original = parse_curated_markdown(path.read_text()).provenance
|
||||
path.write_text(
|
||||
path.read_text().replace(
|
||||
"## Rule\n\nUse order ID and year.",
|
||||
"## Rule\n\nThe approved key also includes company.",
|
||||
)
|
||||
)
|
||||
archive.consolidate(actor="Curator", activate=lambda _: None)
|
||||
current = read_active(archive)
|
||||
assert isinstance(current.provenance, ManualEvidenceProvenance)
|
||||
assert current.provenance.original == original
|
||||
assert current.provenance.declared_by == "Curator"
|
||||
assert (archive.active_snapshot() / "source/orders.md").read_text() == "Use order ID and year."
|
||||
|
||||
|
||||
def test_failed_activation_and_restart_retry_reuse_the_candidate(tmp_path):
|
||||
path = write_unit(tmp_path)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
previous = archive.active_snapshot()
|
||||
path.write_text(
|
||||
path.read_text().replace("Use order ID and year.", "Use the reviewed composite key.")
|
||||
)
|
||||
calls = []
|
||||
|
||||
def fail(candidate):
|
||||
calls.append(candidate)
|
||||
raise RuntimeError("Index unavailable")
|
||||
|
||||
with pytest.raises(RuntimeError, match="Index unavailable"):
|
||||
archive.consolidate(actor="Alice", activate=fail)
|
||||
recovered = LocalEvidenceArchive(tmp_path)
|
||||
assert recovered.active_snapshot() == previous
|
||||
result = recovered.consolidate(actor="Alice", activate=calls.append)
|
||||
assert calls[0] == calls[1]
|
||||
assert result["status"] == "active"
|
||||
|
||||
|
||||
def test_deleted_source_unit_stays_deleted_after_restart_and_refinement_attempt(tmp_path):
|
||||
path = write_unit(tmp_path, source=True)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.initialize()
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
path.unlink()
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
state = yaml.safe_load((tmp_path / "evidence/.local/state.yaml").read_text())
|
||||
assert state["deleted_ids"] == ["evidence:order-key"]
|
||||
assert state["suppressed_sources"] == ["source/orders.md"]
|
||||
|
||||
class MustNotCall:
|
||||
def restructure(self, request):
|
||||
raise AssertionError("Deleted knowledge must not be regenerated")
|
||||
|
||||
with pytest.raises(EvidencePreparationError, match="explicit_source_refresh"):
|
||||
prepare_workspace_evidence(tmp_path, restructurer=MustNotCall(), git_status=lambda _: ())
|
||||
assert not path.exists()
|
||||
assert not list(LocalEvidenceArchive(tmp_path).active_snapshot().rglob("*.md"))
|
||||
|
||||
|
||||
def test_invalid_edit_does_not_change_files_metadata_or_active_snapshot(tmp_path):
|
||||
path = write_unit(tmp_path)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
old = archive.active_snapshot()
|
||||
state = (tmp_path / "evidence/.local/state.yaml").read_bytes()
|
||||
path.write_text(path.read_text().replace("## Rule", "## Wrong heading"))
|
||||
invalid = path.read_bytes()
|
||||
with pytest.raises(ValueError, match="curated/domain/order-key.md.*section"):
|
||||
archive.consolidate(actor="Alice", activate=lambda _: pytest.fail("Must not activate"))
|
||||
assert path.read_bytes() == invalid
|
||||
assert (tmp_path / "evidence/.local/state.yaml").read_bytes() == state
|
||||
assert archive.active_snapshot() == old
|
||||
|
||||
|
||||
def test_workflow_correction_checks_revision_and_never_overwrites_a_manual_edit(tmp_path):
|
||||
path = write_unit(tmp_path)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
before = archive.get("evidence:order-key")
|
||||
proposed = before["unit"].model_copy(update={"title": "Workflow correction"})
|
||||
path.write_text(path.read_text().replace("Use order ID and year.", "Manual edit in progress."))
|
||||
with pytest.raises(ArchiveConflict):
|
||||
archive.save(proposed, expected_revision=before["revision"], actor="Reviewer")
|
||||
assert "Manual edit in progress." in path.read_text()
|
||||
latest = archive.get("evidence:order-key")
|
||||
archive.save(
|
||||
latest["unit"].model_copy(update={"title": "Approved title"}),
|
||||
expected_revision=latest["revision"],
|
||||
actor="Reviewer",
|
||||
)
|
||||
assert archive.get("evidence:order-key")["unit"].title == "Approved title"
|
||||
|
||||
|
||||
def test_source_refresh_does_not_replace_manual_content_or_historical_citation(tmp_path):
|
||||
path = write_unit(tmp_path, source=True)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.initialize()
|
||||
path.write_text(
|
||||
path.read_text().replace("## Rule\n\nUse order ID and year.", "## Rule\n\nManual rule.")
|
||||
)
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
(tmp_path / "evidence/source/orders.md").write_text("Contradictory newer source.")
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
assert read_active(archive).payload.rule == "Manual rule."
|
||||
assert (archive.active_snapshot() / "source/orders.md").read_text() == "Use order ID and year."
|
||||
|
||||
|
||||
def test_symlinks_cannot_escape_the_archive(tmp_path):
|
||||
path = write_unit(tmp_path)
|
||||
path.unlink()
|
||||
outside = tmp_path / "outside.md"
|
||||
outside.write_text("Private unrelated file")
|
||||
path.symlink_to(outside)
|
||||
with pytest.raises(ValueError, match="symlinks"):
|
||||
LocalEvidenceArchive(tmp_path).consolidate(actor="Alice")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("edit_after_interruption", [False, True])
|
||||
def test_interrupted_normalization_recovers_without_losing_new_edits(
|
||||
tmp_path, monkeypatch, edit_after_interruption
|
||||
):
|
||||
import tht.evidence.local_archive as module
|
||||
|
||||
path = write_unit(tmp_path, source=True)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.initialize()
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
old_active = archive.active_snapshot()
|
||||
path.write_text(
|
||||
path.read_text().replace("## Rule\n\nUse order ID and year.", "## Rule\n\nReviewed rule.")
|
||||
)
|
||||
atomic = module._atomic
|
||||
|
||||
def interrupted(target, data):
|
||||
if target == path:
|
||||
raise OSError("Interrupted file write")
|
||||
atomic(target, data)
|
||||
|
||||
monkeypatch.setattr(module, "_atomic", interrupted)
|
||||
with pytest.raises(OSError, match="Interrupted"):
|
||||
archive.consolidate(actor="Bob", activate=lambda _: pytest.fail("Must not activate"))
|
||||
assert archive.active_snapshot() == old_active
|
||||
if edit_after_interruption:
|
||||
path.write_text(path.read_text().replace("Reviewed rule.", "Further manual correction."))
|
||||
monkeypatch.setattr(module, "_atomic", atomic)
|
||||
recovered = LocalEvidenceArchive(tmp_path)
|
||||
recovered.consolidate(actor="Bob", activate=lambda _: None)
|
||||
assert read_active(recovered).payload.rule == (
|
||||
"Further manual correction." if edit_after_interruption else "Reviewed rule."
|
||||
)
|
||||
assert read_active(recovered).provenance.declared_by == "Bob"
|
||||
assert read_active(recovered).provenance.original.source_file == "source/orders.md"
|
||||
|
||||
|
||||
def test_missing_archive_directory_is_not_interpreted_as_deletion(tmp_path):
|
||||
write_unit(tmp_path)
|
||||
archive = LocalEvidenceArchive(tmp_path)
|
||||
archive.consolidate(actor="Alice", activate=lambda _: None)
|
||||
active = archive.active_snapshot()
|
||||
(archive.evidence / "curated").rename(archive.evidence / "unmounted")
|
||||
with pytest.raises(ValueError, match="absence is not deletion"):
|
||||
archive.consolidate(actor="Alice", activate=lambda _: pytest.fail("Must not activate"))
|
||||
assert archive.active_snapshot() == active
|
||||
@@ -9,7 +9,7 @@ from tht.pi_skill_projection import (
|
||||
render_projection,
|
||||
)
|
||||
|
||||
BASELINE_SHA256 = "62bfa0dbc1179b43b2d80dc48155a6119421488a1ffc160e9c664f1fe280ce52"
|
||||
BASELINE_SHA256 = "314d5e62eecda59dc16f00946d1f814829a326e7b117e102fd03c61f1f00a1a1"
|
||||
|
||||
|
||||
def test_modular_pi_skill_renders_the_byte_identical_approved_projection():
|
||||
|
||||
@@ -371,6 +371,7 @@ def test_candidate_evaluation_is_not_required_outside_v2_filesystem_corpora(
|
||||
cfg = SimpleNamespace(
|
||||
evidence=SimpleNamespace(
|
||||
schema_version=schema_version,
|
||||
local_archive_root=None,
|
||||
source_root=None,
|
||||
sources=[SimpleNamespace(type=source_type)],
|
||||
),
|
||||
|
||||
@@ -2,7 +2,6 @@ from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from datetime import UTC, datetime
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
@@ -10,7 +9,6 @@ import pytest
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from tht.cli import app
|
||||
from tht.memory import MemoryRecord, save_registry
|
||||
from tht.ports.vector import VectorStoreError
|
||||
|
||||
|
||||
@@ -102,22 +100,6 @@ def _write_catalog_snapshot(tmp_path: Path) -> None:
|
||||
}))
|
||||
|
||||
|
||||
def _memory_record() -> MemoryRecord:
|
||||
return MemoryRecord(
|
||||
id="mem-0001",
|
||||
ts=datetime(2026, 1, 1, tzinfo=UTC),
|
||||
session_id="s1",
|
||||
decision_seq=7,
|
||||
type="concept_clarified",
|
||||
subject="paziente attivo",
|
||||
detail="flag_attivo = TRUE",
|
||||
rationale="r",
|
||||
question_context="dammi i pazienti attivi",
|
||||
tables=[],
|
||||
concepts=["paziente attivo"],
|
||||
)
|
||||
|
||||
|
||||
def test_vector_index_schema_accepts_qdrant_only_runtime_config(tmp_path, monkeypatch):
|
||||
cfg = _qdrant_runtime_config(tmp_path)
|
||||
_write_catalog_snapshot(tmp_path)
|
||||
@@ -196,42 +178,29 @@ def test_vector_index_schema_json_failure_is_pristine(tmp_path, monkeypatch):
|
||||
}
|
||||
|
||||
|
||||
def test_memory_promote_accepts_qdrant_only_runtime_config(tmp_path, monkeypatch):
|
||||
def test_memory_promote_adapts_qdrant_runtime_to_authoritative_service(tmp_path, monkeypatch):
|
||||
cfg = _qdrant_runtime_config(tmp_path)
|
||||
store = _FakeVectorStore()
|
||||
promoted = [_memory_record()]
|
||||
snapshot = SimpleNamespace(manifest=SimpleNamespace(id="s1"))
|
||||
|
||||
snapshot = object()
|
||||
calls = []
|
||||
service = SimpleNamespace(close=lambda: None, promote=lambda source, seqs:
|
||||
calls.append((source, seqs)) or [{"indexed": True, "card": {"id": "mem-test"}}])
|
||||
monkeypatch.setattr("tht.cli.memory_cmd.load_snapshot_or_exit", lambda cfg, session: snapshot)
|
||||
monkeypatch.setattr("tht.memory.promote_snapshot", lambda *args, **kwargs: promoted)
|
||||
monkeypatch.setattr("tht.memory.load_registry", lambda path: promoted)
|
||||
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store)
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder())
|
||||
|
||||
res = CliRunner().invoke(
|
||||
app,
|
||||
["memory", "promote", "--session", "s1", "--decision", "7", "--json", "-c", str(cfg)],
|
||||
)
|
||||
|
||||
assert res.exit_code == 0, res.output
|
||||
assert json.loads(res.stdout)["indexed"] is True
|
||||
assert store.upserts
|
||||
monkeypatch.setattr("tht.cli.memory_cmd.memory_service", lambda cfg: service)
|
||||
result = CliRunner().invoke(app, [
|
||||
"memory", "promote", "--session", "s1", "--decision", "7", "--json", "-c", str(cfg),
|
||||
])
|
||||
assert result.exit_code == 0, result.output
|
||||
assert json.loads(result.stdout)["indexed"] is True
|
||||
assert calls == [(snapshot, [7])]
|
||||
|
||||
|
||||
def test_memory_index_accepts_qdrant_only_runtime_config(tmp_path, monkeypatch):
|
||||
def test_memory_index_rebuilds_only_through_authoritative_service(tmp_path, monkeypatch):
|
||||
cfg = _qdrant_runtime_config(tmp_path)
|
||||
store = _FakeVectorStore()
|
||||
records = [_memory_record()]
|
||||
|
||||
save_registry(records, tmp_path / "artifacts" / "memory" / "registry.jsonl")
|
||||
monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store)
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda _: _FakeEmbedder())
|
||||
|
||||
res = CliRunner().invoke(app, ["memory", "index", "-c", str(cfg)])
|
||||
|
||||
assert res.exit_code == 0, res.output
|
||||
assert "OK:" in res.output
|
||||
assert store.upserts
|
||||
service = SimpleNamespace(close=lambda: None, rebuild=lambda: [{"indexed": True}])
|
||||
monkeypatch.setattr("tht.cli.memory_cmd.memory_service", lambda cfg: service)
|
||||
result = CliRunner().invoke(app, ["memory", "index", "--json", "-c", str(cfg)])
|
||||
assert result.exit_code == 0, result.output
|
||||
assert json.loads(result.stdout) == [{"indexed": True}]
|
||||
|
||||
|
||||
def test_vector_help_exposes_only_the_supported_qdrant_command():
|
||||
|
||||
@@ -259,95 +259,11 @@ def test_clear_drops_reference_without_deleting_memory_collection():
|
||||
assert not any(method == "DELETE" and url.endswith("/collections/demo-memory") for method, url in calls)
|
||||
|
||||
|
||||
def test_clear_can_create_memory_collection_before_payload_indexes_become_visible():
|
||||
def test_clear_does_not_read_or_import_legacy_memory():
|
||||
calls = []
|
||||
memory_created = False
|
||||
|
||||
def request(method, url, *, json=None, timeout=None):
|
||||
nonlocal memory_created
|
||||
calls.append((method, url, json))
|
||||
if method == "GET" and url.endswith("/collections/demo"):
|
||||
return FakeResponse(200, {"result": {}})
|
||||
if method == "GET" and url.endswith("/collections/demo-memory"):
|
||||
if not memory_created:
|
||||
return FakeResponse(404, {"status": "error"})
|
||||
return FakeResponse(200, {
|
||||
"result": {
|
||||
"config": {"params": {"vectors": {"size": 1024, "distance": "Cosine"}}},
|
||||
# Qdrant exposes newly-created payload indexes asynchronously.
|
||||
"payload_schema": {},
|
||||
}
|
||||
})
|
||||
if method == "PUT" and url.endswith("/collections/demo-memory"):
|
||||
memory_created = True
|
||||
return FakeResponse(200, {"status": "ok"})
|
||||
if method == "PUT" and url.endswith("/collections/demo-memory/index"):
|
||||
return FakeResponse(200, {"status": "ok"})
|
||||
if method == "POST" and url.endswith("/collections/demo/points/scroll"):
|
||||
return FakeResponse(200, {
|
||||
"result": {"points": [], "next_page_offset": None}
|
||||
})
|
||||
if method == "DELETE" and url.endswith("/collections/demo"):
|
||||
return FakeResponse(200, {"status": "ok"})
|
||||
if method == "GET" and url.endswith("/collections/demo-reference"):
|
||||
return FakeResponse(404, {"status": "error"})
|
||||
raise AssertionError((method, url, json))
|
||||
|
||||
store = QdrantVectorStore(
|
||||
base_url="http://qdrant:6333",
|
||||
collections={"reference": "demo-reference", "memory": "demo-memory"},
|
||||
workspace_id="demo",
|
||||
expected_dimension=1024,
|
||||
collection_lifecycle="require_existing",
|
||||
request=request,
|
||||
)
|
||||
|
||||
assert store.clear_reference() is True
|
||||
assert memory_created is True
|
||||
assert any(method == "DELETE" and url.endswith("/collections/demo") for method, url, _ in calls)
|
||||
|
||||
|
||||
def test_clear_migrates_legacy_memory_before_retiring_shared_collection():
|
||||
calls = []
|
||||
legacy_point = {
|
||||
"id": "legacy-memory-id",
|
||||
"vector": [0.25] * 1024,
|
||||
"payload": {
|
||||
"workspace_id": "demo",
|
||||
"kind": "memory",
|
||||
"record_kind": "memory",
|
||||
"record_key": "memory:legacy",
|
||||
"content_hash": "sha256:" + "a" * 64,
|
||||
},
|
||||
}
|
||||
ready_memory = {
|
||||
"result": {
|
||||
"config": {"params": {"vectors": {"size": 1024, "distance": "Cosine"}}},
|
||||
"payload_schema": {
|
||||
field: {"data_type": "keyword"}
|
||||
for field in (
|
||||
"content_hash", "document_id", "kind", "record_key", "record_kind",
|
||||
"vector_generation", "workspace_id", "workspace_revision",
|
||||
)
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
def request(method, url, *, json=None, timeout=None):
|
||||
calls.append((method, url, json))
|
||||
if method == "GET" and url.endswith("/collections/demo"):
|
||||
return FakeResponse(200, {"result": {}})
|
||||
if method == "GET" and url.endswith("/collections/demo-memory"):
|
||||
return FakeResponse(200, ready_memory)
|
||||
if method == "POST" and url.endswith("/collections/demo/points/scroll"):
|
||||
assert json["with_vector"] is True
|
||||
return FakeResponse(200, {
|
||||
"result": {"points": [legacy_point], "next_page_offset": None}
|
||||
})
|
||||
if method == "PUT" and "/collections/demo-memory/points?wait=true" in url:
|
||||
return FakeResponse(200, {"status": "ok"})
|
||||
if method == "DELETE" and url.endswith("/collections/demo"):
|
||||
return FakeResponse(200, {"status": "ok"})
|
||||
calls.append((method, url))
|
||||
if method == "GET" and url.endswith("/collections/demo-reference"):
|
||||
return FakeResponse(200, {"result": {}})
|
||||
if method == "DELETE" and url.endswith("/collections/demo-reference"):
|
||||
@@ -357,23 +273,13 @@ def test_clear_migrates_legacy_memory_before_retiring_shared_collection():
|
||||
store = QdrantVectorStore(
|
||||
base_url="http://qdrant:6333",
|
||||
collections={"reference": "demo-reference", "memory": "demo-memory"},
|
||||
workspace_id="demo",
|
||||
expected_dimension=1024,
|
||||
request=request,
|
||||
workspace_id="demo", expected_dimension=1024, request=request,
|
||||
)
|
||||
|
||||
assert store.clear_reference() is True
|
||||
migrated = next(
|
||||
payload for method, url, payload in calls
|
||||
if method == "PUT" and "/collections/demo-memory/points?wait=true" in url
|
||||
)
|
||||
assert migrated == {"points": [legacy_point]}
|
||||
deleted = [url for method, url, _payload in calls if method == "DELETE"]
|
||||
assert deleted == [
|
||||
"http://qdrant:6333/collections/demo",
|
||||
"http://qdrant:6333/collections/demo-reference",
|
||||
assert calls == [
|
||||
("GET", "http://qdrant:6333/collections/demo-reference"),
|
||||
("DELETE", "http://qdrant:6333/collections/demo-reference"),
|
||||
]
|
||||
assert not any(url.endswith("/collections/demo-memory") for url in deleted)
|
||||
|
||||
|
||||
def _ready_collection_with_bm25(fake: FakeQdrantHttp) -> None:
|
||||
@@ -718,6 +624,40 @@ def test_search_filters_by_workspace_and_allowed_record_kinds():
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("kind,family", [("memory", "domain_clarification"),
|
||||
("solved_question", "solved_question")])
|
||||
def test_memory_hybrid_applies_identical_scope_to_both_prefetch_branches(kind, family):
|
||||
from tht.memory.retrieval import RecallScope
|
||||
|
||||
fake = FakeQdrantHttp()
|
||||
_ready_collection_with_bm25(fake)
|
||||
store = _store(fake)
|
||||
scope = RecallScope(database="dwh", schema_name="sales", table="orders", column="id",
|
||||
scope="Sales", concepts=["grain"])
|
||||
store.search(["memory"], [0.2] * 1024, limit=5, kinds=[kind],
|
||||
query_text="Order grain", query_language="english",
|
||||
metadata_filter=scope.vector_filter(family))
|
||||
query = next(call[2] for call in reversed(fake.calls) if call[1].endswith("/points/query"))
|
||||
dense, lexical = query["prefetch"]
|
||||
assert dense["filter"] == lexical["filter"]
|
||||
must = dense["filter"]["must"]
|
||||
assert {"key": "workspace_id", "match": {"value": "demo"}} in must
|
||||
assert {"key": "memory_family", "match": {"value": family}} in must
|
||||
assert {"key": "memory_format", "match": {"value": 2}} in must
|
||||
assert {"key": "memory_scope", "match": {"value": "Sales"}} in must
|
||||
assert {"key": "memory_concepts", "match": {"value": "grain"}} in must
|
||||
nested = must[-1]["should"][1]["nested"]
|
||||
assert nested["key"] == "memory_dependencies"
|
||||
assert nested["filter"]["must"] == [
|
||||
{"key": "database", "match": {"value": "dwh"}},
|
||||
{"key": "schema_name", "match": {"any": ["", "sales"]}},
|
||||
{"key": "table", "match": {"any": ["", "orders"]}},
|
||||
{"key": "column", "match": {"any": ["", "id"]}},
|
||||
]
|
||||
assert lexical["using"] == "bm25"
|
||||
assert query["query"] == {"rrf": {}}
|
||||
|
||||
|
||||
def test_evidence_search_refuses_dense_only_fallback():
|
||||
store = _store(FakeQdrantHttp())
|
||||
|
||||
|
||||
@@ -1,141 +1,90 @@
|
||||
"""L1: `tht memory solved-search` — degrado gentile e mapping dei risultati.
|
||||
|
||||
SKILL.md prescrive solved-search in F4/F6/F7 di OGNI sessione: se lo store
|
||||
semantico è irraggiungibile (Qdrant/Ollama non disponibili) il comando non deve morire con un
|
||||
traceback grezzo ma degradare a un avviso di una riga su stderr, con stdout
|
||||
puro (`[]` in modalita' --json) ed exit 0, cosi' il modello prosegue senza
|
||||
exemplar. Il finalize-hook gestisce gia' lo stesso scenario in modo analogo.
|
||||
"""
|
||||
"""CLI adaptation: trusted context, verified recall and explicit availability failures."""
|
||||
import json
|
||||
from datetime import UTC, datetime
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
from typer.testing import CliRunner
|
||||
|
||||
from tht.cli import app
|
||||
from tht.memory import MemoryRecord, save_registry
|
||||
from tht.memory.models import MemoryUnavailable
|
||||
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
|
||||
from tht.vectorstore.store import VectorHit
|
||||
|
||||
|
||||
def _cfg(tmp_path):
|
||||
cfg = tmp_path / "workspace.yaml"
|
||||
cfg.write_text(
|
||||
"database: {database: d, schema: s, user: u, password: p, transport: direct}\n"
|
||||
f"paths: {{artifacts: {tmp_path/'a'}, indexes: {tmp_path/'i'}, sessions: {tmp_path/'se'}}}\n"
|
||||
"vector_db: {database: v, schema: vectors, user: u, password: p, transport: direct}\n"
|
||||
"embeddings: {base_url: 'http://localhost:11434', model: qwen3-embedding:0.6b, dim: 1024}\n"
|
||||
)
|
||||
return cfg
|
||||
@pytest.fixture
|
||||
def runtime(monkeypatch):
|
||||
service = SimpleNamespace(close=lambda: None, recall=lambda *a, **kw: [])
|
||||
monkeypatch.setattr("tht.cli.memory_cmd._load_config_or_exit", lambda path: object())
|
||||
# The config is supplied by the runner; mapping to the service is tested here.
|
||||
monkeypatch.setattr("tht.cli.memory_cmd._load_config_or_exit",
|
||||
lambda path: SimpleNamespace(embeddings=object(),
|
||||
database=SimpleNamespace(database="sales", db_schema="public")))
|
||||
monkeypatch.setattr("tht.cli.memory_cmd.memory_service", lambda cfg: service)
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.open_searcher", lambda cfg: object())
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object())
|
||||
return service
|
||||
|
||||
|
||||
def test_solved_search_degrades_when_vectordb_unreachable(tmp_path, monkeypatch):
|
||||
@pytest.mark.parametrize("failure", [VectorStoreError, VectorReadUnavailable])
|
||||
def test_solved_search_warns_when_vector_is_unavailable(runtime, monkeypatch, failure):
|
||||
def boom(cfg):
|
||||
raise VectorStoreError("Qdrant non raggiungibile")
|
||||
|
||||
raise failure("private adapter details")
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.open_searcher", boom)
|
||||
res = CliRunner().invoke(
|
||||
app, ["memory", "solved-search", "quante ablazioni", "--json", "-c", str(_cfg(tmp_path))]
|
||||
)
|
||||
assert res.exit_code == 0, res.output
|
||||
assert json.loads(res.stdout) == [] # stdout puro: JSON valido
|
||||
assert "exemplar non disponibili" in res.stderr
|
||||
result = CliRunner().invoke(app, ["memory", "solved-search", "orders", "--json"])
|
||||
assert result.exit_code == 0
|
||||
assert json.loads(result.stdout) == []
|
||||
assert "exemplar non disponibili" in result.stderr
|
||||
assert "private adapter details" not in result.output
|
||||
|
||||
|
||||
def test_solved_search_degrades_direct_vector_read_error(tmp_path, monkeypatch):
|
||||
def boom(cfg):
|
||||
raise VectorReadUnavailable("Vector read operation unavailable")
|
||||
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.open_searcher", boom)
|
||||
res = CliRunner().invoke(
|
||||
app, ["memory", "solved-search", "quante ablazioni", "--json", "-c", str(_cfg(tmp_path))]
|
||||
)
|
||||
assert res.exit_code == 0, res.output
|
||||
assert json.loads(res.stdout) == []
|
||||
assert "exemplar non disponibili" in res.stderr
|
||||
def test_solved_search_passes_only_verified_service_payload(runtime):
|
||||
def recall(question, **kwargs):
|
||||
assert question == "orders"
|
||||
assert kwargs["solved"] and kwargs["top"] == 3
|
||||
assert kwargs["scope"].database == "sales"
|
||||
assert kwargs["scope"].schema_name == "public"
|
||||
return [{"id": "mem-id", "question": "Orders?", "sql": "select 1", "tables": []}]
|
||||
runtime.recall = recall
|
||||
result = CliRunner().invoke(app, ["memory", "solved-search", "orders", "--json"])
|
||||
assert result.exit_code == 0
|
||||
assert json.loads(result.stdout)[0]["sql"] == "select 1"
|
||||
|
||||
|
||||
def test_solved_search_degrades_human_mode(tmp_path, monkeypatch):
|
||||
def boom(cfg):
|
||||
raise VectorStoreError("Qdrant non raggiungibile")
|
||||
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.open_searcher", boom)
|
||||
res = CliRunner().invoke(
|
||||
app, ["memory", "solved-search", "quante ablazioni", "-c", str(_cfg(tmp_path))]
|
||||
)
|
||||
assert res.exit_code == 0, res.output
|
||||
assert "exemplar non disponibili" in res.stderr
|
||||
assert "Traceback" not in res.stderr
|
||||
def test_archive_failure_is_not_an_empty_success(runtime):
|
||||
def recall(*args, **kwargs):
|
||||
raise MemoryUnavailable("Memory archive is unavailable")
|
||||
runtime.recall = recall
|
||||
result = CliRunner().invoke(app, ["memory", "solved-search", "orders", "--json"])
|
||||
assert result.exit_code == 1
|
||||
assert json.loads(result.stdout)["status"] == 503
|
||||
|
||||
|
||||
def test_solved_search_json_maps_hit_metadata(tmp_path, monkeypatch):
|
||||
hit = VectorHit(
|
||||
id="solved:s1", kind="solved_question", ref="s1", title="quante ablazioni nel 2023",
|
||||
content="quante ablazioni nel 2023",
|
||||
metadata={
|
||||
"session_id": "s1", "question": "quante ablazioni nel 2023",
|
||||
"sql": "SELECT 1", "tables": ["fact_seeablazione"],
|
||||
},
|
||||
similarity=0.91,
|
||||
)
|
||||
|
||||
class FakeSearcher:
|
||||
def search(self, vec, top_n=10, kinds=None):
|
||||
assert kinds == ["solved_question"]
|
||||
return [hit]
|
||||
|
||||
class FakeEmbedder:
|
||||
def embed_query(self, text):
|
||||
return [0.1] * 8
|
||||
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.open_searcher", lambda cfg: FakeSearcher())
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda e: FakeEmbedder())
|
||||
res = CliRunner().invoke(
|
||||
app, ["memory", "solved-search", "quante ablazioni", "--json", "-c", str(_cfg(tmp_path))]
|
||||
)
|
||||
assert res.exit_code == 0, res.output
|
||||
data = json.loads(res.stdout)
|
||||
assert data == [{
|
||||
"session_id": "s1", "question": "quante ablazioni nel 2023",
|
||||
"sql": "SELECT 1", "tables": ["fact_seeablazione"], "score": 0.91,
|
||||
}]
|
||||
def test_missing_principal_cannot_bypass_admin(monkeypatch, tmp_path):
|
||||
monkeypatch.delenv("THT_PRINCIPAL_ISSUER", raising=False)
|
||||
monkeypatch.delenv("THT_PRINCIPAL_SUBJECT", raising=False)
|
||||
request = tmp_path / "request.json"
|
||||
request.write_text(json.dumps({"action": "list", "runtime": {}}))
|
||||
result = CliRunner().invoke(app, ["memory", "admin", "--workspace", "sales", "-c", str(request)])
|
||||
assert result.exit_code == 1
|
||||
assert json.loads(result.stdout)["status"] == 403
|
||||
|
||||
|
||||
def test_memory_search_excludes_legacy_table_records(tmp_path, monkeypatch):
|
||||
records = [
|
||||
MemoryRecord(
|
||||
id="mem-0001", ts=datetime(2026, 1, 1, tzinfo=UTC), session_id="s1",
|
||||
decision_seq=1, type="table_promoted", subject="fact_pazienti",
|
||||
),
|
||||
MemoryRecord(
|
||||
id="mem-0002", ts=datetime(2026, 1, 1, tzinfo=UTC), session_id="s1",
|
||||
decision_seq=2, type="concept_clarified", subject="paziente attivo",
|
||||
detail="flag_attivo = TRUE",
|
||||
),
|
||||
]
|
||||
cfg = _cfg(tmp_path)
|
||||
save_registry(records, tmp_path / "a" / "memory" / "registry.jsonl")
|
||||
@pytest.mark.parametrize("command", ["search", "solved-search"])
|
||||
def test_recall_cannot_override_runtime_database_context(runtime, command):
|
||||
result = CliRunner().invoke(app, ["memory", command, "orders", "--json",
|
||||
"--filters", '{"database":"outside"}'])
|
||||
assert result.exit_code == 1
|
||||
assert json.loads(result.stdout)["status"] == 400
|
||||
|
||||
class FakeSearcher:
|
||||
def search(self, vec, top_n=10, kinds=None):
|
||||
return [
|
||||
VectorHit(
|
||||
id=f"memory:{record.id}", kind="memory", ref=record.id,
|
||||
title=record.subject, content=record.detail, metadata={},
|
||||
similarity=0.9,
|
||||
)
|
||||
for record in records
|
||||
]
|
||||
|
||||
class FakeEmbedder:
|
||||
def embed_query(self, text):
|
||||
return [0.1] * 8
|
||||
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.open_searcher", lambda workspace: FakeSearcher())
|
||||
monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda embeddings: FakeEmbedder())
|
||||
|
||||
res = CliRunner().invoke(
|
||||
app, ["memory", "search", "pazienti", "--json", "-c", str(cfg)]
|
||||
)
|
||||
|
||||
assert res.exit_code == 0, res.output
|
||||
assert [record["id"] for record in json.loads(res.stdout)] == ["mem-0002"]
|
||||
@pytest.mark.parametrize("command", ["search", "solved-search"])
|
||||
def test_recall_passes_explicit_business_and_table_filters(runtime, command):
|
||||
def recall(question, **kwargs):
|
||||
scope = kwargs["scope"]
|
||||
assert (scope.database, scope.schema_name, scope.table, scope.scope) == (
|
||||
"sales", "public", "orders", "Sales")
|
||||
assert scope.concepts == ["grain"]
|
||||
return []
|
||||
runtime.recall = recall
|
||||
result = CliRunner().invoke(app, ["memory", command, "orders", "--json", "--filters",
|
||||
'{"table":"orders","scope":"Sales","concepts":["grain"]}'])
|
||||
assert result.exit_code == 0, result.output
|
||||
|
||||
@@ -34,6 +34,9 @@ def test_built_wheel_omits_vector_sql_migrations_and_discovers_cli(tmp_path):
|
||||
assert not any(name.startswith("tht/migrations/vector/") for name in names)
|
||||
assert "tht/migrations/sessions/001_schema.sql" in names
|
||||
assert "tht/migrations/sessions/002_security.sql" in names
|
||||
assert "tht/migrations/memory/001_memory.sql" in names
|
||||
assert "tht/migrations/memory/002_hybrid_projection.sql" in names
|
||||
assert "tht/migrations/memory/003_review_receipts.sql" in names
|
||||
|
||||
subprocess.run(
|
||||
[sys.executable, "-m", "pip", "install", "--no-deps", "--target", str(target), wheel],
|
||||
|
||||
@@ -111,6 +111,7 @@ def test_workflow_definition_has_the_approved_semantic_contract():
|
||||
"name": "datamart",
|
||||
"advance": "reviewer_decide",
|
||||
"prerequisites": [
|
||||
{"decision_exists": "memory_summary_reviewed"},
|
||||
{
|
||||
"any": [
|
||||
{"decision_exists": "datamart_requested"},
|
||||
@@ -124,6 +125,7 @@ def test_workflow_definition_has_the_approved_semantic_contract():
|
||||
"datamart_declined",
|
||||
"memory_promoted",
|
||||
"memory_promotion_declined",
|
||||
"memory_summary_reviewed",
|
||||
],
|
||||
},
|
||||
]
|
||||
|
||||
@@ -80,9 +80,6 @@ class QdrantVectorStore:
|
||||
if not legacy_constructor and collections["reference"] == collections["memory"]:
|
||||
raise VectorStoreError("Qdrant reference and memory collections must be distinct")
|
||||
self._collections = dict(collections)
|
||||
self._split_collections = collections["reference"] != collections["memory"]
|
||||
self._legacy_collection = workspace_id
|
||||
self._legacy_checked = False
|
||||
self._workspace_id = workspace_id
|
||||
self._workspace_revision = None
|
||||
self._workspace_revision = workspace_revision
|
||||
@@ -177,7 +174,13 @@ class QdrantVectorStore:
|
||||
filter_must.extend(self._revision_filter(allowed_record_kinds))
|
||||
filter_must.append(self._semantic_kind_filter(allowed_record_kinds))
|
||||
filter_must.append({"key": "record_kind", "match": {"any": allowed_record_kinds}})
|
||||
if metadata_filter is not None:
|
||||
if metadata_filter is not None and "memory" in metadata_filter:
|
||||
if set(metadata_filter) != {"memory"} or not set(allowed_record_kinds) <= {
|
||||
"memory", "solved_question",
|
||||
}:
|
||||
raise VectorStoreError("Unsupported Memory metadata filter")
|
||||
filter_must.extend(self._memory_filter(metadata_filter["memory"]))
|
||||
elif metadata_filter is not None:
|
||||
allowed_filters = {
|
||||
"vector_generation", "document_ids", "workspace_id", "purpose",
|
||||
"required_kinds", "required_concepts", "required_tables", "required_columns",
|
||||
@@ -221,9 +224,9 @@ class QdrantVectorStore:
|
||||
raise VectorStoreError("Invalid vector metadata filter")
|
||||
filter_must.extend({"key": payload_key, "match": {"value": item}} for item in values)
|
||||
if retrieval_mode not in {"fused", "dense", "bm25"}:
|
||||
raise VectorStoreError("Evidence retrieval mode is invalid")
|
||||
raise VectorStoreError("Vector retrieval mode is invalid")
|
||||
if retrieval_mode == "dense":
|
||||
if allowed_record_kinds != ["evidence"]:
|
||||
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
|
||||
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
|
||||
response = self._call(
|
||||
"POST",
|
||||
@@ -236,7 +239,7 @@ class QdrantVectorStore:
|
||||
},
|
||||
)
|
||||
elif retrieval_mode == "bm25":
|
||||
if allowed_record_kinds != ["evidence"]:
|
||||
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
|
||||
raise VectorStoreError("Evidence branch diagnostics are only available for Evidence")
|
||||
if query_text is None or query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 query is invalid")
|
||||
@@ -267,8 +270,8 @@ class QdrantVectorStore:
|
||||
},
|
||||
)
|
||||
else:
|
||||
if allowed_record_kinds != ["evidence"]:
|
||||
raise VectorStoreError("Hybrid BM25 is only available for Evidence")
|
||||
if not set(allowed_record_kinds) <= {"evidence", "memory", "solved_question"}:
|
||||
raise VectorStoreError("Hybrid BM25 is only available for Evidence and Memory")
|
||||
if query_text.strip() == "" or query_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 query is invalid")
|
||||
self._ensure_collection(physical_collection, strict=False, require_bm25=True)
|
||||
@@ -323,16 +326,13 @@ class QdrantVectorStore:
|
||||
|
||||
def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int:
|
||||
validate_collection(collection)
|
||||
# The first write with the split configuration is also the upgrade cutover. This keeps
|
||||
# existing runtime memory reachable even when the operator reruns preprocessing without
|
||||
# invoking the explicit clear operation first.
|
||||
if self._split_collections:
|
||||
self._migrate_legacy_memory()
|
||||
# Memory is projected only from its authoritative archive. Never import legacy payloads.
|
||||
physical_collection = self._physical_collection_for_logical(collection)
|
||||
self._ensure_collection(
|
||||
physical_collection,
|
||||
strict=True,
|
||||
require_bm25=any(record.sparse_text is not None for record in records),
|
||||
maintain_bm25=collection == "memory",
|
||||
)
|
||||
points = []
|
||||
for write_record in records:
|
||||
@@ -341,8 +341,8 @@ class QdrantVectorStore:
|
||||
semantic_kind = qdrant_semantic_kind(write_record.record.kind)
|
||||
vector: list[float] | dict = write_record.embedding
|
||||
if write_record.sparse_text is not None:
|
||||
if semantic_kind != "evidence" or write_record.sparse_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("Evidence BM25 document is invalid")
|
||||
if semantic_kind not in {"evidence", "memory"} or write_record.sparse_language not in _BM25_LANGUAGES:
|
||||
raise VectorStoreError("BM25 document is invalid")
|
||||
vector = {
|
||||
"": write_record.embedding,
|
||||
"bm25": self._bm25_document(write_record.sparse_text, write_record.sparse_language),
|
||||
@@ -391,6 +391,26 @@ class QdrantVectorStore:
|
||||
)
|
||||
return before
|
||||
|
||||
def prepare_memory_index(self) -> None:
|
||||
"""Explicit rebuild may recreate a lost collection; reads never do so."""
|
||||
self._ensure_collection(self._collections["memory"], strict=True,
|
||||
require_bm25=True, maintain_bm25=True, allow_create=True)
|
||||
|
||||
def delete_memory_records(self, record_keys: list[str]) -> None:
|
||||
"""Delete exact authoritative Memory projections, never reference vectors."""
|
||||
if not record_keys or any(not key.startswith("card:mem-") for key in record_keys):
|
||||
raise VectorStoreError("Exact Memory card keys are required")
|
||||
collection = self._collections["memory"]
|
||||
if self._call("GET", f"/collections/{collection}", None, allow_missing=True) is None:
|
||||
return
|
||||
self._call("POST", f"/collections/{collection}/points/delete?wait=true", {
|
||||
"filter": {"must": [
|
||||
*self._workspace_filter(),
|
||||
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
|
||||
{"key": "record_key", "match": {"any": record_keys}},
|
||||
]},
|
||||
})
|
||||
|
||||
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int:
|
||||
if collection != "evidence" or _GENERATION.fullmatch(generation) is None:
|
||||
raise VectorStoreError("Only exact Evidence generations may be deleted")
|
||||
@@ -442,60 +462,14 @@ class QdrantVectorStore:
|
||||
return [{"key": "workspace_id", "match": {"value": self._workspace_id}}]
|
||||
|
||||
def clear_reference(self) -> bool:
|
||||
"""Preserve legacy memory, then drop only replaceable schema/Evidence vectors."""
|
||||
legacy_deleted = self._migrate_legacy_memory()
|
||||
"""Drop only replaceable schema/Evidence vectors; do not import legacy Memory."""
|
||||
collection = self._collections["reference"]
|
||||
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
|
||||
if response is None:
|
||||
return legacy_deleted
|
||||
return False
|
||||
self._call("DELETE", f"/collections/{collection}", None)
|
||||
return True
|
||||
|
||||
def _migrate_legacy_memory(self) -> bool:
|
||||
"""Preserve memory from the pre-split collection before retiring it."""
|
||||
legacy = self._legacy_collection
|
||||
if (
|
||||
self._legacy_checked
|
||||
or not self._split_collections
|
||||
or legacy in self._collections.values()
|
||||
):
|
||||
return False
|
||||
response = self._call("GET", f"/collections/{legacy}", None, allow_missing=True)
|
||||
if response is None:
|
||||
self._legacy_checked = True
|
||||
return False
|
||||
memory = self._collections["memory"]
|
||||
self._ensure_collection(memory, strict=True, allow_create=True)
|
||||
points = self._scroll(
|
||||
legacy,
|
||||
[
|
||||
*self._workspace_filter(),
|
||||
self._semantic_kind_filter(["memory", "solved_question"]),
|
||||
{"key": "record_kind", "match": {"any": ["memory", "solved_question"]}},
|
||||
],
|
||||
with_vector=True,
|
||||
)
|
||||
migrated = []
|
||||
for point in points:
|
||||
if not isinstance(point.get("id"), (str, int)) or "vector" not in point:
|
||||
raise VectorStoreError("Qdrant returned malformed legacy memory response")
|
||||
if not isinstance(point.get("payload"), dict):
|
||||
raise VectorStoreError("Qdrant returned malformed legacy memory response")
|
||||
migrated.append({
|
||||
"id": point["id"],
|
||||
"vector": point["vector"],
|
||||
"payload": point["payload"],
|
||||
})
|
||||
for start in range(0, len(migrated), UPSERT_BATCH_SIZE):
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{memory}/points?wait=true",
|
||||
{"points": migrated[start:start + UPSERT_BATCH_SIZE]},
|
||||
)
|
||||
self._call("DELETE", f"/collections/{legacy}", None)
|
||||
self._legacy_checked = True
|
||||
return True
|
||||
|
||||
def _physical_collection_for_logical(self, collection: str) -> str:
|
||||
validate_collection(collection)
|
||||
return self._collections["memory" if collection == "memory" else "reference"]
|
||||
@@ -551,6 +525,34 @@ class QdrantVectorStore:
|
||||
else "Embedding dimension does not match configured dimension"
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _memory_filter(value: object) -> list[dict]:
|
||||
fields = {"scope", "database", "schema_name", "table", "column"}
|
||||
if not isinstance(value, dict) or set(value) != fields | {"family", "concepts", "format"}:
|
||||
raise VectorStoreError("Invalid Memory metadata filter")
|
||||
if (any(not isinstance(value[key], str) for key in fields)
|
||||
or value["family"] is not None and not isinstance(value["family"], str)
|
||||
or value["family"] not in {None, "domain_clarification", "sql_rule",
|
||||
"solved_question", "explained_error"}
|
||||
or type(value["format"]) is not int or value["format"] != 2
|
||||
or not isinstance(value["concepts"], list)
|
||||
or not all(isinstance(c, str) and c for c in value["concepts"])):
|
||||
raise VectorStoreError("Invalid Memory metadata filter")
|
||||
must = [{"key": "memory_format", "match": {"value": value["format"]}}]
|
||||
for key in ("family", "scope"):
|
||||
if value[key]:
|
||||
must.append({"key": f"memory_{key}", "match": {"value": value[key]}})
|
||||
must.extend({"key": "memory_concepts", "match": {"value": c}} for c in value["concepts"])
|
||||
if value["database"]:
|
||||
dependency = [{"key": "database", "match": {"value": value["database"]}}]
|
||||
dependency.extend({"key": key, "match": {"any": ["", value[key]]}}
|
||||
for key in ("schema_name", "table", "column") if value[key])
|
||||
must.append({"should": [
|
||||
{"is_empty": {"key": "memory_dependencies"}},
|
||||
{"nested": {"key": "memory_dependencies", "filter": {"must": dependency}}},
|
||||
]})
|
||||
return must
|
||||
|
||||
@staticmethod
|
||||
def _bm25_compatible(info: dict) -> bool:
|
||||
sparse_vectors = info.get("config", {}).get("params", {}).get("sparse_vectors")
|
||||
@@ -561,7 +563,7 @@ class QdrantVectorStore:
|
||||
|
||||
def _ensure_collection(
|
||||
self, collection: str, *, strict: bool, require_bm25: bool = False,
|
||||
allow_create: bool = False,
|
||||
allow_create: bool = False, maintain_bm25: bool = False,
|
||||
) -> dict | None:
|
||||
created = False
|
||||
response = self._call("GET", f"/collections/{collection}", None, allow_missing=True)
|
||||
@@ -573,7 +575,9 @@ class QdrantVectorStore:
|
||||
self._call(
|
||||
"PUT",
|
||||
f"/collections/{collection}",
|
||||
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"}},
|
||||
{"vectors": {"size": self._expected_dimension or 1024, "distance": "Cosine"},
|
||||
**({"sparse_vectors": {"bm25": {"modifier": "idf"}}}
|
||||
if require_bm25 and maintain_bm25 else {})},
|
||||
)
|
||||
for field_name in _KEYWORD_INDEXES:
|
||||
self._call(
|
||||
@@ -614,7 +618,14 @@ class QdrantVectorStore:
|
||||
{"field_name": field_name, "field_schema": "keyword"},
|
||||
)
|
||||
if require_bm25 and not self._bm25_compatible(result):
|
||||
raise VectorStoreError("Evidence BM25 collection configuration mismatch")
|
||||
sparse = result.get("config", {}).get("params", {}).get("sparse_vectors")
|
||||
if maintain_bm25 and (sparse is None or isinstance(sparse, dict) and "bm25" not in sparse):
|
||||
# Explicit Memory writes may add the missing sparse vector without
|
||||
# touching dense points or the separately managed Reference collection.
|
||||
self._call("PUT", f"/collections/{collection}/vectors/bm25",
|
||||
{"sparse": {"modifier": "idf"}})
|
||||
else:
|
||||
raise VectorStoreError("BM25 collection configuration mismatch")
|
||||
return result
|
||||
|
||||
def _scroll(
|
||||
|
||||
@@ -0,0 +1,260 @@
|
||||
"""Closed session repair choices, durable receipts and authoritative archive activation.
|
||||
|
||||
Memory commits its receipt with the card. Evidence records the choice before writing
|
||||
files and recovers by comparing the approved result, never by repeating a stale write.
|
||||
"""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from sqlalchemy import text
|
||||
|
||||
from tht.evidence.canonical import CuratedEvidence
|
||||
from tht.evidence.local_archive import LocalEvidenceArchive, _content, _digest
|
||||
from tht.memory.models import CardInput, MemoryConflict, MemoryNotFound
|
||||
from tht.memory.review import digest
|
||||
from tht.phase import current_phase, effective_decisions
|
||||
|
||||
|
||||
class RepairOption(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
id: str = Field(pattern=r"^[a-zA-Z0-9_-]{1,80}$")
|
||||
label: str = Field(min_length=1, max_length=1000)
|
||||
archive: Literal["memory", "evidence"]
|
||||
target_id: str = Field(min_length=1, max_length=100)
|
||||
revision: str = Field(min_length=1, max_length=100)
|
||||
content: dict
|
||||
|
||||
|
||||
class RepairProposal(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
reason: str = Field(min_length=1, max_length=10000)
|
||||
options: list[RepairOption] = Field(min_length=1, max_length=5)
|
||||
|
||||
|
||||
def _context(snapshot):
|
||||
return digest({"phase": current_phase(snapshot), "question": snapshot.artifacts.get("question"),
|
||||
"decisions": [d.model_dump(mode="json") for d in effective_decisions(snapshot)]})
|
||||
|
||||
|
||||
def _authorize(service, snapshot):
|
||||
service._session(snapshot)
|
||||
if snapshot.manifest.status in {"finalized", "archived"}:
|
||||
raise MemoryConflict("Archive repair requires an open session")
|
||||
|
||||
|
||||
def _archive(cfg, workspace):
|
||||
root = cfg.evidence.local_archive_root if cfg.evidence else None
|
||||
if not root or Path(root).name != workspace:
|
||||
raise MemoryConflict("Local Evidence is unavailable in this workspace")
|
||||
return LocalEvidenceArchive(root)
|
||||
|
||||
|
||||
def _params(repo, snapshot, repair_id):
|
||||
return {"w": repo.workspace_id, "s": snapshot.manifest.id, "r": repair_id}
|
||||
|
||||
|
||||
def target(service, snapshot, cfg, archive, identity):
|
||||
"""Read complete current content for a proposal, bound to the source session."""
|
||||
_authorize(service, snapshot)
|
||||
if archive == "memory":
|
||||
value = service.repository.get(identity)
|
||||
return {"revision": value.revision, "content": CardInput.model_validate(
|
||||
value.model_dump(include=set(CardInput.model_fields))).model_dump(mode="json")}
|
||||
if archive != "evidence":
|
||||
raise ValueError("Unknown archive")
|
||||
try:
|
||||
value = _archive(cfg, service.repository.workspace_id).get(identity)
|
||||
except KeyError:
|
||||
raise MemoryNotFound("Evidence was not found in this workspace") from None
|
||||
return {"revision": value["revision"], "content": value["unit"].model_dump(mode="json")}
|
||||
|
||||
|
||||
def _read(repo, snapshot, repair_id):
|
||||
with repo.transaction() as connection:
|
||||
value = connection.execute(text("SELECT data FROM thoth_memory.archive_repairs "
|
||||
"WHERE workspace_id=:w AND session_id=:s AND repair_id=:r"),
|
||||
_params(repo, snapshot, repair_id)).scalar_one_or_none()
|
||||
if value is None:
|
||||
raise MemoryNotFound("Repair was not found in this session and workspace")
|
||||
return value
|
||||
|
||||
|
||||
def _write(repo, snapshot, repair_id, value):
|
||||
with repo.transaction() as connection:
|
||||
connection.execute(text("INSERT INTO thoth_memory.archive_repairs "
|
||||
"(workspace_id,session_id,repair_id,data) VALUES (:w,:s,:r,CAST(:data AS jsonb)) "
|
||||
"ON CONFLICT (workspace_id,session_id,repair_id) DO UPDATE SET data=EXCLUDED.data"),
|
||||
{**_params(repo, snapshot, repair_id), "data": json.dumps(value)})
|
||||
|
||||
|
||||
def prepare(service, snapshot, cfg, proposal: RepairProposal):
|
||||
_authorize(service, snapshot)
|
||||
if (len({o.id for o in proposal.options}) != len(proposal.options)
|
||||
or any(o.id in {"reject", "continue"} for o in proposal.options)):
|
||||
raise ValueError("Repair choices must have distinct identities")
|
||||
items = []
|
||||
for option in proposal.options:
|
||||
item = option.model_dump(mode="json")
|
||||
if option.archive == "memory":
|
||||
before = service.repository.get(option.target_id)
|
||||
if before.revision != option.revision:
|
||||
raise MemoryConflict("Memory changed; prepare new choices")
|
||||
item["before"] = before.model_dump(mode="json")
|
||||
item["content"] = CardInput.model_validate(option.content).model_dump(mode="json")
|
||||
else:
|
||||
archive = _archive(cfg, service.repository.workspace_id)
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
if not state["active"] or state["pending"] or state.get("import_writes"):
|
||||
raise MemoryConflict("Consolidate Evidence before proposing a session repair")
|
||||
files = archive._files(archive.evidence)
|
||||
if files != archive._files(archive._snapshot(state["active"])):
|
||||
raise MemoryConflict("Consolidate external Evidence edits before session repair")
|
||||
existing = archive._units(files).get(option.target_id)
|
||||
if existing is None:
|
||||
raise MemoryNotFound("Evidence was not found in this workspace")
|
||||
relative, before = existing
|
||||
if _content(before) != option.revision:
|
||||
raise MemoryConflict("Evidence changed; prepare new choices")
|
||||
value = CuratedEvidence.model_validate(option.content)
|
||||
if (value.schema_version != 4 or value.id != before.id
|
||||
or value.kind != before.kind or value.review_items):
|
||||
raise ValueError("Repair must preserve Evidence identity/kind and resolve review items")
|
||||
# Curators change knowledge, not the source history supplied by the archive.
|
||||
value = value.model_copy(update={"provenance": before.provenance})
|
||||
item.update(content=value.model_dump(mode="json"),
|
||||
before=before.model_dump(mode="json"), file=relative,
|
||||
other_files={p: _digest(v) for p, v in files.items() if p != relative})
|
||||
items.append(item)
|
||||
value = {"reason": proposal.reason, "options": items, "context": _context(snapshot),
|
||||
"choice": None, "status": "proposed", "saved": False, "indexed": False}
|
||||
repair_id = digest({"workspace": service.repository.workspace_id,
|
||||
"session": snapshot.manifest.id, **value})
|
||||
with service.repository.operation() as repo:
|
||||
try:
|
||||
_read(repo, snapshot, repair_id)
|
||||
except MemoryNotFound:
|
||||
_write(repo, snapshot, repair_id, value)
|
||||
return show(service, snapshot, cfg, repair_id)
|
||||
|
||||
|
||||
def show(service, snapshot, cfg, repair_id):
|
||||
_authorize(service, snapshot)
|
||||
value = _read(service.repository, snapshot, repair_id)
|
||||
# A completed receipt describes history; current eligibility is checked afresh.
|
||||
if value.get("saved"):
|
||||
option = next(o for o in value["options"] if o["id"] == value["choice"])
|
||||
try:
|
||||
if option["archive"] == "memory":
|
||||
current = service.repository.get(option["target_id"])
|
||||
same = current.revision == value["saved_revision"]
|
||||
indexed = same and current.indexed
|
||||
else:
|
||||
archive = _archive(cfg, service.repository.workspace_id)
|
||||
with archive.operation():
|
||||
units = archive._units(archive._files(archive.evidence))
|
||||
current = units[option["target_id"]][1]
|
||||
same = _content(current) == _content(
|
||||
CuratedEvidence.model_validate(option["content"]))
|
||||
state = archive._state()
|
||||
active = archive._units(archive._files(archive._snapshot(state["active"]))) \
|
||||
if state["active"] else {}
|
||||
indexed = same and option["target_id"] in active and \
|
||||
active[option["target_id"]][1] == current
|
||||
value.update(indexed=bool(indexed), status="superseded" if not same else
|
||||
"active" if indexed else "pending_activation")
|
||||
except (MemoryNotFound, KeyError, ValueError):
|
||||
value.update(indexed=False, status="superseded")
|
||||
return {**value, "repair_id": repair_id, "can_apply": service.principal.is_admin}
|
||||
|
||||
|
||||
def list_repairs(service, snapshot):
|
||||
_authorize(service, snapshot)
|
||||
with service.repository.transaction() as connection:
|
||||
rows = connection.execute(text("SELECT repair_id,data->>'status' AS recorded_status, "
|
||||
"data->>'reason' AS reason FROM thoth_memory.archive_repairs "
|
||||
"WHERE workspace_id=:w AND session_id=:s ORDER BY created_at"),
|
||||
{"w": service.repository.workspace_id, "s": snapshot.manifest.id}).mappings().all()
|
||||
return {"repairs": [dict(row) for row in rows]}
|
||||
|
||||
|
||||
def apply(service, snapshot, cfg, repair_id, choice, *, activate=None):
|
||||
_authorize(service, snapshot)
|
||||
if choice != "reject":
|
||||
service._admin()
|
||||
with service.repository.operation() as repo:
|
||||
value = _read(repo, snapshot, repair_id)
|
||||
if value["choice"] is not None and value["choice"] != choice:
|
||||
raise MemoryConflict("This repair already has a different recorded choice")
|
||||
if value["choice"] is None and value["context"] != _context(snapshot):
|
||||
raise MemoryConflict("Session decisions changed; reformulate the repair")
|
||||
if choice == "reject":
|
||||
value.update(choice=choice, status="rejected", actor=service.principal.subject)
|
||||
_write(repo, snapshot, repair_id, value)
|
||||
else:
|
||||
option = next((o for o in value["options"] if o["id"] == choice), None)
|
||||
if option is None:
|
||||
raise ValueError("Select one of the reviewed repair choices")
|
||||
if option["archive"] == "memory":
|
||||
with repo.transaction():
|
||||
if not value["saved"]:
|
||||
current = repo.get(option["target_id"])
|
||||
if current.revision != option["revision"]:
|
||||
raise MemoryConflict("Memory changed; reformulate the repair")
|
||||
repo.save(CardInput.model_validate(option["content"]),
|
||||
card_id=option["target_id"])
|
||||
value.update(choice=choice, saved=True, status="pending_activation",
|
||||
actor=service.principal.subject,
|
||||
saved_revision=repo.get(option["target_id"]).revision)
|
||||
_write(repo, snapshot, repair_id, value)
|
||||
elif repo.get(option["target_id"]).revision != value["saved_revision"]:
|
||||
raise MemoryConflict("The repaired Memory was changed again; do not replay it")
|
||||
result = service._propagate(repo, option["target_id"])
|
||||
value.update(indexed=result["indexed"], status="active" if result["indexed"] else
|
||||
"pending_activation")
|
||||
_write(repo, snapshot, repair_id, value)
|
||||
else:
|
||||
_apply_evidence(service, repo, snapshot, cfg, repair_id, choice, value, option,
|
||||
activate)
|
||||
return show(service, snapshot, cfg, repair_id)
|
||||
|
||||
|
||||
def _apply_evidence(service, repo, snapshot, cfg, repair_id, choice, receipt, option, activate):
|
||||
archive = _archive(cfg, repo.workspace_id)
|
||||
proposed = CuratedEvidence.model_validate(option["content"])
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
if state.get("import_writes") or (receipt["choice"] is None and state["pending"]):
|
||||
raise MemoryConflict("Finish the pending Evidence consolidation before this repair")
|
||||
files = archive._files(archive.evidence)
|
||||
if {p: _digest(v) for p, v in files.items() if p != option["file"]} != option["other_files"]:
|
||||
raise MemoryConflict("Other Evidence files changed; reconcile them before retrying")
|
||||
existing = archive._units(files).get(option["target_id"])
|
||||
if existing is None:
|
||||
raise MemoryConflict("Evidence was removed after review")
|
||||
_, current = existing
|
||||
already_written = _content(current) == _content(proposed) and receipt["choice"] == choice
|
||||
if not already_written and (receipt["saved"] or _content(current) != option["revision"]
|
||||
or current.provenance != proposed.provenance):
|
||||
raise MemoryConflict("Evidence changed; the approved correction cannot overwrite it")
|
||||
# Commit approval before touching the filesystem; a restart can recover only this choice.
|
||||
receipt.update(choice=choice, actor=service.principal.subject, status="applying")
|
||||
_write(repo, snapshot, repair_id, receipt)
|
||||
try:
|
||||
if not already_written:
|
||||
archive._save(proposed, expected_revision=option["revision"],
|
||||
actor=service.principal.subject)
|
||||
receipt.update(saved=True, status="pending_activation")
|
||||
_write(repo, snapshot, repair_id, receipt)
|
||||
archive._consolidate(service.principal.subject, activate)
|
||||
except Exception: # noqa: BLE001 - durable approval covers file/index interruption.
|
||||
receipt.update(status="pending_activation" if receipt["saved"] else "applying",
|
||||
indexed=False)
|
||||
_write(repo, snapshot, repair_id, receipt)
|
||||
return
|
||||
receipt.update(indexed=activate is not None,
|
||||
status="active" if activate else "pending_activation")
|
||||
_write(repo, snapshot, repair_id, receipt)
|
||||
@@ -23,6 +23,46 @@ from tht.evidence import (
|
||||
evidence_app = typer.Typer(help="Prepare and validate workspace Evidence", no_args_is_help=True)
|
||||
|
||||
|
||||
@evidence_app.command("sources", hidden=True)
|
||||
def sources_cmd(action: str, config: Path = CONFIG_OPT,
|
||||
source_id: str | None = typer.Option(None), revision: str | None = typer.Option(None),
|
||||
decision: str | None = typer.Option(None), actor: str = typer.Option("installation operator"),
|
||||
json_output: bool = typer.Option(False, "--json")):
|
||||
from tht.evidence.administration import ConsolidationError, source_action
|
||||
try:
|
||||
if action not in {"refresh", "decide"} or len(actor) > 256 or not actor.strip():
|
||||
raise ValueError("Invalid source action")
|
||||
if action == "refresh" and any(v is not None for v in (source_id, revision, decision)):
|
||||
raise ValueError("Refresh does not accept decision options")
|
||||
if action == "decide" and (decision not in {"keep", "replace"} or not source_id or not revision):
|
||||
raise ValueError("A source decision requires source-id, revision and keep or replace")
|
||||
payload = source_action(config, action=action, source_id=source_id, revision=revision,
|
||||
decision=decision, actor=actor)
|
||||
except Exception as error: # noqa: BLE001 - never expose connector/provider exception details
|
||||
safe = isinstance(error, (ValueError, EvidencePreparationError, ConsolidationError))
|
||||
_emit({"status": "failed", "code": "evidence_source_failed",
|
||||
"error": str(error)[:1500] if safe else "Source acquisition or refinement failed; existing Evidence is preserved.",
|
||||
"saved": isinstance(error, ConsolidationError) and error.saved}, json_output)
|
||||
raise typer.Exit(1) from None
|
||||
_emit(payload, json_output)
|
||||
|
||||
|
||||
@evidence_app.command("admin", hidden=True)
|
||||
def admin_cmd(workspace: str = typer.Option(...), config: Path = CONFIG_OPT):
|
||||
from tht.evidence.administration import browse
|
||||
try:
|
||||
if os.environ.get("THT_PRINCIPAL_IS_ADMIN", "").lower() not in {"true", "1"}:
|
||||
raise ValueError("Evidence administration requires an administrator")
|
||||
payload = json.loads(config.read_text())
|
||||
root = Path(payload["root"])
|
||||
if not root.is_absolute() or root.name != workspace:
|
||||
raise ValueError("Invalid workspace archive identity")
|
||||
_emit(browse(root, payload.get("query", {})), True)
|
||||
except (ValueError, OSError) as error:
|
||||
_emit({"code": "evidence_unavailable", "message": str(error)[:1500]}, True)
|
||||
raise typer.Exit(1) from None
|
||||
|
||||
|
||||
def _canonical_worktree(workspace_root: Path) -> Path:
|
||||
requested = workspace_root.absolute()
|
||||
root = workspace_root.resolve()
|
||||
@@ -123,7 +163,8 @@ def prepare_cmd(
|
||||
) -> None:
|
||||
"""Prepare changed Source Evidence without committing or publishing it."""
|
||||
root = _canonical_worktree(workspace_root)
|
||||
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
from tht.evidence.authoring import authoring_skill_path
|
||||
skill_path = authoring_skill_path()
|
||||
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
|
||||
try:
|
||||
try:
|
||||
@@ -180,7 +221,7 @@ def migrate_cmd(
|
||||
workspace_root: Path,
|
||||
json_output: Annotated[bool, typer.Option("--json", help="Write machine JSON to stdout.")] = False,
|
||||
) -> None:
|
||||
"""Rewrite legacy Curated units as table-free v3 Markdown without model calls."""
|
||||
"""Convert legacy units to editable v4 Markdown and establish a local baseline."""
|
||||
root = _canonical_worktree(workspace_root)
|
||||
try:
|
||||
report = migrate_workspace_evidence(root)
|
||||
|
||||
+307
-454
@@ -1,502 +1,355 @@
|
||||
# TODO (drop registry, decisione spec 5): questo modulo e' portato col modello
|
||||
# registry intatto (load_registry/promote/update_record/delete_record). Le memory
|
||||
# dovrebbero vivere SOLO nel vectordb (metadata arricchito con subject/detail/rationale
|
||||
# in Onda 3.1). Riscrivere: promote -> upsert batch vectordb; list/show -> scan
|
||||
# vectordb; delete -> metadata.status="superseded"; index/clear -> droppati.
|
||||
# Task separato: la validazione richiede L2 (vectordb reale).
|
||||
"""Thin command adapters for the authoritative Memory service."""
|
||||
|
||||
import json
|
||||
import re
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
from sqlalchemy.exc import OperationalError, ProgrammingError
|
||||
from pydantic import ValidationError
|
||||
|
||||
from tht.cli._guards import (
|
||||
require_server_profile,
|
||||
require_vector_write_allowed,
|
||||
)
|
||||
from tht.cli.config_cmd import CONFIG_OPT
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
from tht.cli.session_cmd import load_snapshot_or_exit
|
||||
from tht.cli.vector_cmd import require_vector_cfg
|
||||
from tht.memory.models import CardInput, CardQuery, MemoryError
|
||||
from tht.memory.runtime import admin_service, memory_service
|
||||
from tht.ports.vector import VectorStoreError
|
||||
from tht.vectorstore.embeddings import EmbeddingsError
|
||||
|
||||
memory_app = typer.Typer(help="Review memory (registro canonico + indice semantico)")
|
||||
DECISION_OPT = typer.Option(None, "--decision", help="Seq da promuovere (ripetibile).")
|
||||
memory_app = typer.Typer(help="Authoritative Memory cards and verified recall")
|
||||
DECISION_OPT = typer.Option(None, "--decision")
|
||||
|
||||
|
||||
def registry_path(cfg) -> Path:
|
||||
if getattr(cfg.paths, "memory", None) is not None:
|
||||
return cfg.paths.memory / "registry.jsonl"
|
||||
# Legacy location remains readable while old workspaces are retired.
|
||||
return cfg.paths.artifacts / "memory" / "registry.jsonl"
|
||||
def _output(value):
|
||||
typer.echo(json.dumps(value, ensure_ascii=False, default=str))
|
||||
|
||||
|
||||
def _resync_memory(cfg):
|
||||
"""Risincronizza l'indice semantico col registro corrente (incrementale)."""
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.vector_cmd import make_embedder, sync_canonical_records
|
||||
from tht.memory import load_registry, memory_vector_records
|
||||
|
||||
records = memory_vector_records(load_registry(registry_path(cfg)))
|
||||
return sync_canonical_records(
|
||||
"memory",
|
||||
records,
|
||||
store=build_vector_store(cfg, require_write=True),
|
||||
embedder=make_embedder(cfg.embeddings),
|
||||
)
|
||||
|
||||
|
||||
@memory_app.command("promote")
|
||||
def promote_cmd(
|
||||
session: str = typer.Option(..., "--session"),
|
||||
decision: list[int] = DECISION_OPT,
|
||||
preview: bool = typer.Option(False, "--preview", help="Mostra i candidati in JSON, non scrive."),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Promuove le decisioni SCELTE nel registro globale. Usa --preview per vedere i candidati."""
|
||||
import json as _json
|
||||
|
||||
from tht.memory import promote_snapshot
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
|
||||
if preview:
|
||||
from tht.memory import (
|
||||
MAX_PROMOTION_CANDIDATES,
|
||||
preview_promotions_snapshot,
|
||||
reusable_promotions_snapshot,
|
||||
)
|
||||
cand = preview_promotions_snapshot(snapshot, registry_path(cfg))
|
||||
extra = len(reusable_promotions_snapshot(snapshot, registry_path(cfg))) - len(cand)
|
||||
payload = [
|
||||
{"decision_seq": c.decision_seq, "type": c.type, "subject": c.subject,
|
||||
"detail": c.detail, "rationale": c.rationale,
|
||||
"question_context": c.question_context,
|
||||
"tables": c.tables, "concepts": c.concepts}
|
||||
for c in cand
|
||||
]
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(payload, ensure_ascii=False, indent=2))
|
||||
elif not payload:
|
||||
typer.secho("Nessun candidato da promuovere.", fg=typer.colors.YELLOW)
|
||||
else:
|
||||
for c in payload:
|
||||
typer.echo(f" [{c['decision_seq']}] {c['type']}: {c['subject']}")
|
||||
if extra > 0:
|
||||
typer.secho(
|
||||
f"NOTA: mostrati {len(cand)} candidati su {len(cand) + extra} riusabili "
|
||||
f"(cap {MAX_PROMOTION_CANDIDATES}); gli altri non sono proposti.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
return
|
||||
|
||||
if not decision:
|
||||
typer.secho("ERRORE: indica le decisioni con --decision <seq> (vedi `--preview`).",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
require_server_profile(cfg, "memory promote")
|
||||
require_vector_cfg(cfg)
|
||||
promoted = promote_snapshot(snapshot, seqs=list(decision), registry_path=registry_path(cfg))
|
||||
if not promoted:
|
||||
msg = "Nessuna nuova promozione (gia' presenti o seq inesistenti)."
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(
|
||||
{"promoted": [], "indexed": False, "message": msg}, ensure_ascii=False))
|
||||
else:
|
||||
typer.secho(msg, fg=typer.colors.YELLOW)
|
||||
return
|
||||
|
||||
# Promozione nel registro: riuscita. L'indicizzazione semantica puo' fallire
|
||||
# (runtime non pronto o vectordb irraggiungibile da questa postazione): in quel
|
||||
# caso le memorie restano nel registro ma NON sono trovate da `tht memory
|
||||
# search` finche' non si reindicizza sul server. `indexed` rende lo stato
|
||||
# leggibile da Pi, cosi' il reviewer lo vede invece di perderlo nello stderr.
|
||||
ids = [{"id": r.id, "type": r.type, "subject": r.subject} for r in promoted]
|
||||
indexed = True
|
||||
warning = None
|
||||
@contextmanager
|
||||
def _service(config):
|
||||
service = None
|
||||
try:
|
||||
_resync_memory(cfg)
|
||||
except (ProgrammingError, OperationalError):
|
||||
indexed = False
|
||||
warning = (
|
||||
f"{len(promoted)} memorie promosse nel registro, ma l'indice vettoriale "
|
||||
"NON e' stato sincronizzato (runtime vettoriale mancante o irraggiungibile): "
|
||||
"NON saranno trovate da `tht memory search` finche' non reindicizzi sul "
|
||||
"server (`tht memory index` quando il runtime vettoriale è disponibile)."
|
||||
)
|
||||
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(
|
||||
{"promoted": ids, "indexed": indexed,
|
||||
"message": warning or f"{len(promoted)} memorie promosse e indicizzate."},
|
||||
ensure_ascii=False, indent=2))
|
||||
return
|
||||
for r in promoted:
|
||||
typer.echo(f" {r.id}: {r.type} {r.subject}")
|
||||
if indexed:
|
||||
typer.secho(f"OK: {len(promoted)} memorie promosse e indicizzate.",
|
||||
fg=typer.colors.GREEN)
|
||||
else:
|
||||
typer.secho(f"ATTENZIONE: {warning}", fg=typer.colors.YELLOW)
|
||||
cfg = _load_config_or_exit(config)
|
||||
service = memory_service(cfg)
|
||||
yield cfg, service
|
||||
except MemoryError as error:
|
||||
_output({"code": error.code, "message": str(error), "status": error.status})
|
||||
raise typer.Exit(1) from None
|
||||
except (ValidationError, ValueError):
|
||||
_output({"code": "memory_invalid", "message": "Memory request is invalid", "status": 400})
|
||||
raise typer.Exit(1) from None
|
||||
except (VectorStoreError, EmbeddingsError, OSError):
|
||||
_output({"code": "memory_unavailable", "message": "Memory operation is unavailable",
|
||||
"status": 503})
|
||||
raise typer.Exit(1) from None
|
||||
finally:
|
||||
if service:
|
||||
service.close()
|
||||
|
||||
|
||||
@memory_app.command("save-one")
|
||||
def save_one_cmd(
|
||||
session: str = typer.Option(..., "--session"),
|
||||
decision: int = typer.Option(
|
||||
..., "--decision", help="decision_seq della decisione da salvare come memoria."
|
||||
),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Upsert mirato (una riga) della memoria di una decisione nel semantic store (D11).
|
||||
|
||||
Promuove la decisione nel registro locale (idempotente) e fa un singolo upsert
|
||||
con dedup hash client-side -- niente full-resync. Il factory seleziona il writer
|
||||
del runtime vettoriale attivo.
|
||||
"""
|
||||
import json as _json
|
||||
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
from tht.memory import load_registry, promote_snapshot, save_one_memory
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
require_vector_write_allowed(cfg, "memory save-one")
|
||||
store = build_vector_store(cfg, require_write=True)
|
||||
|
||||
# Promuove la decisione scelta nel registro locale (idempotente: salta se gia' presente
|
||||
# o se stale post-rollback, perche' _compute_promotions usa la vista effective).
|
||||
promote_snapshot(snapshot, seqs=[decision], registry_path=registry_path(cfg))
|
||||
records = [r for r in load_registry(registry_path(cfg)) if r.session_id == snapshot.manifest.id]
|
||||
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
count = save_one_memory(records, decision, store=store, embedder=embedder)
|
||||
|
||||
msg = (
|
||||
f"{count} memoria salvata nell'indice semantico (decision_seq {decision})."
|
||||
if count
|
||||
else f"Nessun upsert (decisione {decision} assente/stale o memoria gia' aggiornata)."
|
||||
)
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(
|
||||
{"upserted": count, "decision_seq": decision, "message": msg}, ensure_ascii=False))
|
||||
return
|
||||
typer.secho(f"OK: {msg}", fg=typer.colors.GREEN if count else typer.colors.YELLOW)
|
||||
|
||||
|
||||
@memory_app.command("index")
|
||||
def index_cmd(config: Path = CONFIG_OPT) -> None:
|
||||
"""Sincronizza il registro memory nell'indice semantico (full-resync)."""
|
||||
from tht.cli.vector_cmd import _print_stats
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_write_allowed(cfg, "memory index")
|
||||
require_vector_cfg(cfg)
|
||||
_print_stats(_resync_memory(cfg))
|
||||
@memory_app.command("admin")
|
||||
def admin_cmd(workspace: str = typer.Option(...), config: Path = CONFIG_OPT):
|
||||
"""Backend-owned request snapshot. No DWH or session configuration is required."""
|
||||
service = None
|
||||
try:
|
||||
if not re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", workspace):
|
||||
raise ValueError("Invalid workspace")
|
||||
payload = json.loads(config.read_text())
|
||||
service = admin_service(workspace, payload["runtime"])
|
||||
request = payload.get("request", {})
|
||||
action = payload["action"]
|
||||
if action == "cleanup":
|
||||
from tht.memory.cleanup import CleanupRequest, cleanup
|
||||
result = cleanup(service, CleanupRequest.model_validate(request))
|
||||
elif action == "list":
|
||||
result = service.list(CardQuery.model_validate(request))
|
||||
elif action == "show":
|
||||
result = service.get(request["id"])
|
||||
elif action in {"create", "update"}:
|
||||
result = service.save(CardInput.model_validate(request["card"]),
|
||||
request.get("id") if action == "update" else None)
|
||||
elif action == "delete":
|
||||
result = service.delete(request["id"])
|
||||
elif action == "pending":
|
||||
result = service.pending()
|
||||
elif action == "retry":
|
||||
result = service.retry(request["id"])
|
||||
else:
|
||||
raise ValueError("Unknown Memory action")
|
||||
_output(result)
|
||||
except MemoryError as error:
|
||||
_output({"code": error.code, "message": str(error), "status": error.status})
|
||||
raise typer.Exit(1) from None
|
||||
except (ValueError, KeyError, TypeError, OSError):
|
||||
_output({"code": "memory_invalid", "message": "Memory request is invalid", "status": 400})
|
||||
raise typer.Exit(1) from None
|
||||
finally:
|
||||
if service:
|
||||
service.close()
|
||||
|
||||
|
||||
@memory_app.command("list")
|
||||
def list_cmd(
|
||||
type_: str = typer.Option(None, "--type", help="Filtra per tipo decisione."),
|
||||
session: str = typer.Option(None, "--session", help="Filtra per sessione."),
|
||||
table: str = typer.Option(None, "--table", help="Filtra per tabella coinvolta."),
|
||||
concept: str = typer.Option(None, "--concept", help="Filtra per concetto."),
|
||||
json_out: bool = typer.Option(False, "--json"),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Elenca le memorie del registro (filtri combinati in AND)."""
|
||||
import json as _json
|
||||
|
||||
from rich.console import Console
|
||||
from rich.table import Table
|
||||
|
||||
from tht.memory import load_registry
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
recs = load_registry(registry_path(cfg))
|
||||
if type_:
|
||||
recs = [r for r in recs if r.type == type_]
|
||||
if session:
|
||||
recs = [r for r in recs if r.session_id == session]
|
||||
if table:
|
||||
recs = [r for r in recs if table in r.tables]
|
||||
if concept:
|
||||
recs = [r for r in recs if concept in r.concepts]
|
||||
|
||||
if json_out:
|
||||
typer.echo(_json.dumps([r.model_dump(mode="json") for r in recs],
|
||||
ensure_ascii=False, indent=2))
|
||||
return
|
||||
if not recs:
|
||||
typer.secho("Nessuna memoria nel registro.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
t = Table(title="Review memory")
|
||||
for col in ("Id", "Tipo", "Soggetto", "Sessione"):
|
||||
t.add_column(col)
|
||||
for r in recs:
|
||||
t.add_row(r.id, r.type, r.subject, r.session_id)
|
||||
Console().print(t)
|
||||
def list_cmd(filters: str = typer.Option("{}", "--filters"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.list(CardQuery.model_validate_json(filters)))
|
||||
|
||||
|
||||
@memory_app.command("show")
|
||||
def show_cmd(
|
||||
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
|
||||
json_out: bool = typer.Option(False, "--json"),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Mostra una singola memoria."""
|
||||
import json as _json
|
||||
def show_cmd(mem_id: str, json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.get(mem_id))
|
||||
|
||||
from tht.memory import load_registry
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
rec = {r.id: r for r in load_registry(registry_path(cfg))}.get(mem_id)
|
||||
if rec is None:
|
||||
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=6)
|
||||
if json_out:
|
||||
typer.echo(_json.dumps(rec.model_dump(mode="json"), ensure_ascii=False, indent=2))
|
||||
return
|
||||
for k, v in rec.model_dump(mode="json").items():
|
||||
typer.echo(f"{k}: {v}")
|
||||
@memory_app.command("create")
|
||||
def create_cmd(data: Path = typer.Option(..., "--data"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.save(CardInput.model_validate_json(data.read_text())))
|
||||
|
||||
|
||||
@memory_app.command("update")
|
||||
def update_cmd(
|
||||
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
|
||||
subject: str = typer.Option(None, "--subject"),
|
||||
type_: str = typer.Option(None, "--type"),
|
||||
detail: str = typer.Option(None, "--detail"),
|
||||
rationale: str = typer.Option(None, "--rationale"),
|
||||
question_context: str = typer.Option(None, "--question-context"),
|
||||
tables: str = typer.Option(None, "--tables", help="CSV; \"\" per azzerare."),
|
||||
concepts: str = typer.Option(None, "--concepts", help="CSV; \"\" per azzerare."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Modifica i campi di merito di una memoria (provenienza immutabile)."""
|
||||
from typing import get_args
|
||||
|
||||
from tht.decisions import DecisionType
|
||||
from tht.memory import MemoryNotFound, update_record
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
|
||||
fields: dict = {}
|
||||
for name, val in (("subject", subject), ("type", type_), ("detail", detail),
|
||||
("rationale", rationale), ("question_context", question_context)):
|
||||
if val is not None:
|
||||
fields[name] = val
|
||||
if tables is not None:
|
||||
fields["tables"] = [t.strip() for t in tables.split(",") if t.strip()]
|
||||
if concepts is not None:
|
||||
fields["concepts"] = [c.strip() for c in concepts.split(",") if c.strip()]
|
||||
|
||||
if not fields:
|
||||
typer.secho("ERRORE: nessun campo da modificare indicato.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
if "type" in fields and fields["type"] not in get_args(DecisionType):
|
||||
typer.secho(f"ERRORE: tipo '{fields['type']}' non valido.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1)
|
||||
|
||||
require_server_profile(cfg, "memory update")
|
||||
require_vector_cfg(cfg)
|
||||
try:
|
||||
rec = update_record(registry_path(cfg), mem_id, fields)
|
||||
except MemoryNotFound:
|
||||
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=6)
|
||||
_resync_memory(cfg)
|
||||
typer.secho(f"OK: {rec.id} aggiornata e reindicizzata.", fg=typer.colors.GREEN)
|
||||
def update_cmd(mem_id: str, data: Path = typer.Option(..., "--data"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.save(CardInput.model_validate_json(data.read_text()), mem_id))
|
||||
|
||||
|
||||
@memory_app.command("delete")
|
||||
def delete_cmd(
|
||||
mem_id: str = typer.Argument(..., help="Id memoria (es. mem-0001)."),
|
||||
yes: bool = typer.Option(False, "--yes", "-y", help="Salta la conferma."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Cancella una singola memoria (registro + indice)."""
|
||||
from tht.memory import MemoryNotFound, delete_record
|
||||
def delete_cmd(mem_id: str, yes: bool = typer.Option(False, "--yes", "-y"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
service._admin()
|
||||
if not yes and not typer.confirm(f"Delete Memory card {mem_id} and its links?"):
|
||||
raise typer.Exit(1)
|
||||
_output(service.delete(mem_id))
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_server_profile(cfg, "memory delete")
|
||||
require_vector_cfg(cfg)
|
||||
if not yes and not typer.confirm(f"Cancellare definitivamente la memoria '{mem_id}'?"):
|
||||
typer.secho("Annullato.", fg=typer.colors.YELLOW)
|
||||
raise typer.Exit(code=1)
|
||||
try:
|
||||
delete_record(registry_path(cfg), mem_id)
|
||||
except MemoryNotFound:
|
||||
typer.secho(f"ERRORE: memoria '{mem_id}' non trovata. Usa `tht memory list`.",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=6)
|
||||
_resync_memory(cfg)
|
||||
typer.secho(f"OK: {mem_id} cancellata e deindicizzata.", fg=typer.colors.GREEN)
|
||||
|
||||
@memory_app.command("pending")
|
||||
def pending_cmd(json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.pending())
|
||||
|
||||
|
||||
@memory_app.command("retry")
|
||||
def retry_cmd(mem_id: str, json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.retry(mem_id))
|
||||
|
||||
|
||||
@memory_app.command("index")
|
||||
def index_cmd(json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (_, service):
|
||||
_output(service.rebuild())
|
||||
|
||||
|
||||
@memory_app.command("promote")
|
||||
def promote_cmd(session: str = typer.Option(..., "--session"), decision: list[int] = DECISION_OPT,
|
||||
preview: bool = typer.Option(False, "--preview"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (cfg, service):
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
if preview:
|
||||
_output(service.promotions(snapshot)[:5])
|
||||
elif not decision:
|
||||
raise ValueError("Explicit decisions are required")
|
||||
else:
|
||||
results = service.promote(snapshot, decision)
|
||||
_output({"promoted": [r.get("card") for r in results],
|
||||
"indexed": all(r["indexed"] for r in results), "results": results})
|
||||
|
||||
|
||||
@memory_app.command("propose")
|
||||
def propose_cmd(session: str = typer.Option(..., "--session"),
|
||||
data: Path = typer.Option(..., "--data"), config: Path = CONFIG_OPT):
|
||||
"""Persist reviewer-grounded proposals without changing the Memory archive."""
|
||||
from tht.cli.session_cmd import session_repository
|
||||
from tht.memory.review import validate_proposals
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
service._session(snapshot)
|
||||
proposals = validate_proposals(snapshot, json.loads(data.read_text()))
|
||||
session_repository(cfg).write_artifact(session, "memory_proposals",
|
||||
json.dumps([p.model_dump(mode="json") for p in proposals], ensure_ascii=False))
|
||||
_output({"proposals": len(proposals), "saved_to_archive": False})
|
||||
|
||||
|
||||
@memory_app.command("summary")
|
||||
def summary_cmd(session: str = typer.Option(..., "--session"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.memory.review import prepare
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(prepare(service, load_snapshot_or_exit(cfg, session)))
|
||||
|
||||
|
||||
@memory_app.command("repair-prepare")
|
||||
def repair_prepare_cmd(session: str = typer.Option(..., "--session"),
|
||||
proposal_json: str = typer.Option(..., "--proposal-json"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import RepairProposal, prepare
|
||||
|
||||
if len(proposal_json.encode()) > 1_000_000:
|
||||
_output({"code": "memory_invalid", "message": "Repair proposal is too large", "status": 400})
|
||||
raise typer.Exit(1)
|
||||
with _service(config) as (cfg, service):
|
||||
_output(prepare(service, load_snapshot_or_exit(cfg, session), cfg,
|
||||
RepairProposal.model_validate_json(proposal_json)))
|
||||
|
||||
|
||||
@memory_app.command("repair-target")
|
||||
def repair_target_cmd(session: str = typer.Option(..., "--session"),
|
||||
archive: str = typer.Option(..., "--archive"),
|
||||
target_id: str = typer.Option(..., "--target-id"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import target
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(target(service, load_snapshot_or_exit(cfg, session), cfg, archive, target_id))
|
||||
|
||||
|
||||
@memory_app.command("repairs")
|
||||
def repairs_cmd(session: str = typer.Option(..., "--session"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import list_repairs
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(list_repairs(service, load_snapshot_or_exit(cfg, session)))
|
||||
|
||||
|
||||
@memory_app.command("repair-show")
|
||||
def repair_show_cmd(session: str = typer.Option(..., "--session"),
|
||||
repair_id: str = typer.Option(..., "--repair-id"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import show
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(show(service, load_snapshot_or_exit(cfg, session), cfg, repair_id))
|
||||
|
||||
|
||||
@memory_app.command("repair-apply")
|
||||
def repair_apply_cmd(session: str = typer.Option(..., "--session"),
|
||||
repair_id: str = typer.Option(..., "--repair-id"),
|
||||
choice: str = typer.Option(..., "--choice"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.archive_repair import apply
|
||||
from tht.cli.preprocess_cmd import run_from_config
|
||||
|
||||
def activate(snapshot):
|
||||
result = run_from_config(config, local_snapshot=snapshot)
|
||||
if result.status != "succeeded":
|
||||
raise RuntimeError("Evidence activation did not complete")
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(apply(service, load_snapshot_or_exit(cfg, session), cfg, repair_id, choice,
|
||||
activate=activate))
|
||||
|
||||
|
||||
@memory_app.command("review-apply")
|
||||
def review_apply_cmd(session: str = typer.Option(..., "--session"),
|
||||
review_json: str = typer.Option(..., "--review-json"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.memory.review import ReviewResponse, apply
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
_output(apply(service, load_snapshot_or_exit(cfg, session),
|
||||
ReviewResponse.model_validate_json(review_json)))
|
||||
|
||||
|
||||
@memory_app.command("save-one")
|
||||
def save_one_cmd(session: str = typer.Option(..., "--session"),
|
||||
decision: int = typer.Option(..., "--decision"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
with _service(config) as (cfg, service):
|
||||
results = service.promote(load_snapshot_or_exit(cfg, session), [decision])
|
||||
_output({"upserted": sum(r["indexed"] for r in results), "decision_seq": decision,
|
||||
"indexed": all(r["indexed"] for r in results), "results": results})
|
||||
|
||||
|
||||
@memory_app.command("search")
|
||||
def search_cmd(
|
||||
question: str = typer.Argument(..., help="Domanda o termini di ricerca."),
|
||||
top: int = typer.Option(5, "--top"),
|
||||
session: str = typer.Option(
|
||||
None, "--session",
|
||||
help="Esclude le memorie gia' decise (applicate o rifiutate) in questa sessione.",
|
||||
),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON per Pi."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Cerca memorie riapplicabili, ordinate per similarita'. Mai applicate in automatico.
|
||||
def search_cmd(question: str, top: int = typer.Option(5, "--top"),
|
||||
session: str | None = typer.Option(None, "--session"),
|
||||
filters: str = typer.Option("{}", "--filters"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher
|
||||
with _service(config) as (cfg, service):
|
||||
decisions = load_snapshot_or_exit(cfg, session).decisions if session else []
|
||||
_output(service.recall(question, searcher=open_searcher(cfg),
|
||||
embedder=make_embedder(cfg.embeddings), top=top, decisions=decisions,
|
||||
scope=_recall_scope(cfg, filters)))
|
||||
|
||||
Con `--session` non ripropone le memorie gia' decise in quella sessione (fix:
|
||||
memorie scartate riproposte): rifiutate via `memory_rejected` o gia' applicate."""
|
||||
from rich.console import Console
|
||||
from rich.table import Table
|
||||
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher, require_vector_cfg
|
||||
from tht.memory import load_registry, recall_memories
|
||||
def _recall_scope(cfg, filters):
|
||||
from tht.memory.retrieval import RecallScope
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_cfg(cfg)
|
||||
decisions = load_snapshot_or_exit(cfg, session).decisions if session is not None else []
|
||||
searcher = open_searcher(cfg)
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
results = recall_memories(
|
||||
question,
|
||||
records=load_registry(registry_path(cfg)),
|
||||
decisions=decisions,
|
||||
searcher=searcher,
|
||||
embedder=embedder,
|
||||
top=top,
|
||||
)
|
||||
context = {"database": cfg.database.database, "schema_name": cfg.database.db_schema}
|
||||
supplied = json.loads(filters)
|
||||
if not isinstance(supplied, dict) or any(
|
||||
key in supplied and supplied[key] != value for key, value in context.items()
|
||||
):
|
||||
raise ValueError("Recall cannot override the configured database/schema context")
|
||||
return RecallScope.model_validate({**supplied, **context})
|
||||
|
||||
if json_out:
|
||||
typer.echo(json.dumps(results, ensure_ascii=False, indent=2))
|
||||
return
|
||||
if not results:
|
||||
typer.secho("Nessuna memoria candidata.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
table = Table(title=f"Memorie candidate per: {question}")
|
||||
table.add_column("Id")
|
||||
table.add_column("Tipo")
|
||||
table.add_column("Soggetto")
|
||||
table.add_column("Contesto originale")
|
||||
table.add_column("Score", justify="right")
|
||||
for r in results:
|
||||
table.add_row(r["id"], r["type"], r["subject"],
|
||||
r["question_context"][:60], f"{r['score']:.3f}")
|
||||
Console().print(table)
|
||||
|
||||
@memory_app.command("rules")
|
||||
def rules_cmd(question: str, session: str = typer.Option(..., "--session"),
|
||||
filters: str = typer.Option("{}", "--filters"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
"""Consult SQL rules and explained errors in schema linking and SQL construction."""
|
||||
from types import SimpleNamespace
|
||||
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher
|
||||
from tht.phase import current_phase
|
||||
|
||||
with _service(config) as (cfg, service):
|
||||
snapshot = load_snapshot_or_exit(cfg, session)
|
||||
service._session(snapshot)
|
||||
if current_phase(snapshot) not in {4, 6, 7}:
|
||||
raise ValueError("Memory rules are consulted in schema linking or SQL construction")
|
||||
scope = _recall_scope(cfg, filters)
|
||||
vector = make_embedder(cfg.embeddings).embed_query(question)
|
||||
embedder = SimpleNamespace(embed_query=lambda _: vector)
|
||||
searcher = open_searcher(cfg)
|
||||
candidates = []
|
||||
for family in ("sql_rule", "explained_error"):
|
||||
candidates.extend(service.retrieve(question, searcher=searcher, embedder=embedder,
|
||||
scope=scope, family=family, top=5))
|
||||
_output([{**candidate.card.model_dump(mode="json"), "score": candidate.score,
|
||||
"retrieval_path": candidate.path, "consultative": True}
|
||||
for candidate in sorted(candidates, key=lambda c: (-c.score, c.card.id))[:10]])
|
||||
|
||||
|
||||
def index_solved_session(cfg, session_id: str) -> int:
|
||||
"""Indicizza la coppia domanda->SQL della sessione (kind solved_question).
|
||||
|
||||
Solleva SolvedIndexError se mancano gli artefatti: il finalize lo degrada a warning,
|
||||
il comando CLI lo converte in errore esplicito."""
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.sql_cmd import promoted_tables_for
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
from tht.memory import index_solved_question
|
||||
|
||||
store = build_vector_store(cfg, require_write=True)
|
||||
return index_solved_question(
|
||||
load_snapshot_or_exit(cfg, session_id),
|
||||
promoted_tables_for(cfg, session_id),
|
||||
store=store,
|
||||
embedder=make_embedder(cfg.embeddings),
|
||||
)
|
||||
"""Recovery only: never recreate a deleted card from historical session artifacts."""
|
||||
service = memory_service(cfg)
|
||||
try:
|
||||
result = service.retry_solved(load_snapshot_or_exit(cfg, session_id))
|
||||
if not result["indexed"]:
|
||||
raise RuntimeError(result["error"])
|
||||
return int(result["action"] == "upsert")
|
||||
finally:
|
||||
service.close()
|
||||
|
||||
|
||||
@memory_app.command("solved-index")
|
||||
def solved_index_cmd(
|
||||
session_id: str = typer.Argument(..., help="Id sessione con sql_final.sql approvato."),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Indicizza la coppia domanda->SQL nel semantic store (backfill; il finalize lo fa da solo)."""
|
||||
import json as _json
|
||||
|
||||
from tht.memory import SolvedIndexError
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_write_allowed(cfg, "memory solved-index")
|
||||
try:
|
||||
count = index_solved_session(cfg, session_id)
|
||||
except RuntimeError as e:
|
||||
typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=4)
|
||||
except SolvedIndexError as e:
|
||||
typer.secho(f"ERRORE: sessione {session_id} non indicizzabile: {e}",
|
||||
fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=3)
|
||||
msg = (
|
||||
f"1 coppia domanda->SQL indicizzata (solved:{session_id})."
|
||||
if count else "Nessun upsert: coppia gia' aggiornata."
|
||||
)
|
||||
if json_out:
|
||||
typer.echo(_json.dumps({"upserted": count, "id": f"solved:{session_id}"},
|
||||
ensure_ascii=False))
|
||||
return
|
||||
typer.secho(f"OK: {msg}", fg=typer.colors.GREEN)
|
||||
def solved_index_cmd(session_id: str, json_out: bool = typer.Option(False, "--json"),
|
||||
config: Path = CONFIG_OPT):
|
||||
"""Retry an existing authoritative exemplar's Qdrant projection; never import a session."""
|
||||
with _service(config) as (cfg, service):
|
||||
_output(service.retry_solved(load_snapshot_or_exit(cfg, session_id)))
|
||||
|
||||
|
||||
@memory_app.command("solved-search")
|
||||
def solved_search_cmd(
|
||||
question: str = typer.Argument(..., help="Domanda da confrontare con quelle risolte."),
|
||||
top: int = typer.Option(3, "--top"),
|
||||
json_out: bool = typer.Option(False, "--json", help="Output JSON (per Pi)."),
|
||||
config: Path = CONFIG_OPT,
|
||||
) -> None:
|
||||
"""Domande gia' risolte simili (kind solved_question): domanda, SQL e tabelle."""
|
||||
from rich.console import Console
|
||||
from rich.table import Table
|
||||
|
||||
def solved_search_cmd(question: str, top: int = typer.Option(3, "--top"),
|
||||
filters: str = typer.Option("{}", "--filters"),
|
||||
json_out: bool = typer.Option(False, "--json"), config: Path = CONFIG_OPT):
|
||||
from tht.cli.vector_cmd import make_embedder, open_searcher
|
||||
from tht.memory import search_solved_questions
|
||||
from tht.ports.vector import VectorReadUnavailable, VectorStoreError
|
||||
from tht.vectorstore.embeddings import EmbeddingsError
|
||||
|
||||
cfg = _load_config_or_exit(config)
|
||||
require_vector_cfg(cfg)
|
||||
# Degrado gentile: SKILL.md prescrive solved-search in F4/F6/F7 di ogni sessione,
|
||||
# quindi vectordb/Ollama irraggiungibili non devono produrre un traceback grezzo
|
||||
# nel transcript: avviso di una riga su stderr, stdout puro ([] in --json), exit 0.
|
||||
try:
|
||||
searcher = open_searcher(cfg)
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
results = search_solved_questions(
|
||||
question,
|
||||
searcher=searcher,
|
||||
embedder=embedder,
|
||||
top=top,
|
||||
)
|
||||
except (VectorStoreError, VectorReadUnavailable, EmbeddingsError, OperationalError) as e:
|
||||
typer.secho(
|
||||
f"ATTENZIONE: exemplar non disponibili ({e}). Prosegui senza.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
if json_out:
|
||||
typer.echo("[]")
|
||||
return
|
||||
if json_out:
|
||||
typer.echo(json.dumps(results, ensure_ascii=False, indent=2))
|
||||
return
|
||||
if not results:
|
||||
typer.secho("Nessuna domanda risolta simile.", fg=typer.colors.YELLOW)
|
||||
return
|
||||
table = Table(title=f"Domande risolte simili a: {question}")
|
||||
table.add_column("Sessione")
|
||||
table.add_column("Domanda")
|
||||
table.add_column("Tabelle")
|
||||
table.add_column("Score", justify="right")
|
||||
for r in results:
|
||||
table.add_row(r["session_id"], r["question"][:60],
|
||||
", ".join(r["tables"]), f"{r['score']:.3f}")
|
||||
Console().print(table)
|
||||
with _service(config) as (cfg, service):
|
||||
try:
|
||||
result = service.recall(question, searcher=open_searcher(cfg),
|
||||
embedder=make_embedder(cfg.embeddings), top=top, solved=True,
|
||||
scope=_recall_scope(cfg, filters))
|
||||
except (VectorStoreError, EmbeddingsError):
|
||||
typer.echo("Avviso: exemplar non disponibili; ricerca semantica non riuscita.", err=True)
|
||||
result = []
|
||||
_output(result)
|
||||
|
||||
@@ -50,6 +50,10 @@ def _evaluation_workspace_root(cfg) -> Path:
|
||||
|
||||
def _requires_candidate_evaluation(cfg) -> bool:
|
||||
evidence = cfg.evidence
|
||||
if evidence and evidence.local_archive_root and (
|
||||
evidence.local_archive_root / "evidence/.local/state.yaml"
|
||||
).exists():
|
||||
return False
|
||||
if evidence is None or evidence.schema_version != 2:
|
||||
return False
|
||||
return evidence.source_root is not None or any(
|
||||
@@ -224,7 +228,8 @@ def _parse_dwh_steps(value: str) -> tuple[str, ...]:
|
||||
return steps
|
||||
|
||||
|
||||
def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None):
|
||||
def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None,
|
||||
local_snapshot: Path | None = None):
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.schema_cmd import _load_config_or_exit
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
@@ -239,8 +244,11 @@ def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None =
|
||||
corpus_root = cfg.paths.artifacts.parent / "corpus"
|
||||
vector_store = build_vector_store(cfg, require_write=True)
|
||||
embedder = make_embedder(cfg.embeddings)
|
||||
from tht.evidence.adapters import FilesystemEvidenceSource
|
||||
sources = [FilesystemEvidenceSource(local_snapshot, patterns=("curated/**/*.md",))] \
|
||||
if local_snapshot is not None else build_sources(cfg.evidence)
|
||||
pipeline = build_preprocessing_pipeline(
|
||||
store=CorpusStore(corpus_root), sources=build_sources(cfg.evidence),
|
||||
store=CorpusStore(corpus_root), sources=sources,
|
||||
embedder=embedder,
|
||||
vector_store=vector_store,
|
||||
embedding_id=cfg.embeddings.id or f"ollama/{cfg.embeddings.model}",
|
||||
@@ -376,9 +384,17 @@ def evidence_cmd(
|
||||
dry_run: bool = typer.Option(False, "--dry-run"),
|
||||
resume: str | None = typer.Option(None, "--resume"),
|
||||
json_output: bool = typer.Option(False, "--json"),
|
||||
consolidate: bool = typer.Option(False, "--consolidate"),
|
||||
) -> None:
|
||||
if action is not None and action != "gc":
|
||||
raise typer.BadParameter("only the optional 'gc' action is supported")
|
||||
if consolidate and (dry_run or resume is not None or action is not None):
|
||||
message = "Consolidation cannot be combined with dry-run, resume or gc"
|
||||
if json_output:
|
||||
typer.echo(json.dumps({"status": "failed", "code": "invalid_consolidation", "error": message}))
|
||||
else:
|
||||
typer.secho(message, fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=2)
|
||||
if action == "gc":
|
||||
try:
|
||||
payload = gc_from_config(config, dry_run=dry_run)
|
||||
@@ -406,21 +422,29 @@ def evidence_cmd(
|
||||
typer.secho("ERRORE: resume requires a preprocessing run id", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=2)
|
||||
try:
|
||||
result = run_from_config(config, dry_run=dry_run, resume=resume)
|
||||
except Exception: # noqa: BLE001
|
||||
if consolidate:
|
||||
from tht.evidence.administration import consolidate_from_config
|
||||
result = consolidate_from_config(config)
|
||||
else:
|
||||
result = run_from_config(config, dry_run=dry_run, resume=resume)
|
||||
except Exception as error: # noqa: BLE001
|
||||
from tht.evidence.administration import ConsolidationError
|
||||
detail = str(error) if isinstance(error, ConsolidationError) else "preprocessing failed"
|
||||
payload = {"status": "failed"}
|
||||
if isinstance(error, ConsolidationError):
|
||||
payload["saved"] = error.saved
|
||||
if json_output:
|
||||
typer.echo(json.dumps(
|
||||
_evidence_json_payload(
|
||||
cfg,
|
||||
payload,
|
||||
code="preprocessing_failed",
|
||||
error="preprocessing failed",
|
||||
error=detail,
|
||||
),
|
||||
sort_keys=True,
|
||||
))
|
||||
else:
|
||||
typer.secho("ERRORE: preprocessing failed", fg=typer.colors.RED, err=True)
|
||||
typer.secho(f"ERRORE: {detail}", fg=typer.colors.RED, err=True)
|
||||
raise typer.Exit(code=1) from None
|
||||
payload = result.model_dump(mode="json")
|
||||
if payload.get("status") != "succeeded":
|
||||
|
||||
@@ -624,44 +624,7 @@ def finalize_cmd(session_id: str = typer.Argument(...), config: Path = CONFIG_OP
|
||||
repository, session_id, validation_report=report, evidence=evidence
|
||||
)
|
||||
# --- memoria attiva (parte B): indicizza la coppia domanda->SQL, best-effort ---
|
||||
# Qualunque errore (writer key assente, VPN giu', Ollama spento) NON deve
|
||||
# bloccare il finalize: l'indice e' derivato e recuperabile con
|
||||
# `tht memory solved-index <id>`. Memory owns this best-effort policy; core
|
||||
# has already committed the authoritative finalized snapshot above.
|
||||
try:
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
from tht.memory import index_solved_question_best_effort
|
||||
|
||||
finalized_snapshot = repository.get(session_id)
|
||||
outcome = index_solved_question_best_effort(
|
||||
finalized_snapshot,
|
||||
promoted_tables,
|
||||
store_factory=lambda: build_vector_store(cfg, require_write=True),
|
||||
embedder_factory=lambda: make_embedder(cfg.embeddings),
|
||||
)
|
||||
if outcome.error is not None:
|
||||
typer.secho(
|
||||
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({outcome.error}). "
|
||||
f"Recupera con `tht memory solved-index {session_id}`.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
elif outcome.upserted:
|
||||
typer.secho(
|
||||
"OK: coppia domanda->SQL indicizzata nel vectordb (solved_question).",
|
||||
fg=typer.colors.GREEN,
|
||||
)
|
||||
else:
|
||||
typer.secho(
|
||||
"Coppia domanda->SQL gia' aggiornata nel vectordb (nessun upsert).",
|
||||
fg=typer.colors.CYAN,
|
||||
)
|
||||
except Exception as e: # noqa: BLE001 - solved-question indexing is explicitly best effort
|
||||
typer.secho(
|
||||
f"ATTENZIONE: coppia domanda->SQL non indicizzata ({e}). "
|
||||
f"Recupera con `tht memory solved-index {session_id}`.",
|
||||
fg=typer.colors.YELLOW, err=True,
|
||||
)
|
||||
# Memory cards, including exemplars, are saved only by the explicit final review.
|
||||
typer.secho(f"OK: sessione {session_id} finalizzata. Artefatti:", fg=typer.colors.GREEN)
|
||||
for name in ARTIFACT_FILES:
|
||||
state = "presente" if name not in {"session_manifest.yaml", "review_decisions.jsonl"} else "persistito"
|
||||
|
||||
@@ -512,6 +512,8 @@ EvidenceSourceConfig = Annotated[
|
||||
|
||||
|
||||
class EvidenceSourcesConfig(BaseModel):
|
||||
# Persistent curator checkout; only its activated snapshot is used by preprocessing.
|
||||
local_archive_root: Path | None = None
|
||||
# Version 2 is the materialized source/curated authoring layout.
|
||||
schema_version: Literal[1, 2] = 1
|
||||
# Legacy curated-tree configuration remains accepted during migration.
|
||||
|
||||
@@ -43,6 +43,7 @@ DecisionType = Literal[
|
||||
# declined_promotion_seqs per non riproporre i candidati rifiutati).
|
||||
"memory_promoted",
|
||||
"memory_promotion_declined",
|
||||
"memory_summary_reviewed",
|
||||
# D15: marker di ritrazione. subject = "phase:N", retracts = decision_seq ritirata.
|
||||
# Resta nel log di audit (append-only); effective_decisions() la esclude dalla vista.
|
||||
"decision_retracted",
|
||||
|
||||
@@ -22,6 +22,7 @@ from tht.evidence.authoring import (
|
||||
)
|
||||
from tht.evidence.canonical import (
|
||||
CuratedEvidence,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
load_curated_tree,
|
||||
parse_curated_markdown,
|
||||
@@ -37,6 +38,7 @@ from tht.evidence.contracts import (
|
||||
validate_namespaced_value,
|
||||
validate_safe_metadata,
|
||||
)
|
||||
from tht.evidence.local_archive import ArchiveConflict, LocalEvidenceArchive
|
||||
from tht.evidence.preprocessing import EvidenceEmbedder, build_preprocessing_pipeline
|
||||
from tht.evidence.search import (
|
||||
ActiveEvidenceSearcher,
|
||||
@@ -58,6 +60,7 @@ from tht.evidence.sources import build_sources
|
||||
__all__ = [
|
||||
"AcquiredDocument",
|
||||
"ActiveEvidenceSearcher",
|
||||
"ArchiveConflict",
|
||||
"CorpusWorkspaceMismatchError",
|
||||
"CuratedEvidence",
|
||||
"EvidenceEmbedder",
|
||||
@@ -75,6 +78,8 @@ __all__ = [
|
||||
"EvidenceSource",
|
||||
"EvidenceSourceError",
|
||||
"EvidenceSourceErrorCategory",
|
||||
"LocalEvidenceArchive",
|
||||
"ManualEvidenceProvenance",
|
||||
"PiEvidenceRestructurer",
|
||||
"RestructureCandidate",
|
||||
"RestructureRequest",
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
"""Local Evidence browsing and explicit consolidation, independent of DWH access."""
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from .canonical import parse_curated_markdown
|
||||
from .local_archive import LocalEvidenceArchive, _content
|
||||
|
||||
|
||||
class ConsolidationError(RuntimeError):
|
||||
def __init__(self, message, *, saved=False):
|
||||
super().__init__(message)
|
||||
self.saved = saved
|
||||
|
||||
|
||||
def consolidate_from_config(config: Path):
|
||||
from tht.cli.preprocess_cmd import run_from_config
|
||||
from tht.config import load_config
|
||||
|
||||
from .authoring import EvidencePreparationError, migrate_workspace_evidence
|
||||
|
||||
cfg = load_config(config)
|
||||
if not cfg.evidence or not cfg.evidence.local_archive_root:
|
||||
raise ConsolidationError("Local Evidence is not configured for this workspace")
|
||||
root = cfg.evidence.local_archive_root
|
||||
archive = LocalEvidenceArchive(root)
|
||||
if not (archive.metadata / "state.yaml").exists():
|
||||
try:
|
||||
# Explicit first consolidation performs the one-time legacy conversion.
|
||||
if (archive.evidence / "manifest.yaml").is_file():
|
||||
migrate_workspace_evidence(root)
|
||||
else:
|
||||
archive.initialize()
|
||||
except (ValueError, OSError, EvidencePreparationError) as error:
|
||||
raise ConsolidationError(str(error)) from error
|
||||
result = None
|
||||
|
||||
def activate(snapshot):
|
||||
nonlocal result
|
||||
try:
|
||||
result = run_from_config(config, local_snapshot=snapshot)
|
||||
if result.status != "succeeded":
|
||||
raise ConsolidationError("Evidence indexing is blocked; check unit size and review items", saved=True)
|
||||
except ConsolidationError:
|
||||
raise
|
||||
except Exception as error:
|
||||
raise ConsolidationError("Evidence files were saved, but indexing failed. Retry consolidation.", saved=True) from error
|
||||
|
||||
try:
|
||||
archive.consolidate(actor=os.environ.get("THT_PRINCIPAL_SUBJECT") or "installation operator",
|
||||
activate=activate)
|
||||
except (ValueError, OSError) as error:
|
||||
raise ConsolidationError(str(error)) from error
|
||||
return result
|
||||
|
||||
|
||||
def browse(root: Path, query: dict):
|
||||
"""Read complete working units and their active status without opening an index."""
|
||||
archive = LocalEvidenceArchive(root)
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
active = archive._snapshot(state["active"]) if state["active"] else None
|
||||
active_units = archive._units(archive._files(active), allow_review=True) if active else {}
|
||||
files = archive._files(archive.evidence)
|
||||
items, errors = [], []
|
||||
seen = set()
|
||||
for relative, data in files.items():
|
||||
try:
|
||||
unit = parse_curated_markdown(data.decode(), path=Path(relative))
|
||||
if unit.id in seen:
|
||||
raise ValueError("Duplicate Evidence identifier")
|
||||
seen.add(unit.id)
|
||||
old = active_units.get(unit.id)
|
||||
status = "review_required" if unit.review_items else "legacy" if unit.schema_version != 4 \
|
||||
else "active" if old and _content(old[1]) == _content(unit) and old[1].provenance == unit.provenance else "modified" if old else "new"
|
||||
items.append({**unit.model_dump(mode="json"), "file": relative, "status": status,
|
||||
"revision": _content(unit)})
|
||||
except (ValueError, UnicodeError) as error:
|
||||
errors.append({"file": relative, "message": str(error)[:1500]})
|
||||
for identity, (relative, unit) in active_units.items():
|
||||
if identity not in seen:
|
||||
items.append({**unit.model_dump(mode="json"), "file": relative,
|
||||
"status": "invalid" if relative in files else "removed", "revision": _content(unit)})
|
||||
def matches(item):
|
||||
for field in ("kind", "status", "language"):
|
||||
if query.get(field) and item[field] != query[field]:
|
||||
return False
|
||||
if query.get("purpose") and query["purpose"] not in item["purposes"]:
|
||||
return False
|
||||
for key, field in (("concept", "concepts"), ("table", "tables"), ("column", "columns")):
|
||||
if query.get(key) and not any(query[key].casefold() in v.casefold() for v in item["applies_to"][field]):
|
||||
return False
|
||||
provenance = item["provenance"]
|
||||
if query.get("source") and query["source"].casefold() not in str(provenance).casefold():
|
||||
return False
|
||||
return not query.get("q") or query["q"].casefold() in str(item).casefold()
|
||||
selected = [item for item in items if matches(item)]
|
||||
selected.sort(key=lambda item: (str(item.get(query.get("sort", "title"), "")).casefold(), item["id"]),
|
||||
reverse=query.get("direction") == "desc")
|
||||
page, size = int(query.get("page", 1)), int(query.get("page_size", 25))
|
||||
if page < 1 or not 1 <= size <= 100:
|
||||
raise ValueError("Invalid Evidence page")
|
||||
result = {"items": selected[(page-1)*size:page*size], "total": len(selected), "page": page,
|
||||
"page_size": size, "errors": errors, "active_revision": state["active"],
|
||||
"pending_revision": state["pending"], "initialized": bool(state.get("baseline") or state["pending"] or state["active"])}
|
||||
if query.get("id"):
|
||||
result["item"] = next((item for item in items if item["id"] == query["id"]), None)
|
||||
from .imports import reviews
|
||||
result["source_reviews"] = reviews(archive)
|
||||
return result
|
||||
|
||||
|
||||
def source_action(config, *, action, source_id=None, revision=None, decision=None, actor="installation operator"):
|
||||
from tht.config import load_config
|
||||
|
||||
from .authoring import PiEvidenceRestructurer, authoring_skill_path, migrate_workspace_evidence
|
||||
from .imports import acquisition_sources, decide, refresh
|
||||
|
||||
cfg = load_config(config)
|
||||
if not cfg.evidence or not cfg.evidence.local_archive_root:
|
||||
raise ValueError("Local Evidence is not configured")
|
||||
archive = LocalEvidenceArchive(cfg.evidence.local_archive_root)
|
||||
if not (archive.metadata / "state.yaml").exists():
|
||||
if (archive.evidence / "manifest.yaml").is_file():
|
||||
migrate_workspace_evidence(archive.root)
|
||||
else:
|
||||
(archive.evidence / "curated").mkdir(parents=True, exist_ok=True)
|
||||
archive.initialize()
|
||||
if action == "refresh":
|
||||
skill = authoring_skill_path()
|
||||
return refresh(archive, acquisition_sources(cfg), PiEvidenceRestructurer(
|
||||
os.environ.get("THT_PI_EXECUTABLE", "pi"), skill))
|
||||
if action != "decide":
|
||||
raise ValueError("Unknown source action")
|
||||
|
||||
def activate(snapshot):
|
||||
from tht.cli.preprocess_cmd import run_from_config
|
||||
try:
|
||||
result = run_from_config(config, local_snapshot=snapshot)
|
||||
if result.status != "succeeded":
|
||||
raise RuntimeError("Indexing did not succeed")
|
||||
except Exception as error:
|
||||
raise ConsolidationError("Source decision saved, but indexing failed. Retry the same decision.", saved=True) from error
|
||||
|
||||
try:
|
||||
return decide(archive, source_id=source_id, revision=revision, decision=decision,
|
||||
actor=actor, activate=activate)
|
||||
except (ValueError, OSError) as error:
|
||||
from .imports import reviews
|
||||
if any(r["id"] == source_id and r["status"] == "applying" for r in reviews(archive)):
|
||||
raise ConsolidationError(f"Source decision saved. {str(error)[:1200]}. Retry the same decision.", saved=True) from error
|
||||
raise
|
||||
@@ -25,6 +25,7 @@ from tht.evidence.canonical import (
|
||||
EvidenceKind,
|
||||
EvidencePurpose,
|
||||
EvidenceScope,
|
||||
ManualEvidenceProvenance,
|
||||
ReviewItem,
|
||||
StrictModel,
|
||||
dump_curated_markdown,
|
||||
@@ -189,6 +190,12 @@ def _restore_exact_source_excerpts(
|
||||
})
|
||||
|
||||
|
||||
def authoring_skill_path() -> Path:
|
||||
"""The installed wheel and the deployment's Pi resources live in different roots."""
|
||||
root = Path(os.environ.get("THT_HARNESS_DIR", str(Path(__file__).resolve().parents[2])))
|
||||
return root / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
|
||||
|
||||
class PiEvidenceRestructurer:
|
||||
"""Invoke Pi once, without tools or session state, for one changed source."""
|
||||
|
||||
@@ -389,6 +396,14 @@ def dump_manifest(manifest: EvidenceManifest) -> str:
|
||||
def validate_workspace_evidence(workspace_root: Path) -> ValidationReport:
|
||||
"""Validate the curated corpus without writing the workspace."""
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local" / "state.yaml").is_file():
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
try:
|
||||
LocalEvidenceArchive(workspace_root).validate()
|
||||
return ValidationReport(())
|
||||
except (OSError, ValueError) as error:
|
||||
return ValidationReport((ValidationFinding("error", "local_evidence_invalid",
|
||||
"evidence/curated", str(error)),))
|
||||
findings: list[ValidationFinding] = []
|
||||
manifest_path = evidence_root / "manifest.yaml"
|
||||
if not manifest_path.is_file():
|
||||
@@ -485,6 +500,9 @@ def _validate_manifest_source(
|
||||
def _validate_unit(
|
||||
manifest: EvidenceManifest, evidence: CuratedEvidence, source: str | None,
|
||||
) -> list[ValidationFinding]:
|
||||
if isinstance(evidence.provenance, ManualEvidenceProvenance):
|
||||
return [ValidationFinding("error", "unresolved_review_item", evidence.id, item.message)
|
||||
for item in evidence.review_items]
|
||||
if evidence.id in manifest.orphans:
|
||||
return []
|
||||
path = evidence.provenance.source_file
|
||||
@@ -560,6 +578,8 @@ def prepare_workspace_evidence(
|
||||
raise EvidencePreparationError("authoring_workers_invalid")
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local" / "state.yaml").exists():
|
||||
raise EvidencePreparationError("local_archive_requires_explicit_source_refresh")
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
manifest_path = evidence_root / "manifest.yaml"
|
||||
try:
|
||||
@@ -695,10 +715,9 @@ def migrate_workspace_evidence(
|
||||
*,
|
||||
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
||||
) -> EvidenceMigrationReport:
|
||||
"""Rewrite legacy Curated units as table-free v3 Markdown without changing semantics."""
|
||||
"""Convert legacy Curated units to editable v4 and preserve a local baseline."""
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
try:
|
||||
manifest = load_manifest(evidence_root / "manifest.yaml")
|
||||
documents = load_curated_tree(evidence_root / "curated")
|
||||
@@ -709,7 +728,7 @@ def migrate_workspace_evidence(
|
||||
if len(documents_by_id) != len(documents):
|
||||
raise EvidencePreparationError("duplicate_evidence_id")
|
||||
upgraded = {
|
||||
evidence_id: document.model_copy(update={"schema_version": 3})
|
||||
evidence_id: document.model_copy(update={"schema_version": 4})
|
||||
for evidence_id, document in documents_by_id.items()
|
||||
}
|
||||
migrated_ids: list[str] = []
|
||||
@@ -728,12 +747,16 @@ def migrate_workspace_evidence(
|
||||
migrated = tuple(sorted(migrated_ids))
|
||||
unchanged = tuple(sorted(unchanged_ids))
|
||||
if not migrated:
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
LocalEvidenceArchive(workspace_root).initialize()
|
||||
return EvidenceMigrationReport(
|
||||
migrated=(),
|
||||
unchanged=unchanged,
|
||||
findings=validate_workspace_evidence(workspace_root).findings,
|
||||
)
|
||||
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
|
||||
from .local_archive import LocalEvidenceArchive
|
||||
LocalEvidenceArchive(workspace_root).initialize()
|
||||
return EvidenceMigrationReport(
|
||||
migrated=migrated,
|
||||
unchanged=unchanged,
|
||||
@@ -762,6 +785,8 @@ def resolve_workspace_evidence(
|
||||
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
if (evidence_root / ".local/state.yaml").exists():
|
||||
raise EvidencePreparationError("local_archive_requires_explicit_local_resolution")
|
||||
_reject_dirty_worktree(workspace_root, git_status or _git_status)
|
||||
try:
|
||||
manifest = load_manifest(evidence_root / "manifest.yaml")
|
||||
@@ -1017,7 +1042,7 @@ def _candidate_to_evidence(
|
||||
mode="json",
|
||||
exclude={"schema_version", "existing_id", "supporting_excerpts"},
|
||||
)
|
||||
data["schema_version"] = 3
|
||||
data["schema_version"] = 4
|
||||
data["id"] = evidence_id
|
||||
data["provenance"] = {
|
||||
"source_file": source_file,
|
||||
@@ -1038,7 +1063,7 @@ def _unsupported_unit(
|
||||
message="The current source no longer supports this Evidence unit.",
|
||||
),)
|
||||
return evidence.model_copy(update={
|
||||
"schema_version": 3,
|
||||
"schema_version": 4,
|
||||
"provenance": evidence.provenance.model_copy(update={
|
||||
"source_file": source_file,
|
||||
"source_sha256": source_hash,
|
||||
|
||||
@@ -98,6 +98,19 @@ class ReviewItem(StrictModel):
|
||||
field: str | None = None
|
||||
|
||||
|
||||
class ManualEvidenceProvenance(StrictModel):
|
||||
"""The curator supports the current content; an earlier document is only its origin."""
|
||||
|
||||
model_config = ConfigDict(extra="forbid", frozen=True)
|
||||
kind: Literal["manual"] = "manual"
|
||||
declared_by: str = Field(min_length=1)
|
||||
original: EvidenceProvenance | None = None
|
||||
|
||||
@property
|
||||
def source_file(self) -> str:
|
||||
return f"Manual declaration: {self.declared_by}"
|
||||
|
||||
|
||||
class FormulaPayload(StrictModel):
|
||||
concept: str
|
||||
columns: tuple[str, ...]
|
||||
@@ -229,19 +242,23 @@ _EVIDENCE_ID = re.compile(r"^evidence:[a-z0-9]+(?:-[a-z0-9]+)*$")
|
||||
|
||||
|
||||
class CuratedEvidence(StrictModel):
|
||||
schema_version: Literal[1, 2, 3]
|
||||
schema_version: Literal[1, 2, 3, 4]
|
||||
id: str
|
||||
title: str
|
||||
kind: EvidenceKind
|
||||
purposes: tuple[EvidencePurpose, ...]
|
||||
applies_to: EvidenceScope = Field(default_factory=EvidenceScope)
|
||||
language: str
|
||||
provenance: EvidenceProvenance
|
||||
provenance: EvidenceProvenance | ManualEvidenceProvenance
|
||||
review_items: tuple[ReviewItem, ...] = ()
|
||||
payload: EvidencePayload
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _validate_kind_payload(self) -> CuratedEvidence:
|
||||
if isinstance(self.provenance, ManualEvidenceProvenance) and self.schema_version != 4:
|
||||
raise ValueError("manual declarations require Curated unit schema v4")
|
||||
if self.schema_version == 4 and (not self.purposes or not self.language.strip()):
|
||||
raise ValueError("editable Evidence requires a language and at least one purpose")
|
||||
if not is_evidence_id(self.id):
|
||||
raise ValueError("id must use the evidence:<slug> form")
|
||||
expected = _PAYLOAD_TYPE_BY_KIND.get(self.kind)
|
||||
@@ -1056,13 +1073,19 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
_, frontmatter, body = text.split("---\n", 2)
|
||||
except ValueError as error:
|
||||
raise ValueError("curated evidence frontmatter is malformed") from error
|
||||
raw = yaml.safe_load(frontmatter)
|
||||
try:
|
||||
raw = yaml.safe_load(frontmatter)
|
||||
except yaml.YAMLError as error:
|
||||
raise ValueError("curated evidence frontmatter is malformed") from error
|
||||
try:
|
||||
data = dict(raw)
|
||||
except (TypeError, ValueError) as error:
|
||||
raise ValueError("curated evidence frontmatter must be a mapping") from error
|
||||
if data.get("schema_version") == 2:
|
||||
data = _parse_v2_body(data, body)
|
||||
elif data.get("schema_version") == 4:
|
||||
from tht.evidence.editable import parse_document, parse_metadata
|
||||
data = parse_document(parse_metadata(frontmatter), body)
|
||||
else:
|
||||
if body.strip():
|
||||
raise ValueError("curated evidence must not contain an ignored body")
|
||||
@@ -1081,6 +1104,9 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
|
||||
def dump_curated_markdown(value: CuratedEvidence) -> str:
|
||||
"""Render one canonical Curated Evidence Markdown document."""
|
||||
if value.schema_version == 4:
|
||||
from tht.evidence.editable import render_document
|
||||
return render_document(value)
|
||||
if value.schema_version == 3:
|
||||
return f"{_render_v3_metadata(value)}\n{_render_v3_body(value)}"
|
||||
if value.schema_version == 2:
|
||||
|
||||
@@ -8,7 +8,12 @@ from typing import Self
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, JsonValue, field_validator, model_validator
|
||||
|
||||
from tht.evidence.canonical import EVIDENCE_KINDS, EVIDENCE_PURPOSES
|
||||
from tht.evidence.canonical import (
|
||||
EVIDENCE_KINDS,
|
||||
EVIDENCE_PURPOSES,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
)
|
||||
from tht.evidence.contracts import (
|
||||
canonical_provenance_uri,
|
||||
normalize_aware_datetime,
|
||||
@@ -76,10 +81,10 @@ def _validate_evidence_metadata(metadata: Mapping[str, JsonValue]) -> None:
|
||||
if not isinstance(metadata["language"], str) or not metadata["language"]:
|
||||
raise ValueError("typed Evidence metadata must contain language")
|
||||
provenance = metadata["provenance"]
|
||||
if not isinstance(provenance, dict) or set(provenance) != {
|
||||
"source_file", "source_sha256", "supporting_excerpts",
|
||||
}:
|
||||
raise ValueError("typed Evidence metadata must contain canonical provenance")
|
||||
if not isinstance(provenance, dict):
|
||||
raise ValueError("typed Evidence metadata must contain canonical provenance") # noqa: TRY004
|
||||
model = ManualEvidenceProvenance if provenance.get("kind") == "manual" else EvidenceProvenance
|
||||
model.model_validate(provenance)
|
||||
|
||||
|
||||
class _CanonicalValue(BaseModel):
|
||||
|
||||
@@ -0,0 +1,177 @@
|
||||
"""Curated unit v4: visible Markdown fields are the sole human-content authority."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
|
||||
import yaml
|
||||
|
||||
from .canonical import (
|
||||
_PAYLOAD_TYPE_BY_KIND,
|
||||
CuratedEvidence,
|
||||
ManualEvidenceProvenance,
|
||||
_v2_labels,
|
||||
)
|
||||
|
||||
_LIST_FIELDS = {"synonyms", "variants", "columns", "tables"}
|
||||
|
||||
|
||||
def parse_metadata(text: str) -> dict:
|
||||
class UniqueKeysLoader(yaml.SafeLoader):
|
||||
pass
|
||||
|
||||
def mapping(loader, node):
|
||||
pairs = loader.construct_pairs(node, deep=True)
|
||||
result = {}
|
||||
for key, value in pairs:
|
||||
if key in result:
|
||||
raise ValueError(f"Duplicate metadata key: {key}")
|
||||
result[key] = value
|
||||
return result
|
||||
|
||||
UniqueKeysLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, mapping)
|
||||
try:
|
||||
return yaml.load(text, Loader=UniqueKeysLoader)
|
||||
except (yaml.YAMLError, TypeError) as error:
|
||||
raise ValueError("Evidence metadata is malformed") from error
|
||||
|
||||
|
||||
def _sections(body: str, headings: dict[str, str]) -> dict[str, str]:
|
||||
"""Recognize structural H2s outside code fences; all other Markdown is content."""
|
||||
sections: dict[str, list[str]] = {}
|
||||
field = None
|
||||
fence = None
|
||||
for line in body.splitlines():
|
||||
match = re.match(r"^\s{0,3}(`{3,}|~{3,})", line)
|
||||
if match:
|
||||
marker = match[1]
|
||||
if fence is None:
|
||||
fence = marker
|
||||
elif marker[0] == fence[0] and len(marker) >= len(fence):
|
||||
fence = None
|
||||
heading = headings.get(line[3:]) if line.startswith("## ") and fence is None else None
|
||||
if heading:
|
||||
if heading in sections:
|
||||
raise ValueError(f"Duplicate section: {line[3:]}")
|
||||
field = heading
|
||||
sections[field] = []
|
||||
elif field is not None:
|
||||
sections[field].append(line)
|
||||
elif line.strip():
|
||||
raise ValueError("Content must follow a documented section heading")
|
||||
if fence:
|
||||
raise ValueError("Unclosed Markdown code fence")
|
||||
return {key: "\n".join(lines).strip() for key, lines in sections.items()}
|
||||
|
||||
|
||||
def _list(text: str) -> list[str]:
|
||||
if not text:
|
||||
return []
|
||||
values = []
|
||||
for line in text.splitlines():
|
||||
if not line.startswith("- ") or not line[2:].strip():
|
||||
raise ValueError("List entries must use '- value', one per line")
|
||||
value = line[2:]
|
||||
# Quoted strings preserve multiline and unusual values during migration.
|
||||
values.append(json.loads(value) if value.startswith('"') else value)
|
||||
return values
|
||||
|
||||
|
||||
def _values(text: str) -> dict[str, str]:
|
||||
values = {}
|
||||
key = None
|
||||
lines = []
|
||||
for line in text.splitlines():
|
||||
if line.startswith("### "):
|
||||
if key is not None:
|
||||
values[key] = "\n".join(lines).strip()
|
||||
label = line[4:]
|
||||
key = json.loads(label) if label.startswith('"') else label
|
||||
if key in values:
|
||||
raise ValueError("Duplicate enum value")
|
||||
lines = []
|
||||
elif key is None:
|
||||
if line.strip():
|
||||
raise ValueError("Enum values require '### value' headings")
|
||||
else:
|
||||
lines.append(line)
|
||||
if key is not None:
|
||||
values[key] = "\n".join(lines).strip()
|
||||
return values
|
||||
|
||||
|
||||
def parse_document(metadata: dict, body: str) -> dict:
|
||||
data = dict(metadata)
|
||||
if {"title", "payload", *(_PAYLOAD_TYPE_BY_KIND)}.intersection(data):
|
||||
raise ValueError("Title and payload must be edited only in the Markdown body")
|
||||
lines = body.strip().splitlines()
|
||||
if not lines or not lines[0].startswith("# ") or not lines[0][2:].strip():
|
||||
raise ValueError("A title starting with '# ' is required")
|
||||
data["title"] = lines[0][2:].strip()
|
||||
kind = data.get("kind")
|
||||
if kind not in _PAYLOAD_TYPE_BY_KIND:
|
||||
raise ValueError("Unknown Evidence kind")
|
||||
labels = _v2_labels(str(data.get("language", "")))
|
||||
fields = _PAYLOAD_TYPE_BY_KIND[kind].model_fields
|
||||
sections = _sections("\n".join(lines[1:]), {labels[key]: key for key in fields})
|
||||
required = {name for name, field in fields.items() if field.is_required()}
|
||||
if not required <= sections.keys():
|
||||
raise ValueError(
|
||||
"Missing sections: " + ", ".join(labels[k] for k in sorted(required - sections.keys()))
|
||||
)
|
||||
payload = {}
|
||||
for key, content in sections.items():
|
||||
if key in _LIST_FIELDS:
|
||||
payload[key] = _list(content)
|
||||
elif key == "values":
|
||||
payload[key] = _values(content)
|
||||
elif key == "sql" and content.startswith("```sql\n") and content.endswith("\n```"):
|
||||
payload[key] = content[7:-4]
|
||||
else:
|
||||
payload[key] = content
|
||||
if fields[key].is_required() and not payload[key] and key != "values":
|
||||
raise ValueError(f"Section {labels[key]} must not be empty")
|
||||
data["payload"] = payload
|
||||
data.setdefault(
|
||||
"provenance", ManualEvidenceProvenance(declared_by="local curator").model_dump()
|
||||
)
|
||||
return data
|
||||
|
||||
|
||||
def render_document(value: CuratedEvidence) -> str:
|
||||
metadata = value.model_dump(mode="json", exclude={"title", "payload"})
|
||||
labels = _v2_labels(value.language)
|
||||
parts = [
|
||||
f"---\n{yaml.safe_dump(metadata, allow_unicode=True, sort_keys=False)}---\n\n# {value.title}"
|
||||
]
|
||||
for key, content in value.payload.model_dump(mode="json").items():
|
||||
if key in _LIST_FIELDS:
|
||||
rendered = "\n".join(
|
||||
"- "
|
||||
+ (
|
||||
json.dumps(v, ensure_ascii=False)
|
||||
if "\n" in v or v.startswith('"') or v != v.strip()
|
||||
else v
|
||||
)
|
||||
for v in content
|
||||
)
|
||||
elif key == "values":
|
||||
rendered = "\n\n".join(
|
||||
f"### {json.dumps(k, ensure_ascii=False)}\n\n{v}" for k, v in content.items()
|
||||
)
|
||||
elif key == "sql":
|
||||
rendered = f"```sql\n{content}\n```"
|
||||
else:
|
||||
rendered = content
|
||||
parts.append(f"## {labels[key]}\n\n{rendered}")
|
||||
rendered = "\n\n".join(parts) + "\n"
|
||||
# Migration must fail explicitly rather than silently changing unrepresentable content.
|
||||
restored = CuratedEvidence.model_validate(
|
||||
parse_document(metadata, rendered.split("---\n", 2)[2])
|
||||
)
|
||||
if restored != value:
|
||||
raise ValueError(
|
||||
f"{value.id}: content cannot be represented losslessly in editable Markdown"
|
||||
)
|
||||
return rendered
|
||||
@@ -0,0 +1,235 @@
|
||||
"""Explicit acquisition and durable source comparisons; never implicit runtime refresh."""
|
||||
|
||||
import base64
|
||||
import json
|
||||
|
||||
from .authoring import (
|
||||
RestructureRequest,
|
||||
_allocate_evidence_id,
|
||||
_candidate_to_evidence,
|
||||
normalize_source_text,
|
||||
)
|
||||
from .canonical import (
|
||||
CuratedEvidence,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
)
|
||||
from .corpus.normalize import _decode
|
||||
from .local_archive import ArchiveConflict, LocalEvidenceArchive, _atomic, _digest
|
||||
|
||||
MAX_DOCUMENTS = 200
|
||||
MAX_TOTAL_BYTES = 100 * 1024 * 1024
|
||||
|
||||
|
||||
def _origin(unit):
|
||||
return unit.provenance.original if isinstance(unit.provenance, ManualEvidenceProvenance) else unit.provenance
|
||||
|
||||
|
||||
def _records(archive):
|
||||
path = archive.metadata / "sources.json"
|
||||
if path.is_symlink():
|
||||
raise ValueError("Source metadata must not use symlinks")
|
||||
return json.loads(path.read_text()) if path.exists() else {}
|
||||
|
||||
|
||||
def _write_records(archive, records):
|
||||
_atomic(archive.metadata / "sources.json", json.dumps(records, ensure_ascii=False, sort_keys=True))
|
||||
|
||||
|
||||
def reviews(archive):
|
||||
return [{k: v for k, v in row.items() if k not in {"expected", "text", "writes", "approved"}}
|
||||
for row in _records(archive).values()]
|
||||
|
||||
|
||||
def acquisition_sources(cfg):
|
||||
"""Local drafts and original files, plus configured read-only remote connectors."""
|
||||
from .adapters import FilesystemEvidenceSource
|
||||
from .sources import build_sources
|
||||
|
||||
root = cfg.evidence.local_archive_root / "evidence"
|
||||
# Canonical acquired versions are immutable lineage, never new input documents.
|
||||
patterns = [str(p.relative_to(root)) for folder in ("incoming", "source")
|
||||
for p in sorted((root / folder).rglob("*.md"))
|
||||
if not p.is_relative_to(root / "source/acquired")]
|
||||
result = [FilesystemEvidenceSource(root, patterns=patterns)] if patterns else []
|
||||
# Filesystem descriptors select the installation's local authoring tree after E2.
|
||||
remote = cfg.evidence.model_copy(update={"source_root": None,
|
||||
"sources": [s for s in cfg.evidence.sources if s.type != "filesystem"]})
|
||||
result.extend(build_sources(remote, acquisition=True))
|
||||
return result
|
||||
|
||||
|
||||
def refresh(archive: LocalEvidenceArchive, sources, restructurer):
|
||||
"""Acquire everything successfully before recording proposals. Missing is never deletion."""
|
||||
with archive.operation():
|
||||
state = archive._state()
|
||||
if state.get("pending") or state.get("import_writes"):
|
||||
raise ArchiveConflict("Complete the pending consolidation before refreshing sources")
|
||||
records = _records(archive)
|
||||
if any(r["status"] == "applying" for r in records.values()):
|
||||
raise ArchiveConflict("Retry the pending source decision before refreshing")
|
||||
files = archive._files(archive.evidence)
|
||||
units = archive._units(files, allow_review=True)
|
||||
documents, total = {}, 0
|
||||
for adapter in sources:
|
||||
for item in adapter.discover():
|
||||
document = adapter.acquire(item)
|
||||
total += len(document.content)
|
||||
if len(documents) >= MAX_DOCUMENTS or total > MAX_TOTAL_BYTES:
|
||||
raise ValueError("Source refresh exceeds the local acquisition limit")
|
||||
relative = item.metadata.get("relative_path") if item.uri.startswith("file:") else None
|
||||
identity = f"local:{relative}" if relative else item.uri
|
||||
key = _digest(identity.encode())
|
||||
if key in documents:
|
||||
raise ValueError("Duplicate acquisition identity")
|
||||
documents[key] = (document, relative)
|
||||
reserved = set(units) | set(state["deleted_ids"])
|
||||
for record in records.values():
|
||||
reserved.update(u["id"] for u in record.get("proposed", []))
|
||||
changed, unchanged = 0, 0
|
||||
acquired = {}
|
||||
for key, (document, relative) in documents.items():
|
||||
text = normalize_source_text(_decode(document))
|
||||
sha = "sha256:" + _digest(text.encode())
|
||||
old = records.get(key)
|
||||
stale = old and old["status"] == "review" and any(
|
||||
p not in files or _digest(files[p]) != h for p, h in old["expected"].items())
|
||||
if old and old["sha256"] == sha and not stale:
|
||||
old["availability"] = "available"
|
||||
unchanged += 1
|
||||
continue
|
||||
current = {i: pair for i, pair in units.items()
|
||||
if (old and i in old["unit_ids"]) or
|
||||
(_origin(pair[1]) and _origin(pair[1]).source_file == relative)}
|
||||
# Seed imported E2 document identity without asking the model to recurate unchanged text.
|
||||
if old is None and current and all(_origin(u).source_sha256 == sha for _, u in current.values()):
|
||||
records[key] = {"id": key, "uri": document.source.uri, "legacy_file": relative,
|
||||
"sha256": sha, "unit_ids": sorted(current), "status": "accepted",
|
||||
"availability": "available", "revision": sha[7:], "proposed": []}
|
||||
unchanged += 1
|
||||
continue
|
||||
source_file = f"source/acquired/{key}/{sha[7:]}.md"
|
||||
request = RestructureRequest(source_file=source_file, source_sha256=sha,
|
||||
normalized_text=text, previous_units=tuple(u for _, u in current.values()))
|
||||
proposed = []
|
||||
seen = set()
|
||||
suppressed = relative in state["suppressed_sources"] or any(
|
||||
p.startswith(f"source/acquired/{key}/") for p in state["suppressed_sources"]) or (old and (
|
||||
old.get("suppressed", False) or any(i in state["deleted_ids"] for i in old["unit_ids"])))
|
||||
for candidate in restructurer.restructure(request):
|
||||
identity = candidate.existing_id
|
||||
if identity in state["deleted_ids"]:
|
||||
continue
|
||||
if identity is not None and identity not in current:
|
||||
raise ValueError("Source proposal refers to an unrelated Evidence identity")
|
||||
if identity is None:
|
||||
if suppressed:
|
||||
continue # A model-created identifier cannot bypass a curated deletion.
|
||||
identity = _allocate_evidence_id(candidate.title, reserved)
|
||||
reserved.add(identity)
|
||||
if identity in seen:
|
||||
raise ValueError("Source proposal repeats an Evidence identity")
|
||||
seen.add(identity)
|
||||
unit = _candidate_to_evidence(candidate, identity, source_file, sha)
|
||||
dump_curated_markdown(unit) # Refuse an uneditable proposal before saving any review.
|
||||
if any(excerpt not in text for excerpt in unit.provenance.supporting_excerpts):
|
||||
raise ValueError("Source proposal contains an excerpt absent from the acquired document")
|
||||
proposed.append(unit.model_dump(mode="json"))
|
||||
expected = {p: _digest(files[p]) for p, _ in current.values()}
|
||||
row = {"id": key, "uri": document.source.uri, "legacy_file": relative,
|
||||
"sha256": sha, "source_file": source_file, "text": text,
|
||||
"status": "review", "availability": "available", "suppressed": bool(suppressed),
|
||||
"unit_ids": sorted(current), "expected": expected,
|
||||
"current": [u.model_dump(mode="json") for _, u in current.values()], "proposed": proposed,
|
||||
"removed_ids": sorted(set(current) - seen)}
|
||||
row["revision"] = _digest(json.dumps(row, sort_keys=True).encode())
|
||||
records[key] = row
|
||||
acquired[key] = {"source": document.source.model_dump(mode="json"),
|
||||
"media_type": document.media_type, "raw_base64": base64.b64encode(document.content).decode()}
|
||||
changed += 1
|
||||
for key, row in records.items():
|
||||
if key not in documents:
|
||||
row["availability"] = "missing"
|
||||
# No writes above: an access/model failure preserves every previous review and active unit.
|
||||
for key, value in acquired.items():
|
||||
directory = archive.metadata / "acquisitions" / key
|
||||
if any(p.is_symlink() for p in [directory, directory.parent]):
|
||||
raise ValueError("Acquisition metadata must not use symlinks")
|
||||
_atomic(directory / f"{records[key]['sha256'][7:]}.json", json.dumps(value, ensure_ascii=False))
|
||||
_write_records(archive, records)
|
||||
return {"status": "succeeded", "counts": {"changed": changed, "unchanged": unchanged,
|
||||
"review": sum(r["status"] == "review" for r in records.values())}}
|
||||
|
||||
|
||||
def decide(archive, *, source_id, revision, decision, actor, activate):
|
||||
if decision not in {"keep", "replace"} or not actor.strip():
|
||||
raise ValueError("Choose keep or replace and supply a curator")
|
||||
with archive.operation():
|
||||
records = _records(archive)
|
||||
row = records.get(source_id)
|
||||
if row is None or row["revision"] != revision:
|
||||
raise ArchiveConflict("Source comparison changed; refresh the page")
|
||||
if row["status"] == "applying":
|
||||
if row["decision"] != decision:
|
||||
raise ArchiveConflict("Retry the saved source decision before changing it")
|
||||
state = archive._state()
|
||||
state.update(import_writes=row["writes"], approved_imports=row["approved"])
|
||||
archive._write_state(state)
|
||||
result = archive._consolidate(actor, activate)
|
||||
else:
|
||||
if row["status"] != "review":
|
||||
raise ArchiveConflict("This source comparison was already decided")
|
||||
state = archive._state()
|
||||
if state.get("pending") or state.get("import_writes"):
|
||||
raise ArchiveConflict("Complete the pending consolidation first")
|
||||
files = archive._files(archive.evidence)
|
||||
units = archive._units(files, allow_review=True)
|
||||
if any(_digest(files[p]) != h if p in files else True for p, h in row["expected"].items()):
|
||||
raise ArchiveConflict("Curated files changed since source review; refresh the source comparison")
|
||||
selected = [CuratedEvidence.model_validate(v) for v in row["proposed"]] if decision == "replace" else [units[i][1] for i in row["unit_ids"]]
|
||||
writes, approved = {}, {}
|
||||
def write(path, content):
|
||||
current = archive.evidence / path
|
||||
writes[path] = {"before": _digest(current.read_bytes()) if current.exists() else None, "after": content}
|
||||
if decision == "replace":
|
||||
write(row["source_file"], row["text"])
|
||||
for identity in row["unit_ids"]:
|
||||
write(units[identity][0], None)
|
||||
for value in selected:
|
||||
if value.review_items:
|
||||
raise ValueError("The proposal needs review; correct the input draft and refresh, or keep local content")
|
||||
if decision == "keep":
|
||||
origin = _origin(value)
|
||||
if origin:
|
||||
old = archive._snapshot(state["active"] or state["baseline"])
|
||||
candidate = old / origin.source_file
|
||||
if not candidate.is_file():
|
||||
candidate = archive.evidence / origin.source_file
|
||||
text = normalize_source_text(candidate.read_text())
|
||||
if "sha256:" + _digest(text.encode()) != origin.source_sha256:
|
||||
raise ValueError("The original source version is unavailable")
|
||||
path = f"source/acquired/{source_id}/{origin.source_sha256[7:]}.md"
|
||||
write(path, text)
|
||||
origin = EvidenceProvenance(source_file=path, source_sha256=origin.source_sha256,
|
||||
supporting_excerpts=origin.supporting_excerpts)
|
||||
value = value.model_copy(update={"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)})
|
||||
if value.id in units and value.id not in row["unit_ids"]:
|
||||
raise ArchiveConflict("A proposed Evidence identity was created elsewhere")
|
||||
path = units[value.id][0] if value.id in units and units[value.id][1].kind == value.kind else f"curated/{value.kind}/{value.id[9:]}.md"
|
||||
if path in files and value.id not in units:
|
||||
raise ArchiveConflict("The proposed file path is occupied")
|
||||
write(path, dump_curated_markdown(value))
|
||||
approved[value.id] = _digest(dump_curated_markdown(value).encode())
|
||||
# Journal before applying files; retry never silently clobbers an external edit.
|
||||
row.update(status="applying", decision=decision, decided_by=actor,
|
||||
next_unit_ids=[u.id for u in selected], writes=writes, approved=approved)
|
||||
_write_records(archive, records)
|
||||
state.update(import_writes=writes, approved_imports=approved)
|
||||
archive._write_state(state)
|
||||
result = archive._consolidate(actor, activate)
|
||||
if result["status"] != "active":
|
||||
raise ValueError("Source decision was saved but not activated; retry")
|
||||
row.update(status="accepted" if decision == "replace" else "kept", unit_ids=row["next_unit_ids"])
|
||||
_write_records(archive, records)
|
||||
return {"status": "succeeded", "counts": {"units": result["units"]}}
|
||||
@@ -0,0 +1,472 @@
|
||||
"""Persistent curated files, immutable consolidation candidates and explicit activation.
|
||||
|
||||
The working tree is primary data. Core consumers use only active_snapshot(); an
|
||||
editor save or failed activation never switches that pointer. No Git/network/DWH I/O.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import fcntl
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
from collections.abc import Callable
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
|
||||
from .canonical import (
|
||||
MAX_CURATED_FILE_BYTES,
|
||||
CuratedEvidence,
|
||||
EvidenceProvenance,
|
||||
ManualEvidenceProvenance,
|
||||
dump_curated_markdown,
|
||||
parse_curated_markdown,
|
||||
)
|
||||
|
||||
|
||||
class ArchiveConflict(ValueError):
|
||||
"""The curator must reconcile a concurrent change before replacing it."""
|
||||
|
||||
|
||||
def _digest(value: bytes) -> str:
|
||||
return hashlib.sha256(value).hexdigest()
|
||||
|
||||
|
||||
def _content(unit: CuratedEvidence) -> str:
|
||||
return _digest(
|
||||
json.dumps(
|
||||
unit.model_dump(mode="json", exclude={"schema_version", "provenance"}),
|
||||
sort_keys=True,
|
||||
ensure_ascii=False,
|
||||
).encode()
|
||||
)
|
||||
|
||||
|
||||
def _atomic(path: Path, data: str) -> None:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
fd, temporary = tempfile.mkstemp(prefix=".write-", dir=path.parent)
|
||||
try:
|
||||
with os.fdopen(fd, "w") as handle:
|
||||
handle.write(data)
|
||||
handle.flush()
|
||||
os.fsync(handle.fileno())
|
||||
os.replace(temporary, path)
|
||||
finally:
|
||||
if os.path.exists(temporary):
|
||||
os.unlink(temporary)
|
||||
|
||||
|
||||
class LocalEvidenceArchive:
|
||||
def __init__(self, workspace_root: Path):
|
||||
self.root = workspace_root.resolve()
|
||||
self.evidence = self.root / "evidence"
|
||||
self.metadata = self.evidence / ".local"
|
||||
if self.evidence.is_symlink() or self.metadata.is_symlink():
|
||||
raise ValueError("The local Evidence archive must use persistent regular directories")
|
||||
|
||||
@contextmanager
|
||||
def operation(self):
|
||||
if any(
|
||||
path.is_symlink()
|
||||
for path in (
|
||||
self.evidence,
|
||||
self.metadata,
|
||||
self.metadata / "snapshots",
|
||||
self.metadata / "state.yaml",
|
||||
)
|
||||
):
|
||||
raise ValueError("Evidence archive metadata must not use symlinks")
|
||||
self.root.mkdir(parents=True, exist_ok=True)
|
||||
lock = self.root / ".evidence-archive.lock"
|
||||
if lock.is_symlink():
|
||||
raise ValueError("Evidence lock must not be a symlink")
|
||||
with lock.open("a") as handle:
|
||||
fcntl.flock(handle, fcntl.LOCK_EX)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
fcntl.flock(handle, fcntl.LOCK_UN)
|
||||
|
||||
def _state(self):
|
||||
path = self.metadata / "state.yaml"
|
||||
if not path.exists():
|
||||
return {
|
||||
"schema_version": 1,
|
||||
"active": None,
|
||||
"pending": None,
|
||||
"baseline": None,
|
||||
"deleted_ids": [],
|
||||
"suppressed_sources": [],
|
||||
}
|
||||
value = yaml.safe_load(path.read_text())
|
||||
if not isinstance(value, dict) or value.get("schema_version") != 1:
|
||||
raise ValueError("Unsupported local Evidence archive state")
|
||||
return value
|
||||
|
||||
def _write_state(self, state):
|
||||
_atomic(self.metadata / "state.yaml", yaml.safe_dump(state, sort_keys=True))
|
||||
|
||||
def _snapshot(self, revision: str) -> Path:
|
||||
if (
|
||||
not isinstance(revision, str)
|
||||
or len(revision) != 64
|
||||
or any(c not in "0123456789abcdef" for c in revision)
|
||||
):
|
||||
raise ValueError("Invalid Evidence snapshot identity")
|
||||
path = self.metadata / "snapshots" / revision
|
||||
if not path.is_dir() or path.is_symlink():
|
||||
raise ValueError("Consolidated Evidence snapshot is missing")
|
||||
return path
|
||||
|
||||
def active_snapshot(self) -> Path | None:
|
||||
"""Only this immutable source is eligible for core consumption."""
|
||||
with self.operation():
|
||||
revision = self._state()["active"]
|
||||
return self._snapshot(revision) if revision else None
|
||||
|
||||
def validate(self):
|
||||
"""Check the working files without modifying them or changing active content."""
|
||||
with self.operation():
|
||||
units = self._units(self._files(self.evidence))
|
||||
state = self._state()
|
||||
revision = state["pending"] or state["active"] or state.get("baseline")
|
||||
previous = (
|
||||
self._units(self._files(self._snapshot(revision)), allow_review=True)
|
||||
if revision
|
||||
else {}
|
||||
)
|
||||
for identity, (_, unit) in units.items():
|
||||
old = previous.get(identity)
|
||||
if isinstance(unit.provenance, EvidenceProvenance) and (
|
||||
old is None or _content(old[1]) == _content(unit)
|
||||
):
|
||||
self._validate_document_source(unit.provenance)
|
||||
return len(units)
|
||||
|
||||
def _files(self, root: Path):
|
||||
result = {}
|
||||
curated = root / "curated"
|
||||
if curated.is_symlink():
|
||||
raise ValueError("Curated directory must not be a symlink")
|
||||
if not curated.is_dir():
|
||||
raise ValueError(f"{curated}: curated archive is unavailable; absence is not deletion")
|
||||
for path in sorted(curated.rglob("*")):
|
||||
if path.is_symlink():
|
||||
raise ValueError(f"{path}: symlinks are not supported in the curated archive")
|
||||
if path.suffix != ".md" or path.name.upper().startswith("README"):
|
||||
continue
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{path}: Evidence file exceeds the size limit")
|
||||
result[path.relative_to(root).as_posix()] = path.read_bytes()
|
||||
return result
|
||||
|
||||
def _units(self, files, *, allow_review=False):
|
||||
units = {}
|
||||
for relative, data in files.items():
|
||||
try:
|
||||
unit = parse_curated_markdown(data.decode("utf-8"), path=Path(relative))
|
||||
if unit.schema_version != 4:
|
||||
raise ValueError("Run evidence migrate before consolidating legacy units")
|
||||
if unit.id in units:
|
||||
raise ValueError(f"Duplicate Evidence identity {unit.id}")
|
||||
if unit.review_items and not allow_review:
|
||||
raise ValueError("Resolve review items before consolidation")
|
||||
units[unit.id] = (relative, unit)
|
||||
except ValueError as error:
|
||||
raise ValueError(f"{relative}: {error}") from error
|
||||
return units
|
||||
|
||||
def initialize(self):
|
||||
"""Capture migrated/refined content before edits, without activating review items."""
|
||||
with self.operation():
|
||||
state = self._state()
|
||||
if state.get("baseline") or state["active"] or state["pending"]:
|
||||
return
|
||||
files = self._files(self.evidence)
|
||||
self._units(files, allow_review=True)
|
||||
source = self.evidence / "source"
|
||||
if source.is_symlink():
|
||||
raise ValueError("Source symlinks are not supported")
|
||||
if source.exists():
|
||||
for path in sorted(source.rglob("*")):
|
||||
if path.is_symlink():
|
||||
raise ValueError("Source symlinks are not supported")
|
||||
if path.is_file():
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{path}: source exceeds the size limit")
|
||||
files[path.relative_to(self.evidence).as_posix()] = path.read_bytes()
|
||||
revision = _digest(b"".join(k.encode() + b"\0" + v for k, v in sorted(files.items())))
|
||||
snapshots = self.metadata / "snapshots"
|
||||
snapshots.mkdir(parents=True, exist_ok=True)
|
||||
destination = snapshots / revision
|
||||
if not destination.exists():
|
||||
with tempfile.TemporaryDirectory(prefix=".baseline-", dir=snapshots) as tmp:
|
||||
candidate = Path(tmp) / "snapshot"
|
||||
(candidate / "curated").mkdir(parents=True)
|
||||
for relative, data in files.items():
|
||||
path = candidate / relative
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(data)
|
||||
os.replace(candidate, destination)
|
||||
state["baseline"] = revision
|
||||
self._write_state(state)
|
||||
|
||||
def consolidate(self, *, actor: str, activate: Callable[[Path], None] | None = None):
|
||||
"""Persist one valid candidate; optionally activate it through the existing index stage.
|
||||
|
||||
A missing callback deliberately reports pending_activation, never success.
|
||||
A retry of unchanged files reuses the same candidate after an index failure.
|
||||
"""
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
return self._consolidate(actor, activate)
|
||||
|
||||
def _consolidate(self, actor, activate):
|
||||
state = self._state()
|
||||
self._finish_import(state)
|
||||
self._finish_normalization(state)
|
||||
original_files = self._files(self.evidence)
|
||||
units = self._units(original_files)
|
||||
previous_revision = state["pending"] or state["active"] or state.get("baseline")
|
||||
previous_root = self._snapshot(previous_revision) if previous_revision else None
|
||||
previous = (
|
||||
self._units(self._files(previous_root), allow_review=True) if previous_root else {}
|
||||
)
|
||||
deleted = set(state["deleted_ids"]) | (previous.keys() - units.keys())
|
||||
suppressed = set(state["suppressed_sources"])
|
||||
for identity in previous.keys() - units.keys():
|
||||
provenance = previous[identity][1].provenance
|
||||
source = (
|
||||
provenance.original
|
||||
if isinstance(provenance, ManualEvidenceProvenance)
|
||||
else provenance
|
||||
)
|
||||
if source:
|
||||
suppressed.add(source.source_file)
|
||||
files = {}
|
||||
for identity, (relative, unit) in units.items():
|
||||
old = previous.get(identity)
|
||||
changed = old is not None and _content(old[1]) != _content(unit)
|
||||
approved = state.get("approved_imports", {}).get(identity) == _digest(
|
||||
dump_curated_markdown(unit).encode()
|
||||
)
|
||||
if approved:
|
||||
pass # An explicit source decision authorized this exact content and provenance.
|
||||
elif changed or (
|
||||
isinstance(unit.provenance, ManualEvidenceProvenance)
|
||||
and (old is None or unit.provenance.declared_by == "local curator")
|
||||
):
|
||||
origin = old[1].provenance if old else unit.provenance
|
||||
origin = origin.original if isinstance(origin, ManualEvidenceProvenance) else origin
|
||||
unit = unit.model_copy(
|
||||
update={
|
||||
"provenance": ManualEvidenceProvenance(declared_by=actor, original=origin)
|
||||
}
|
||||
)
|
||||
elif old and unit.provenance != old[1].provenance:
|
||||
raise ArchiveConflict(
|
||||
f"{relative}: provenance is managed; edit the content instead"
|
||||
)
|
||||
if isinstance(unit.provenance, EvidenceProvenance):
|
||||
self._validate_document_source(unit.provenance)
|
||||
files[relative] = dump_curated_markdown(unit).encode()
|
||||
origin = (
|
||||
unit.provenance.original
|
||||
if isinstance(unit.provenance, ManualEvidenceProvenance)
|
||||
else unit.provenance
|
||||
)
|
||||
if origin:
|
||||
candidates = [previous_root / origin.source_file] if previous_root else []
|
||||
candidates.append(self.evidence / origin.source_file)
|
||||
from .authoring import normalize_source_text
|
||||
|
||||
for source in candidates:
|
||||
if source.is_file() and not source.is_symlink():
|
||||
if source.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{source}: source exceeds the size limit")
|
||||
raw = source.read_bytes()
|
||||
if (
|
||||
"sha256:" + _digest(normalize_source_text(raw.decode()).encode())
|
||||
== origin.source_sha256
|
||||
):
|
||||
if origin.source_file in files and files[origin.source_file] != raw:
|
||||
raise ArchiveConflict(
|
||||
"Different source revisions require explicit source resolution"
|
||||
)
|
||||
files[origin.source_file] = raw
|
||||
break
|
||||
else:
|
||||
raise ValueError(
|
||||
f"{origin.source_file}: the recorded original document is unavailable"
|
||||
)
|
||||
# Record current declarations and original document lineage distinctly.
|
||||
manifest = {
|
||||
"schema_version": 1,
|
||||
"units": {
|
||||
identity: {
|
||||
"file": relative,
|
||||
"content_hash": _content(parse_curated_markdown(files[relative].decode())),
|
||||
"provenance": parse_curated_markdown(
|
||||
files[relative].decode()
|
||||
).provenance.model_dump(mode="json"),
|
||||
}
|
||||
for identity, (relative, _) in sorted(units.items())
|
||||
},
|
||||
"deleted_ids": sorted(deleted),
|
||||
"suppressed_sources": sorted(suppressed),
|
||||
}
|
||||
files["local-manifest.yaml"] = yaml.safe_dump(
|
||||
manifest, allow_unicode=True, sort_keys=True
|
||||
).encode()
|
||||
revision = _digest(
|
||||
b"".join(path.encode() + b"\0" + data + b"\0" for path, data in sorted(files.items()))
|
||||
)
|
||||
snapshots = self.metadata / "snapshots"
|
||||
snapshots.mkdir(parents=True, exist_ok=True)
|
||||
destination = snapshots / revision
|
||||
if not destination.exists():
|
||||
with tempfile.TemporaryDirectory(prefix=".candidate-", dir=snapshots) as tmp:
|
||||
candidate = Path(tmp) / "snapshot"
|
||||
(candidate / "curated").mkdir(parents=True)
|
||||
for relative, data in files.items():
|
||||
path = candidate / relative
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_bytes(data)
|
||||
os.replace(candidate, destination)
|
||||
if self._files(self.evidence) != original_files:
|
||||
raise ArchiveConflict(
|
||||
"Evidence files changed during consolidation; retry with the current files"
|
||||
)
|
||||
# The candidate exists first. Pending state makes every subsequent interruption recoverable.
|
||||
state.update(
|
||||
pending=revision,
|
||||
deleted_ids=sorted(deleted),
|
||||
suppressed_sources=sorted(suppressed),
|
||||
normalization={relative: _digest(data) for relative, data in original_files.items()},
|
||||
)
|
||||
state.pop("approved_imports", None)
|
||||
self._write_state(state)
|
||||
self._finish_normalization(state)
|
||||
if activate is None:
|
||||
return {
|
||||
"status": "pending_activation",
|
||||
"revision": revision,
|
||||
"snapshot": str(destination),
|
||||
}
|
||||
activate(destination)
|
||||
state.update(active=revision, pending=None)
|
||||
self._write_state(state)
|
||||
return {
|
||||
"status": "active",
|
||||
"revision": revision,
|
||||
"snapshot": str(destination),
|
||||
"units": len(units),
|
||||
"deleted": len(previous.keys() - units.keys()),
|
||||
}
|
||||
|
||||
def _finish_import(self, state):
|
||||
"""Replay an explicit source decision, rejecting intervening external edits."""
|
||||
writes = state.get("import_writes")
|
||||
if writes is None:
|
||||
return
|
||||
for relative, change in writes.items():
|
||||
path = self.evidence / relative
|
||||
if not relative.startswith(("curated/", "source/acquired/")) or ".." in Path(relative).parts:
|
||||
raise ValueError("Invalid import destination")
|
||||
if any(p.is_symlink() for p in [path, *path.parents] if p != self.root.parent):
|
||||
raise ValueError("Import destinations must not use symlinks")
|
||||
current = _digest(path.read_bytes()) if path.exists() else None
|
||||
after = _digest(change["after"].encode()) if change["after"] is not None else None
|
||||
if current not in (change["before"], after):
|
||||
raise ArchiveConflict("Evidence changed during a source decision; restore or review the file")
|
||||
for relative, change in writes.items():
|
||||
path = self.evidence / relative
|
||||
if change["after"] is None:
|
||||
path.unlink(missing_ok=True)
|
||||
else:
|
||||
_atomic(path, change["after"])
|
||||
del state["import_writes"]
|
||||
self._write_state(state)
|
||||
|
||||
def _finish_normalization(self, state):
|
||||
"""Replay interrupted managed writes only where the user's bytes are unchanged."""
|
||||
if "normalization" not in state:
|
||||
return
|
||||
snapshot = self._snapshot(state["pending"])
|
||||
current = self._files(self.evidence)
|
||||
for relative, original_hash in state["normalization"].items():
|
||||
if relative in current and _digest(current[relative]) == original_hash:
|
||||
_atomic(self.evidence / relative, (snapshot / relative).read_text())
|
||||
_atomic(
|
||||
self.evidence / "local-manifest.yaml", (snapshot / "local-manifest.yaml").read_text()
|
||||
)
|
||||
del state["normalization"]
|
||||
self._write_state(state)
|
||||
|
||||
def _validate_document_source(self, provenance):
|
||||
from .authoring import _normalize, normalize_source_text
|
||||
|
||||
path = self.evidence / provenance.source_file
|
||||
if (
|
||||
path.is_symlink()
|
||||
or not path.is_file()
|
||||
or not path.resolve().is_relative_to(self.evidence.resolve())
|
||||
):
|
||||
raise ValueError(f"{provenance.source_file}: source document is unavailable")
|
||||
if path.stat().st_size > MAX_CURATED_FILE_BYTES:
|
||||
raise ValueError(f"{provenance.source_file}: source exceeds the size limit")
|
||||
source = normalize_source_text(path.read_text())
|
||||
if "sha256:" + _digest(source.encode()) != provenance.source_sha256:
|
||||
raise ArchiveConflict(
|
||||
f"{provenance.source_file}: source changed; explicit source refresh is required"
|
||||
)
|
||||
if any(_normalize(excerpt) not in source for excerpt in provenance.supporting_excerpts):
|
||||
raise ValueError(f"{provenance.source_file}: supporting excerpt is missing")
|
||||
|
||||
def save(self, value: CuratedEvidence, *, expected_revision: str | None, actor: str):
|
||||
"""Explicit workflow correction with optimistic concurrency; activation is a separate boundary."""
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
return self._save(value, expected_revision=expected_revision, actor=actor)
|
||||
|
||||
def _save(self, value, *, expected_revision, actor):
|
||||
"""Save while the caller holds operation(), including workflow receipt recovery."""
|
||||
self._finish_normalization(self._state())
|
||||
units = self._units(self._files(self.evidence))
|
||||
existing = units.get(value.id)
|
||||
if (existing is None) != (expected_revision is None):
|
||||
raise ArchiveConflict("Evidence was created or removed since review")
|
||||
if existing and _content(existing[1]) != expected_revision:
|
||||
raise ArchiveConflict("Evidence changed since review")
|
||||
relative = (
|
||||
existing[0]
|
||||
if existing
|
||||
else f"curated/{value.kind}/{value.id.removeprefix('evidence:')}.md"
|
||||
)
|
||||
if existing and existing[1].kind != value.kind:
|
||||
raise ValueError("An update cannot change the Evidence kind")
|
||||
if existing:
|
||||
value = value.model_copy(update={"provenance": existing[1].provenance})
|
||||
_atomic(self.evidence / relative, dump_curated_markdown(value))
|
||||
return self._consolidate(actor, None)
|
||||
|
||||
def get(self, identity: str):
|
||||
with self.operation():
|
||||
relative, unit = self._units(self._files(self.evidence))[identity]
|
||||
return {"unit": unit, "revision": _content(unit), "path": str(self.evidence / relative)}
|
||||
|
||||
def remove(self, identity: str, *, expected_revision: str, actor: str):
|
||||
if not actor.strip():
|
||||
raise ValueError("A curator identity is required")
|
||||
with self.operation():
|
||||
self._finish_normalization(self._state())
|
||||
relative, unit = self._units(self._files(self.evidence))[identity]
|
||||
if _content(unit) != expected_revision:
|
||||
raise ArchiveConflict("Evidence changed since review")
|
||||
(self.evidence / relative).unlink()
|
||||
return self._consolidate(actor, None)
|
||||
@@ -12,10 +12,18 @@ if TYPE_CHECKING:
|
||||
from tht.config import EvidenceSourcesConfig
|
||||
|
||||
|
||||
def build_sources(evidence: EvidenceSourcesConfig | None) -> list[EvidenceSource]:
|
||||
def build_sources(evidence: EvidenceSourcesConfig | None, *, acquisition: bool = False) -> list[EvidenceSource]:
|
||||
"""Build configured Evidence adapters in the existing deterministic order."""
|
||||
if evidence is None:
|
||||
return []
|
||||
if evidence.local_archive_root is not None and not acquisition:
|
||||
from tht.evidence.local_archive import LocalEvidenceArchive
|
||||
archive = LocalEvidenceArchive(evidence.local_archive_root)
|
||||
if (archive.metadata / "state.yaml").exists():
|
||||
snapshot = archive.active_snapshot()
|
||||
if snapshot is None:
|
||||
raise ValueError("Local Evidence has not been activated; run Evidence consolidation")
|
||||
return [FilesystemEvidenceSource(snapshot, patterns=("curated/**/*.md",))]
|
||||
sources: list[EvidenceSource] = []
|
||||
if evidence.source_root is not None:
|
||||
legacy_root = evidence.source_root / evidence.evidence_dir
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
"""Apply only removals confirmed by a successful Catalog physical synchronization."""
|
||||
|
||||
import json
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from sqlalchemy import text
|
||||
|
||||
from .models import MemoryConflict
|
||||
from .review import digest
|
||||
|
||||
|
||||
class RemovedColumn(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
table: str = Field(min_length=1, max_length=200)
|
||||
column: str = Field(min_length=1, max_length=200)
|
||||
|
||||
|
||||
class CleanupRequest(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
sync_id: str = Field(min_length=1, max_length=200)
|
||||
database: str = Field(min_length=1, max_length=200)
|
||||
schema_name: str = Field(min_length=1, max_length=200)
|
||||
removed_tables: list[str] = Field(default_factory=list, max_length=10000)
|
||||
removed_columns: list[RemovedColumn] = Field(default_factory=list, max_length=100000)
|
||||
|
||||
|
||||
def cleanup(service, request: CleanupRequest):
|
||||
service._admin()
|
||||
fingerprint = digest(request.model_dump(mode="json"))
|
||||
with service.repository.operation() as repo:
|
||||
with repo.transaction() as connection:
|
||||
params = {"w": repo.workspace_id, "sync": request.sync_id}
|
||||
receipt = (
|
||||
connection.execute(
|
||||
text(
|
||||
"SELECT request_hash,card_ids "
|
||||
"FROM thoth_memory.cleanup_receipts WHERE workspace_id=:w AND sync_id=:sync"
|
||||
),
|
||||
params,
|
||||
)
|
||||
.mappings()
|
||||
.first()
|
||||
)
|
||||
if receipt:
|
||||
if receipt["request_hash"] != fingerprint:
|
||||
raise MemoryConflict(
|
||||
"Physical cleanup identity was reused with different removals"
|
||||
)
|
||||
identities = receipt["card_ids"]
|
||||
else:
|
||||
identities = list(
|
||||
connection.execute(
|
||||
text(
|
||||
"SELECT DISTINCT card_id "
|
||||
"FROM thoth_memory.dependencies d WHERE workspace_id=:w "
|
||||
"AND database_id=:db AND schema_name=:schema AND (table_name=ANY(:tables) "
|
||||
"OR EXISTS (SELECT 1 FROM jsonb_to_recordset(CAST(:columns AS jsonb)) "
|
||||
'AS removed("table" text, "column" text) WHERE '
|
||||
'removed."table"=d.table_name AND removed."column"=d.column_name))'
|
||||
),
|
||||
{
|
||||
**params,
|
||||
"db": request.database,
|
||||
"schema": request.schema_name,
|
||||
"tables": request.removed_tables,
|
||||
"columns": json.dumps(
|
||||
[c.model_dump() for c in request.removed_columns]
|
||||
),
|
||||
},
|
||||
).scalars()
|
||||
)
|
||||
for identity in identities:
|
||||
repo.delete(identity)
|
||||
connection.execute(
|
||||
text(
|
||||
"INSERT INTO thoth_memory.cleanup_receipts "
|
||||
"VALUES (:w,:sync,:hash,CAST(:ids AS jsonb))"
|
||||
),
|
||||
{**params, "hash": fingerprint, "ids": json.dumps(identities)},
|
||||
)
|
||||
results = [service._propagate(repo, identity) for identity in identities]
|
||||
return {"deleted": len(identities), "indexed": all(r["indexed"] for r in results)}
|
||||
@@ -25,7 +25,7 @@ class MemoryRecord(BaseModel):
|
||||
concepts: list[str] = []
|
||||
|
||||
|
||||
_MEM_ID_RE = re.compile(r"\bmem-\d{4,}\b")
|
||||
_MEM_ID_RE = re.compile(r"\bmem-(?:[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}|\d{4,})\b")
|
||||
|
||||
|
||||
def decided_memory_ids(decisions: list[DecisionRecord]) -> set[str]:
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
"""Versioned Memory migration pack, run only by installation preparation."""
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
from functools import lru_cache
|
||||
from importlib.resources import files
|
||||
from pathlib import Path
|
||||
|
||||
from sqlalchemy import URL, create_engine, text
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
|
||||
def installation_url(*, migrator: bool = False) -> str:
|
||||
role = "MIGRATOR" if migrator else "RUNTIME"
|
||||
direct = os.environ.get(f"THT_CATALOG_{role}_DATABASE_URL")
|
||||
if not migrator:
|
||||
direct = direct or os.environ.get("THT_CATALOG_DATABASE_URL")
|
||||
if direct:
|
||||
return direct.replace("postgresql://", "postgresql+psycopg2://", 1)
|
||||
prefix = "THT_CATALOG_"
|
||||
try:
|
||||
password = Path(os.environ[prefix + role + "_PASSWORD_FILE"]).read_text().strip()
|
||||
url = URL.create(
|
||||
"postgresql+psycopg2", host=os.environ[prefix + "DB_HOST"],
|
||||
port=int(os.environ.get(prefix + "DB_PORT", "5432")),
|
||||
database=os.environ[prefix + "DB_NAME"],
|
||||
username=os.environ[prefix + role + "_USER"], password=password,
|
||||
)
|
||||
return url.render_as_string(hide_password=False)
|
||||
except (KeyError, OSError, ValueError):
|
||||
raise ValueError("Memory PostgreSQL installation configuration is unavailable") from None
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def expected_migrations() -> dict[str, str]:
|
||||
return {p.name: hashlib.sha256(p.read_text().encode()).hexdigest()
|
||||
for p in files("tht").joinpath("migrations/memory").iterdir()
|
||||
if p.name.endswith(".sql")}
|
||||
|
||||
|
||||
def migrate(database_url: str) -> None:
|
||||
engine = create_engine(database_url, poolclass=NullPool)
|
||||
try:
|
||||
with engine.begin() as connection:
|
||||
connection.execute(text("SELECT pg_advisory_xact_lock(792114203)"))
|
||||
connection.execute(text("CREATE SCHEMA IF NOT EXISTS thoth_memory"))
|
||||
connection.execute(text("CREATE TABLE IF NOT EXISTS thoth_memory.migrations "
|
||||
"(version text PRIMARY KEY, checksum text NOT NULL)"))
|
||||
applied = dict(connection.execute(text(
|
||||
"SELECT version, checksum FROM thoth_memory.migrations"
|
||||
)).all())
|
||||
pack = sorted(files("tht").joinpath("migrations/memory").iterdir(), key=lambda p: p.name)
|
||||
known = {p.name for p in pack if p.name.endswith(".sql")}
|
||||
if set(applied) - known:
|
||||
raise ValueError("Memory schema is newer than this application")
|
||||
for path in pack:
|
||||
if path.name not in known:
|
||||
continue
|
||||
sql = path.read_text()
|
||||
digest = hashlib.sha256(sql.encode()).hexdigest()
|
||||
if path.name in applied:
|
||||
if applied[path.name] != digest:
|
||||
raise ValueError("Memory migration checksum mismatch")
|
||||
continue
|
||||
connection.execute(text(sql))
|
||||
connection.execute(text("INSERT INTO thoth_memory.migrations VALUES (:v, :c)"),
|
||||
{"v": path.name, "c": digest})
|
||||
finally:
|
||||
engine.dispose()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
migrate(installation_url(migrator=True))
|
||||
print("Memory migrations: ready")
|
||||
@@ -0,0 +1,112 @@
|
||||
"""Authoritative Memory contracts, independent of workflow decision kinds."""
|
||||
|
||||
from datetime import datetime
|
||||
from typing import Literal, Self
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
|
||||
Family = Literal["domain_clarification", "sql_rule", "solved_question", "explained_error"]
|
||||
|
||||
|
||||
class Dependency(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
database: str = Field(min_length=1, max_length=200)
|
||||
schema_name: str = Field(default="", max_length=200)
|
||||
table: str = Field(default="", max_length=200)
|
||||
column: str = Field(default="", max_length=200)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def structured(self) -> Self:
|
||||
if self.column and not self.table:
|
||||
raise ValueError("A column dependency requires a table")
|
||||
if self.table and not self.schema_name:
|
||||
raise ValueError("A table dependency requires a schema")
|
||||
return self
|
||||
|
||||
|
||||
class LinkInput(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
target_id: str = Field(min_length=1, max_length=100)
|
||||
meaning: str = Field(min_length=1, max_length=1000)
|
||||
|
||||
|
||||
class CardInput(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
family: Family
|
||||
subject: str = Field(min_length=1, max_length=1000)
|
||||
detail: str = Field(default="", max_length=50000)
|
||||
scope: str = Field(min_length=1, max_length=10000)
|
||||
rationale: str = Field(default="", max_length=10000)
|
||||
question: str = Field(default="", max_length=10000)
|
||||
sql: str = Field(default="", max_length=100000)
|
||||
concepts: list[str] = Field(default_factory=list, max_length=100)
|
||||
dependencies: list[Dependency] = Field(default_factory=list, max_length=200)
|
||||
links: list[LinkInput] = Field(default_factory=list, max_length=200)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def valid_family(self) -> Self:
|
||||
if self.family == "solved_question" and (not self.question or not self.sql):
|
||||
raise ValueError("A solved question requires its question and approved SQL")
|
||||
if self.family == "explained_error" and (not self.detail or not self.rationale):
|
||||
raise ValueError("An explained error requires a correction and rationale")
|
||||
if self.family != "solved_question" and self.sql:
|
||||
raise ValueError("Only solved questions carry exemplar SQL")
|
||||
if any(not c.strip() or len(c) > 200 for c in self.concepts):
|
||||
raise ValueError("Concepts must contain between 1 and 200 characters")
|
||||
if len({link.target_id for link in self.links}) != len(self.links):
|
||||
raise ValueError("Each linked destination must be unique")
|
||||
return self
|
||||
|
||||
|
||||
class Card(CardInput):
|
||||
id: str
|
||||
workspace_id: str
|
||||
origin: Literal["manual", "workflow"]
|
||||
session_id: str | None = None
|
||||
decision_seq: int | None = None
|
||||
created_at: datetime
|
||||
updated_at: datetime
|
||||
revision: str
|
||||
indexed: bool = False
|
||||
|
||||
|
||||
class CardQuery(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
q: str = Field(default="", max_length=1000)
|
||||
family: Family | None = None
|
||||
concept: str = Field(default="", max_length=200)
|
||||
database: str = Field(default="", max_length=200)
|
||||
table: str = Field(default="", max_length=200)
|
||||
column: str = Field(default="", max_length=200)
|
||||
origin: Literal["manual", "workflow"] | None = None
|
||||
updated_after: datetime | None = None
|
||||
updated_before: datetime | None = None
|
||||
page: int = Field(default=1, ge=1)
|
||||
page_size: int = Field(default=25, ge=1, le=100)
|
||||
sort: Literal["updated_at", "created_at", "subject", "family"] = "updated_at"
|
||||
direction: Literal["asc", "desc"] = "desc"
|
||||
|
||||
|
||||
class MemoryError(Exception):
|
||||
code = "memory_operation_failed"
|
||||
status = 500
|
||||
|
||||
|
||||
class MemoryUnavailable(MemoryError):
|
||||
code = "memory_unavailable"
|
||||
status = 503
|
||||
|
||||
|
||||
class MemoryNotFound(MemoryError):
|
||||
code = "memory_not_found"
|
||||
status = 404
|
||||
|
||||
|
||||
class MemoryForbidden(MemoryError):
|
||||
code = "memory_forbidden"
|
||||
status = 403
|
||||
|
||||
|
||||
class MemoryConflict(MemoryError):
|
||||
code = "memory_conflict"
|
||||
status = 409
|
||||
@@ -0,0 +1,245 @@
|
||||
"""Workspace-scoped PostgreSQL persistence and durable projection work."""
|
||||
|
||||
import json
|
||||
from contextlib import contextmanager
|
||||
from uuid import uuid4
|
||||
|
||||
from sqlalchemy import create_engine, text
|
||||
from sqlalchemy.exc import IntegrityError, SQLAlchemyError
|
||||
from sqlalchemy.pool import NullPool
|
||||
|
||||
from .migrate import expected_migrations
|
||||
from .models import Card, CardInput, CardQuery, MemoryConflict, MemoryNotFound, MemoryUnavailable
|
||||
|
||||
|
||||
class MemoryRepository:
|
||||
def __init__(self, database_url: str, workspace_id: str, *, engine=None, connection=None):
|
||||
self.workspace_id = workspace_id
|
||||
self.engine = engine or create_engine(
|
||||
database_url, poolclass=NullPool, connect_args={"connect_timeout": 5},
|
||||
)
|
||||
self.connection = connection
|
||||
|
||||
def close(self):
|
||||
self.engine.dispose()
|
||||
|
||||
@contextmanager
|
||||
def transaction(self):
|
||||
connection = self.connection
|
||||
try:
|
||||
connection = connection or self.engine.connect()
|
||||
with (connection.begin_nested() if connection.in_transaction() else connection.begin()):
|
||||
connection.execute(text("SET LOCAL ROLE thoth_memory_runtime"))
|
||||
connection.execute(text("SELECT set_config('thoth.memory_workspace', :w, true)"),
|
||||
{"w": self.workspace_id})
|
||||
connection.execute(text("SET LOCAL statement_timeout = '15s'"))
|
||||
installed = dict(connection.execute(text(
|
||||
"SELECT version, checksum FROM thoth_memory.migrations"
|
||||
)).all())
|
||||
if installed != expected_migrations():
|
||||
raise MemoryUnavailable("Memory schema is incompatible; run installation migrations")
|
||||
yield connection
|
||||
except IntegrityError:
|
||||
raise MemoryConflict("Memory references conflict with the current archive") from None
|
||||
except SQLAlchemyError:
|
||||
raise MemoryUnavailable("Memory archive is unavailable; check its migrations and access") \
|
||||
from None
|
||||
finally:
|
||||
if self.connection is None and connection is not None:
|
||||
connection.close()
|
||||
|
||||
@contextmanager
|
||||
def operation(self):
|
||||
"""Serialize each workspace across SQL commits and the bounded vector call."""
|
||||
try:
|
||||
connection = self.engine.connect()
|
||||
connection.execute(text("SET statement_timeout = '15s'"))
|
||||
connection.execute(text("SELECT pg_advisory_lock(hashtextextended(:w, 792114204))"),
|
||||
{"w": self.workspace_id})
|
||||
connection.commit()
|
||||
except SQLAlchemyError:
|
||||
if 'connection' in locals():
|
||||
connection.close()
|
||||
raise MemoryUnavailable("Memory archive is busy or unavailable") from None
|
||||
try:
|
||||
yield MemoryRepository("", self.workspace_id, engine=self.engine, connection=connection)
|
||||
finally:
|
||||
# NullPool closes the physical connection, releasing the session advisory lock.
|
||||
connection.close()
|
||||
|
||||
def _card(self, connection, row) -> Card:
|
||||
params = {"w": self.workspace_id, "id": row["id"]}
|
||||
links = connection.execute(text(
|
||||
"SELECT target_id, meaning FROM thoth_memory.links "
|
||||
"WHERE workspace_id=:w AND source_id=:id ORDER BY target_id"
|
||||
), params).mappings().all()
|
||||
dependencies = connection.execute(text(
|
||||
'SELECT database_id AS database, schema_name, table_name AS "table", '
|
||||
'column_name AS "column" FROM thoth_memory.dependencies '
|
||||
"WHERE workspace_id=:w AND card_id=:id "
|
||||
"ORDER BY database_id, schema_name, table_name, column_name"
|
||||
), params).mappings().all()
|
||||
return Card.model_validate({
|
||||
**row["data"], "id": row["id"], "workspace_id": self.workspace_id,
|
||||
"family": row["family"], "subject": row["subject"], "origin": row["origin"],
|
||||
"created_at": row["created_at"], "updated_at": row["updated_at"],
|
||||
"revision": row["revision"], "indexed": row["indexed"],
|
||||
"links": [dict(v) for v in links], "dependencies": [dict(v) for v in dependencies],
|
||||
})
|
||||
|
||||
@staticmethod
|
||||
def _selection():
|
||||
return ("SELECT c.*, (p.revision=c.revision AND NOT p.pending "
|
||||
"AND p.action='upsert' AND p.format=2) AS indexed FROM thoth_memory.cards c "
|
||||
"JOIN thoth_memory.projections p ON p.workspace_id=c.workspace_id "
|
||||
"AND p.card_id=c.id ")
|
||||
|
||||
def get(self, card_id: str) -> Card:
|
||||
with self.transaction() as c:
|
||||
row = c.execute(text(self._selection()+"WHERE c.workspace_id=:w AND c.id=:id"),
|
||||
{"w": self.workspace_id, "id": card_id}).mappings().first()
|
||||
if row is None:
|
||||
raise MemoryNotFound("Memory card was not found in this workspace")
|
||||
return self._card(c, row)
|
||||
|
||||
def list(self, query: CardQuery) -> dict:
|
||||
conditions = ["c.workspace_id=:w"]
|
||||
params = {"w": self.workspace_id}
|
||||
if query.q:
|
||||
conditions.append("(c.id ILIKE :q OR c.subject ILIKE :q OR c.data::text ILIKE :q)")
|
||||
params["q"] = "%" + query.q.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_") + "%"
|
||||
for key in ("family", "origin"):
|
||||
if value := getattr(query, key):
|
||||
conditions.append(f"c.{key}=:{key}")
|
||||
params[key] = value
|
||||
if query.concept:
|
||||
conditions.append("c.data->'concepts' @> CAST(:concept AS jsonb)")
|
||||
params["concept"] = json.dumps([query.concept])
|
||||
refs = []
|
||||
for key, column in [("database", "database_id"), ("table", "table_name"),
|
||||
("column", "column_name")]:
|
||||
if value := getattr(query, key):
|
||||
refs.append(f"d.{column}=:{key}")
|
||||
params[key] = value
|
||||
if refs:
|
||||
conditions.append("EXISTS (SELECT 1 FROM thoth_memory.dependencies d WHERE "
|
||||
"d.workspace_id=c.workspace_id AND d.card_id=c.id AND "
|
||||
+ " AND ".join(refs) + ")")
|
||||
for key, comparison in [("updated_after", ">="), ("updated_before", "<=")]:
|
||||
if value := getattr(query, key):
|
||||
conditions.append(f"c.updated_at {comparison} :{key}")
|
||||
params[key] = value
|
||||
where = " WHERE " + " AND ".join(conditions)
|
||||
with self.transaction() as c:
|
||||
total = c.execute(text("SELECT count(*) FROM thoth_memory.cards c" + where),
|
||||
params).scalar_one()
|
||||
rows = c.execute(text(self._selection() + where
|
||||
+ f" ORDER BY c.{query.sort} {query.direction}, c.id ASC LIMIT :limit OFFSET :offset"),
|
||||
{**params, "limit": query.page_size, "offset": (query.page - 1) * query.page_size},
|
||||
).mappings().all()
|
||||
return {"items": [self._card(c, row).model_dump(mode="json") for row in rows],
|
||||
"total": total, "page": query.page, "page_size": query.page_size}
|
||||
|
||||
def exact_match(self, value: CardInput) -> Card | None:
|
||||
"""Match authored content only; provenance and index state do not create new knowledge."""
|
||||
data = value.model_dump(mode="json", exclude={"dependencies", "links"})
|
||||
with self.transaction() as c:
|
||||
rows = c.execute(text(self._selection() +
|
||||
"WHERE c.workspace_id=:w AND c.family=:family AND c.subject=:subject "
|
||||
"AND c.data - 'session_id' - 'decision_seq'=CAST(:data AS jsonb) ORDER BY c.id"),
|
||||
{"w": self.workspace_id, "family": value.family, "subject": value.subject,
|
||||
"data": json.dumps(data)}).mappings()
|
||||
for row in rows:
|
||||
candidate = self._card(c, row)
|
||||
def ordered(values):
|
||||
return sorted(json.dumps(v.model_dump(), sort_keys=True) for v in values)
|
||||
if (ordered(candidate.dependencies) == ordered(value.dependencies)
|
||||
and ordered(candidate.links) == ordered(value.links)):
|
||||
return candidate
|
||||
return None
|
||||
|
||||
def source(self, source_key: str):
|
||||
with self.transaction() as c:
|
||||
row = c.execute(text("SELECT card_id, action FROM thoth_memory.projections "
|
||||
"WHERE workspace_id=:w AND source_key=:s"),
|
||||
{"w": self.workspace_id, "s": source_key}).mappings().first()
|
||||
return dict(row) if row else None
|
||||
|
||||
def save(self, value: CardInput, *, card_id: str | None = None, source_key: str | None = None,
|
||||
session_id: str | None = None, decision_seq: int | None = None,
|
||||
new_id: str | None = None) -> str:
|
||||
creating = card_id is None
|
||||
card_id = card_id or new_id or "mem-" + str(uuid4())
|
||||
revision = str(uuid4())
|
||||
data = value.model_dump(mode="json", exclude={"links", "dependencies"})
|
||||
with self.transaction() as c:
|
||||
if not creating:
|
||||
old = c.execute(text("SELECT data FROM thoth_memory.cards "
|
||||
"WHERE workspace_id=:w AND id=:id"),
|
||||
{"w": self.workspace_id, "id": card_id}).scalar_one_or_none()
|
||||
if old is None:
|
||||
raise MemoryNotFound("Memory card was not found in this workspace")
|
||||
data.update({k: old.get(k) for k in ("session_id", "decision_seq")})
|
||||
else:
|
||||
if c.execute(text("SELECT 1 FROM thoth_memory.projections "
|
||||
"WHERE workspace_id=:w AND card_id=:id"),
|
||||
{"w": self.workspace_id, "id": card_id}).first():
|
||||
raise MemoryConflict("A proposed card identity was already used")
|
||||
data.update(session_id=session_id, decision_seq=decision_seq)
|
||||
params = {"w": self.workspace_id, "id": card_id, "r": revision,
|
||||
"data": json.dumps(data), "family": value.family, "subject": value.subject,
|
||||
"origin": "workflow" if source_key else "manual", "source": source_key}
|
||||
c.execute(text("INSERT INTO thoth_memory.cards "
|
||||
"(workspace_id,id,family,subject,origin,data,revision) "
|
||||
"VALUES (:w,:id,:family,:subject,:origin,CAST(:data AS jsonb),:r) "
|
||||
"ON CONFLICT (workspace_id,id) DO UPDATE SET family=EXCLUDED.family, "
|
||||
"subject=EXCLUDED.subject,data=EXCLUDED.data,revision=EXCLUDED.revision, "
|
||||
"updated_at=clock_timestamp()"), params)
|
||||
c.execute(text("DELETE FROM thoth_memory.links WHERE workspace_id=:w AND source_id=:id"), params)
|
||||
for link in value.links:
|
||||
c.execute(text("INSERT INTO thoth_memory.links VALUES (:w,:id,:target,:meaning)"),
|
||||
{**params, "target": link.target_id, "meaning": link.meaning})
|
||||
c.execute(text("DELETE FROM thoth_memory.dependencies "
|
||||
"WHERE workspace_id=:w AND card_id=:id"), params)
|
||||
for dep in {tuple(d.model_dump().values()) for d in value.dependencies}:
|
||||
c.execute(text("INSERT INTO thoth_memory.dependencies VALUES "
|
||||
"(:w,:id,:database,:schema,:table,:column)"),
|
||||
{**params, **dict(zip(("database", "schema", "table", "column"), dep))})
|
||||
c.execute(text("INSERT INTO thoth_memory.projections "
|
||||
"(workspace_id,card_id,revision,action,source_key) VALUES (:w,:id,:r,'upsert',:source) "
|
||||
"ON CONFLICT (workspace_id,card_id) DO UPDATE SET revision=EXCLUDED.revision, "
|
||||
"action='upsert',pending=true,error=NULL,updated_at=clock_timestamp()"), params)
|
||||
return card_id
|
||||
|
||||
def delete(self, card_id: str):
|
||||
with self.transaction() as c:
|
||||
params = {"w": self.workspace_id, "id": card_id, "r": str(uuid4())}
|
||||
deleted = c.execute(text("DELETE FROM thoth_memory.cards "
|
||||
"WHERE workspace_id=:w AND id=:id"), params).rowcount
|
||||
if not deleted:
|
||||
raise MemoryNotFound("Memory card was not found in this workspace")
|
||||
c.execute(text("UPDATE thoth_memory.projections SET action='delete',revision=:r,"
|
||||
"pending=true,error=NULL,updated_at=clock_timestamp() "
|
||||
"WHERE workspace_id=:w AND card_id=:id"), params)
|
||||
|
||||
def projections(self, *, pending: bool = True):
|
||||
with self.transaction() as c:
|
||||
needs_update = "(pending OR (action='upsert' AND format<>2))"
|
||||
rows = c.execute(text("SELECT card_id,revision,action,"+needs_update+" AS pending,"
|
||||
"error,updated_at "
|
||||
"FROM thoth_memory.projections WHERE workspace_id=:w "
|
||||
+ ("AND "+needs_update+" " if pending else "") + "ORDER BY updated_at,card_id"),
|
||||
{"w": self.workspace_id}).mappings().all()
|
||||
return [dict(row) for row in rows]
|
||||
|
||||
def projection_result(self, card_id: str, revision: str, error: str | None):
|
||||
with self.transaction() as c:
|
||||
c.execute(text("UPDATE thoth_memory.projections SET pending=:p,error=:error,format=2,"
|
||||
"updated_at=clock_timestamp() WHERE workspace_id=:w AND card_id=:id AND revision=:r"),
|
||||
{"p": error is not None, "error": error, "w": self.workspace_id,
|
||||
"id": card_id, "r": revision})
|
||||
|
||||
def invalidate_all(self):
|
||||
with self.transaction() as c:
|
||||
c.execute(text("UPDATE thoth_memory.projections SET pending=true,error=NULL "
|
||||
"WHERE workspace_id=:w"), {"w": self.workspace_id})
|
||||
@@ -0,0 +1,131 @@
|
||||
"""Bounded graph recall over current Memory cards; no model or approval side effects."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
||||
|
||||
from .models import Card, Family, MemoryNotFound
|
||||
|
||||
PROJECTION_FORMAT = 2
|
||||
MAX_SEEDS = 100
|
||||
MAX_DEPTH = 2
|
||||
MAX_LINKS_PER_CARD = 20
|
||||
MAX_VISITED = 200
|
||||
MAX_EDGES = 400
|
||||
RRF_K = 60
|
||||
|
||||
|
||||
class RecallScope(BaseModel):
|
||||
"""Exact business scope/concepts and hierarchical physical context.
|
||||
|
||||
Cards without dependencies are workspace-wide. A dependency applies to its
|
||||
database and every descendant of the schema/table/column it names.
|
||||
All physical fields must match the SAME dependency, never separate entries.
|
||||
"""
|
||||
|
||||
model_config = ConfigDict(extra="forbid", str_strip_whitespace=True)
|
||||
scope: str = Field(default="", max_length=10000)
|
||||
database: str = Field(default="", max_length=200)
|
||||
schema_name: str = Field(default="", max_length=200)
|
||||
table: str = Field(default="", max_length=200)
|
||||
column: str = Field(default="", max_length=200)
|
||||
concepts: list[str] = Field(default_factory=list, max_length=100)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def validate_context(self):
|
||||
if ((self.schema_name and not self.database) or (self.table and not self.schema_name)
|
||||
or (self.column and not self.table)):
|
||||
raise ValueError("Physical recall scope requires its database/schema/table ancestors")
|
||||
if any(not value.strip() or len(value) > 200 for value in self.concepts):
|
||||
raise ValueError("Recall concepts must contain between 1 and 200 characters")
|
||||
return self
|
||||
|
||||
def matches(self, card: Card) -> bool:
|
||||
if self.scope and self.scope != card.scope:
|
||||
return False
|
||||
if not set(self.concepts) <= set(card.concepts):
|
||||
return False
|
||||
if not self.database or not card.dependencies:
|
||||
return True
|
||||
return any(all(not getattr(self, key) or getattr(dep, key) in ("", getattr(self, key))
|
||||
for key in ("database", "schema_name", "table", "column"))
|
||||
for dep in card.dependencies)
|
||||
|
||||
def vector_filter(self, family: Family | None) -> dict:
|
||||
return {"memory": {**self.model_dump(), "family": family,
|
||||
"format": PROJECTION_FORMAT}}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecalledCard:
|
||||
card: Card
|
||||
score: float
|
||||
path: tuple[str, ...]
|
||||
|
||||
|
||||
def expand_and_rank(repo, hits, *, scope: RecallScope, family: Family | None,
|
||||
excluded: set[str], top: int) -> list[RecalledCard]:
|
||||
"""RRF direct rank + strongest link path, decayed by 0.5 per outgoing hop.
|
||||
|
||||
Roots, nodes, fan-out and depth are all bounded. Repeated paths do not add
|
||||
votes: cycles and highly connected cards cannot amplify their own relevance.
|
||||
The caller holds the workspace operation lock while resolving authority.
|
||||
"""
|
||||
cache: dict[str, Card | None] = {}
|
||||
|
||||
def current(identity):
|
||||
if identity not in cache:
|
||||
if len(cache) >= MAX_VISITED:
|
||||
return None
|
||||
try:
|
||||
card = repo.get(identity)
|
||||
except MemoryNotFound:
|
||||
card = None
|
||||
if card is not None and (not card.indexed or card.id in excluded
|
||||
or (family and card.family != family) or not scope.matches(card)):
|
||||
card = None
|
||||
cache[identity] = card
|
||||
return cache[identity]
|
||||
|
||||
direct: dict[str, float] = {}
|
||||
graph: dict[str, tuple[float, tuple[str, ...]]] = {}
|
||||
seeds = []
|
||||
for rank, hit in enumerate(hits[:MAX_SEEDS], 1):
|
||||
card = current(hit.ref)
|
||||
if (card is None or hit.metadata.get("memory_revision") != card.revision
|
||||
or hit.metadata.get("memory_format") != PROJECTION_FORMAT
|
||||
or card.id in direct):
|
||||
continue
|
||||
direct[card.id] = 1 / (RRF_K + rank)
|
||||
seeds.append(card)
|
||||
|
||||
traversed = 0
|
||||
for seed in seeds:
|
||||
frontier = [(seed, (seed.id,))]
|
||||
visited = {seed.id}
|
||||
for depth in range(1, MAX_DEPTH + 1):
|
||||
next_frontier = []
|
||||
for source, path in frontier:
|
||||
for link in sorted(source.links, key=lambda link: link.target_id)[:MAX_LINKS_PER_CARD]:
|
||||
if traversed >= MAX_EDGES:
|
||||
break
|
||||
traversed += 1
|
||||
if link.target_id in visited:
|
||||
continue
|
||||
visited.add(link.target_id)
|
||||
target = current(link.target_id)
|
||||
if target is None:
|
||||
continue
|
||||
target_path = (*path, target.id)
|
||||
score = direct[seed.id] * 0.5 ** depth
|
||||
previous = graph.get(target.id)
|
||||
if previous is None or (-score, target_path) < (-previous[0], previous[1]):
|
||||
graph[target.id] = (score, target_path)
|
||||
next_frontier.append((target, target_path))
|
||||
frontier = next_frontier
|
||||
|
||||
ranked = []
|
||||
for identity in direct.keys() | graph.keys():
|
||||
graph_score, path = graph.get(identity, (0, (identity,)))
|
||||
ranked.append(RecalledCard(cache[identity], direct.get(identity, 0) + graph_score, path))
|
||||
return sorted(ranked, key=lambda result: (-result.score, result.card.id))[:top]
|
||||
@@ -0,0 +1,313 @@
|
||||
"""Reviewer-edited additions/updates, grounded in effective approved session decisions."""
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from uuid import NAMESPACE_URL, uuid5
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
from sqlalchemy import text
|
||||
|
||||
from tht.phase import current_phase, effective_decisions
|
||||
|
||||
from .models import CardInput, MemoryConflict
|
||||
from .solved import _build_solved_snapshot
|
||||
|
||||
APPROVED_SOURCES = {
|
||||
"concept_clarified",
|
||||
"join_modified",
|
||||
"column_corrected",
|
||||
"concept_formula_approved",
|
||||
"cte_corrected",
|
||||
"cte_approved",
|
||||
"sql_approved",
|
||||
}
|
||||
|
||||
|
||||
class Proposal(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
id: str = Field(pattern=r"^[a-zA-Z0-9_-]{1,80}$")
|
||||
source_seqs: list[int] = Field(min_length=1, max_length=20)
|
||||
card: CardInput
|
||||
target_id: str | None = None
|
||||
target_revision: str | None = None
|
||||
reason: str = Field(min_length=1, max_length=10000)
|
||||
|
||||
|
||||
class Selection(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
id: str
|
||||
card: CardInput
|
||||
|
||||
|
||||
class ReviewResponse(BaseModel):
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
summary_id: str
|
||||
items: list[Selection] = Field(max_length=20)
|
||||
|
||||
|
||||
def digest(value):
|
||||
return hashlib.sha256(
|
||||
json.dumps(value, sort_keys=True, ensure_ascii=False).encode()
|
||||
).hexdigest()
|
||||
|
||||
|
||||
def context_hash(snapshot):
|
||||
return digest(
|
||||
{
|
||||
"decisions": [
|
||||
d.model_dump(mode="json")
|
||||
for d in effective_decisions(snapshot)
|
||||
if d.type != "memory_summary_reviewed"
|
||||
],
|
||||
"proposals": snapshot.artifacts.get("memory_proposals"),
|
||||
"sql": snapshot.artifacts.get("sql_final"),
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def solved_snapshot(snapshot):
|
||||
linking = json.loads(snapshot.artifacts.get("schema_linking", "{}"))
|
||||
tables = {
|
||||
c["name"]
|
||||
for c in linking.get("candidates", [])
|
||||
if c.get("kind") == "table" and c.get("decision") == "promoted"
|
||||
}
|
||||
return _build_solved_snapshot(snapshot, tables)
|
||||
|
||||
|
||||
def validate_proposals(snapshot, raw):
|
||||
if not isinstance(raw, list) or len(raw) > 20:
|
||||
raise ValueError("Memory proposals must be a list of at most 20 cards")
|
||||
proposals = [Proposal.model_validate(p) for p in raw]
|
||||
effective = {d.seq: d for d in effective_decisions(snapshot)}
|
||||
if len({p.id for p in proposals}) != len(proposals):
|
||||
raise ValueError("Memory proposal identities must be unique")
|
||||
for p in proposals:
|
||||
sources = [effective.get(seq) for seq in p.source_seqs]
|
||||
if any(d is None or d.type not in APPROVED_SOURCES for d in sources):
|
||||
raise MemoryConflict("Memory proposals require effective approved source decisions")
|
||||
if p.card.family == "explained_error" and not any(d.rationale.strip() for d in sources):
|
||||
raise MemoryConflict("Explained errors require an approved explanation")
|
||||
if p.card.family == "solved_question":
|
||||
solved = solved_snapshot(snapshot)
|
||||
if p.card.sql != solved.metadata["sql"]:
|
||||
raise MemoryConflict("Exemplar SQL must match the current approved solution")
|
||||
if bool(p.target_id) != bool(p.target_revision):
|
||||
raise ValueError("Updates require the identity and revision of the card being replaced")
|
||||
return proposals
|
||||
|
||||
|
||||
def prepare(service, snapshot):
|
||||
service._session(snapshot)
|
||||
if current_phase(snapshot) != 8 or snapshot.manifest.status in {"finalized", "archived"}:
|
||||
raise MemoryConflict("The Memory summary is reviewed at the end of F8")
|
||||
with service.repository.transaction() as connection:
|
||||
receipt = (
|
||||
connection.execute(
|
||||
text(
|
||||
"SELECT summary_id,result FROM thoth_memory.reviews "
|
||||
"WHERE workspace_id=:w AND session_id=:s AND result->>'context_hash'=:h "
|
||||
"ORDER BY created_at DESC LIMIT 1"
|
||||
),
|
||||
{
|
||||
"w": service.repository.workspace_id,
|
||||
"s": snapshot.manifest.id,
|
||||
"h": context_hash(snapshot),
|
||||
},
|
||||
)
|
||||
.mappings()
|
||||
.first()
|
||||
)
|
||||
if receipt:
|
||||
return {
|
||||
"reviewed": True,
|
||||
"summary_id": receipt["summary_id"],
|
||||
"saved": len(receipt["result"]["saved"]),
|
||||
}
|
||||
raw = json.loads(snapshot.artifacts.get("memory_proposals", "[]"))
|
||||
proposals = validate_proposals(snapshot, raw)
|
||||
covered = {seq for p in proposals for seq in p.source_seqs}
|
||||
for d in service.promotions(snapshot):
|
||||
if d["decision_seq"] not in covered:
|
||||
proposals.append(
|
||||
Proposal(
|
||||
id=f"decision-{d['decision_seq']}",
|
||||
source_seqs=[d["decision_seq"]],
|
||||
reason="Reusable domain clarification",
|
||||
card=CardInput(
|
||||
family="domain_clarification",
|
||||
subject=d["subject"],
|
||||
detail=d["detail"],
|
||||
rationale=d["rationale"],
|
||||
question=d["question_context"],
|
||||
scope=service.repository.workspace_id,
|
||||
concepts=[d["subject"]],
|
||||
),
|
||||
)
|
||||
)
|
||||
if not any(p.card.family == "solved_question" for p in proposals):
|
||||
solved = solved_snapshot(snapshot)
|
||||
approved = [d for d in effective_decisions(snapshot) if d.type == "sql_approved"][-1]
|
||||
proposals.append(
|
||||
Proposal(
|
||||
id="solved-question",
|
||||
source_seqs=[approved.seq],
|
||||
reason="Approved solution, for consultation in future questions",
|
||||
card=CardInput(
|
||||
family="solved_question",
|
||||
subject=solved.title,
|
||||
question=solved.content,
|
||||
sql=solved.metadata["sql"],
|
||||
scope=service.repository.workspace_id,
|
||||
dependencies=[
|
||||
{
|
||||
"database": snapshot.manifest.database,
|
||||
"schema_name": snapshot.manifest.db_schema,
|
||||
"table": table,
|
||||
}
|
||||
for table in solved.metadata["tables"]
|
||||
],
|
||||
),
|
||||
)
|
||||
)
|
||||
if len(proposals) > 20:
|
||||
raise MemoryConflict("Reduce the final Memory summary to at most 20 cards")
|
||||
items = []
|
||||
targets = set()
|
||||
content = {}
|
||||
aliases = {}
|
||||
for p in proposals:
|
||||
before = None
|
||||
if p.target_id:
|
||||
old = service.repository.get(p.target_id)
|
||||
if old.revision != p.target_revision:
|
||||
raise MemoryConflict(
|
||||
"A Memory card changed; refresh the proposed update before review"
|
||||
)
|
||||
if p.target_id in targets:
|
||||
raise MemoryConflict("Propose only one update to each Memory card")
|
||||
targets.add(p.target_id)
|
||||
before = old.model_dump(mode="json")
|
||||
key = digest(p.card.model_dump(mode="json"))
|
||||
duplicate = service.repository.exact_match(p.card)
|
||||
if duplicate and (not p.target_id or duplicate.id == p.target_id):
|
||||
aliases["proposal:" + p.id] = duplicate.id
|
||||
continue
|
||||
if not p.target_id and key in content:
|
||||
aliases["proposal:" + p.id] = "proposal:" + content[key]
|
||||
continue
|
||||
content[key] = p.id
|
||||
items.append({**p.model_dump(mode="json"), "before": before})
|
||||
for item in items:
|
||||
for link in item["card"]["links"]:
|
||||
link["target_id"] = aliases.get(link["target_id"], link["target_id"])
|
||||
return {
|
||||
"summary_id": digest(
|
||||
{
|
||||
"items": items,
|
||||
"decisions": [d.model_dump(mode="json") for d in effective_decisions(snapshot)],
|
||||
}
|
||||
),
|
||||
"items": items,
|
||||
}
|
||||
|
||||
|
||||
def apply(service, snapshot, response: ReviewResponse):
|
||||
service._session(snapshot)
|
||||
request_hash = digest(response.model_dump(mode="json"))
|
||||
session_id = snapshot.manifest.id
|
||||
with service.repository.operation() as repo:
|
||||
with repo.transaction() as connection:
|
||||
params = {"w": repo.workspace_id, "s": session_id, "id": response.summary_id}
|
||||
receipt = (
|
||||
connection.execute(
|
||||
text(
|
||||
"SELECT request_hash,result FROM thoth_memory.reviews "
|
||||
"WHERE workspace_id=:w AND session_id=:s AND summary_id=:id"
|
||||
),
|
||||
params,
|
||||
)
|
||||
.mappings()
|
||||
.first()
|
||||
)
|
||||
if receipt:
|
||||
if receipt["request_hash"] != request_hash:
|
||||
raise MemoryConflict(
|
||||
"This summary was already reviewed with different selections"
|
||||
)
|
||||
saved = receipt["result"]["saved"]
|
||||
else:
|
||||
# Resolve the same locked repository for preview and optimistic update checks.
|
||||
original = service.repository
|
||||
service.repository = repo
|
||||
try:
|
||||
summary = prepare(service, snapshot)
|
||||
finally:
|
||||
service.repository = original
|
||||
if summary.get("reviewed") or summary["summary_id"] != response.summary_id:
|
||||
raise MemoryConflict("The Memory summary changed; review it again")
|
||||
choices = {choice.id: choice for choice in response.items}
|
||||
candidates = {p["id"]: p for p in summary["items"]}
|
||||
if len(choices) != len(response.items) or choices.keys() - candidates.keys():
|
||||
raise ValueError("Memory review contains duplicate or unknown choices")
|
||||
selected = validate_proposals(
|
||||
snapshot,
|
||||
[
|
||||
{
|
||||
**{k: v for k, v in candidates[key].items() if k != "before"},
|
||||
"card": choice.card.model_dump(mode="json"),
|
||||
}
|
||||
for key, choice in choices.items()
|
||||
],
|
||||
)
|
||||
identities = {
|
||||
p.id: p.target_id
|
||||
or "mem-"
|
||||
+ str(uuid5(NAMESPACE_URL, f"thothii:{repo.workspace_id}:{session_id}:{p.id}"))
|
||||
for p in selected
|
||||
}
|
||||
saved = []
|
||||
for p in selected:
|
||||
# Create/update every selected card before inserting links among new cards.
|
||||
identity = repo.save(
|
||||
p.card.model_copy(update={"links": []}),
|
||||
card_id=p.target_id,
|
||||
new_id=identities[p.id],
|
||||
source_key=f"review:{session_id}:{p.id}",
|
||||
session_id=session_id,
|
||||
decision_seq=p.source_seqs[0],
|
||||
)
|
||||
saved.append({"id": identity, "proposal_id": p.id})
|
||||
for p in selected:
|
||||
value = p.card.model_dump(mode="json")
|
||||
for link in value["links"]:
|
||||
if link["target_id"].startswith("proposal:"):
|
||||
target = link["target_id"].removeprefix("proposal:")
|
||||
if target not in identities:
|
||||
raise ValueError("Select the linked card or remove its link")
|
||||
link["target_id"] = identities[target]
|
||||
repo.save(CardInput.model_validate(value), card_id=identities[p.id])
|
||||
connection.execute(
|
||||
text(
|
||||
"INSERT INTO thoth_memory.reviews "
|
||||
"(workspace_id,session_id,summary_id,request_hash,result) "
|
||||
"VALUES (:w,:s,:id,:hash,CAST(:result AS jsonb))"
|
||||
),
|
||||
{
|
||||
**params,
|
||||
"hash": request_hash,
|
||||
"result": json.dumps(
|
||||
{
|
||||
"saved": saved,
|
||||
"context_hash": context_hash(snapshot),
|
||||
}
|
||||
),
|
||||
},
|
||||
)
|
||||
results = [service._propagate(repo, item["id"]) for item in saved]
|
||||
return {
|
||||
"saved": len(saved),
|
||||
"declined": len(summary["items"]) - len(saved) if not receipt else None,
|
||||
"indexed": all(r["indexed"] for r in results),
|
||||
"results": results,
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
"""Installation binding for Memory; vector dependencies are opened only when needed."""
|
||||
|
||||
import os
|
||||
|
||||
from tht.session.models import PrincipalContext
|
||||
|
||||
from .migrate import installation_url
|
||||
from .models import MemoryForbidden, MemoryUnavailable
|
||||
from .repository import MemoryRepository
|
||||
from .service import MemoryService
|
||||
|
||||
|
||||
def _principal():
|
||||
issuer = os.environ.get("THT_PRINCIPAL_ISSUER", "").strip()
|
||||
subject = os.environ.get("THT_PRINCIPAL_SUBJECT", "").strip()
|
||||
if not issuer or not subject:
|
||||
raise MemoryForbidden("A trusted runtime principal is required for Memory")
|
||||
return PrincipalContext(
|
||||
issuer=issuer, subject=subject,
|
||||
is_admin=os.environ.get("THT_PRINCIPAL_IS_ADMIN", "").lower() in {"1", "true"},
|
||||
)
|
||||
|
||||
|
||||
def _repository(workspace_id):
|
||||
try:
|
||||
url = installation_url()
|
||||
except ValueError:
|
||||
raise MemoryUnavailable("Memory PostgreSQL installation configuration is unavailable") \
|
||||
from None
|
||||
return MemoryRepository(url, workspace_id)
|
||||
|
||||
|
||||
def memory_service(cfg):
|
||||
from tht.adapters.factory import build_vector_store
|
||||
from tht.cli.vector_cmd import make_embedder
|
||||
|
||||
principal = _principal()
|
||||
return MemoryService(_repository(cfg._workspace_id), principal,
|
||||
language=cfg.language,
|
||||
store_factory=lambda: build_vector_store(cfg, require_write=True),
|
||||
embedder_factory=lambda: make_embedder(cfg.embeddings))
|
||||
|
||||
|
||||
def admin_service(workspace_id, runtime):
|
||||
"""Admin access needs no DWH binding, active session or Evidence materialization."""
|
||||
from tht.adapters.vector.qdrant import QdrantVectorStore
|
||||
from tht.config import EmbeddingsConfig
|
||||
from tht.vectorstore.embeddings import OllamaEmbeddings
|
||||
|
||||
principal = _principal()
|
||||
if not principal.is_admin:
|
||||
raise MemoryForbidden("Memory administration requires an administrator")
|
||||
return MemoryService(_repository(workspace_id), principal,
|
||||
language=runtime.get("memoryLanguage", "en"),
|
||||
store_factory=lambda: QdrantVectorStore(
|
||||
base_url=runtime["internalQdrantUrl"], workspace_id=workspace_id,
|
||||
collections={"reference": workspace_id+"-reference", "memory": workspace_id+"-memory"},
|
||||
expected_dimension=runtime["internalEmbeddingDimensions"],
|
||||
),
|
||||
embedder_factory=lambda: OllamaEmbeddings(EmbeddingsConfig(
|
||||
base_url=runtime["internalEmbeddingUrl"], model=runtime["internalEmbeddingModel"],
|
||||
dimensions=runtime["internalEmbeddingDimensions"], timeout=30,
|
||||
)))
|
||||
@@ -0,0 +1,240 @@
|
||||
"""Memory operations: SQL authority, explicit projection recovery, verified recall."""
|
||||
|
||||
import hashlib
|
||||
|
||||
from tht.phase import effective_decisions
|
||||
from tht.ports.vector import VectorWriteRecord
|
||||
from tht.session.models import PrincipalContext
|
||||
from tht.vectorstore.records import VectorRecord
|
||||
|
||||
from .core import decided_memory_ids, declined_promotion_seqs, question_context
|
||||
from .models import (
|
||||
CardInput,
|
||||
CardQuery,
|
||||
Dependency,
|
||||
MemoryConflict,
|
||||
MemoryForbidden,
|
||||
MemoryNotFound,
|
||||
)
|
||||
from .repository import MemoryRepository
|
||||
from .retrieval import MAX_SEEDS, PROJECTION_FORMAT, RecallScope, expand_and_rank
|
||||
from .solved import _build_solved_snapshot
|
||||
|
||||
|
||||
class MemoryService:
|
||||
def __init__(self, repository: MemoryRepository, principal: PrincipalContext,
|
||||
*, store_factory, embedder_factory, language: str = "en"):
|
||||
self.repository = repository
|
||||
self.principal = principal
|
||||
self.store_factory = store_factory
|
||||
self.embedder_factory = embedder_factory
|
||||
if language not in {"en", "it"}:
|
||||
raise ValueError("Memory language must be en or it")
|
||||
self.query_language = {"en": "english", "it": "italian"}[language]
|
||||
|
||||
def close(self):
|
||||
self.repository.close()
|
||||
|
||||
def _admin(self):
|
||||
if not self.principal.is_admin:
|
||||
raise MemoryForbidden("Memory administration requires an administrator")
|
||||
|
||||
def list(self, query: CardQuery):
|
||||
self._admin()
|
||||
return self.repository.list(query)
|
||||
|
||||
def get(self, card_id: str):
|
||||
self._admin()
|
||||
return self.repository.get(card_id).model_dump(mode="json")
|
||||
|
||||
def pending(self):
|
||||
self._admin()
|
||||
return self.repository.projections()
|
||||
|
||||
def save(self, value: CardInput, card_id: str | None = None):
|
||||
self._admin()
|
||||
with self.repository.operation() as repo:
|
||||
card_id = repo.save(value, card_id=card_id)
|
||||
return self._propagate(repo, card_id)
|
||||
|
||||
def delete(self, card_id: str):
|
||||
self._admin()
|
||||
with self.repository.operation() as repo:
|
||||
repo.delete(card_id)
|
||||
return self._propagate(repo, card_id)
|
||||
|
||||
def retry(self, card_id: str):
|
||||
self._admin()
|
||||
with self.repository.operation() as repo:
|
||||
return self._propagate(repo, card_id)
|
||||
|
||||
def _propagate(self, repo, card_id):
|
||||
operation = next((p for p in repo.projections(pending=False)
|
||||
if p["card_id"] == card_id), None)
|
||||
if operation is None:
|
||||
raise MemoryNotFound("Memory operation was not found in this workspace")
|
||||
error = None
|
||||
if operation["pending"]:
|
||||
try:
|
||||
store = self.store_factory()
|
||||
if operation["action"] == "delete":
|
||||
store.delete_memory_records([f"card:{card_id}"])
|
||||
else:
|
||||
card = repo.get(card_id)
|
||||
kind = "solved_question" if card.family == "solved_question" else "memory"
|
||||
content = "\n".join(filter(None, [card.subject, card.detail, card.scope,
|
||||
card.rationale, card.question, card.sql,
|
||||
" ".join(card.concepts),
|
||||
"\n".join(".".join(filter(None, [d.database,
|
||||
d.schema_name, d.table, d.column]))
|
||||
for d in card.dependencies)]))
|
||||
record = VectorRecord(
|
||||
id=f"card:{card.id}", kind=kind, ref=card.id, title=card.subject,
|
||||
content=content, metadata={"memory_revision": card.revision,
|
||||
"memory_format": PROJECTION_FORMAT, "memory_family": card.family,
|
||||
"memory_scope": card.scope, "memory_concepts": card.concepts,
|
||||
"memory_dependencies": [d.model_dump() for d in card.dependencies]},
|
||||
)
|
||||
embedding = self.embedder_factory().embed_documents([content])[0]
|
||||
# A family change may change the vector kind (and point identity).
|
||||
store.delete_memory_records([f"card:{card_id}"])
|
||||
store.upsert("memory", [VectorWriteRecord(
|
||||
record=record, embedding=embedding,
|
||||
content_hash=hashlib.sha256(card.revision.encode()).hexdigest(),
|
||||
sparse_text=content, sparse_language=self.query_language,
|
||||
)])
|
||||
except Exception: # noqa: BLE001 - durable pending state covers adapter/factory failures.
|
||||
error = "Memory change is saved; index update is incomplete. Retry the index update."
|
||||
repo.projection_result(card_id, operation["revision"], error)
|
||||
result = {"id": card_id, "saved": True, "indexed": error is None,
|
||||
"action": operation["action"], "error": error}
|
||||
if operation["action"] != "delete":
|
||||
result["card"] = repo.get(card_id).model_dump(mode="json")
|
||||
return result
|
||||
|
||||
def rebuild(self):
|
||||
self._admin()
|
||||
with self.repository.operation() as repo:
|
||||
# Persist invalidation before deleting anything. A crash remains recoverable.
|
||||
repo.invalidate_all()
|
||||
store = self.store_factory()
|
||||
store.prepare_memory_index()
|
||||
store.delete_kinds("memory", ["memory", "solved_question"])
|
||||
return [self._propagate(repo, p["card_id"]) for p in repo.projections()]
|
||||
|
||||
def retrieve(self, question: str, *, searcher, embedder, top: int = 5,
|
||||
family=None, scope: RecallScope | None = None, excluded=()):
|
||||
if type(top) is not int or not 1 <= top <= 100:
|
||||
raise ValueError("Recall limit must be between 1 and 100")
|
||||
if not question.strip():
|
||||
raise ValueError("Recall question must not be empty")
|
||||
scope = scope or RecallScope()
|
||||
self.repository.list(CardQuery(page_size=1))
|
||||
kinds = (["solved_question"] if family == "solved_question" else
|
||||
["memory"] if family else ["memory", "solved_question"])
|
||||
hits = searcher.search(embedder.embed_query(question),
|
||||
top_n=min(MAX_SEEDS, max(20, top * 3)), kinds=kinds,
|
||||
query_text=question, query_language=self.query_language,
|
||||
metadata_filter=scope.vector_filter(family))
|
||||
# Mutations use the same lock: links, eligibility and payload are resolved
|
||||
# together against current authority, after the potentially slow vector call.
|
||||
with self.repository.operation() as repo:
|
||||
return expand_and_rank(repo, hits, scope=scope, family=family,
|
||||
excluded=set(excluded), top=top)
|
||||
|
||||
def recall(self, question: str, *, searcher, embedder, top: int = 5,
|
||||
solved: bool = False, decisions=(), scope: RecallScope | None = None):
|
||||
family = "solved_question" if solved else "domain_clarification"
|
||||
candidates = self.retrieve(question, searcher=searcher, embedder=embedder, top=top,
|
||||
family=family, scope=scope, excluded=decided_memory_ids(list(decisions)))
|
||||
result = []
|
||||
for candidate in candidates:
|
||||
card = candidate.card
|
||||
common = {"id": card.id, "session_id": card.session_id,
|
||||
"revision": card.revision, "family": card.family, "scope": card.scope,
|
||||
"dependencies": [d.model_dump() for d in card.dependencies],
|
||||
"tables": sorted({d.table for d in card.dependencies if d.table}),
|
||||
"score": round(candidate.score, 6),
|
||||
"retrieval": {"path": list(candidate.path), "method": "hybrid_links"}}
|
||||
if solved:
|
||||
result.append({**common, "question": card.question, "sql": card.sql})
|
||||
else:
|
||||
# New workflow category consumption belongs to M3.
|
||||
result.append({**common, "type": "concept_clarified",
|
||||
"subject": card.subject, "detail": card.detail, "rationale": card.rationale,
|
||||
"question_context": card.question, "scope": card.scope,
|
||||
"concepts": card.concepts})
|
||||
return result
|
||||
|
||||
def _session(self, snapshot):
|
||||
if (snapshot.manifest.workspace_id != self.repository.workspace_id
|
||||
or (not self.principal.is_admin
|
||||
and snapshot.manifest.author != self.principal.subject)):
|
||||
raise MemoryForbidden("Memory source session is outside the authorized context")
|
||||
|
||||
def promotions(self, snapshot):
|
||||
self._session(snapshot)
|
||||
decisions = effective_decisions(snapshot)
|
||||
declined = declined_promotion_seqs(decisions)
|
||||
result = []
|
||||
seen = set()
|
||||
for d in decisions:
|
||||
if d.type != "concept_clarified" or d.seq in declined:
|
||||
continue
|
||||
source_key = f"decision:{snapshot.manifest.id}:{d.seq}"
|
||||
if self.repository.source(source_key) is not None:
|
||||
continue
|
||||
content = (d.subject, d.detail, d.rationale)
|
||||
if content in seen:
|
||||
continue
|
||||
seen.add(content)
|
||||
result.append({"decision_seq": d.seq, "type": d.type, "subject": d.subject,
|
||||
"detail": d.detail, "rationale": d.rationale,
|
||||
"question_context": question_context(decisions, snapshot.manifest)})
|
||||
return result
|
||||
|
||||
def promote(self, snapshot, seqs):
|
||||
self._session(snapshot)
|
||||
decisions = effective_decisions(snapshot)
|
||||
selected = [d for d in decisions if d.seq in seqs and d.type == "concept_clarified"]
|
||||
results = []
|
||||
with self.repository.operation() as repo:
|
||||
for d in selected:
|
||||
key = f"decision:{snapshot.manifest.id}:{d.seq}"
|
||||
existing = repo.source(key)
|
||||
if existing and existing["action"] == "delete":
|
||||
continue
|
||||
card_id = existing["card_id"] if existing else repo.save(CardInput(
|
||||
family="domain_clarification", subject=d.subject, detail=d.detail,
|
||||
rationale=d.rationale, scope=self.repository.workspace_id,
|
||||
question=question_context(decisions, snapshot.manifest), concepts=[d.subject],
|
||||
), source_key=key, session_id=snapshot.manifest.id, decision_seq=d.seq)
|
||||
results.append(self._propagate(repo, card_id))
|
||||
return results
|
||||
|
||||
def save_solved(self, snapshot, promoted_tables=None):
|
||||
self._session(snapshot)
|
||||
if snapshot.manifest.status != "finalized":
|
||||
raise MemoryConflict("Only a finalized session can produce a solved-question card")
|
||||
record = _build_solved_snapshot(snapshot, promoted_tables)
|
||||
with self.repository.operation() as repo:
|
||||
existing = repo.source(f"solved:{snapshot.manifest.id}")
|
||||
if existing and existing["action"] == "delete":
|
||||
return {"saved": True, "indexed": True, "action": "delete", "error": None}
|
||||
card_id = existing["card_id"] if existing else repo.save(CardInput(
|
||||
family="solved_question", subject=record.title,
|
||||
scope=self.repository.workspace_id, question=record.content,
|
||||
sql=record.metadata["sql"],
|
||||
dependencies=[Dependency(database=snapshot.manifest.database,
|
||||
schema_name=snapshot.manifest.db_schema, table=table)
|
||||
for table in record.metadata["tables"]],
|
||||
), source_key=f"solved:{snapshot.manifest.id}", session_id=snapshot.manifest.id)
|
||||
return self._propagate(repo, card_id)
|
||||
|
||||
def retry_solved(self, snapshot):
|
||||
self._session(snapshot)
|
||||
with self.repository.operation() as repo:
|
||||
existing = repo.source(f"solved:{snapshot.manifest.id}")
|
||||
if existing is None:
|
||||
raise MemoryNotFound("No authoritative exemplar exists for this session")
|
||||
return self._propagate(repo, existing["card_id"])
|
||||
@@ -0,0 +1,72 @@
|
||||
CREATE SCHEMA IF NOT EXISTS thoth_memory;
|
||||
DO $$ BEGIN
|
||||
IF NOT EXISTS (SELECT FROM pg_roles WHERE rolname = 'thoth_memory_runtime') THEN
|
||||
CREATE ROLE thoth_memory_runtime NOLOGIN;
|
||||
END IF;
|
||||
IF EXISTS (SELECT FROM pg_roles WHERE rolname = 'thothii_catalog_runtime') THEN
|
||||
GRANT thoth_memory_runtime TO thothii_catalog_runtime;
|
||||
END IF;
|
||||
END $$;
|
||||
|
||||
CREATE TABLE thoth_memory.cards (
|
||||
workspace_id text NOT NULL,
|
||||
id text NOT NULL,
|
||||
family text NOT NULL,
|
||||
subject text NOT NULL,
|
||||
origin text NOT NULL CHECK (origin IN ('manual', 'workflow')),
|
||||
data jsonb NOT NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
updated_at timestamptz NOT NULL DEFAULT now(),
|
||||
revision text NOT NULL,
|
||||
PRIMARY KEY (workspace_id, id)
|
||||
);
|
||||
CREATE INDEX ON thoth_memory.cards (workspace_id, updated_at, id);
|
||||
CREATE INDEX ON thoth_memory.cards (workspace_id, family);
|
||||
CREATE TABLE thoth_memory.links (
|
||||
workspace_id text NOT NULL,
|
||||
source_id text NOT NULL,
|
||||
target_id text NOT NULL,
|
||||
meaning text NOT NULL,
|
||||
PRIMARY KEY (workspace_id, source_id, target_id),
|
||||
CHECK (source_id <> target_id),
|
||||
FOREIGN KEY (workspace_id, source_id) REFERENCES thoth_memory.cards ON DELETE CASCADE,
|
||||
FOREIGN KEY (workspace_id, target_id) REFERENCES thoth_memory.cards ON DELETE CASCADE
|
||||
);
|
||||
CREATE TABLE thoth_memory.dependencies (
|
||||
workspace_id text NOT NULL,
|
||||
card_id text NOT NULL,
|
||||
database_id text NOT NULL,
|
||||
schema_name text NOT NULL,
|
||||
table_name text NOT NULL,
|
||||
column_name text NOT NULL,
|
||||
PRIMARY KEY (workspace_id, card_id, database_id, schema_name, table_name, column_name),
|
||||
FOREIGN KEY (workspace_id, card_id) REFERENCES thoth_memory.cards ON DELETE CASCADE
|
||||
);
|
||||
-- No FK to cards: deletion recovery and source receipts survive removal of the card.
|
||||
CREATE TABLE thoth_memory.projections (
|
||||
workspace_id text NOT NULL,
|
||||
card_id text NOT NULL,
|
||||
revision text NOT NULL,
|
||||
action text NOT NULL CHECK (action IN ('upsert', 'delete')),
|
||||
pending boolean NOT NULL DEFAULT true,
|
||||
error text,
|
||||
source_key text,
|
||||
updated_at timestamptz NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (workspace_id, card_id),
|
||||
UNIQUE (workspace_id, source_key)
|
||||
);
|
||||
DO $$ DECLARE t text; BEGIN
|
||||
FOREACH t IN ARRAY ARRAY['cards', 'links', 'dependencies', 'projections'] LOOP
|
||||
EXECUTE format('ALTER TABLE thoth_memory.%I ENABLE ROW LEVEL SECURITY', t);
|
||||
EXECUTE format('ALTER TABLE thoth_memory.%I FORCE ROW LEVEL SECURITY', t);
|
||||
EXECUTE format(
|
||||
'CREATE POLICY workspace_isolation ON thoth_memory.%I USING '
|
||||
'(workspace_id = current_setting(''thoth.memory_workspace'', true)) '
|
||||
'WITH CHECK (workspace_id = current_setting(''thoth.memory_workspace'', true))', t);
|
||||
EXECUTE format('GRANT SELECT, INSERT, UPDATE, DELETE ON thoth_memory.%I '
|
||||
'TO thoth_memory_runtime', t);
|
||||
END LOOP;
|
||||
END $$;
|
||||
GRANT USAGE ON SCHEMA thoth_memory TO thoth_memory_runtime;
|
||||
GRANT SELECT ON thoth_memory.migrations TO thoth_memory_runtime;
|
||||
REVOKE ALL ON SCHEMA thoth_memory FROM PUBLIC;
|
||||
@@ -0,0 +1,3 @@
|
||||
-- Existing dense projections remain recoverable, but are not hybrid-ready.
|
||||
-- No cross-workspace data update or weakening of FORCE RLS is necessary.
|
||||
ALTER TABLE thoth_memory.projections ADD COLUMN format integer NOT NULL DEFAULT 1;
|
||||
@@ -0,0 +1,29 @@
|
||||
CREATE TABLE thoth_memory.reviews (
|
||||
workspace_id text NOT NULL,
|
||||
session_id text NOT NULL,
|
||||
summary_id text NOT NULL,
|
||||
request_hash text NOT NULL,
|
||||
result jsonb NOT NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (workspace_id, session_id, summary_id)
|
||||
);
|
||||
ALTER TABLE thoth_memory.reviews ENABLE ROW LEVEL SECURITY;
|
||||
ALTER TABLE thoth_memory.reviews FORCE ROW LEVEL SECURITY;
|
||||
CREATE POLICY workspace_isolation ON thoth_memory.reviews
|
||||
USING (workspace_id = current_setting('thoth.memory_workspace', true))
|
||||
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
|
||||
GRANT SELECT, INSERT ON thoth_memory.reviews TO thoth_memory_runtime;
|
||||
|
||||
CREATE TABLE thoth_memory.cleanup_receipts (
|
||||
workspace_id text NOT NULL,
|
||||
sync_id text NOT NULL,
|
||||
request_hash text NOT NULL,
|
||||
card_ids jsonb NOT NULL,
|
||||
PRIMARY KEY (workspace_id, sync_id)
|
||||
);
|
||||
ALTER TABLE thoth_memory.cleanup_receipts ENABLE ROW LEVEL SECURITY;
|
||||
ALTER TABLE thoth_memory.cleanup_receipts FORCE ROW LEVEL SECURITY;
|
||||
CREATE POLICY workspace_isolation ON thoth_memory.cleanup_receipts
|
||||
USING (workspace_id = current_setting('thoth.memory_workspace', true))
|
||||
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
|
||||
GRANT SELECT, INSERT ON thoth_memory.cleanup_receipts TO thoth_memory_runtime;
|
||||
@@ -0,0 +1,14 @@
|
||||
CREATE TABLE thoth_memory.archive_repairs (
|
||||
workspace_id text NOT NULL,
|
||||
session_id text NOT NULL,
|
||||
repair_id text NOT NULL,
|
||||
data jsonb NOT NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
PRIMARY KEY (workspace_id, session_id, repair_id)
|
||||
);
|
||||
ALTER TABLE thoth_memory.archive_repairs ENABLE ROW LEVEL SECURITY;
|
||||
ALTER TABLE thoth_memory.archive_repairs FORCE ROW LEVEL SECURITY;
|
||||
CREATE POLICY workspace_isolation ON thoth_memory.archive_repairs
|
||||
USING (workspace_id = current_setting('thoth.memory_workspace', true))
|
||||
WITH CHECK (workspace_id = current_setting('thoth.memory_workspace', true));
|
||||
GRANT SELECT, INSERT, UPDATE ON thoth_memory.archive_repairs TO thoth_memory_runtime;
|
||||
@@ -230,6 +230,8 @@ def advance_problems(source: Path | SessionSnapshot, phase: int) -> list[str]:
|
||||
problems.append(f"CTE non ancora approvato: {nc} (Fase 6)")
|
||||
if phase == 7 and not _has_decision(source, "sql_approved"):
|
||||
problems.append("manca la decisione sql_approved (Fase 7)")
|
||||
if phase == 8 and not _has_decision(source, "memory_summary_reviewed"):
|
||||
problems.append("Fase 8: il riepilogo Memory deve essere revisionato prima della chiusura")
|
||||
if phase == 8 and not any(
|
||||
d.type in ("datamart_requested", "datamart_declined")
|
||||
for d in effective_decisions(source)
|
||||
|
||||
@@ -88,6 +88,10 @@ class VectorStore(Protocol):
|
||||
|
||||
def delete_kinds(self, collection: str, kinds: list[str]) -> int: ...
|
||||
|
||||
def prepare_memory_index(self) -> None: ...
|
||||
|
||||
def delete_memory_records(self, record_keys: list[str]) -> None: ...
|
||||
|
||||
def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int: ...
|
||||
|
||||
def list_evidence_generations(self, collection: str, workspace_id: str) -> list[str]: ...
|
||||
|
||||
@@ -30,6 +30,7 @@ _ARTIFACT_FILES = {
|
||||
"cte_tests": "cte_tests.json",
|
||||
"cte_plan": "cte_plan.json",
|
||||
"cte_plan_doc": "cte_plan_doc.json",
|
||||
"memory_proposals": "memory_proposals.json",
|
||||
}
|
||||
_ARTIFACT_KEYS = {filename: key for key, filename in _ARTIFACT_FILES.items()}
|
||||
_SAFE_CTE_NAME = re.compile(r"[A-Za-z0-9_-]+\Z")
|
||||
|
||||
@@ -31,6 +31,7 @@ _ARTIFACT_KEYS = {
|
||||
"evidence",
|
||||
"evidence_receipts",
|
||||
"sql_final",
|
||||
"memory_proposals",
|
||||
"validation_report",
|
||||
"retrieval_pack",
|
||||
"cte_tests",
|
||||
|
||||
@@ -64,11 +64,12 @@ phases:
|
||||
name: datamart
|
||||
advance: reviewer_decide
|
||||
prerequisites:
|
||||
- decision_exists: memory_summary_reviewed
|
||||
- any:
|
||||
- decision_exists: datamart_requested
|
||||
- decision_exists: datamart_declined
|
||||
artifacts_out: []
|
||||
emits: [datamart_requested, datamart_declined, memory_promoted, memory_promotion_declined]
|
||||
emits: [datamart_requested, datamart_declined, memory_promoted, memory_promotion_declined, memory_summary_reviewed]
|
||||
|
||||
decision_min_phase: auto
|
||||
max_phase: auto
|
||||
|
||||
Reference in New Issue
Block a user