feat: refine metadata catalog workflows

This commit is contained in:
Codex
2026-09-02 11:38:47 +02:00
parent 076c9742c5
commit 4531746038
48 changed files with 1345 additions and 493 deletions
@@ -218,6 +218,12 @@ test("suggests sensitive flags from structural metadata without persisting them"
runId: responseBody.run.id,
sequence: 2,
level: "info",
message: "Classified 1 of 1 columns.",
},
{
runId: responseBody.run.id,
sequence: 3,
level: "info",
message: "Sensitive-field suggestion generation completed for 1 column.",
},
]);
@@ -671,6 +677,7 @@ test("generates one selected Catalog Column from a single JSON code fence", asyn
errorSummary: null,
});
expect(Object.keys(start.json()).sort()).toEqual([
"cacheReadTokens",
"createdAt",
"databaseId",
"errorSummary",
@@ -678,9 +685,11 @@ test("generates one selected Catalog Column from a single JSON code fence", asyn
"finishedAt",
"generated",
"id",
"inputTokens",
"language",
"modelId",
"nonGeneratable",
"outputTokens",
"processed",
"scope",
"startedAt",
@@ -1272,7 +1281,7 @@ test("Stop aborts source sampling before any model request", async () => {
test("an isolated exhausted technical batch failure allows completion with errors", async () => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async (request) => {
if (vi.mocked(modelCompleter.complete).mock.calls.length === 1) {
if (vi.mocked(modelCompleter.complete).mock.calls.length <= 2) {
throw new ModelCompletionProviderError();
}
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
@@ -1320,7 +1329,7 @@ test("an isolated exhausted technical batch failure allows completion with error
failed: 10,
errorSummary: "Description generation completed with errors.",
});
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
expect(modelCompleter.complete).toHaveBeenCalledTimes(3);
const updated = new Map(
(await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]),
);
@@ -1346,10 +1355,10 @@ test("success resets the technical-failure streak and the third later failure st
const modelCompleter: ModelCompleter = {
complete: vi.fn(async (request) => {
const call = vi.mocked(modelCompleter.complete).mock.calls.length;
if ([1, 2, 4, 5, 6].includes(call)) {
if ([1, 2, 4, 5, 6, 7, 8, 9].includes(call)) {
throw Object.assign(new ModelCompletionProviderError(), { message: sensitiveDiagnostic });
}
if (call > 6) throw new Error("a later batch must not start");
if (call > 9) throw new Error("a later batch must not start");
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
return JSON.stringify({
results: context.targets.map((target: { targetId: string }) => ({
@@ -1389,25 +1398,25 @@ test("success resets the technical-failure streak and the third later failure st
expect(run).toMatchObject({
status: "failed",
total: 61,
processed: 60,
processed: 50,
generated: 10,
nonGeneratable: 0,
failed: 50,
failed: 40,
errorSummary: "Description generation stopped after three consecutive technical batch failures.",
});
expect(modelCompleter.complete).toHaveBeenCalledTimes(6);
expect(modelCompleter.complete).toHaveBeenCalledTimes(9);
const requests = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => request);
expect(new Set(requests.map((request) => request.signal)).size).toBe(1);
expect(new Set(requests.map((request) => request.model.id))).toEqual(new Set([configuredModel.id]));
const updated = new Map(
(await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]),
);
targetIds.slice(20, 30).forEach((targetId) => {
targetIds.slice(10, 20).forEach((targetId) => {
expect(updated.get(targetId)?.generatedDescription).toBe("Successful reset batch.");
});
expect(updated.get(targetIds[60]!)?.generatedDescription).toBeNull();
expect(updated.get(targetIds[50]!)?.generatedDescription).toBeNull();
const events = await repository.listDescriptionGenerationEvents(run.id);
expect(events.filter((event) => event.message.includes("model provider request failed"))).toHaveLength(5);
expect(events.filter((event) => event.message.includes("model provider request failed"))).toHaveLength(8);
expect(events.at(-1)).toEqual(expect.objectContaining({
level: "error",
message: "Description generation stopped after three consecutive technical batch failures.",
@@ -1660,6 +1669,7 @@ test("Description Generation history is newest-first, bounded, and exposes only
expect(response.statusCode).toBe(200);
expect(response.json().map((run: { id: string }) => run.id)).toEqual(ids.slice(1).reverse());
expect(Object.keys(response.json()[0]).sort()).toEqual([
"cacheReadTokens",
"createdAt",
"databaseId",
"errorSummary",
@@ -1667,9 +1677,11 @@ test("Description Generation history is newest-first, bounded, and exposes only
"finishedAt",
"generated",
"id",
"inputTokens",
"language",
"modelId",
"nonGeneratable",
"outputTokens",
"processed",
"scope",
"startedAt",
@@ -1945,7 +1957,7 @@ test("retains completed batch writes when a later batch response is malformed",
});
const { run } = await waitForTerminalRun(app, start.json().id);
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
expect(modelCompleter.complete).toHaveBeenCalledTimes(3);
expect(run).toMatchObject({
status: "completed_with_errors",
total: 11,
@@ -1972,9 +1984,14 @@ test("retains completed batch writes when a later batch response is malformed",
expect(events.slice(2, 12).map((event) => event.message)).toEqual(
orderedIds.slice(0, 10).map((targetId) => `Generated description for Catalog Column ${targetId}.`),
);
expect(events.find((event) => event.level === "warning" && event.message.includes("Retrying batch"))).toEqual(
expect.objectContaining({
message: "The model response did not match the required schema. Retrying batch (attempt 2 of 2).",
}),
);
expect(events.find((event) => event.level === "error")).toEqual(expect.objectContaining({
level: "error",
message: `The model response was invalid. Affected Catalog Column target: ${orderedIds[10]}.`,
message: `The model response did not match the required schema. Affected Catalog Column target: ${orderedIds[10]}.`,
}));
} finally {
await app.close();
@@ -2186,7 +2203,7 @@ test("Generate Missing skips prior partial results and includes null, empty, and
const metadata = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
if (mode === "partial") {
partialCall += 1;
if (partialCall === 2) throw new ModelCompletionProviderError();
if (partialCall === 2 || partialCall === 3) throw new ModelCompletionProviderError();
return JSON.stringify({
results: metadata.targets.map((target: { targetId: string }) => ({
targetId: target.targetId,
@@ -2512,6 +2529,58 @@ test("localizes valid non-generatable Catalog Column results in English", async
}
});
test("retries invalid JSON once and completes the batch when the second response is valid", async () => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async (request) => {
if (vi.mocked(modelCompleter.complete).mock.calls.length === 1) {
return { content: "not-json", usage: { input: 11, cacheRead: 3, output: 2 } };
}
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
return {
content: JSON.stringify({
results: context.targets.map((target: { targetId: string }) => ({
targetId: target.targetId,
outcome: "generated",
description: "Generated after the application retry.",
})),
}),
usage: { input: 7, cacheRead: 1, output: 5 },
};
}),
};
const { app, repository, database, table, column } = await setup(modelCompleter);
try {
const start = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/description-generation-runs`,
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] },
});
const { run } = await waitForTerminalRun(app, start.json().id);
expect(run).toMatchObject({
status: "completed",
processed: 1,
generated: 1,
failed: 0,
inputTokens: 18,
cacheReadTokens: 4,
outputTokens: 7,
});
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
expect(await repository.getColumn(database.id, table.id, column.id)).toMatchObject({
generatedDescription: "Generated after the application retry.",
});
expect(await repository.listDescriptionGenerationEvents(run.id)).toEqual(expect.arrayContaining([
expect.objectContaining({
level: "warning",
message: "The model response was not valid JSON. Retrying batch (attempt 2 of 2).",
}),
]));
} finally {
await app.close();
}
});
test("fails safely when the provider fails and redacts provider diagnostics", async () => {
const sensitiveDiagnostic = "test-provider-secret private prompt raw provider payload";
const modelCompleter: ModelCompleter = {
@@ -2528,6 +2597,7 @@ test("fails safely when the provider fails and redacts provider diagnostics", as
});
const { run } = await waitForTerminalRun(app, start.json().id);
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
expect(run).toMatchObject({
status: "completed_with_errors",
processed: 1,
@@ -2565,7 +2635,7 @@ test.each([
["duplicate mappings", (targetIds: readonly string[]) => ({ results: [
{ targetId: targetIds[0], outcome: "generated", description: "First valid value" },
{ targetId: targetIds[0], outcome: "non_generatable" },
] })],
] }), "The model response was missing one or more requested targets."],
["unknown mappings", (targetIds: readonly string[]) => ({ results: [
{ targetId: targetIds[0], outcome: "non_generatable" },
{
@@ -2573,15 +2643,15 @@ test.each([
outcome: "generated",
description: "Unknown target value",
},
] })],
] }), "The model response was missing one or more requested targets."],
["missing mappings", (targetIds: readonly string[]) => ({ results: [
{ targetId: targetIds[0], outcome: "generated", description: "Only one result" },
] })],
] }), "The model response was missing one or more requested targets."],
["malformed mappings", (targetIds: readonly string[]) => ({ results: [
{ targetId: targetIds[0], outcome: "generated", description: "First valid value" },
{ targetId: targetIds[1], outcome: "generated", description: " " },
] })],
] as const)("rejects %s without applying any result from the batch", async (_name, responseFor) => {
] }), "The model response did not match the required schema."],
] as const)("rejects %s without applying any result from the batch", async (_name, responseFor, failureMessage) => {
let selectedColumnIds: string[] = [];
const modelCompleter: ModelCompleter = {
complete: vi.fn(async () => JSON.stringify(responseFor(selectedColumnIds))),
@@ -2625,6 +2695,7 @@ test.each([
});
const { run } = await waitForTerminalRun(app, start.json().id);
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
expect(run).toMatchObject({
status: "completed_with_errors",
processed: 2,
@@ -2644,7 +2715,7 @@ test.each([
expect((await repository.listDescriptionGenerationEvents(run.id)).find((event) => event.level === "error")).toEqual(
expect.objectContaining({
level: "error",
message: `The model response was invalid. Affected Catalog Column targets: ${selectedColumnIds.join(", ")}.`,
message: `${failureMessage} Affected Catalog Column targets: ${selectedColumnIds.join(", ")}.`,
}),
);
} finally {
@@ -13,6 +13,7 @@ import { up as upSchemaSync } from "../src/catalog/migrations/003_catalog_schema
import { up as upDescriptionGeneration } from "../src/catalog/migrations/005_description_generation_runs.js";
import { up as upSensitiveDataFlag } from "../src/catalog/migrations/006_sensitive_data_flag.js";
import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_sensitive_data_suggestion_runs.js";
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
import { KyselyCatalogRepository, type CatalogDatabase } from "../src/catalog/repository.js";
import { loadConfig } from "../src/config.js";
import type { WorkspaceRegistry } from "../src/workspaces/registry.js";
@@ -48,6 +49,7 @@ test.skipIf(!dockerAvailable)("Fastify persists Description Generation success a
await upSensitiveDataFlag(db);
await upDescriptionGeneration(db);
await upSensitiveSuggestionRuns(db);
await upAiTokenUsage(db);
const repository = new KyselyCatalogRepository(db);
const database = await repository.create({
workspaceId: "psd-clinical",
@@ -124,7 +126,7 @@ test.skipIf(!dockerAvailable)("Fastify persists Description Generation success a
description: "Elenco dei pazienti e dei loro dati clinici.",
}] });
}
if (call === 4) {
if (call === 5) {
return JSON.stringify({ results: [
{
targetId: birthDate.id,
@@ -138,14 +140,14 @@ test.skipIf(!dockerAvailable)("Fastify persists Description Generation success a
},
] });
}
if (call === 5) {
if (call === 6) {
return JSON.stringify({ results: [{
targetId: table.id,
outcome: "generated",
description: "Descrizione rigenerata della tabella pazienti.",
}] });
}
if (call === 6) {
if (call === 7) {
return JSON.stringify({ results: [{
targetId: birthDate.id,
outcome: "generated",
@@ -340,7 +342,7 @@ test.skipIf(!dockerAvailable)("Fastify persists Description Generation success a
expect(await repository.getColumn(database.id, table.id, birthDate.id)).toMatchObject({
generatedDescription: "Descrizione recuperata della data di nascita.",
});
expect(modelCompleter.complete).toHaveBeenCalledTimes(6);
expect(modelCompleter.complete).toHaveBeenCalledTimes(7);
expect(JSON.stringify(vi.mocked(modelCompleter.complete).mock.calls)).toContain(persistedSampleSecret);
const runIds = [
@@ -1,6 +1,9 @@
import { expect, test, vi } from "vitest";
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import {
PostgresDescriptionSourceSampler,
ConcreteDescriptionSourceSampler,
type DescriptionSourceSamplingTarget,
} from "../src/catalog/description-source-sampler.js";
import type {
@@ -8,6 +11,8 @@ import type {
CatalogPostgresAccess,
} from "../src/catalog/postgres-access.js";
import type { WorkspaceDatabase } from "../src/catalog/types.js";
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
const database: WorkspaceDatabase = {
id: "11111111-1111-4111-8111-111111111111",
@@ -51,7 +56,7 @@ test("samples at most five source rows and five distinct non-null examples in a
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const sampler = new PostgresDescriptionSourceSampler(access);
const sampler = new ConcreteDescriptionSourceSampler(access);
const controller = new AbortController();
const samples = await sampler.sample(database, [target], controller.signal);
@@ -92,13 +97,71 @@ test("samples at most five source rows and five distinct non-null examples in a
expect(end).toHaveBeenCalledOnce();
});
test("samples source rows through the configured REST run_query binding", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-source-rest-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
const release = vi.fn();
const secretStore = {
materialize: vi.fn(() => ({
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
release,
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
{ 'status"code': "active", ward: null },
{ 'status"code': "pending", ward: "A" },
]), { status: 200, headers: { "content-type": "application/json" } }));
vi.stubGlobal("fetch", fetchMock);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
};
const sampler = new ConcreteDescriptionSourceSampler(access, secretStore);
const restDatabase: WorkspaceDatabase = {
...database,
binding: {
transport: "rest_api",
baseUrl: "https://dwh.example.test/root/",
restPath: "/health",
restAuth: "x-api-key",
},
};
try {
await expect(sampler.sample(restDatabase, [target], new AbortController().signal)).resolves.toEqual([{
targetId: target.targetId,
tableName: target.tableName,
rows: [
{ fields: [{ name: 'status"code', value: "active" }, { name: "ward", value: null }] },
{ fields: [{ name: 'status"code', value: "pending" }, { name: "ward", value: "A" }] },
],
representativeValues: [
{ column: 'status"code', values: ["active", "pending"] },
{ column: "ward", values: ["A"] },
],
}]);
expect(access.connect).not.toHaveBeenCalled();
expect(fetchMock).toHaveBeenCalledWith("https://dwh.example.test/root/rpc/run_query", expect.objectContaining({
method: "POST",
headers: { "content-type": "application/json", "x-api-key": "test-api-key" },
body: JSON.stringify({
query_text: 'SELECT LEFT(("status""code")::text, 256) AS "status""code", LEFT(("ward")::text, 256) AS "ward" FROM "clinical""data"."patient""facts" LIMIT 5',
}),
}));
expect(release).toHaveBeenCalledOnce();
} finally {
vi.unstubAllGlobals();
rmSync(root, { recursive: true, force: true });
}
});
test("does not issue a SELECT when a protected target has no source columns", async () => {
const query = vi.fn(async () => ({ rows: [] }));
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const sampler = new PostgresDescriptionSourceSampler(access);
const sampler = new ConcreteDescriptionSourceSampler(access);
const samples = await sampler.sample(database, [{
targetId: target.targetId,
@@ -129,7 +192,7 @@ test("rolls back and closes the source connection when sampling fails", async ()
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const sampler = new PostgresDescriptionSourceSampler(access);
const sampler = new ConcreteDescriptionSourceSampler(access);
const controller = new AbortController();
await expect(sampler.sample(database, [target], controller.signal)).rejects.toThrow();
@@ -13,6 +13,7 @@ import { up as upDescriptionGeneration } from "../src/catalog/migrations/005_des
import { up as upSensitiveDataFlag } from "../src/catalog/migrations/006_sensitive_data_flag.js";
import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_sensitive_data_suggestion_runs.js";
import { up as upLogicalRelationships } from "../src/catalog/migrations/008_catalog_logical_relationships.js";
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
const dockerAvailable = spawnSync("docker", ["info"], { stdio: "ignore" }).status === 0;
@@ -28,6 +29,9 @@ test.skipIf(!dockerAvailable)("PostgreSQL migration enforces one database per wo
await upSchemaSync(db);
await upSensitiveDataFlag(db);
await upLogicalRelationships(db);
await upDescriptionGeneration(db);
await upSensitiveSuggestionRuns(db);
await upAiTokenUsage(db);
await sql`CREATE ROLE thothii_catalog_runtime`.execute(db);
await upRuntimeSequencePrivileges(db);
const sequencePrivilege = await sql<{ allowed: boolean }>`
@@ -386,6 +390,7 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
await upLogicalRelationships(db);
await upDescriptionGeneration(db);
await upSensitiveSuggestionRuns(db);
await upAiTokenUsage(db);
const repository = new KyselyCatalogRepository(db);
const firstDatabase = await repository.create({
workspaceId: "generation-one",
@@ -127,7 +127,7 @@ test("resolves only a configured selection for the later generation boundary", (
expect(() => models.resolve("unknown-model")).toThrow(MetadataGenerationModelUnavailableError);
});
test("loads DeepSeek, GLM, and an explicit keyless Qwen endpoint from installation setup", () => {
test("loads DeepSeek models, GLM, and an explicit keyless Qwen endpoint from installation setup", () => {
const { installationFile, secretsFile } = metadataConfiguration(`metadataGeneration:
default: glm-53
models:
@@ -135,6 +135,10 @@ test("loads DeepSeek, GLM, and an explicit keyless Qwen endpoint from installati
label: DeepSeek V4 Pro
litellm: {provider: deepseek, model: deepseek-v4-pro}
apiKeyEnv: DEEPSEEK_API_KEY
- id: deepseek-v4-flash
label: DeepSeek V4 Flash
litellm: {provider: deepseek, model: deepseek-v4-flash}
apiKeyEnv: DEEPSEEK_API_KEY
- id: glm-53
label: GLM 5.3
litellm:
@@ -156,6 +160,7 @@ test("loads DeepSeek, GLM, and an explicit keyless Qwen endpoint from installati
expect(models.catalog()).toEqual({
models: [
{ id: "deepseek-v4-pro", label: "DeepSeek V4 Pro" },
{ id: "deepseek-v4-flash", label: "DeepSeek V4 Flash" },
{ id: "glm-53", label: "GLM 5.3" },
{ id: "qwen-36", label: "Qwen 3.6" },
],
+2 -2
View File
@@ -64,7 +64,7 @@ sys.stdout.write(json.dumps({"ok": True, "content": "Descrizione italiana"}))
signal: new AbortController().signal,
});
expect(content).toBe("Descrizione italiana");
expect(content).toEqual({ content: "Descrizione italiana", usage: { input: 0, cacheRead: 0, output: 0 } });
const captured = JSON.parse(readFileSync(join(roots[0]!, "request.json"), "utf8"));
expect(captured.request).toEqual({
model: "openai/gpt-4.1-mini",
@@ -100,7 +100,7 @@ sys.stdout.write(json.dumps({"ok": True, "content": "Descrizione Qwen"}))
},
messages: [{ role: "user", content: "Describe invented metadata." }],
signal: new AbortController().signal,
})).resolves.toBe("Descrizione Qwen");
})).resolves.toEqual({ content: "Descrizione Qwen", usage: { input: 0, cacheRead: 0, output: 0 } });
expect(JSON.parse(readFileSync(join(roots[0]!, "request.json"), "utf8"))).toEqual({
model: "openai/qwen3.6-35b-a3b",