3148 lines
118 KiB
TypeScript
3148 lines
118 KiB
TypeScript
import { expect, test, vi } from "vitest";
|
|
import { buildApp } from "../src/app.js";
|
|
import type { DescriptionSourceSampler } from "../src/catalog/description-source-sampler.js";
|
|
import { MemoryCatalogRepository } from "../src/catalog/memory-repository.js";
|
|
import {
|
|
MetadataGenerationModelUnavailableError,
|
|
type MetadataGenerationModels,
|
|
type ResolvedMetadataGenerationModel,
|
|
} from "../src/catalog/metadata-generation-models.js";
|
|
import {
|
|
ModelCompletionCancelledError,
|
|
ModelCompletionProviderError,
|
|
type ModelCompleter,
|
|
type ModelCompletionRequest,
|
|
} from "../src/catalog/model-completer.js";
|
|
import { CatalogOperationCoordinator } from "../src/catalog/operation-coordinator.js";
|
|
import type {
|
|
CatalogDatabaseClient,
|
|
CatalogPostgresAccess,
|
|
} from "../src/catalog/postgres-access.js";
|
|
import { loadConfig } from "../src/config.js";
|
|
import type { WorkspaceRegistry, WorkspaceRevision } from "../src/workspaces/registry.js";
|
|
import type { WorkspaceDescriptor } from "../src/workspaces/schema.js";
|
|
|
|
const workspace: WorkspaceDescriptor = {
|
|
workspace: {
|
|
schema_version: 4,
|
|
id: "psd-clinical",
|
|
name: "Policlinico San Donato",
|
|
language: "it",
|
|
},
|
|
dwh: {
|
|
engine: "postgres",
|
|
database: "warehouse",
|
|
schema: "datawarehouse",
|
|
port: 5432,
|
|
supported_transports: ["postgres_direct"],
|
|
},
|
|
};
|
|
const revision: WorkspaceRevision = {
|
|
id: "psd-clinical",
|
|
commit: "a".repeat(40),
|
|
blob: "b".repeat(40),
|
|
snapshotPath: "/tmp/psd.yaml",
|
|
};
|
|
const configuredModel: ResolvedMetadataGenerationModel = {
|
|
id: "openai/gpt-4.1-mini",
|
|
provider: "openai",
|
|
model: "gpt-4.1-mini",
|
|
apiKeyEnv: "OPENAI_API_KEY",
|
|
apiKey: "test-provider-secret",
|
|
};
|
|
|
|
function models(): MetadataGenerationModels {
|
|
return {
|
|
catalog: () => ({ models: [{ id: configuredModel.id, label: "OpenAI Mini" }], default: configuredModel.id }),
|
|
resolve: (selection) => {
|
|
if (selection !== configuredModel.id) throw new MetadataGenerationModelUnavailableError();
|
|
return configuredModel;
|
|
},
|
|
};
|
|
}
|
|
|
|
async function setup(
|
|
modelCompleter: ModelCompleter,
|
|
env: Record<string, string> = {},
|
|
language: "en" | "it" = "it",
|
|
descriptionSourceSampler: DescriptionSourceSampler | null = {
|
|
sample: vi.fn(async () => []),
|
|
},
|
|
catalogPostgresAccess?: CatalogPostgresAccess,
|
|
) {
|
|
const repository = new MemoryCatalogRepository();
|
|
const database = await repository.create({
|
|
workspaceId: workspace.workspace.id,
|
|
engine: "postgres",
|
|
databaseName: "warehouse",
|
|
schema: "datawarehouse",
|
|
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
|
|
});
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: "Clinical patients" }],
|
|
columns: [{
|
|
tableName: "patients",
|
|
name: "birth_date",
|
|
ordinalPosition: 1,
|
|
dataType: "date",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: "Patient date of birth",
|
|
}],
|
|
relationships: [],
|
|
});
|
|
const table = (await repository.listTables(database.id))[0]!;
|
|
const column = (await repository.listColumns(database.id, table.id))[0]!;
|
|
const registry = {
|
|
list: vi.fn(async () => [revision]),
|
|
read: vi.fn(async () => ({
|
|
workspace: {
|
|
...workspace,
|
|
workspace: { ...workspace.workspace, language },
|
|
},
|
|
revision,
|
|
})),
|
|
} as unknown as WorkspaceRegistry;
|
|
const operations = new CatalogOperationCoordinator();
|
|
const app = buildApp(loadConfig({ NODE_ENV: "test", THT_HARNESS_DIR: "/missing", ...env }), {
|
|
thtRunner: {} as never,
|
|
workspaceRegistry: registry,
|
|
workspaceDiagnoser: vi.fn(),
|
|
catalogRepository: repository,
|
|
catalogOperationCoordinator: operations,
|
|
metadataGenerationModels: models(),
|
|
modelCompleter,
|
|
...(descriptionSourceSampler ? { descriptionSourceSampler } : {}),
|
|
...(catalogPostgresAccess ? { catalogPostgresAccess } : {}),
|
|
});
|
|
return { app, repository, database, table, column, operations };
|
|
}
|
|
|
|
async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: string) {
|
|
for (let attempt = 0; attempt < 100; attempt += 1) {
|
|
const response = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/description-generation-runs/${runId}`,
|
|
});
|
|
const run = response.json();
|
|
if (["completed", "completed_with_errors", "cancelled", "failed", "interrupted"].includes(run.status)) {
|
|
return { response, run };
|
|
}
|
|
await new Promise((resolve) => setTimeout(resolve, 5));
|
|
}
|
|
throw new Error(`Description Generation Run ${runId} did not finish`);
|
|
}
|
|
|
|
test("suggests sensitive flags from structural metadata without persisting them", async () => {
|
|
const modelCompleter = {
|
|
complete: vi.fn(async () => JSON.stringify({
|
|
suggestions: [{ columnId: expect.any(String), sensitive: true }],
|
|
})),
|
|
};
|
|
const { app, repository, database, table, column } = await setup(modelCompleter);
|
|
modelCompleter.complete.mockResolvedValueOnce(JSON.stringify({
|
|
suggestions: [{ columnId: column.id, sensitive: true }],
|
|
}));
|
|
|
|
try {
|
|
const response = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: { modelId: configuredModel.id, scope: "all" },
|
|
});
|
|
|
|
expect(response.statusCode).toBe(200);
|
|
const responseBody = response.json();
|
|
expect(responseBody).toMatchObject({
|
|
run: {
|
|
databaseId: database.id,
|
|
scope: "all",
|
|
modelId: configuredModel.id,
|
|
status: "completed",
|
|
total: 1,
|
|
suggestedSensitive: 1,
|
|
suggestedNonSensitive: 0,
|
|
errorSummary: null,
|
|
},
|
|
suggestions: [{
|
|
columnId: column.id,
|
|
tableId: table.id,
|
|
tableName: table.name,
|
|
columnName: column.name,
|
|
version: column.version,
|
|
currentSensitive: false,
|
|
sensitive: true,
|
|
}],
|
|
});
|
|
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
|
.toMatchObject({ sensitive: false });
|
|
expect(responseBody.run).not.toHaveProperty("suggestions");
|
|
|
|
const history = await app.inject({
|
|
method: "GET",
|
|
url: "/catalog/sensitive-data-suggestion-runs?limit=1",
|
|
});
|
|
expect(history.statusCode).toBe(200);
|
|
expect(history.json()).toEqual([responseBody.run]);
|
|
expect(history.body).not.toContain(column.id);
|
|
|
|
const detail = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/sensitive-data-suggestion-runs/${responseBody.run.id}`,
|
|
});
|
|
expect(detail.statusCode).toBe(200);
|
|
expect(detail.json()).toEqual(responseBody.run);
|
|
|
|
const events = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/sensitive-data-suggestion-runs/${responseBody.run.id}/events-list`,
|
|
});
|
|
expect(events.statusCode).toBe(200);
|
|
expect(events.body).not.toContain(column.id);
|
|
expect(events.json()).toMatchObject([
|
|
{
|
|
runId: responseBody.run.id,
|
|
sequence: 1,
|
|
level: "info",
|
|
message: "Sensitive-field suggestion generation started.",
|
|
},
|
|
{
|
|
runId: responseBody.run.id,
|
|
sequence: 2,
|
|
level: "info",
|
|
message: "Classified 1 of 1 columns.",
|
|
},
|
|
{
|
|
runId: responseBody.run.id,
|
|
sequence: 3,
|
|
level: "info",
|
|
message: "Sensitive-field suggestion generation completed for 1 column.",
|
|
},
|
|
]);
|
|
|
|
const request = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest;
|
|
const prompt = request.messages.map((message) => message.content).join("\n");
|
|
expect(prompt).toContain("patients");
|
|
expect(prompt).toContain("birth_date");
|
|
expect(prompt).toContain("date");
|
|
expect(prompt).not.toContain("Patient date of birth");
|
|
expect(prompt).not.toContain("test-provider-secret");
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("limits sensitive-data suggestions to the selected tables or columns", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
const payload = JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
|
|
columns: Array<{ columnId: string; column: string }>;
|
|
};
|
|
return JSON.stringify({
|
|
suggestions: payload.columns.map((column) => ({
|
|
columnId: column.columnId,
|
|
sensitive: column.column.includes("name") || column.column.includes("note"),
|
|
})),
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database } = await setup(modelCompleter);
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [
|
|
{ name: "patients", sourceComment: null },
|
|
{ name: "visits", sourceComment: null },
|
|
{ name: "billing", sourceComment: null },
|
|
],
|
|
columns: [
|
|
{ tableName: "patients", name: "patient_name", ordinalPosition: 1, dataType: "text", isNullable: false, defaultExpression: null, primaryKeyPosition: null, sourceComment: null },
|
|
{ tableName: "patients", name: "status", ordinalPosition: 2, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null },
|
|
{ tableName: "visits", name: "clinical_note", ordinalPosition: 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null },
|
|
{ tableName: "billing", name: "invoice_total", ordinalPosition: 1, dataType: "numeric", isNullable: false, defaultExpression: null, primaryKeyPosition: null, sourceComment: null },
|
|
],
|
|
relationships: [],
|
|
});
|
|
const tables = await repository.listTables(database.id);
|
|
const patients = tables.find((table) => table.name === "patients")!;
|
|
const visits = tables.find((table) => table.name === "visits")!;
|
|
const billing = tables.find((table) => table.name === "billing")!;
|
|
const patientColumns = await repository.listColumns(database.id, patients.id);
|
|
const visitColumns = await repository.listColumns(database.id, visits.id);
|
|
const billingColumns = await repository.listColumns(database.id, billing.id);
|
|
const status = patientColumns.find((column) => column.name === "status")!;
|
|
const clinicalNote = visitColumns.find((column) => column.name === "clinical_note")!;
|
|
|
|
try {
|
|
const tableResponse = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_tables",
|
|
targetIds: [visits.id, patients.id],
|
|
},
|
|
});
|
|
expect(tableResponse.statusCode).toBe(200);
|
|
expect(tableResponse.json().suggestions).toHaveLength(3);
|
|
expect(tableResponse.json().suggestions).toEqual(expect.arrayContaining([
|
|
expect.objectContaining({ tableId: patients.id, columnName: "patient_name", sensitive: true }),
|
|
expect.objectContaining({ tableId: patients.id, columnName: "status", sensitive: false }),
|
|
expect.objectContaining({ tableId: visits.id, columnName: "clinical_note", sensitive: true }),
|
|
]));
|
|
expect(tableResponse.json().suggestions).not.toEqual(expect.arrayContaining([
|
|
expect.objectContaining({ tableId: billing.id }),
|
|
]));
|
|
|
|
const columnResponse = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [clinicalNote.id, status.id],
|
|
},
|
|
});
|
|
expect(columnResponse.statusCode).toBe(200);
|
|
expect(columnResponse.json().suggestions).toHaveLength(2);
|
|
expect(columnResponse.json().suggestions).toEqual(expect.arrayContaining([
|
|
expect.objectContaining({ tableId: patients.id, columnId: status.id, sensitive: false }),
|
|
expect.objectContaining({ tableId: visits.id, columnId: clinicalNote.id, sensitive: true }),
|
|
]));
|
|
|
|
const prompts = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => (
|
|
JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
|
|
columns: Array<{ columnId: string }>;
|
|
}
|
|
));
|
|
expect(prompts[0]!.columns.map((column) => column.columnId).sort()).toEqual(
|
|
[...patientColumns, ...visitColumns].map((column) => column.id).sort(),
|
|
);
|
|
expect(prompts[0]!.columns.map((column) => column.columnId)).not.toContain(billingColumns[0]!.id);
|
|
expect(prompts[1]!.columns.map((column) => column.columnId).sort()).toEqual(
|
|
[status.id, clinicalNote.id].sort(),
|
|
);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("explains invalid sensitive-data suggestion selections without calling the model", async () => {
|
|
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
|
|
const { app, database, table } = await setup(modelCompleter);
|
|
|
|
try {
|
|
const empty = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_tables", targetIds: [] },
|
|
});
|
|
expect(empty.statusCode).toBe(400);
|
|
expect(empty.json()).toEqual({
|
|
code: "sensitive_data_suggestion_request_invalid",
|
|
message: "Choose a database, one or more tables, or one or more columns to classify.",
|
|
});
|
|
|
|
const duplicate = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_tables",
|
|
targetIds: [table.id, table.id],
|
|
},
|
|
});
|
|
expect(duplicate.statusCode).toBe(400);
|
|
expect(duplicate.json()).toEqual({
|
|
code: "sensitive_data_suggestion_target_ids_duplicate",
|
|
message: "Each selected table or column must appear only once.",
|
|
});
|
|
|
|
const missingTable = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_tables",
|
|
targetIds: ["00000000-0000-4000-8000-000000000001"],
|
|
},
|
|
});
|
|
expect(missingTable.statusCode).toBe(404);
|
|
expect(missingTable.json()).toEqual({
|
|
code: "catalog_table_not_found",
|
|
message: "One or more selected Catalog Tables were not found in this database.",
|
|
});
|
|
|
|
const missingColumn = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: ["00000000-0000-4000-8000-000000000002"],
|
|
},
|
|
});
|
|
expect(missingColumn.statusCode).toBe(404);
|
|
expect(missingColumn.json()).toEqual({
|
|
code: "catalog_column_not_found",
|
|
message: "One or more selected Catalog Columns were not found in this database.",
|
|
});
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("batches sensitive-data suggestions for schemas larger than one helper message", async () => {
|
|
const maxHelperMessageBytes = 64 * 1024;
|
|
const seenColumnIds: string[] = [];
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
const userMessage = request.messages.find((message) => message.role === "user")!;
|
|
expect(Buffer.byteLength(userMessage.content, "utf8")).toBeLessThanOrEqual(maxHelperMessageBytes);
|
|
const payload = JSON.parse(userMessage.content) as {
|
|
columns: Array<{ columnId: string; column: string }>;
|
|
};
|
|
expect(payload.columns.length).toBeLessThanOrEqual(10);
|
|
seenColumnIds.push(...payload.columns.map((column) => column.columnId));
|
|
return JSON.stringify({
|
|
suggestions: payload.columns.map((column) => ({
|
|
columnId: column.columnId,
|
|
sensitive: column.column.endsWith("_private"),
|
|
})),
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database } = await setup(modelCompleter);
|
|
const columnCount = 900;
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "wide_table", sourceComment: null }],
|
|
columns: Array.from({ length: columnCount }, (_, index) => ({
|
|
tableName: "wide_table",
|
|
name: `field_${index.toString().padStart(4, "0")}${index % 10 === 0 ? "_private" : ""}`,
|
|
ordinalPosition: index + 1,
|
|
dataType: "character varying(255)",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const wideTable = (await repository.listTables(database.id)).find((table) => table.name === "wide_table")!;
|
|
const expectedColumnIds = (await repository.listColumns(database.id, wideTable.id)).map((column) => column.id);
|
|
|
|
try {
|
|
const response = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: { modelId: configuredModel.id, scope: "all" },
|
|
});
|
|
|
|
expect(response.statusCode).toBe(200);
|
|
const suggestions = response.json().suggestions as Array<{
|
|
columnName: string;
|
|
currentSensitive: boolean;
|
|
sensitive: boolean;
|
|
}>;
|
|
expect(suggestions).toHaveLength(columnCount);
|
|
expect(suggestions).toEqual(expect.arrayContaining([
|
|
expect.objectContaining({ columnName: "field_0000_private", currentSensitive: false, sensitive: true }),
|
|
expect.objectContaining({ columnName: "field_0001", currentSensitive: false, sensitive: false }),
|
|
]));
|
|
expect(vi.mocked(modelCompleter.complete).mock.calls.length).toBeGreaterThan(1);
|
|
expect(seenColumnIds.slice().sort()).toEqual(expectedColumnIds.slice().sort());
|
|
expect(new Set(seenColumnIds).size).toBe(columnCount);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("retries one invalid sensitive-data classification before returning the review draft", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => "unused"),
|
|
};
|
|
const { app, database, column } = await setup(modelCompleter);
|
|
vi.mocked(modelCompleter.complete)
|
|
.mockResolvedValueOnce("not-json")
|
|
.mockResolvedValueOnce(JSON.stringify({
|
|
suggestions: [{ columnId: column.id, sensitive: true }],
|
|
}));
|
|
|
|
try {
|
|
const response = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: { modelId: configuredModel.id, scope: "all" },
|
|
});
|
|
|
|
expect(response.statusCode).toBe(200);
|
|
expect(response.json().suggestions).toEqual([
|
|
expect.objectContaining({ columnId: column.id, sensitive: true }),
|
|
]);
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test.each(["malformed", "incomplete", "duplicate"] as const)(
|
|
"fails safely when sensitive-data suggestions are %s",
|
|
async (kind) => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => "unused"),
|
|
};
|
|
const { app, repository, database, column } = await setup(modelCompleter);
|
|
const rawResponse = kind === "malformed"
|
|
? "RAW_PROVIDER_RESPONSE_DO_NOT_EXPOSE_{"
|
|
: kind === "incomplete"
|
|
? JSON.stringify({ suggestions: [] })
|
|
: JSON.stringify({
|
|
suggestions: [
|
|
{ columnId: column.id, sensitive: true },
|
|
{ columnId: column.id, sensitive: true },
|
|
],
|
|
});
|
|
vi.mocked(modelCompleter.complete).mockResolvedValueOnce(rawResponse);
|
|
|
|
try {
|
|
const response = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: { modelId: configuredModel.id, scope: "all" },
|
|
});
|
|
|
|
expect(response.statusCode).toBe(502);
|
|
expect(response.json()).toEqual({
|
|
code: "sensitive_data_suggestion_invalid_response",
|
|
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
|
|
});
|
|
expect(response.body).not.toContain(rawResponse);
|
|
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
|
.toMatchObject({ sensitive: false });
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
},
|
|
);
|
|
|
|
test("explains a sensitive-data suggestion provider failure without exposing provider details", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => {
|
|
throw new ModelCompletionProviderError();
|
|
}),
|
|
};
|
|
const { app, repository, database, column } = await setup(modelCompleter);
|
|
|
|
try {
|
|
const response = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
|
payload: { modelId: configuredModel.id, scope: "all" },
|
|
});
|
|
|
|
expect(response.statusCode).toBe(502);
|
|
expect(response.json()).toEqual({
|
|
code: "sensitive_data_suggestion_provider_unavailable",
|
|
message: "The selected LLM service could not complete the request. No suggestions were applied.",
|
|
});
|
|
expect(response.body).not.toContain("model completion failed");
|
|
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
|
.toMatchObject({ sensitive: false });
|
|
|
|
const history = await app.inject({
|
|
method: "GET",
|
|
url: "/catalog/sensitive-data-suggestion-runs",
|
|
});
|
|
expect(history.statusCode).toBe(200);
|
|
const [failedRun] = history.json();
|
|
expect(failedRun).toMatchObject({
|
|
databaseId: database.id,
|
|
status: "failed",
|
|
total: 1,
|
|
suggestedSensitive: 0,
|
|
suggestedNonSensitive: 0,
|
|
errorSummary: "Sensitive-field suggestion generation failed.",
|
|
});
|
|
|
|
const events = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/sensitive-data-suggestion-runs/${failedRun.id}/events-list`,
|
|
});
|
|
expect(events.statusCode).toBe(200);
|
|
expect(events.json()).toMatchObject([
|
|
{
|
|
runId: failedRun.id,
|
|
sequence: 1,
|
|
level: "info",
|
|
message: "Sensitive-field suggestion generation started.",
|
|
},
|
|
{
|
|
runId: failedRun.id,
|
|
sequence: 2,
|
|
level: "error",
|
|
message: "Sensitive-field suggestion generation failed.",
|
|
},
|
|
]);
|
|
expect(events.body).not.toContain("model completion failed");
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
interface SseFrame {
|
|
id?: string;
|
|
event?: string;
|
|
data: string;
|
|
}
|
|
|
|
function sseFrameReader(response: Response) {
|
|
const reader = response.body!.getReader();
|
|
const decoder = new TextDecoder();
|
|
const queued: SseFrame[] = [];
|
|
let buffer = "";
|
|
const parseAvailable = () => {
|
|
let boundary = buffer.indexOf("\n\n");
|
|
while (boundary >= 0) {
|
|
const block = buffer.slice(0, boundary);
|
|
buffer = buffer.slice(boundary + 2);
|
|
const frame: SseFrame = { data: "" };
|
|
const data: string[] = [];
|
|
for (const line of block.split("\n")) {
|
|
if (line.startsWith("id: ")) frame.id = line.slice(4);
|
|
else if (line.startsWith("event: ")) frame.event = line.slice(7);
|
|
else if (line.startsWith("data: ")) data.push(line.slice(6));
|
|
}
|
|
frame.data = data.join("\n");
|
|
if (frame.event || frame.id || frame.data) queued.push(frame);
|
|
boundary = buffer.indexOf("\n\n");
|
|
}
|
|
};
|
|
return {
|
|
async next(): Promise<SseFrame> {
|
|
while (queued.length === 0) {
|
|
const chunk = await reader.read();
|
|
if (chunk.done) throw new Error("SSE stream ended before the expected event arrived");
|
|
buffer += decoder.decode(chunk.value, { stream: true }).replaceAll("\r\n", "\n");
|
|
parseAvailable();
|
|
}
|
|
return queued.shift()!;
|
|
},
|
|
async cancel(): Promise<void> {
|
|
await reader.cancel();
|
|
},
|
|
};
|
|
}
|
|
|
|
test("generates one selected Catalog Column from a single JSON code fence", async () => {
|
|
let resolveCompletion!: (content: string) => void;
|
|
const completion = new Promise<string>((resolve) => { resolveCompletion = resolve; });
|
|
const modelCompleter = { complete: vi.fn(async () => await completion) };
|
|
const { app, database, table, column } = await setup(modelCompleter);
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
},
|
|
});
|
|
|
|
expect(start.statusCode).toBe(202);
|
|
expect(start.json()).toMatchObject({
|
|
databaseId: database.id,
|
|
scope: "selected_columns",
|
|
modelId: configuredModel.id,
|
|
language: "it",
|
|
status: "queued",
|
|
total: 1,
|
|
processed: 0,
|
|
generated: 0,
|
|
nonGeneratable: 0,
|
|
failed: 0,
|
|
startedAt: null,
|
|
finishedAt: null,
|
|
errorSummary: null,
|
|
});
|
|
expect(Object.keys(start.json()).sort()).toEqual([
|
|
"cacheReadTokens",
|
|
"createdAt",
|
|
"databaseId",
|
|
"errorSummary",
|
|
"failed",
|
|
"finishedAt",
|
|
"generated",
|
|
"id",
|
|
"inputTokens",
|
|
"language",
|
|
"modelId",
|
|
"nonGeneratable",
|
|
"outputTokens",
|
|
"processed",
|
|
"scope",
|
|
"startedAt",
|
|
"status",
|
|
"total",
|
|
"updatedAt",
|
|
]);
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(1);
|
|
const completionRequest = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest;
|
|
expect(completionRequest.model).toEqual(configuredModel);
|
|
expect(completionRequest.messages).toHaveLength(2);
|
|
expect(completionRequest.messages[0]?.content).toContain('workspace language "it"');
|
|
expect(completionRequest.messages[0]?.content).toContain('{"results":[');
|
|
expect(completionRequest.messages[1]?.content).toContain(`"targetId":"${column.id}"`);
|
|
expect(completionRequest.messages[1]?.content).not.toMatch(/source rows|samples|example values/i);
|
|
expect(start.body).not.toMatch(/test-provider-secret|Catalog metadata/);
|
|
|
|
resolveCompletion(`\`\`\`json\n${JSON.stringify({
|
|
results: [{
|
|
targetId: column.id,
|
|
outcome: "generated",
|
|
description: "Data di nascita del paziente.",
|
|
}],
|
|
})}\n\`\`\``);
|
|
const { response: completedResponse, run } = await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(completedResponse.statusCode).toBe(200);
|
|
expect(run).toMatchObject({
|
|
status: "completed",
|
|
total: 1,
|
|
processed: 1,
|
|
generated: 1,
|
|
nonGeneratable: 0,
|
|
failed: 0,
|
|
errorSummary: null,
|
|
});
|
|
expect(run.startedAt).toEqual(expect.any(String));
|
|
expect(run.finishedAt).toEqual(expect.any(String));
|
|
|
|
const columns = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/databases/${database.id}/tables/${table.id}/columns`,
|
|
});
|
|
expect(columns.statusCode).toBe(200);
|
|
expect(columns.json()).toEqual([
|
|
expect.objectContaining({
|
|
id: column.id,
|
|
description: null,
|
|
generatedDescription: "Data di nascita del paziente.",
|
|
version: column.version + 1,
|
|
}),
|
|
]);
|
|
|
|
const events = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/description-generation-runs/${run.id}/events-list?after=0`,
|
|
});
|
|
expect(events.statusCode).toBe(200);
|
|
expect(events.json().map((event: { sequence: number }) => event.sequence)).toEqual([1, 2, 3, 4]);
|
|
expect(Object.keys(events.json()[0]).sort()).toEqual(["createdAt", "level", "message", "sequence"]);
|
|
expect(events.json()).toEqual([
|
|
expect.objectContaining({ sequence: 1, level: "info", message: "Description generation queued." }),
|
|
expect.objectContaining({ sequence: 2, level: "info", message: "Description generation started." }),
|
|
expect.objectContaining({
|
|
sequence: 3,
|
|
level: "info",
|
|
message: `Generated description for Catalog Column ${column.id}.`,
|
|
}),
|
|
expect.objectContaining({ sequence: 4, level: "info", message: "Description generation completed." }),
|
|
]);
|
|
expect(events.body).not.toMatch(/test-provider-secret|gpt-4\.1|openai\/gpt|Catalog metadata/);
|
|
const laterEvents = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/description-generation-runs/${run.id}/events-list?after=2`,
|
|
});
|
|
expect(laterEvents.json().map((event: { sequence: number }) => event.sequence)).toEqual([3, 4]);
|
|
} finally {
|
|
resolveCompletion(JSON.stringify({
|
|
results: [{
|
|
targetId: column.id,
|
|
outcome: "generated",
|
|
description: "Data di nascita del paziente.",
|
|
}],
|
|
}));
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("keeps real source samples transient across the Fastify API and application logs", async () => {
|
|
let selectedColumnId = "";
|
|
const sampleSecret = "FASTIFY_TRANSIENT_SAMPLE_9ca73e";
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => JSON.stringify({
|
|
results: [{
|
|
targetId: selectedColumnId,
|
|
outcome: "generated",
|
|
description: "Data di nascita del paziente.",
|
|
}],
|
|
})),
|
|
};
|
|
const descriptionSourceSampler: DescriptionSourceSampler = {
|
|
sample: vi.fn(async (_database, targets) => [{
|
|
targetId: targets[0]!.targetId,
|
|
tableName: targets[0]!.tableName,
|
|
rows: [{ fields: [{ name: targets[0]!.columnNames[0]!, value: sampleSecret }] }],
|
|
representativeValues: [{
|
|
column: targets[0]!.columnNames[0]!,
|
|
values: [sampleSecret],
|
|
}],
|
|
}]),
|
|
};
|
|
const { app, repository, database, table, column } = await setup(
|
|
modelCompleter,
|
|
{},
|
|
"it",
|
|
descriptionSourceSampler,
|
|
);
|
|
selectedColumnId = column.id;
|
|
const logSpies = [
|
|
vi.spyOn(app.log, "info"),
|
|
vi.spyOn(app.log, "warn"),
|
|
vi.spyOn(app.log, "error"),
|
|
];
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
},
|
|
});
|
|
expect(start.statusCode).toBe(202);
|
|
await waitForTerminalRun(app, start.json().id);
|
|
|
|
const completionRequest = vi.mocked(modelCompleter.complete).mock.calls[0]![0];
|
|
expect(JSON.stringify(completionRequest.messages)).toContain(sampleSecret);
|
|
|
|
const apiResponses = await Promise.all([
|
|
app.inject({ method: "GET", url: `/catalog/description-generation-runs/${start.json().id}` }),
|
|
app.inject({
|
|
method: "GET",
|
|
url: `/catalog/description-generation-runs/${start.json().id}/events-list`,
|
|
}),
|
|
app.inject({ method: "GET", url: `/catalog/databases/${database.id}` }),
|
|
app.inject({ method: "GET", url: `/catalog/databases/${database.id}/tables` }),
|
|
app.inject({
|
|
method: "GET",
|
|
url: `/catalog/databases/${database.id}/tables/${table.id}/columns`,
|
|
}),
|
|
]);
|
|
expect(apiResponses.every((response) => response.statusCode === 200)).toBe(true);
|
|
expect(apiResponses.map((response) => response.body).join("\n")).not.toContain(sampleSecret);
|
|
|
|
const persisted = JSON.stringify({
|
|
run: await repository.getDescriptionGenerationRun(start.json().id),
|
|
events: await repository.listDescriptionGenerationEvents(start.json().id),
|
|
database: await repository.get(database.id),
|
|
table: await repository.getTable(database.id, table.id),
|
|
column: await repository.getColumn(database.id, table.id, column.id),
|
|
});
|
|
expect(persisted).not.toContain(sampleSecret);
|
|
expect(JSON.stringify(logSpies.flatMap((spy) => spy.mock.calls))).not.toContain(sampleSecret);
|
|
} finally {
|
|
for (const spy of logSpies) spy.mockRestore();
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("never exposes a protected source value to the model, persistence, logs, or browser APIs", async () => {
|
|
let selectedColumnId = "";
|
|
const protectedValue = "PROTECTED_SOURCE_VALUE_8f4c2a";
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => JSON.stringify({
|
|
results: [{
|
|
targetId: selectedColumnId,
|
|
outcome: "generated",
|
|
description: "Data di nascita del paziente.",
|
|
}],
|
|
})),
|
|
};
|
|
const descriptionSourceSampler: DescriptionSourceSampler = {
|
|
sample: vi.fn(async (_database, targets) => [{
|
|
targetId: targets[0]!.targetId,
|
|
tableName: targets[0]!.tableName,
|
|
rows: [{ fields: [{ name: "birth_date", value: protectedValue }] }],
|
|
representativeValues: [{ column: "birth_date", values: [protectedValue] }],
|
|
}]),
|
|
};
|
|
const { app, repository, database, table, column } = await setup(
|
|
modelCompleter,
|
|
{},
|
|
"it",
|
|
descriptionSourceSampler,
|
|
);
|
|
selectedColumnId = column.id;
|
|
await repository.updateColumnMetadata(
|
|
database.id,
|
|
table.id,
|
|
column.id,
|
|
column.version,
|
|
column.description,
|
|
column.generatedDescription,
|
|
true,
|
|
);
|
|
const logSpies = [
|
|
vi.spyOn(app.log, "info"),
|
|
vi.spyOn(app.log, "warn"),
|
|
vi.spyOn(app.log, "error"),
|
|
];
|
|
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
},
|
|
});
|
|
expect(start.statusCode).toBe(202);
|
|
await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(descriptionSourceSampler.sample).toHaveBeenCalledWith(
|
|
expect.anything(),
|
|
[expect.objectContaining({ targetId: column.id, columnNames: [] })],
|
|
expect.any(AbortSignal),
|
|
);
|
|
const completionRequest = vi.mocked(modelCompleter.complete).mock.calls[0]![0];
|
|
const providerPayload = JSON.stringify(completionRequest.messages);
|
|
expect(providerPayload).not.toContain(protectedValue);
|
|
expect(providerPayload).toContain("1981-01-01");
|
|
|
|
const apiResponses = await Promise.all([
|
|
app.inject({ method: "GET", url: `/catalog/description-generation-runs/${start.json().id}` }),
|
|
app.inject({
|
|
method: "GET",
|
|
url: `/catalog/description-generation-runs/${start.json().id}/events-list`,
|
|
}),
|
|
app.inject({ method: "GET", url: `/catalog/databases/${database.id}` }),
|
|
app.inject({ method: "GET", url: `/catalog/databases/${database.id}/tables` }),
|
|
app.inject({
|
|
method: "GET",
|
|
url: `/catalog/databases/${database.id}/tables/${table.id}/columns`,
|
|
}),
|
|
]);
|
|
expect(apiResponses.every((response) => response.statusCode === 200)).toBe(true);
|
|
expect(apiResponses.map((response) => response.body).join("\n")).not.toContain(protectedValue);
|
|
|
|
const persisted = JSON.stringify({
|
|
run: await repository.getDescriptionGenerationRun(start.json().id),
|
|
events: await repository.listDescriptionGenerationEvents(start.json().id),
|
|
database: await repository.get(database.id),
|
|
table: await repository.getTable(database.id, table.id),
|
|
column: await repository.getColumn(database.id, table.id, column.id),
|
|
});
|
|
expect(persisted).not.toContain(protectedValue);
|
|
expect(JSON.stringify(logSpies.flatMap((spy) => spy.mock.calls))).not.toContain(protectedValue);
|
|
} finally {
|
|
for (const spy of logSpies) spy.mockRestore();
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("wires the production sampler to the same injected CatalogPostgresAccess instance", async () => {
|
|
let selectedColumnId = "";
|
|
const sampleSecret = "PRODUCTION_WIRING_SAMPLE_f2986a";
|
|
const query = vi.fn(async (sql: string) => ({
|
|
rows: sql.startsWith("SELECT") ? [{ birth_date: sampleSecret }] : [],
|
|
}));
|
|
const end = vi.fn(async () => undefined);
|
|
const catalogPostgresAccess: CatalogPostgresAccess = {
|
|
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
|
};
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => JSON.stringify({
|
|
results: [{
|
|
targetId: selectedColumnId,
|
|
outcome: "generated",
|
|
description: "Data di nascita del paziente.",
|
|
}],
|
|
})),
|
|
};
|
|
const { app, database, column } = await setup(
|
|
modelCompleter,
|
|
{},
|
|
"it",
|
|
null,
|
|
catalogPostgresAccess,
|
|
);
|
|
selectedColumnId = column.id;
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
},
|
|
});
|
|
expect(start.statusCode).toBe(202);
|
|
expect((await waitForTerminalRun(app, start.json().id)).run.status).toBe("completed");
|
|
|
|
expect(catalogPostgresAccess.connect).toHaveBeenCalledWith(
|
|
expect.objectContaining({ id: database.id }),
|
|
expect.any(AbortSignal),
|
|
);
|
|
expect(query.mock.calls.map(([sql]) => String(sql).split(" ")[0])).toEqual([
|
|
"BEGIN",
|
|
"SELECT",
|
|
"ROLLBACK",
|
|
]);
|
|
expect(end).toHaveBeenCalledOnce();
|
|
const completionRequest = vi.mocked(modelCompleter.complete).mock.calls[0]![0];
|
|
expect(JSON.stringify(completionRequest.messages)).toContain(sampleSecret);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("generates selected Catalog Columns in caller order through sequential batches of ten", async () => {
|
|
const pending = Array.from({ length: 2 }, () => {
|
|
let resolve!: (content: string) => void;
|
|
const promise = new Promise<string>((done) => { resolve = done; });
|
|
return { promise, resolve };
|
|
});
|
|
let active = 0;
|
|
let maxActive = 0;
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => {
|
|
const callIndex = (modelCompleter.complete as ReturnType<typeof vi.fn>).mock.calls.length - 1;
|
|
active += 1;
|
|
maxActive = Math.max(maxActive, active);
|
|
try {
|
|
return await pending[callIndex]!.promise;
|
|
} finally {
|
|
active -= 1;
|
|
}
|
|
}),
|
|
};
|
|
const { app, repository, database, table } = await setup(modelCompleter);
|
|
let orderedIds: string[] = [];
|
|
const responseFor = (targetIds: readonly string[], offset: number) => JSON.stringify({
|
|
results: targetIds.map((targetId, index) => ({
|
|
targetId,
|
|
outcome: "generated",
|
|
description: `Generated ${offset + index + 1}`,
|
|
})).reverse(),
|
|
});
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: "Clinical patients" }],
|
|
columns: Array.from({ length: 11 }, (_, index) => ({
|
|
tableName: "patients",
|
|
name: `column_${String(index + 1).padStart(2, "0")}`,
|
|
ordinalPosition: index + 1,
|
|
dataType: "text",
|
|
isNullable: index % 2 === 0,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: index === 0 ? 1 : null,
|
|
sourceComment: `Column ${index + 1}`,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const columns = await repository.listColumns(database.id, table.id);
|
|
orderedIds = columns.map((column) => column.id).reverse();
|
|
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: orderedIds,
|
|
},
|
|
});
|
|
|
|
expect(start.statusCode).toBe(202);
|
|
expect(start.json()).toMatchObject({ total: 11, scope: "selected_columns" });
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(1);
|
|
expect(maxActive).toBe(1);
|
|
|
|
const firstRequest = (modelCompleter.complete as ReturnType<typeof vi.fn>).mock.calls[0]![0] as ModelCompletionRequest;
|
|
const firstMetadata = JSON.parse(firstRequest.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
expect(firstMetadata.targets.map((target: { targetId: string }) => target.targetId)).toEqual(orderedIds.slice(0, 10));
|
|
|
|
pending[0]!.resolve(responseFor(orderedIds.slice(0, 10), 0));
|
|
for (let attempt = 0; attempt < 100 && (modelCompleter.complete as ReturnType<typeof vi.fn>).mock.calls.length < 2; attempt += 1) {
|
|
await new Promise((resolve) => setTimeout(resolve, 1));
|
|
}
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
|
|
expect(maxActive).toBe(1);
|
|
const secondRequest = (modelCompleter.complete as ReturnType<typeof vi.fn>).mock.calls[1]![0] as ModelCompletionRequest;
|
|
const secondMetadata = JSON.parse(secondRequest.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
expect(secondMetadata.targets.map((target: { targetId: string }) => target.targetId)).toEqual(orderedIds.slice(10));
|
|
|
|
pending[1]!.resolve(responseFor(orderedIds.slice(10), 10));
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
expect(run).toMatchObject({
|
|
status: "completed",
|
|
total: 11,
|
|
processed: 11,
|
|
generated: 11,
|
|
nonGeneratable: 0,
|
|
failed: 0,
|
|
});
|
|
const updatedById = new Map(
|
|
(await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]),
|
|
);
|
|
orderedIds.forEach((targetId, index) => {
|
|
expect(updatedById.get(targetId)).toMatchObject({
|
|
description: null,
|
|
generatedDescription: `Generated ${index + 1}`,
|
|
});
|
|
});
|
|
const events = await repository.listDescriptionGenerationEvents(run.id);
|
|
expect(events.map((event) => event.sequence)).toEqual(Array.from({ length: 14 }, (_, index) => index + 1));
|
|
expect(events.slice(2, -1).map((event) => event.message)).toEqual(
|
|
orderedIds.map((targetId) => `Generated description for Catalog Column ${targetId}.`),
|
|
);
|
|
} finally {
|
|
pending[0]!.resolve(responseFor(orderedIds.slice(0, 10), 0));
|
|
pending[1]!.resolve(responseFor(orderedIds.slice(10), 10));
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("Stop cancels the current helper, retains prior writes, and prevents later batches", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
const targetIds = context.targets.map((target: { targetId: string }) => target.targetId);
|
|
if (vi.mocked(modelCompleter.complete).mock.calls.length === 1) {
|
|
return JSON.stringify({
|
|
results: targetIds.map((targetId: string, index: number) => ({
|
|
targetId,
|
|
outcome: "generated",
|
|
description: `Completed before Stop ${index + 1}`,
|
|
})),
|
|
});
|
|
}
|
|
return await new Promise<string>((_resolve, reject) => {
|
|
const cancel = () => reject(new ModelCompletionCancelledError());
|
|
if (request.signal.aborted) cancel();
|
|
else request.signal.addEventListener("abort", cancel, { once: true });
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database, table, operations } = await setup(modelCompleter);
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: null }],
|
|
columns: Array.from({ length: 21 }, (_, index) => ({
|
|
tableName: "patients",
|
|
name: `column_${String(index + 1).padStart(2, "0")}`,
|
|
ordinalPosition: index + 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const columns = await repository.listColumns(database.id, table.id);
|
|
const targetIds = columns.map((column) => column.id);
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds,
|
|
},
|
|
});
|
|
expect(start.statusCode).toBe(202);
|
|
for (let attempt = 0; attempt < 100 && vi.mocked(modelCompleter.complete).mock.calls.length < 2; attempt += 1) {
|
|
await new Promise((resolve) => setTimeout(resolve, 2));
|
|
}
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
|
|
|
|
const cancel = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/description-generation-runs/${start.json().id}/cancel`,
|
|
});
|
|
|
|
expect(cancel.statusCode).toBe(200);
|
|
expect(cancel.json()).toMatchObject({
|
|
status: "cancelled",
|
|
total: 21,
|
|
processed: 10,
|
|
generated: 10,
|
|
nonGeneratable: 0,
|
|
failed: 0,
|
|
errorSummary: null,
|
|
});
|
|
const requests = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => request);
|
|
expect(requests).toHaveLength(2);
|
|
expect(requests[0]!.signal).toBe(requests[1]!.signal);
|
|
expect(requests[1]!.signal.aborted).toBe(true);
|
|
|
|
const updated = new Map(
|
|
(await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]),
|
|
);
|
|
targetIds.slice(0, 10).forEach((targetId, index) => {
|
|
expect(updated.get(targetId)?.generatedDescription).toBe(`Completed before Stop ${index + 1}`);
|
|
});
|
|
targetIds.slice(10).forEach((targetId) => {
|
|
expect(updated.get(targetId)?.generatedDescription).toBeNull();
|
|
});
|
|
const events = await repository.listDescriptionGenerationEvents(start.json().id);
|
|
expect(events.map((event) => event.sequence)).toEqual(
|
|
Array.from({ length: events.length }, (_, index) => index + 1),
|
|
);
|
|
expect(events.at(-1)).toEqual(expect.objectContaining({
|
|
level: "warning",
|
|
message: "Description generation cancelled.",
|
|
}));
|
|
const release = operations.reserve(database.id);
|
|
release();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("Stop aborts source sampling before any model request", async () => {
|
|
let samplingStarted!: () => void;
|
|
const started = new Promise<void>((resolve) => { samplingStarted = resolve; });
|
|
let samplingSignal: AbortSignal | undefined;
|
|
const sourceSampler: DescriptionSourceSampler = {
|
|
sample: vi.fn(async (_database, _targets, signal) => {
|
|
samplingSignal = signal;
|
|
samplingStarted();
|
|
return await new Promise((resolve, reject) => {
|
|
const cancel = () => reject(new Error("sampling aborted"));
|
|
if (signal.aborted) cancel();
|
|
else signal.addEventListener("abort", cancel, { once: true });
|
|
});
|
|
}),
|
|
};
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("cancelled sampling must not call the model"); }),
|
|
};
|
|
const { app, repository, database, column } = await setup(
|
|
modelCompleter,
|
|
{},
|
|
"it",
|
|
sourceSampler,
|
|
);
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
},
|
|
});
|
|
expect(start.statusCode).toBe(202);
|
|
await started;
|
|
|
|
const cancel = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/description-generation-runs/${start.json().id}/cancel`,
|
|
});
|
|
|
|
expect(cancel.statusCode).toBe(200);
|
|
expect(cancel.json()).toMatchObject({ status: "cancelled", processed: 0 });
|
|
expect(samplingSignal?.aborted).toBe(true);
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
expect((await repository.listDescriptionGenerationEvents(start.json().id)).map(
|
|
(event) => event.message,
|
|
)).not.toContain(
|
|
"Source samples unavailable for this batch; generation continued with catalog metadata only.",
|
|
);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("an isolated exhausted technical batch failure allows completion with errors", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
if (vi.mocked(modelCompleter.complete).mock.calls.length <= 2) {
|
|
throw new ModelCompletionProviderError();
|
|
}
|
|
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
return JSON.stringify({
|
|
results: context.targets.map((target: { targetId: string }) => ({
|
|
targetId: target.targetId,
|
|
outcome: "generated",
|
|
description: "Generated after an isolated technical failure.",
|
|
})),
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database, table } = await setup(modelCompleter);
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: null }],
|
|
columns: Array.from({ length: 11 }, (_, index) => ({
|
|
tableName: "patients",
|
|
name: `column_${String(index + 1).padStart(2, "0")}`,
|
|
ordinalPosition: index + 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const targetIds = (await repository.listColumns(database.id, table.id)).map((column) => column.id);
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds },
|
|
});
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(run).toMatchObject({
|
|
status: "completed_with_errors",
|
|
total: 11,
|
|
processed: 11,
|
|
generated: 1,
|
|
nonGeneratable: 0,
|
|
failed: 10,
|
|
errorSummary: "Description generation completed with errors.",
|
|
});
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(3);
|
|
const updated = new Map(
|
|
(await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]),
|
|
);
|
|
targetIds.slice(0, 10).forEach((targetId) => {
|
|
expect(updated.get(targetId)?.generatedDescription).toBeNull();
|
|
});
|
|
expect(updated.get(targetIds[10]!)?.generatedDescription).toBe(
|
|
"Generated after an isolated technical failure.",
|
|
);
|
|
expect((await repository.listDescriptionGenerationEvents(run.id)).at(-1)).toEqual(
|
|
expect.objectContaining({
|
|
level: "warning",
|
|
message: "Description generation completed with errors.",
|
|
}),
|
|
);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("success resets the technical-failure streak and the third later failure stops the run", async () => {
|
|
const sensitiveDiagnostic = "provider payload must remain redacted 7d41";
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
const call = vi.mocked(modelCompleter.complete).mock.calls.length;
|
|
if ([1, 2, 4, 5, 6, 7, 8, 9].includes(call)) {
|
|
throw Object.assign(new ModelCompletionProviderError(), { message: sensitiveDiagnostic });
|
|
}
|
|
if (call > 9) throw new Error("a later batch must not start");
|
|
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
return JSON.stringify({
|
|
results: context.targets.map((target: { targetId: string }) => ({
|
|
targetId: target.targetId,
|
|
outcome: "generated",
|
|
description: "Successful reset batch.",
|
|
})),
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database, table } = await setup(modelCompleter);
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: null }],
|
|
columns: Array.from({ length: 61 }, (_, index) => ({
|
|
tableName: "patients",
|
|
name: `column_${String(index + 1).padStart(2, "0")}`,
|
|
ordinalPosition: index + 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const targetIds = (await repository.listColumns(database.id, table.id)).map((column) => column.id);
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds },
|
|
});
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(run).toMatchObject({
|
|
status: "failed",
|
|
total: 61,
|
|
processed: 50,
|
|
generated: 10,
|
|
nonGeneratable: 0,
|
|
failed: 40,
|
|
errorSummary: "Description generation stopped after three consecutive technical batch failures.",
|
|
});
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(9);
|
|
const requests = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => request);
|
|
expect(new Set(requests.map((request) => request.signal)).size).toBe(1);
|
|
expect(new Set(requests.map((request) => request.model.id))).toEqual(new Set([configuredModel.id]));
|
|
const updated = new Map(
|
|
(await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]),
|
|
);
|
|
targetIds.slice(10, 20).forEach((targetId) => {
|
|
expect(updated.get(targetId)?.generatedDescription).toBe("Successful reset batch.");
|
|
});
|
|
expect(updated.get(targetIds[50]!)?.generatedDescription).toBeNull();
|
|
const events = await repository.listDescriptionGenerationEvents(run.id);
|
|
expect(events.filter((event) => event.message.includes("model provider request failed"))).toHaveLength(8);
|
|
expect(events.at(-1)).toEqual(expect.objectContaining({
|
|
level: "error",
|
|
message: "Description generation stopped after three consecutive technical batch failures.",
|
|
}));
|
|
expect(JSON.stringify({ run, events })).not.toContain(sensitiveDiagnostic);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test.each(["queued", "running"] as const)(
|
|
"backend startup marks a persisted %s Description Generation Run interrupted without resuming it",
|
|
async (status) => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("startup must not resume generation"); }),
|
|
};
|
|
const { app, repository, database } = await setup(modelCompleter);
|
|
try {
|
|
const created = await repository.createDescriptionGenerationRun(
|
|
database.id,
|
|
"missing",
|
|
configuredModel.id,
|
|
"it",
|
|
1,
|
|
);
|
|
if (status === "running") {
|
|
await repository.updateDescriptionGenerationRun(created.id, {
|
|
status: "running",
|
|
startedAt: "2026-08-28T08:00:00.000Z",
|
|
});
|
|
}
|
|
|
|
await app.ready();
|
|
|
|
expect(await repository.getDescriptionGenerationRun(created.id)).toMatchObject({
|
|
status: "interrupted",
|
|
errorSummary: "Description generation was interrupted by backend restart.",
|
|
finishedAt: expect.any(String),
|
|
});
|
|
expect(await repository.listDescriptionGenerationEvents(created.id)).toEqual([
|
|
expect.objectContaining({
|
|
sequence: 1,
|
|
level: "warning",
|
|
message: "Description generation was interrupted by backend restart.",
|
|
}),
|
|
]);
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
},
|
|
);
|
|
|
|
test("Unlock is rejected while a local Description Generation worker is live", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => await new Promise<string>((_resolve, reject) => {
|
|
const cancel = () => reject(new ModelCompletionCancelledError());
|
|
if (request.signal.aborted) cancel();
|
|
else request.signal.addEventListener("abort", cancel, { once: true });
|
|
})),
|
|
};
|
|
const { app, database, column } = await setup(modelCompleter);
|
|
let runId = "";
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
},
|
|
});
|
|
runId = start.json().id;
|
|
expect(modelCompleter.complete).toHaveBeenCalledOnce();
|
|
|
|
const unlock = await app.inject({
|
|
method: "POST",
|
|
url: "/catalog/description-generation-runs/unlock",
|
|
});
|
|
|
|
expect(unlock.statusCode).toBe(409);
|
|
expect(unlock.json()).toEqual({
|
|
code: "description_generation_run_live",
|
|
message: "A local Description Generation worker or helper is still running.",
|
|
});
|
|
expect(vi.mocked(modelCompleter.complete).mock.calls[0]![0].signal.aborted).toBe(false);
|
|
} finally {
|
|
if (runId) {
|
|
await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/description-generation-runs/${runId}/cancel`,
|
|
});
|
|
}
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("Unlock is rejected while a Description Generation start is still launching its worker", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => await new Promise<string>((_resolve, reject) => {
|
|
const cancel = () => reject(new ModelCompletionCancelledError());
|
|
if (request.signal.aborted) cancel();
|
|
else request.signal.addEventListener("abort", cancel, { once: true });
|
|
})),
|
|
};
|
|
const { app, repository, database, column } = await setup(modelCompleter);
|
|
const appendEvent = repository.appendDescriptionGenerationEvent.bind(repository);
|
|
let queuedEventEntered!: () => void;
|
|
const queuedEventStarted = new Promise<void>((resolve) => { queuedEventEntered = resolve; });
|
|
let releaseQueuedEvent!: () => void;
|
|
const queuedEventGate = new Promise<void>((resolve) => { releaseQueuedEvent = resolve; });
|
|
let blockQueuedEvent = true;
|
|
vi.spyOn(repository, "appendDescriptionGenerationEvent").mockImplementation(
|
|
async (runId, level, message) => {
|
|
if (blockQueuedEvent && message === "Description generation queued.") {
|
|
blockQueuedEvent = false;
|
|
queuedEventEntered();
|
|
await queuedEventGate;
|
|
}
|
|
return await appendEvent(runId, level, message);
|
|
},
|
|
);
|
|
let runId = "";
|
|
try {
|
|
const startPromise = app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
},
|
|
});
|
|
await queuedEventStarted;
|
|
|
|
const unlock = await app.inject({
|
|
method: "POST",
|
|
url: "/catalog/description-generation-runs/unlock",
|
|
});
|
|
releaseQueuedEvent();
|
|
const start = await startPromise;
|
|
runId = start.json().id;
|
|
|
|
expect(start.statusCode).toBe(202);
|
|
expect(unlock.statusCode).toBe(409);
|
|
expect(unlock.json()).toEqual({
|
|
code: "description_generation_run_live",
|
|
message: "A local Description Generation worker or helper is still running.",
|
|
});
|
|
} finally {
|
|
releaseQueuedEvent();
|
|
if (runId) {
|
|
await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/description-generation-runs/${runId}/cancel`,
|
|
});
|
|
}
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("Unlock interrupts a stale active run and releases its stale local catalog reservation", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("stale Unlock must not invoke the model"); }),
|
|
};
|
|
const { app, repository, database, operations } = await setup(modelCompleter);
|
|
try {
|
|
await app.ready();
|
|
const stale = await repository.createDescriptionGenerationRun(
|
|
database.id,
|
|
"all",
|
|
configuredModel.id,
|
|
"it",
|
|
2,
|
|
);
|
|
await repository.updateDescriptionGenerationRun(stale.id, {
|
|
status: "running",
|
|
startedAt: "2026-08-28T08:00:00.000Z",
|
|
});
|
|
operations.reserve(database.id, "description_generation");
|
|
|
|
const unlock = await app.inject({
|
|
method: "POST",
|
|
url: "/catalog/description-generation-runs/unlock",
|
|
});
|
|
|
|
expect(unlock.statusCode).toBe(200);
|
|
expect(unlock.json()).toMatchObject({
|
|
id: stale.id,
|
|
databaseId: database.id,
|
|
status: "interrupted",
|
|
errorSummary: "Description generation was interrupted by Unlock.",
|
|
finishedAt: expect.any(String),
|
|
});
|
|
expect((await repository.listDescriptionGenerationEvents(stale.id)).at(-1)).toEqual(
|
|
expect.objectContaining({
|
|
level: "warning",
|
|
message: "Description generation was interrupted by Unlock.",
|
|
}),
|
|
);
|
|
const release = operations.reserve(database.id);
|
|
release();
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("Description Generation history is newest-first, bounded, and exposes only the safe run shape", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("history must not invoke the model"); }),
|
|
};
|
|
const { app, repository, database } = await setup(modelCompleter);
|
|
try {
|
|
const ids: string[] = [];
|
|
for (const status of ["completed", "cancelled", "failed"] as const) {
|
|
const run = await repository.createDescriptionGenerationRun(
|
|
database.id,
|
|
"missing",
|
|
configuredModel.id,
|
|
"it",
|
|
3,
|
|
);
|
|
ids.push(run.id);
|
|
await repository.updateDescriptionGenerationRun(run.id, {
|
|
status,
|
|
processed: status === "completed" ? 3 : 1,
|
|
generated: status === "completed" ? 3 : 1,
|
|
failed: status === "failed" ? 1 : 0,
|
|
finishedAt: new Date().toISOString(),
|
|
errorSummary: status === "failed" ? "Safe technical summary." : null,
|
|
...({
|
|
provider: "openai",
|
|
providerPayload: "raw-provider-payload",
|
|
apiKey: "history-secret-key",
|
|
prompt: "private persisted prompt",
|
|
sample: "private source sample",
|
|
stack: "private stack trace",
|
|
} as object),
|
|
} as never);
|
|
await new Promise((resolve) => setTimeout(resolve, 2));
|
|
}
|
|
|
|
const response = await app.inject({
|
|
method: "GET",
|
|
url: "/catalog/description-generation-runs?limit=2",
|
|
});
|
|
|
|
expect(response.statusCode).toBe(200);
|
|
expect(response.json().map((run: { id: string }) => run.id)).toEqual(ids.slice(1).reverse());
|
|
expect(Object.keys(response.json()[0]).sort()).toEqual([
|
|
"cacheReadTokens",
|
|
"createdAt",
|
|
"databaseId",
|
|
"errorSummary",
|
|
"failed",
|
|
"finishedAt",
|
|
"generated",
|
|
"id",
|
|
"inputTokens",
|
|
"language",
|
|
"modelId",
|
|
"nonGeneratable",
|
|
"outputTokens",
|
|
"processed",
|
|
"scope",
|
|
"startedAt",
|
|
"status",
|
|
"total",
|
|
"updatedAt",
|
|
]);
|
|
expect(response.body).not.toMatch(
|
|
/raw-provider-payload|history-secret-key|private persisted prompt|private source sample|private stack trace/,
|
|
);
|
|
expect((await app.inject({
|
|
method: "GET",
|
|
url: "/catalog/description-generation-runs?limit=0",
|
|
})).statusCode).toBe(400);
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("Description Generation SSE replays, subscribes, and reconnects in persisted sequence order", async () => {
|
|
let resolveCompletion!: (content: string) => void;
|
|
const completion = new Promise<string>((resolve) => { resolveCompletion = resolve; });
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => await completion),
|
|
};
|
|
const { app, database, column } = await setup(modelCompleter);
|
|
const controllers: AbortController[] = [];
|
|
try {
|
|
const baseUrl = await app.listen({ host: "127.0.0.1", port: 0 });
|
|
const accepted = await fetch(`${baseUrl}/catalog/databases/${database.id}/description-generation-runs`, {
|
|
method: "POST",
|
|
headers: { "content-type": "application/json" },
|
|
body: JSON.stringify({
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
}),
|
|
});
|
|
expect(accepted.status).toBe(202);
|
|
const runId = (await accepted.json() as { id: string }).id;
|
|
|
|
const firstController = new AbortController();
|
|
controllers.push(firstController);
|
|
const firstResponse = await fetch(
|
|
`${baseUrl}/catalog/description-generation-runs/${runId}/events?after=0`,
|
|
{ signal: firstController.signal },
|
|
);
|
|
expect(firstResponse.status).toBe(200);
|
|
expect(firstResponse.headers.get("content-type")).toBe("text/event-stream; charset=utf-8");
|
|
const firstReader = sseFrameReader(firstResponse);
|
|
const firstLogs: Array<{ sequence: number; message: string }> = [];
|
|
while (firstLogs.length < 2) {
|
|
const frame = await firstReader.next();
|
|
if (frame.event === "log") firstLogs.push(JSON.parse(frame.data));
|
|
}
|
|
expect(firstLogs.map((event) => event.sequence)).toEqual([1, 2]);
|
|
|
|
resolveCompletion(JSON.stringify({
|
|
results: [{
|
|
targetId: column.id,
|
|
outcome: "generated",
|
|
description: "Descrizione ricevuta via subscription.",
|
|
}],
|
|
}));
|
|
let terminalRun: { status: string } | undefined;
|
|
while (firstLogs.length < 4 || terminalRun?.status !== "completed") {
|
|
const frame = await firstReader.next();
|
|
if (frame.event === "log") firstLogs.push(JSON.parse(frame.data));
|
|
if (frame.event === "run") terminalRun = JSON.parse(frame.data);
|
|
}
|
|
expect(firstLogs.map((event) => event.sequence)).toEqual([1, 2, 3, 4]);
|
|
expect(firstLogs.at(-1)?.message).toBe("Description generation completed.");
|
|
expect(terminalRun).toMatchObject({ status: "completed" });
|
|
await firstReader.cancel();
|
|
firstController.abort();
|
|
|
|
const reconnectController = new AbortController();
|
|
controllers.push(reconnectController);
|
|
const reconnectResponse = await fetch(
|
|
`${baseUrl}/catalog/description-generation-runs/${runId}/events?after=2`,
|
|
{ signal: reconnectController.signal },
|
|
);
|
|
const reconnectReader = sseFrameReader(reconnectResponse);
|
|
const replayed: Array<{ sequence: number }> = [];
|
|
let replayedRun: { status: string } | undefined;
|
|
while (replayed.length < 2 || replayedRun?.status !== "completed") {
|
|
const frame = await reconnectReader.next();
|
|
if (frame.event === "log") replayed.push(JSON.parse(frame.data));
|
|
if (frame.event === "run") replayedRun = JSON.parse(frame.data);
|
|
}
|
|
expect(replayed.map((event) => event.sequence)).toEqual([3, 4]);
|
|
expect(replayedRun).toMatchObject({ status: "completed" });
|
|
await reconnectReader.cancel();
|
|
reconnectController.abort();
|
|
} finally {
|
|
controllers.forEach((controller) => controller.abort());
|
|
resolveCompletion(JSON.stringify({ results: [] }));
|
|
await app.close();
|
|
}
|
|
}, 10_000);
|
|
|
|
test("SSE exposes a terminal run signal for every outcome that may leave partial catalog writes", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("terminal replay must not invoke the model"); }),
|
|
};
|
|
const { app, repository, database } = await setup(modelCompleter);
|
|
try {
|
|
const baseUrl = await app.listen({ host: "127.0.0.1", port: 0 });
|
|
for (const status of [
|
|
"completed",
|
|
"completed_with_errors",
|
|
"cancelled",
|
|
"failed",
|
|
"interrupted",
|
|
] as const) {
|
|
const created = await repository.createDescriptionGenerationRun(
|
|
database.id,
|
|
"missing",
|
|
configuredModel.id,
|
|
"it",
|
|
1,
|
|
);
|
|
await repository.updateDescriptionGenerationRun(created.id, {
|
|
status,
|
|
processed: status === "cancelled" || status === "interrupted" ? 0 : 1,
|
|
generated: status === "completed" ? 1 : 0,
|
|
failed: status === "completed_with_errors" || status === "failed" ? 1 : 0,
|
|
finishedAt: new Date().toISOString(),
|
|
errorSummary: status === "completed" || status === "cancelled" ? null : `Safe ${status} summary.`,
|
|
});
|
|
await repository.appendDescriptionGenerationEvent(created.id, "warning", `Terminal ${status}.`);
|
|
|
|
const response = await fetch(
|
|
`${baseUrl}/catalog/description-generation-runs/${created.id}/events?after=0`,
|
|
);
|
|
const reader = sseFrameReader(response);
|
|
let observed: { databaseId: string; status: string } | undefined;
|
|
while (!observed) {
|
|
const frame = await reader.next();
|
|
if (frame.event === "run") observed = JSON.parse(frame.data);
|
|
}
|
|
expect(observed).toMatchObject({ databaseId: database.id, status });
|
|
}
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("emits one safe metadata-only warning for each batch whose optional sampling fails", async () => {
|
|
const samplingFailureSecret = "BATCH_SAMPLE_FAILURE_DETAIL_d0139a";
|
|
const descriptionSourceSampler: DescriptionSourceSampler = {
|
|
sample: vi.fn(async () => { throw new Error(samplingFailureSecret); }),
|
|
};
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
const userContent = request.messages.find((message) => message.role === "user")!.content;
|
|
const context = JSON.parse(userContent.slice(userContent.indexOf("\n") + 1));
|
|
return JSON.stringify({
|
|
results: context.targets.map((target: { targetId: string }) => ({
|
|
targetId: target.targetId,
|
|
outcome: "generated",
|
|
description: "Descrizione metadata-only.",
|
|
})),
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database, table } = await setup(
|
|
modelCompleter,
|
|
{},
|
|
"it",
|
|
descriptionSourceSampler,
|
|
);
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: null }],
|
|
columns: Array.from({ length: 11 }, (_, index) => ({
|
|
tableName: "patients",
|
|
name: `column_${index}`,
|
|
ordinalPosition: index + 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const columnIds = (await repository.listColumns(database.id, table.id)).map((column) => column.id);
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: columnIds,
|
|
},
|
|
});
|
|
expect(start.statusCode).toBe(202);
|
|
expect((await waitForTerminalRun(app, start.json().id)).run.status).toBe("completed");
|
|
|
|
expect(descriptionSourceSampler.sample).toHaveBeenCalledTimes(2);
|
|
const events = await repository.listDescriptionGenerationEvents(start.json().id);
|
|
expect(events.filter((event) => event.level === "warning").map((event) => event.message)).toEqual([
|
|
"Source samples unavailable for this batch; generation continued with catalog metadata only.",
|
|
"Source samples unavailable for this batch; generation continued with catalog metadata only.",
|
|
]);
|
|
expect(JSON.stringify(events)).not.toContain(samplingFailureSecret);
|
|
expect(vi.mocked(modelCompleter.complete).mock.calls.every(([request]) => (
|
|
!JSON.stringify(request.messages).includes(samplingFailureSecret)
|
|
))).toBe(true);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("retains completed batch writes when a later batch response is malformed", async () => {
|
|
let orderedIds: string[] = [];
|
|
let callIndex = 0;
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => {
|
|
const currentCall = callIndex;
|
|
callIndex += 1;
|
|
if (currentCall === 0) {
|
|
return JSON.stringify({
|
|
results: orderedIds.slice(0, 10).map((targetId, index) => ({
|
|
targetId,
|
|
outcome: "generated",
|
|
description: `Retained ${index + 1}`,
|
|
})),
|
|
});
|
|
}
|
|
return JSON.stringify({
|
|
results: [{
|
|
targetId: orderedIds[10],
|
|
outcome: "generated",
|
|
description: " ",
|
|
}],
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database, table } = await setup(modelCompleter);
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: null }],
|
|
columns: Array.from({ length: 11 }, (_, index) => ({
|
|
tableName: "patients",
|
|
name: `retained_${String(index + 1).padStart(2, "0")}`,
|
|
ordinalPosition: index + 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const originalColumns = await repository.listColumns(database.id, table.id);
|
|
orderedIds = originalColumns.map((column) => column.id);
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: orderedIds,
|
|
},
|
|
});
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(3);
|
|
expect(run).toMatchObject({
|
|
status: "completed_with_errors",
|
|
total: 11,
|
|
processed: 11,
|
|
generated: 10,
|
|
nonGeneratable: 0,
|
|
failed: 1,
|
|
errorSummary: "Description generation completed with errors.",
|
|
});
|
|
const updatedById = new Map(
|
|
(await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]),
|
|
);
|
|
orderedIds.slice(0, 10).forEach((targetId, index) => {
|
|
expect(updatedById.get(targetId)).toMatchObject({
|
|
generatedDescription: `Retained ${index + 1}`,
|
|
version: originalColumns[index]!.version + 1,
|
|
});
|
|
});
|
|
expect(updatedById.get(orderedIds[10]!)).toMatchObject({
|
|
generatedDescription: null,
|
|
version: originalColumns[10]!.version,
|
|
});
|
|
const events = await repository.listDescriptionGenerationEvents(run.id);
|
|
expect(events.slice(2, 12).map((event) => event.message)).toEqual(
|
|
orderedIds.slice(0, 10).map((targetId) => `Generated description for Catalog Column ${targetId}.`),
|
|
);
|
|
expect(events.find((event) => event.level === "warning" && event.message.includes("Retrying batch"))).toEqual(
|
|
expect.objectContaining({
|
|
message: "The model response did not match the required schema. Retrying batch (attempt 2 of 2).",
|
|
}),
|
|
);
|
|
expect(events.find((event) => event.level === "error")).toEqual(expect.objectContaining({
|
|
level: "error",
|
|
message: `The model response did not match the required schema. Affected Catalog Column target: ${orderedIds[10]}.`,
|
|
}));
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("rejects duplicate selected target IDs before launching generation", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("must not be called"); }),
|
|
};
|
|
const { app, database, column, operations } = await setup(modelCompleter);
|
|
try {
|
|
const response = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id, column.id],
|
|
},
|
|
});
|
|
|
|
expect(response.statusCode).toBe(400);
|
|
expect(response.json()).toEqual({
|
|
code: "description_generation_target_ids_duplicate",
|
|
message: "Description generation target IDs must be unique.",
|
|
});
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
const release = operations.reserve(database.id);
|
|
release();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("requires strict database-wide scopes and rejects zero eligible targets before creating a run", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("must not be called"); }),
|
|
};
|
|
const { app, repository, database, column, operations } = await setup(modelCompleter);
|
|
try {
|
|
const unexpectedTargetIds = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "all", targetIds: [column.id] },
|
|
});
|
|
expect(unexpectedTargetIds.statusCode).toBe(400);
|
|
expect(unexpectedTargetIds.json()).toEqual({
|
|
code: "description_generation_request_invalid",
|
|
message: "Description generation request is invalid.",
|
|
});
|
|
|
|
await repository.deleteDatabaseMetadata([database.id], "tables");
|
|
for (const scope of ["all", "missing"] as const) {
|
|
const response = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope },
|
|
});
|
|
expect(response.statusCode).toBe(409);
|
|
expect(response.json()).toEqual({
|
|
code: "description_generation_no_eligible_targets",
|
|
message: scope === "all"
|
|
? "No Catalog Tables or Catalog Columns are available for description generation."
|
|
: "No Catalog Tables or Catalog Columns have a missing Generated Description.",
|
|
});
|
|
}
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
const release = operations.reserve(database.id);
|
|
release();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("Generate All replaces every description in column-first homogeneous batches with refreshed table context", async () => {
|
|
const requests: ModelCompletionRequest[] = [];
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
requests.push(request);
|
|
const metadata = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
const kind = metadata.targets.every((target: Record<string, unknown>) => "column" in target)
|
|
? "column"
|
|
: metadata.targets.every((target: Record<string, unknown>) => "columns" in target)
|
|
? "table"
|
|
: "mixed";
|
|
return JSON.stringify({
|
|
results: metadata.targets.map((target: { targetId: string }) => ({
|
|
targetId: target.targetId,
|
|
outcome: "generated",
|
|
description: `Generated ${kind} ${target.targetId}`,
|
|
})),
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database } = await setup(modelCompleter);
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: Array.from({ length: 11 }, (_, index) => ({
|
|
name: `table_${String(index + 1).padStart(2, "0")}`,
|
|
sourceComment: `Table ${index + 1}`,
|
|
})),
|
|
columns: Array.from({ length: 11 }, (_, index) => ({
|
|
tableName: `table_${String(index + 1).padStart(2, "0")}`,
|
|
name: `column_${String(index + 1).padStart(2, "0")}`,
|
|
ordinalPosition: 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: `Column ${index + 1}`,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const originalTables = await repository.listTables(database.id);
|
|
const originalColumns = (await Promise.all(
|
|
originalTables.map(async (table) => await repository.listColumns(database.id, table.id)),
|
|
)).flat();
|
|
await repository.updateTableMetadata(
|
|
database.id,
|
|
originalTables[0]!.id,
|
|
originalTables[0]!.version,
|
|
originalTables[0]!.description,
|
|
"Existing generated table description",
|
|
);
|
|
await repository.updateColumnMetadata(
|
|
database.id,
|
|
originalTables[0]!.id,
|
|
originalColumns[0]!.id,
|
|
originalColumns[0]!.version,
|
|
originalColumns[0]!.description,
|
|
"Existing generated column description",
|
|
);
|
|
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "all" },
|
|
});
|
|
expect(start.statusCode).toBe(202);
|
|
expect(start.json()).toMatchObject({ scope: "all", total: 22 });
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
expect(run).toMatchObject({
|
|
scope: "all",
|
|
status: "completed",
|
|
total: 22,
|
|
processed: 22,
|
|
generated: 22,
|
|
nonGeneratable: 0,
|
|
failed: 0,
|
|
});
|
|
|
|
expect(requests).toHaveLength(4);
|
|
const requestMetadata = requests.map((request) => ({
|
|
system: request.messages[0]!.content,
|
|
metadata: JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")),
|
|
}));
|
|
expect(requestMetadata.map(({ metadata }) => metadata.targets.length)).toEqual([10, 1, 10, 1]);
|
|
expect(requestMetadata.slice(0, 2).every(({ system, metadata }) => (
|
|
system.includes("Catalog Column")
|
|
&& metadata.targets.every((target: Record<string, unknown>) => "column" in target && !("columns" in target))
|
|
))).toBe(true);
|
|
expect(requestMetadata.slice(2).every(({ system, metadata }) => (
|
|
system.includes("Catalog Table")
|
|
&& metadata.targets.every((target: Record<string, unknown>) => "columns" in target && !("column" in target))
|
|
))).toBe(true);
|
|
expect(requestMetadata.slice(0, 2).flatMap(({ metadata }) => (
|
|
metadata.targets.map((target: { targetId: string }) => target.targetId)
|
|
))).toEqual(originalColumns.map((column) => column.id));
|
|
expect(requestMetadata.slice(2).flatMap(({ metadata }) => (
|
|
metadata.targets.map((target: { targetId: string }) => target.targetId)
|
|
))).toEqual(originalTables.map((table) => table.id));
|
|
|
|
const columnByTable = new Map(originalColumns.map((column) => [column.tableId, column]));
|
|
for (const target of requestMetadata.slice(2).flatMap(({ metadata }) => metadata.targets)) {
|
|
const column = columnByTable.get(target.targetId)!;
|
|
expect(target.columns[0].currentDescription).toBe(`Generated column ${column.id}`);
|
|
}
|
|
const updatedTables = await repository.listTables(database.id);
|
|
const updatedColumns = (await Promise.all(
|
|
updatedTables.map(async (table) => await repository.listColumns(database.id, table.id)),
|
|
)).flat();
|
|
expect(updatedTables.map((table) => table.generatedDescription)).toEqual(
|
|
updatedTables.map((table) => `Generated table ${table.id}`),
|
|
);
|
|
expect(updatedColumns.map((column) => column.generatedDescription)).toEqual(
|
|
updatedColumns.map((column) => `Generated column ${column.id}`),
|
|
);
|
|
expect(updatedTables[0]!.generatedDescription).not.toBe("Existing generated table description");
|
|
expect(updatedColumns[0]!.generatedDescription).not.toBe("Existing generated column description");
|
|
|
|
const events = await repository.listDescriptionGenerationEvents(run.id);
|
|
expect(events[0]!.message).toBe("Description generation queued (scope: all).");
|
|
expect(events[1]!.message).toBe("Description generation started (scope: all).");
|
|
expect(events.at(-1)!.message).toBe("Description generation completed (scope: all).");
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("Generate Missing skips prior partial results and includes null, empty, and whitespace descriptions", async () => {
|
|
let mode: "partial" | "missing" = "partial";
|
|
let partialCall = 0;
|
|
const missingRequests: ModelCompletionRequest[] = [];
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
const metadata = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
if (mode === "partial") {
|
|
partialCall += 1;
|
|
if (partialCall === 2 || partialCall === 3) throw new ModelCompletionProviderError();
|
|
return JSON.stringify({
|
|
results: metadata.targets.map((target: { targetId: string }) => ({
|
|
targetId: target.targetId,
|
|
outcome: "generated",
|
|
description: `Retained partial ${target.targetId}`,
|
|
})),
|
|
});
|
|
}
|
|
missingRequests.push(request);
|
|
return JSON.stringify({
|
|
results: metadata.targets.map((target: { targetId: string }) => ({
|
|
targetId: target.targetId,
|
|
outcome: "generated",
|
|
description: `Recovered missing ${target.targetId}`,
|
|
})),
|
|
});
|
|
}),
|
|
};
|
|
const { app, repository, database } = await setup(modelCompleter);
|
|
try {
|
|
const tableNames = ["table_01", "table_02", "table_03"];
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: tableNames.map((name) => ({ name, sourceComment: null })),
|
|
columns: Array.from({ length: 11 }, (_, index) => ({
|
|
tableName: index < 9 ? tableNames[0]! : tableNames[index - 8]!,
|
|
name: `column_${String(index + 1).padStart(2, "0")}`,
|
|
ordinalPosition: index < 9 ? index + 1 : 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
})),
|
|
relationships: [],
|
|
});
|
|
const tables = await repository.listTables(database.id);
|
|
const columns = (await Promise.all(
|
|
tables.map(async (table) => await repository.listColumns(database.id, table.id)),
|
|
)).flat();
|
|
const seededTables = [
|
|
await repository.updateTableMetadata(
|
|
database.id, tables[0]!.id, tables[0]!.version, null, "Existing valid table result",
|
|
),
|
|
await repository.updateTableMetadata(
|
|
database.id, tables[1]!.id, tables[1]!.version, null, " ",
|
|
),
|
|
await repository.updateTableMetadata(
|
|
database.id, tables[2]!.id, tables[2]!.version, null, "",
|
|
),
|
|
];
|
|
|
|
const partialStart = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: columns.map((column) => column.id),
|
|
},
|
|
});
|
|
const { run: partialRun } = await waitForTerminalRun(app, partialStart.json().id);
|
|
expect(partialRun).toMatchObject({
|
|
status: "completed_with_errors",
|
|
total: 11,
|
|
processed: 11,
|
|
generated: 10,
|
|
failed: 1,
|
|
});
|
|
const afterPartial = (await Promise.all(
|
|
tables.map(async (table) => await repository.listColumns(database.id, table.id)),
|
|
)).flat();
|
|
expect(afterPartial.slice(0, 10).every((column) => (
|
|
column.generatedDescription === `Retained partial ${column.id}`
|
|
))).toBe(true);
|
|
expect(afterPartial[10]!.generatedDescription).toBeNull();
|
|
|
|
mode = "missing";
|
|
const missingStart = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "missing" },
|
|
});
|
|
expect(missingStart.statusCode).toBe(202);
|
|
expect(missingStart.json()).toMatchObject({ scope: "missing", total: 3 });
|
|
const { run: missingRun } = await waitForTerminalRun(app, missingStart.json().id);
|
|
expect(missingRun).toMatchObject({
|
|
scope: "missing",
|
|
status: "completed",
|
|
total: 3,
|
|
processed: 3,
|
|
generated: 3,
|
|
nonGeneratable: 0,
|
|
failed: 0,
|
|
});
|
|
|
|
expect(missingRequests).toHaveLength(2);
|
|
const missingMetadata = missingRequests.map((request) => (
|
|
JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"))
|
|
));
|
|
expect(missingMetadata[0]!.targets.map((target: { targetId: string }) => target.targetId)).toEqual([
|
|
afterPartial[10]!.id,
|
|
]);
|
|
expect(missingMetadata[0]!.targets.every((target: Record<string, unknown>) => "column" in target)).toBe(true);
|
|
expect(missingMetadata[1]!.targets.map((target: { targetId: string }) => target.targetId)).toEqual([
|
|
tables[1]!.id,
|
|
tables[2]!.id,
|
|
]);
|
|
expect(missingMetadata[1]!.targets.every((target: Record<string, unknown>) => "columns" in target)).toBe(true);
|
|
const recoveredTableThree = missingMetadata[1]!.targets.find(
|
|
(target: { targetId: string }) => target.targetId === tables[2]!.id,
|
|
);
|
|
expect(recoveredTableThree.columns[0].currentDescription).toBe(
|
|
`Recovered missing ${afterPartial[10]!.id}`,
|
|
);
|
|
|
|
const finalColumns = (await Promise.all(
|
|
tables.map(async (table) => await repository.listColumns(database.id, table.id)),
|
|
)).flat();
|
|
expect(finalColumns.slice(0, 10)).toEqual(afterPartial.slice(0, 10));
|
|
expect(finalColumns[10]!.generatedDescription).toBe(`Recovered missing ${afterPartial[10]!.id}`);
|
|
const finalTables = await repository.listTables(database.id);
|
|
expect(finalTables[0]).toEqual(seededTables[0]);
|
|
expect(finalTables[1]!.generatedDescription).toBe(`Recovered missing ${tables[1]!.id}`);
|
|
expect(finalTables[2]!.generatedDescription).toBe(`Recovered missing ${tables[2]!.id}`);
|
|
|
|
const events = await repository.listDescriptionGenerationEvents(missingRun.id);
|
|
expect(events[0]!.message).toBe("Description generation queued (scope: missing).");
|
|
expect(events.at(-1)!.message).toBe("Description generation completed (scope: missing).");
|
|
|
|
const noMissingTargets = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "missing" },
|
|
});
|
|
expect(noMissingTargets.statusCode).toBe(409);
|
|
expect(noMissingTargets.json()).toEqual({
|
|
code: "description_generation_no_eligible_targets",
|
|
message: "No Catalog Tables or Catalog Columns have a missing Generated Description.",
|
|
});
|
|
expect(missingRequests).toHaveLength(2);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("generates selected Catalog Tables with structural column context and localized non-generatable text", async () => {
|
|
let selectedTableId = "";
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => JSON.stringify({
|
|
results: [{ targetId: selectedTableId, outcome: "non_generatable" }],
|
|
})),
|
|
};
|
|
const { app, repository, database, table } = await setup(modelCompleter);
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: "Clinical patients" }],
|
|
columns: [
|
|
{
|
|
tableName: "patients",
|
|
name: "patient_id",
|
|
ordinalPosition: 1,
|
|
dataType: "uuid",
|
|
isNullable: false,
|
|
defaultExpression: "gen_random_uuid()",
|
|
primaryKeyPosition: 1,
|
|
sourceComment: "Patient identifier",
|
|
},
|
|
{
|
|
tableName: "patients",
|
|
name: "status",
|
|
ordinalPosition: 2,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: "Patient status",
|
|
},
|
|
],
|
|
relationships: [],
|
|
});
|
|
const currentTable = (await repository.getTable(database.id, table.id))!;
|
|
const curatedTable = (await repository.updateTableMetadata(
|
|
database.id,
|
|
currentTable.id,
|
|
currentTable.version,
|
|
"Elenco curato dei pazienti.",
|
|
null,
|
|
))!;
|
|
const columns = await repository.listColumns(database.id, table.id);
|
|
const patientId = columns.find((column) => column.name === "patient_id")!;
|
|
const status = columns.find((column) => column.name === "status")!;
|
|
await repository.updateColumnMetadata(
|
|
database.id,
|
|
table.id,
|
|
patientId.id,
|
|
patientId.version,
|
|
"Identificativo curato.",
|
|
"Identificativo generato.",
|
|
);
|
|
await repository.updateColumnMetadata(
|
|
database.id,
|
|
table.id,
|
|
status.id,
|
|
status.version,
|
|
"Stato curato.",
|
|
null,
|
|
);
|
|
selectedTableId = table.id;
|
|
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_tables",
|
|
targetIds: [table.id],
|
|
},
|
|
});
|
|
|
|
expect(start.statusCode).toBe(202);
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
expect(run).toMatchObject({
|
|
scope: "selected_tables",
|
|
language: "it",
|
|
status: "completed",
|
|
total: 1,
|
|
processed: 1,
|
|
generated: 0,
|
|
nonGeneratable: 1,
|
|
failed: 0,
|
|
});
|
|
const request = (modelCompleter.complete as ReturnType<typeof vi.fn>).mock.calls[0]![0] as ModelCompletionRequest;
|
|
expect(request.messages[0]!.content).toContain("Catalog Table");
|
|
expect(request.messages[0]!.content).toContain('workspace language "it"');
|
|
expect(request.messages[0]!.content).toContain("non_generatable");
|
|
expect(request.messages[0]!.content).toContain("untrusted");
|
|
const metadata = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
expect(metadata.targets).toEqual([expect.objectContaining({
|
|
targetId: table.id,
|
|
table: expect.objectContaining({
|
|
name: "patients",
|
|
sourceComment: "Clinical patients",
|
|
description: "Elenco curato dei pazienti.",
|
|
generatedDescription: null,
|
|
}),
|
|
columns: [
|
|
expect.objectContaining({
|
|
name: "patient_id",
|
|
ordinalPosition: 1,
|
|
dataType: "uuid",
|
|
isNullable: false,
|
|
defaultExpression: "gen_random_uuid()",
|
|
isPrimaryKey: true,
|
|
currentDescription: "Identificativo generato.",
|
|
}),
|
|
expect.objectContaining({
|
|
name: "status",
|
|
ordinalPosition: 2,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
currentDescription: "Stato curato.",
|
|
}),
|
|
],
|
|
})]);
|
|
expect(await repository.getTable(database.id, table.id)).toMatchObject({
|
|
description: "Elenco curato dei pazienti.",
|
|
generatedDescription: "Non generabile",
|
|
version: curatedTable.version + 1,
|
|
});
|
|
expect((await repository.listDescriptionGenerationEvents(run.id)).map((event) => event.message)).toEqual([
|
|
"Description generation queued.",
|
|
"Description generation started.",
|
|
`Stored non-generatable result for Catalog Table ${table.id}.`,
|
|
"Description generation completed.",
|
|
]);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("localizes valid non-generatable Catalog Column results in English", async () => {
|
|
let selectedColumnId = "";
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => JSON.stringify({
|
|
results: [{ targetId: selectedColumnId, outcome: "non_generatable" }],
|
|
})),
|
|
};
|
|
const { app, repository, database, table, column } = await setup(modelCompleter, {}, "en");
|
|
selectedColumnId = column.id;
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_columns",
|
|
targetIds: [column.id],
|
|
},
|
|
});
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(run).toMatchObject({
|
|
language: "en",
|
|
status: "completed",
|
|
processed: 1,
|
|
generated: 0,
|
|
nonGeneratable: 1,
|
|
failed: 0,
|
|
});
|
|
expect(await repository.getColumn(database.id, table.id, column.id)).toMatchObject({
|
|
description: null,
|
|
generatedDescription: "Not generatable",
|
|
version: column.version + 1,
|
|
});
|
|
const request = (modelCompleter.complete as ReturnType<typeof vi.fn>).mock.calls[0]![0] as ModelCompletionRequest;
|
|
expect(request.messages[0]!.content).toContain('workspace language "en"');
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("retries invalid JSON once and completes the batch when the second response is valid", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
if (vi.mocked(modelCompleter.complete).mock.calls.length === 1) {
|
|
return { content: "not-json", usage: { input: 11, cacheRead: 3, output: 2 } };
|
|
}
|
|
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
|
|
return {
|
|
content: JSON.stringify({
|
|
results: context.targets.map((target: { targetId: string }) => ({
|
|
targetId: target.targetId,
|
|
outcome: "generated",
|
|
description: "Generated after the application retry.",
|
|
})),
|
|
}),
|
|
usage: { input: 7, cacheRead: 1, output: 5 },
|
|
};
|
|
}),
|
|
};
|
|
const { app, repository, database, table, column } = await setup(modelCompleter);
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] },
|
|
});
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(run).toMatchObject({
|
|
status: "completed",
|
|
processed: 1,
|
|
generated: 1,
|
|
failed: 0,
|
|
inputTokens: 18,
|
|
cacheReadTokens: 4,
|
|
outputTokens: 7,
|
|
});
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
|
|
expect(await repository.getColumn(database.id, table.id, column.id)).toMatchObject({
|
|
generatedDescription: "Generated after the application retry.",
|
|
});
|
|
expect(await repository.listDescriptionGenerationEvents(run.id)).toEqual(expect.arrayContaining([
|
|
expect.objectContaining({
|
|
level: "warning",
|
|
message: "The model response was not valid JSON. Retrying batch (attempt 2 of 2).",
|
|
}),
|
|
]));
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("fails safely when the provider fails and redacts provider diagnostics", async () => {
|
|
const sensitiveDiagnostic = "test-provider-secret private prompt raw provider payload";
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => {
|
|
throw Object.assign(new ModelCompletionProviderError(), { message: sensitiveDiagnostic });
|
|
}),
|
|
};
|
|
const { app, database, table, column } = await setup(modelCompleter);
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] },
|
|
});
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
|
|
expect(run).toMatchObject({
|
|
status: "completed_with_errors",
|
|
processed: 1,
|
|
generated: 0,
|
|
nonGeneratable: 0,
|
|
failed: 1,
|
|
errorSummary: "Description generation completed with errors.",
|
|
});
|
|
const columns = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/databases/${database.id}/tables/${table.id}/columns`,
|
|
});
|
|
expect(columns.json()).toEqual([
|
|
expect.objectContaining({
|
|
id: column.id,
|
|
generatedDescription: null,
|
|
version: column.version,
|
|
}),
|
|
]);
|
|
const events = await app.inject({
|
|
method: "GET",
|
|
url: `/catalog/description-generation-runs/${run.id}/events-list`,
|
|
});
|
|
expect(`${JSON.stringify(run)}${events.body}`).not.toContain(sensitiveDiagnostic);
|
|
expect(events.json().find((event: { level: string }) => event.level === "error")).toEqual(expect.objectContaining({
|
|
level: "error",
|
|
message: `The model provider request failed. Affected Catalog Column target: ${column.id}.`,
|
|
}));
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test.each([
|
|
["duplicate mappings", (targetIds: readonly string[]) => ({ results: [
|
|
{ targetId: targetIds[0], outcome: "generated", description: "First valid value" },
|
|
{ targetId: targetIds[0], outcome: "non_generatable" },
|
|
] }), "The model response was missing one or more requested targets."],
|
|
["unknown mappings", (targetIds: readonly string[]) => ({ results: [
|
|
{ targetId: targetIds[0], outcome: "non_generatable" },
|
|
{
|
|
targetId: "99999999-9999-4999-8999-999999999999",
|
|
outcome: "generated",
|
|
description: "Unknown target value",
|
|
},
|
|
] }), "The model response was missing one or more requested targets."],
|
|
["missing mappings", (targetIds: readonly string[]) => ({ results: [
|
|
{ targetId: targetIds[0], outcome: "generated", description: "Only one result" },
|
|
] }), "The model response was missing one or more requested targets."],
|
|
["malformed mappings", (targetIds: readonly string[]) => ({ results: [
|
|
{ targetId: targetIds[0], outcome: "generated", description: "First valid value" },
|
|
{ targetId: targetIds[1], outcome: "generated", description: " " },
|
|
] }), "The model response did not match the required schema."],
|
|
] as const)("rejects %s without applying any result from the batch", async (_name, responseFor, failureMessage) => {
|
|
let selectedColumnIds: string[] = [];
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => JSON.stringify(responseFor(selectedColumnIds))),
|
|
};
|
|
const { app, repository, database, table } = await setup(modelCompleter);
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: null }],
|
|
columns: [
|
|
{
|
|
tableName: "patients",
|
|
name: "first_column",
|
|
ordinalPosition: 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
},
|
|
{
|
|
tableName: "patients",
|
|
name: "second_column",
|
|
ordinalPosition: 2,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
},
|
|
],
|
|
relationships: [],
|
|
});
|
|
const originalColumns = await repository.listColumns(database.id, table.id);
|
|
selectedColumnIds = originalColumns.map((column) => column.id);
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: selectedColumnIds },
|
|
});
|
|
const { run } = await waitForTerminalRun(app, start.json().id);
|
|
|
|
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
|
|
expect(run).toMatchObject({
|
|
status: "completed_with_errors",
|
|
processed: 2,
|
|
generated: 0,
|
|
nonGeneratable: 0,
|
|
failed: 2,
|
|
errorSummary: "Description generation completed with errors.",
|
|
});
|
|
expect(await repository.listColumns(database.id, table.id)).toEqual(
|
|
originalColumns.map((column) => expect.objectContaining({
|
|
id: column.id,
|
|
description: column.description,
|
|
generatedDescription: null,
|
|
version: column.version,
|
|
})),
|
|
);
|
|
expect((await repository.listDescriptionGenerationEvents(run.id)).find((event) => event.level === "error")).toEqual(
|
|
expect.objectContaining({
|
|
level: "error",
|
|
message: `${failureMessage} Affected Catalog Column targets: ${selectedColumnIds.join(", ")}.`,
|
|
}),
|
|
);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("returns an explicit global conflict while another run is active", async () => {
|
|
let resolveCompletion!: (content: string) => void;
|
|
const completion = new Promise<string>((resolve) => { resolveCompletion = resolve; });
|
|
const modelCompleter = { complete: vi.fn(async () => await completion) };
|
|
const { app, database, column } = await setup(modelCompleter);
|
|
let firstRunId = "";
|
|
try {
|
|
const first = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] },
|
|
});
|
|
firstRunId = first.json().id;
|
|
const second = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "missing" },
|
|
});
|
|
|
|
expect(second.statusCode).toBe(409);
|
|
expect(second.json()).toEqual({
|
|
code: "description_generation_run_active",
|
|
message: "A Description Generation Run is already active.",
|
|
});
|
|
} finally {
|
|
resolveCompletion(JSON.stringify({
|
|
targetId: column.id,
|
|
outcome: "generated",
|
|
description: "Descrizione valida.",
|
|
}));
|
|
if (firstRunId) await waitForTerminalRun(app, firstRunId);
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("rejects generation while another catalog operation owns the database", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("must not be called"); }),
|
|
};
|
|
const { app, database, operations } = await setup(modelCompleter);
|
|
const release = operations.reserve(database.id);
|
|
try {
|
|
const response = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "all" },
|
|
});
|
|
|
|
expect(response.statusCode).toBe(409);
|
|
expect(response.json()).toEqual({
|
|
code: "database_operation_in_progress",
|
|
message: "A database operation is already in progress.",
|
|
});
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
} finally {
|
|
release();
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("requires database.manage for every Description Generation route", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("must not be called"); }),
|
|
};
|
|
const { app, database } = await setup(modelCompleter, { AUTH_MODE: "upstream" });
|
|
const headers = {
|
|
"x-thoth-principal-issuer": "portal",
|
|
"x-thoth-principal-subject": "catalog-reader",
|
|
"x-thoth-is-admin": "0",
|
|
};
|
|
try {
|
|
const responses = await Promise.all([
|
|
app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
headers,
|
|
payload: { modelId: configuredModel.id, scope: "all" },
|
|
}),
|
|
app.inject({
|
|
method: "GET",
|
|
url: "/catalog/description-generation-runs/99999999-9999-4999-8999-999999999999",
|
|
headers,
|
|
}),
|
|
app.inject({
|
|
method: "GET",
|
|
url: "/catalog/description-generation-runs/99999999-9999-4999-8999-999999999999/events-list",
|
|
headers,
|
|
}),
|
|
app.inject({
|
|
method: "GET",
|
|
url: "/catalog/description-generation-runs",
|
|
headers,
|
|
}),
|
|
app.inject({
|
|
method: "POST",
|
|
url: "/catalog/description-generation-runs/99999999-9999-4999-8999-999999999999/cancel",
|
|
headers,
|
|
}),
|
|
app.inject({
|
|
method: "POST",
|
|
url: "/catalog/description-generation-runs/unlock",
|
|
headers,
|
|
}),
|
|
app.inject({
|
|
method: "GET",
|
|
url: "/catalog/description-generation-runs/99999999-9999-4999-8999-999999999999/events",
|
|
headers,
|
|
}),
|
|
app.inject({
|
|
method: "GET",
|
|
url: "/catalog/sensitive-data-suggestion-runs",
|
|
headers,
|
|
}),
|
|
app.inject({
|
|
method: "GET",
|
|
url: "/catalog/sensitive-data-suggestion-runs/99999999-9999-4999-8999-999999999999",
|
|
headers,
|
|
}),
|
|
app.inject({
|
|
method: "GET",
|
|
url: "/catalog/sensitive-data-suggestion-runs/99999999-9999-4999-8999-999999999999/events-list",
|
|
headers,
|
|
}),
|
|
]);
|
|
|
|
expect(responses.map((response) => response.statusCode)).toEqual([
|
|
403, 403, 403, 403, 403, 403, 403, 403, 403, 403,
|
|
]);
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("validates selected targets and resolves every requested target before launching", async () => {
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async () => { throw new Error("must not be called"); }),
|
|
};
|
|
const { app, repository, database, column, operations } = await setup(modelCompleter);
|
|
try {
|
|
const foreignDatabase = await repository.create({
|
|
workspaceId: "other-workspace",
|
|
engine: "postgres",
|
|
databaseName: "other",
|
|
schema: "public",
|
|
binding: { transport: "postgres_direct", host: "other.internal", port: 5432, username: "reader" },
|
|
});
|
|
await repository.applySchemaSync(foreignDatabase.id, foreignDatabase.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "foreign_table", sourceComment: null }],
|
|
columns: [{
|
|
tableName: "foreign_table",
|
|
name: "foreign_column",
|
|
ordinalPosition: 1,
|
|
dataType: "text",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: null,
|
|
}],
|
|
relationships: [],
|
|
});
|
|
const foreignTable = (await repository.listTables(foreignDatabase.id))[0]!;
|
|
const foreignColumn = (await repository.listColumns(foreignDatabase.id, foreignTable.id))[0]!;
|
|
const invalidPayloads = await Promise.all([
|
|
app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [] },
|
|
}),
|
|
app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id, foreignColumn.id] },
|
|
}),
|
|
]);
|
|
expect(invalidPayloads.map((response) => response.statusCode)).toEqual([400, 404]);
|
|
expect(invalidPayloads[1]!.json()).toEqual({
|
|
code: "catalog_column_not_found",
|
|
message: "One or more selected Catalog Columns were not found.",
|
|
});
|
|
|
|
const unknownModel = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: "openai/unknown-model", scope: "selected_columns", targetIds: [column.id] },
|
|
});
|
|
expect(unknownModel.statusCode).toBe(409);
|
|
expect(unknownModel.json().code).toBe("metadata_generation_model_unavailable");
|
|
|
|
const wrongDatabase = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [foreignColumn.id] },
|
|
});
|
|
expect(wrongDatabase.statusCode).toBe(404);
|
|
expect(wrongDatabase.json().code).toBe("catalog_column_not_found");
|
|
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
|
const release = operations.reserve(database.id);
|
|
release();
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("blocks table and column metadata edits while generation owns the database", async () => {
|
|
let resolveCompletion!: (content: string) => void;
|
|
const completion = new Promise<string>((resolve) => { resolveCompletion = resolve; });
|
|
const modelCompleter = { complete: vi.fn(async () => await completion) };
|
|
const { app, database, table, column } = await setup(modelCompleter);
|
|
let runId = "";
|
|
try {
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] },
|
|
});
|
|
runId = start.json().id;
|
|
const responses = await Promise.all([
|
|
app.inject({
|
|
method: "PATCH",
|
|
url: `/catalog/databases/${database.id}/tables/${table.id}`,
|
|
payload: { version: table.version, description: "Concurrent table edit" },
|
|
}),
|
|
app.inject({
|
|
method: "PATCH",
|
|
url: `/catalog/databases/${database.id}/tables/${table.id}/columns/${column.id}`,
|
|
payload: {
|
|
version: column.version,
|
|
description: "Concurrent column edit",
|
|
generatedDescription: null,
|
|
},
|
|
}),
|
|
]);
|
|
|
|
expect(responses.map((response) => response.statusCode)).toEqual([409, 409]);
|
|
expect(responses.map((response) => response.json().code)).toEqual([
|
|
"database_operation_in_progress",
|
|
"database_operation_in_progress",
|
|
]);
|
|
} finally {
|
|
resolveCompletion(JSON.stringify({
|
|
targetId: column.id,
|
|
outcome: "generated",
|
|
description: "Descrizione valida.",
|
|
}));
|
|
if (runId) await waitForTerminalRun(app, runId);
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("bounds the metadata-only prompt before calling the selected model", async () => {
|
|
let selectedColumnId = "";
|
|
const modelCompleter = {
|
|
complete: vi.fn(async () => JSON.stringify({
|
|
targetId: selectedColumnId,
|
|
outcome: "generated",
|
|
description: "Descrizione valida.",
|
|
})),
|
|
};
|
|
const { app, repository, database } = await setup(modelCompleter);
|
|
const omittedTail = "OMITTED_UNTRUSTED_METADATA_TAIL";
|
|
const oversizedMetadata = `Beginning ${"x".repeat(2_500)} ${omittedTail}`;
|
|
try {
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: [{ name: "patients", sourceComment: oversizedMetadata }],
|
|
columns: [{
|
|
tableName: "patients",
|
|
name: "birth_date",
|
|
ordinalPosition: 1,
|
|
dataType: "date",
|
|
isNullable: true,
|
|
defaultExpression: null,
|
|
primaryKeyPosition: null,
|
|
sourceComment: oversizedMetadata,
|
|
}],
|
|
relationships: [],
|
|
});
|
|
const table = (await repository.listTables(database.id))[0]!;
|
|
const column = (await repository.listColumns(database.id, table.id))[0]!;
|
|
selectedColumnId = column.id;
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] },
|
|
});
|
|
await waitForTerminalRun(app, start.json().id);
|
|
|
|
const request = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest;
|
|
const serializedMessages = JSON.stringify(request.messages);
|
|
expect(serializedMessages.length).toBeLessThan(10_000);
|
|
expect(serializedMessages).toContain("Beginning");
|
|
expect(serializedMessages).not.toContain(omittedTail);
|
|
expect(serializedMessages).not.toMatch(/sourceRows|sampleRows|sampleValues/i);
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|
|
|
|
test("bounds aggregate structural columns and source samples below helper message limits", async () => {
|
|
const requests: ModelCompletionRequest[] = [];
|
|
const modelCompleter: ModelCompleter = {
|
|
complete: vi.fn(async (request) => {
|
|
requests.push(request);
|
|
const userContent = request.messages.find((message) => message.role === "user")!.content;
|
|
const context = JSON.parse(userContent.slice(userContent.indexOf("\n") + 1));
|
|
return JSON.stringify({
|
|
results: context.targets.map((target: { targetId: string }) => ({
|
|
targetId: target.targetId,
|
|
outcome: "generated",
|
|
description: "Descrizione bounded.",
|
|
})),
|
|
});
|
|
}),
|
|
};
|
|
const hostileText = `FIRST_USEFUL_FACT_${'"\\\n'.repeat(900)}`;
|
|
const omittedTail = "LAST_STRUCTURAL_COLUMN_MUST_BE_OMITTED";
|
|
const descriptionSourceSampler: DescriptionSourceSampler = {
|
|
sample: vi.fn(async (_database, targets) => targets.map((target) => {
|
|
const sampleColumns = target.columnNames.slice(0, 8);
|
|
return {
|
|
targetId: target.targetId,
|
|
tableName: target.tableName,
|
|
rows: Array.from({ length: 6 }, (_, rowIndex) => ({
|
|
fields: sampleColumns.map((name, columnIndex) => ({
|
|
name,
|
|
value: `SAMPLE_${rowIndex}_${columnIndex}_${'"\\\n'.repeat(300)}`,
|
|
})),
|
|
})),
|
|
representativeValues: sampleColumns.map((column, columnIndex) => ({
|
|
column,
|
|
values: Array.from(
|
|
{ length: 7 },
|
|
(_, valueIndex) => `EXAMPLE_${columnIndex}_${valueIndex}_${'"\\\n'.repeat(300)}`,
|
|
),
|
|
})),
|
|
};
|
|
})),
|
|
};
|
|
const { app, repository, database } = await setup(
|
|
modelCompleter,
|
|
{},
|
|
"it",
|
|
descriptionSourceSampler,
|
|
);
|
|
try {
|
|
const tableInputs = Array.from({ length: 10 }, (_, tableIndex) => ({
|
|
name: `table_${String(tableIndex).padStart(2, "0")}`,
|
|
sourceComment: tableIndex === 0 ? hostileText : `Table ${tableIndex}`,
|
|
}));
|
|
const columnInputs = tableInputs.flatMap((table, tableIndex) => (
|
|
Array.from({ length: 80 }, (_, columnIndex) => ({
|
|
tableName: table.name,
|
|
name: tableIndex === 9 && columnIndex === 79
|
|
? omittedTail
|
|
: `column_${String(columnIndex).padStart(2, "0")}`,
|
|
ordinalPosition: columnIndex + 1,
|
|
dataType: `text_${'x'.repeat(300)}`,
|
|
isNullable: true,
|
|
defaultExpression: hostileText,
|
|
primaryKeyPosition: null,
|
|
sourceComment: columnIndex === 0 ? `COLUMN_FIRST_${hostileText}` : hostileText,
|
|
}))
|
|
));
|
|
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
|
schemaVersion: 1,
|
|
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
|
tables: tableInputs,
|
|
columns: columnInputs,
|
|
relationships: [],
|
|
});
|
|
const tables = await repository.listTables(database.id);
|
|
const start = await app.inject({
|
|
method: "POST",
|
|
url: `/catalog/databases/${database.id}/description-generation-runs`,
|
|
payload: {
|
|
modelId: configuredModel.id,
|
|
scope: "selected_tables",
|
|
targetIds: tables.map((table) => table.id),
|
|
},
|
|
});
|
|
expect(start.statusCode).toBe(202);
|
|
expect((await waitForTerminalRun(app, start.json().id)).run.status).toBe("completed");
|
|
|
|
expect(requests).toHaveLength(1);
|
|
const contentBytes = requests[0]!.messages.map((message) => Buffer.byteLength(message.content, "utf8"));
|
|
expect(contentBytes.every((bytes) => bytes <= 64 * 1024)).toBe(true);
|
|
expect(contentBytes.reduce((total, bytes) => total + bytes, 0)).toBeLessThanOrEqual(128 * 1024);
|
|
const serializedMessages = JSON.stringify(requests[0]!.messages);
|
|
expect(serializedMessages).toContain("FIRST_USEFUL_FACT");
|
|
expect(serializedMessages).toContain("COLUMN_FIRST");
|
|
expect(serializedMessages).toContain("sourceSample");
|
|
expect(serializedMessages).not.toContain(omittedTail);
|
|
const userMessage = requests[0]!.messages.find((message) => message.role === "user")!.content;
|
|
const context = JSON.parse(userMessage.slice(userMessage.indexOf("\n") + 1));
|
|
expect(context.targets.every((target: { columns?: unknown[] }) => (
|
|
(target.columns?.length ?? 0) <= 24
|
|
))).toBe(true);
|
|
for (const target of context.targets as Array<{
|
|
sourceSample?: {
|
|
rows: Array<{ fields: Array<{ name: string; value: unknown }> }>;
|
|
representativeValues: Array<{ column: string; values: unknown[] }>;
|
|
};
|
|
}>) {
|
|
if (!target.sourceSample) continue;
|
|
expect(Buffer.byteLength(JSON.stringify(target.sourceSample), "utf8")).toBeLessThanOrEqual(8 * 1024);
|
|
expect(target.sourceSample.rows.length).toBeLessThanOrEqual(5);
|
|
expect(target.sourceSample.rows.every((row) => row.fields.length <= 4)).toBe(true);
|
|
expect(target.sourceSample.representativeValues.every((examples) => (
|
|
examples.values.length <= 5
|
|
))).toBe(true);
|
|
for (const row of target.sourceSample.rows) {
|
|
for (const field of row.fields) {
|
|
expect(Buffer.byteLength(JSON.stringify(field.name).slice(1, -1), "utf8")).toBeLessThanOrEqual(128);
|
|
if (typeof field.value === "string") {
|
|
expect(Buffer.byteLength(JSON.stringify(field.value).slice(1, -1), "utf8")).toBeLessThanOrEqual(192);
|
|
}
|
|
}
|
|
}
|
|
}
|
|
} finally {
|
|
await app.close();
|
|
}
|
|
});
|