import { expect, test, vi } from "vitest"; import { buildApp } from "../src/app.js"; import type { DescriptionSourceSampler } from "../src/catalog/description-source-sampler.js"; import { MemoryCatalogRepository } from "../src/catalog/memory-repository.js"; import { MetadataGenerationModelUnavailableError, type MetadataGenerationModels, type ResolvedMetadataGenerationModel, } from "../src/catalog/metadata-generation-models.js"; import { ModelCompletionCancelledError, ModelCompletionProviderError, type ModelCompleter, type ModelCompletionRequest, } from "../src/catalog/model-completer.js"; import { CatalogOperationCoordinator } from "../src/catalog/operation-coordinator.js"; import type { SensitivityValueSource } from "../src/catalog/sensitivity-classifier.js"; import type { CatalogDatabaseClient, CatalogPostgresAccess, } from "../src/catalog/postgres-access.js"; import { loadConfig } from "../src/config.js"; import type { WorkspaceRegistry, WorkspaceRevision } from "../src/workspaces/registry.js"; import type { WorkspaceDescriptor } from "../src/workspaces/schema.js"; const workspace: WorkspaceDescriptor = { workspace: { schema_version: 4, id: "psd-clinical", name: "Policlinico San Donato", language: "it", }, dwh: { engine: "postgres", database: "warehouse", schema: "datawarehouse", port: 5432, supported_transports: ["postgres_direct"], }, }; const revision: WorkspaceRevision = { id: "psd-clinical", commit: "a".repeat(40), blob: "b".repeat(40), snapshotPath: "/tmp/psd.yaml", }; const configuredModel: ResolvedMetadataGenerationModel = { id: "openai/gpt-4.1-mini", provider: "openai", model: "gpt-4.1-mini", apiKeyEnv: "OPENAI_API_KEY", apiKey: "test-provider-secret", }; function models(): MetadataGenerationModels { return { catalog: () => ({ models: [{ id: configuredModel.id, label: "OpenAI Mini" }], default: configuredModel.id }), resolve: (selection) => { if (selection !== configuredModel.id) throw new MetadataGenerationModelUnavailableError(); return configuredModel; }, }; } async function setup( modelCompleter: ModelCompleter, env: Record = {}, language: "en" | "it" = "it", descriptionSourceSampler: DescriptionSourceSampler | null = { sample: vi.fn(async () => []), }, catalogPostgresAccess?: CatalogPostgresAccess, sensitivityValueSource: SensitivityValueSource = { scanTable: vi.fn(async (request, consume) => { await consume(request.columns.map((column) => ({ columnId: column.id, value: "ordinary", characterLength: 8, }))); return { kind: "complete", observedRows: 1 }; }), }, ) { const repository = new MemoryCatalogRepository(); const database = await repository.create({ workspaceId: workspace.workspace.id, engine: "postgres", databaseName: "warehouse", schema: "datawarehouse", binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" }, }); await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: "Clinical patients" }], columns: [{ tableName: "patients", name: "birth_date", ordinalPosition: 1, dataType: "date", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: "Patient date of birth", }], relationships: [], }); const table = (await repository.listTables(database.id))[0]!; const column = (await repository.listColumns(database.id, table.id))[0]!; const registry = { list: vi.fn(async () => [revision]), read: vi.fn(async () => ({ workspace: { ...workspace, workspace: { ...workspace.workspace, language }, }, revision, })), } as unknown as WorkspaceRegistry; const operations = new CatalogOperationCoordinator(); const app = buildApp(loadConfig({ NODE_ENV: "test", THT_HARNESS_DIR: "/missing", ...env }), { thtRunner: {} as never, workspaceRegistry: registry, workspaceDiagnoser: vi.fn(), catalogRepository: repository, catalogOperationCoordinator: operations, metadataGenerationModels: models(), modelCompleter, sensitivityValueSource, ...(descriptionSourceSampler ? { descriptionSourceSampler } : {}), ...(catalogPostgresAccess ? { catalogPostgresAccess } : {}), }); return { app, repository, database, table, column, operations, sensitivityValueSource }; } async function waitForTerminalRun(app: ReturnType, runId: string) { for (let attempt = 0; attempt < 100; attempt += 1) { const response = await app.inject({ method: "GET", url: `/catalog/description-generation-runs/${runId}`, }); const run = response.json(); if (["completed", "completed_with_errors", "cancelled", "failed", "interrupted"].includes(run.status)) { return { response, run }; } await new Promise((resolve) => setTimeout(resolve, 5)); } throw new Error(`Description Generation Run ${runId} did not finish`); } test("assesses sensitive flags locally without persisting them or calling an LLM", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") }; const { app, repository, database, table, column } = await setup(modelCompleter); try { const response = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/sensitive-data-suggestions`, payload: { scope: "all" }, }); expect(response.statusCode).toBe(200); const responseBody = response.json(); expect(responseBody).toMatchObject({ run: { databaseId: database.id, scope: "all", engine: "local", modelId: null, policyVersion: "sensitivity-v1", status: "completed", total: 1, suggestedSensitive: 1, suggestedNonSensitive: 0, unknown: 0, errorSummary: null, }, suggestions: [{ columnId: column.id, tableId: table.id, tableName: table.name, columnName: column.name, version: column.version, currentSensitive: false, sensitive: true, assessment: "sensitive", evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }], }], }); expect(await repository.getColumn(database.id, column.tableId, column.id)) .toMatchObject({ sensitive: false }); expect(responseBody.run).not.toHaveProperty("suggestions"); const history = await app.inject({ method: "GET", url: "/catalog/sensitive-data-suggestion-runs?limit=1", }); expect(history.statusCode).toBe(200); expect(history.json()).toEqual([responseBody.run]); expect(history.body).not.toContain(column.id); const detail = await app.inject({ method: "GET", url: `/catalog/sensitive-data-suggestion-runs/${responseBody.run.id}`, }); expect(detail.statusCode).toBe(200); expect(detail.json()).toEqual(responseBody.run); const events = await app.inject({ method: "GET", url: `/catalog/sensitive-data-suggestion-runs/${responseBody.run.id}/events-list`, }); expect(events.statusCode).toBe(200); expect(events.body).not.toContain(column.id); expect(events.json()).toMatchObject([ { runId: responseBody.run.id, sequence: 1, level: "info", message: "Local sensitivity analysis started.", }, { runId: responseBody.run.id, sequence: 2, level: "info", message: "Assessed 1 of 1 columns locally.", }, { runId: responseBody.run.id, sequence: 3, level: "info", message: "Local sensitivity analysis completed for 1 column.", }, ]); expect(modelCompleter.complete).not.toHaveBeenCalled(); } finally { await app.close(); } }); test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => { const controller = new AbortController(); controller.abort(); const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal); const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") }); try { const response = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/sensitive-data-suggestions`, payload: { scope: "all" }, }); expect(response.statusCode).toBe(504); expect(response.json()).toEqual({ code: "sensitivity_analysis_timeout", message: "Sensitivity analysis reached its time limit. No assessments were applied.", }); expect(await repository.listSensitivityAnalysisRuns()).toEqual([]); } finally { timeout.mockRestore(); await app.close(); } }); test("limits sensitivity analysis to the selected tables or columns", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") }; const { app, repository, database, sensitivityValueSource } = await setup(modelCompleter); await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [ { name: "patients", sourceComment: null }, { name: "visits", sourceComment: null }, { name: "billing", sourceComment: null }, ], columns: [ { tableName: "patients", name: "patient_name", ordinalPosition: 1, dataType: "text", isNullable: false, defaultExpression: null, primaryKeyPosition: null, sourceComment: null }, { tableName: "patients", name: "status", ordinalPosition: 2, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null }, { tableName: "visits", name: "clinical_note", ordinalPosition: 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null }, { tableName: "billing", name: "invoice_total", ordinalPosition: 1, dataType: "numeric", isNullable: false, defaultExpression: null, primaryKeyPosition: null, sourceComment: null }, ], relationships: [], }); const tables = await repository.listTables(database.id); const patients = tables.find((table) => table.name === "patients")!; const visits = tables.find((table) => table.name === "visits")!; const billing = tables.find((table) => table.name === "billing")!; const patientColumns = await repository.listColumns(database.id, patients.id); const visitColumns = await repository.listColumns(database.id, visits.id); const billingColumns = await repository.listColumns(database.id, billing.id); const status = patientColumns.find((column) => column.name === "status")!; const clinicalNote = visitColumns.find((column) => column.name === "clinical_note")!; try { const tableResponse = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/sensitive-data-suggestions`, payload: { scope: "selected_tables", targetIds: [visits.id, patients.id], }, }); expect(tableResponse.statusCode).toBe(200); expect(tableResponse.json().suggestions).toHaveLength(3); expect(tableResponse.json().suggestions).toEqual(expect.arrayContaining([ expect.objectContaining({ tableId: patients.id, columnName: "patient_name", sensitive: true }), expect.objectContaining({ tableId: patients.id, columnName: "status", sensitive: false }), expect.objectContaining({ tableId: visits.id, columnName: "clinical_note", sensitive: true }), ])); expect(tableResponse.json().suggestions).not.toEqual(expect.arrayContaining([ expect.objectContaining({ tableId: billing.id }), ])); const columnResponse = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/sensitive-data-suggestions`, payload: { scope: "selected_columns", targetIds: [clinicalNote.id, status.id], }, }); expect(columnResponse.statusCode).toBe(200); expect(columnResponse.json().suggestions).toHaveLength(2); expect(columnResponse.json().suggestions).toEqual(expect.arrayContaining([ expect.objectContaining({ tableId: patients.id, columnId: status.id, sensitive: false }), expect.objectContaining({ tableId: visits.id, columnId: clinicalNote.id, sensitive: true }), ])); const scannedColumnIds = vi.mocked(sensitivityValueSource.scanTable).mock.calls.flatMap( ([request]) => request.columns.map((column) => column.id), ); expect(scannedColumnIds).toEqual([status.id, status.id]); expect(scannedColumnIds).not.toContain(patientColumns.find( (column) => column.name === "patient_name", )!.id); expect(scannedColumnIds).not.toContain(clinicalNote.id); expect(scannedColumnIds).not.toContain(billingColumns[0]!.id); expect(modelCompleter.complete).not.toHaveBeenCalled(); } finally { await app.close(); } }); test("explains invalid sensitivity-analysis selections without reading source values", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") }; const { app, database, table, sensitivityValueSource } = await setup(modelCompleter); try { const empty = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/sensitive-data-suggestions`, payload: { scope: "selected_tables", targetIds: [] }, }); expect(empty.statusCode).toBe(400); expect(empty.json()).toEqual({ code: "sensitive_data_suggestion_request_invalid", message: "Choose a database, one or more tables, or one or more columns to assess.", }); const duplicate = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/sensitive-data-suggestions`, payload: { scope: "selected_tables", targetIds: [table.id, table.id], }, }); expect(duplicate.statusCode).toBe(400); expect(duplicate.json()).toEqual({ code: "sensitive_data_suggestion_target_ids_duplicate", message: "Each selected table or column must appear only once.", }); const missingTable = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/sensitive-data-suggestions`, payload: { scope: "selected_tables", targetIds: ["00000000-0000-4000-8000-000000000001"], }, }); expect(missingTable.statusCode).toBe(404); expect(missingTable.json()).toEqual({ code: "catalog_table_not_found", message: "One or more selected Catalog Tables were not found in this database.", }); const missingColumn = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/sensitive-data-suggestions`, payload: { scope: "selected_columns", targetIds: ["00000000-0000-4000-8000-000000000002"], }, }); expect(missingColumn.statusCode).toBe(404); expect(missingColumn.json()).toEqual({ code: "catalog_column_not_found", message: "One or more selected Catalog Columns were not found in this database.", }); expect(modelCompleter.complete).not.toHaveBeenCalled(); expect(sensitivityValueSource.scanTable).not.toHaveBeenCalled(); } finally { await app.close(); } }); interface SseFrame { id?: string; event?: string; data: string; } function sseFrameReader(response: Response) { const reader = response.body!.getReader(); const decoder = new TextDecoder(); const queued: SseFrame[] = []; let buffer = ""; const parseAvailable = () => { let boundary = buffer.indexOf("\n\n"); while (boundary >= 0) { const block = buffer.slice(0, boundary); buffer = buffer.slice(boundary + 2); const frame: SseFrame = { data: "" }; const data: string[] = []; for (const line of block.split("\n")) { if (line.startsWith("id: ")) frame.id = line.slice(4); else if (line.startsWith("event: ")) frame.event = line.slice(7); else if (line.startsWith("data: ")) data.push(line.slice(6)); } frame.data = data.join("\n"); if (frame.event || frame.id || frame.data) queued.push(frame); boundary = buffer.indexOf("\n\n"); } }; return { async next(): Promise { while (queued.length === 0) { const chunk = await reader.read(); if (chunk.done) throw new Error("SSE stream ended before the expected event arrived"); buffer += decoder.decode(chunk.value, { stream: true }).replaceAll("\r\n", "\n"); parseAvailable(); } return queued.shift()!; }, async cancel(): Promise { await reader.cancel(); }, }; } test("generates one selected Catalog Column from a single JSON code fence", async () => { let resolveCompletion!: (content: string) => void; const completion = new Promise((resolve) => { resolveCompletion = resolve; }); const modelCompleter = { complete: vi.fn(async () => await completion) }; const { app, database, table, column } = await setup(modelCompleter); try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }, }); expect(start.statusCode).toBe(202); expect(start.json()).toMatchObject({ databaseId: database.id, scope: "selected_columns", modelId: configuredModel.id, language: "it", status: "queued", total: 1, processed: 0, generated: 0, nonGeneratable: 0, failed: 0, startedAt: null, finishedAt: null, errorSummary: null, }); expect(Object.keys(start.json()).sort()).toEqual([ "cacheReadTokens", "createdAt", "databaseId", "errorSummary", "failed", "finishedAt", "generated", "id", "inputTokens", "language", "modelId", "nonGeneratable", "outputTokens", "processed", "scope", "startedAt", "status", "total", "updatedAt", ]); expect(modelCompleter.complete).toHaveBeenCalledTimes(1); const completionRequest = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest; expect(completionRequest.model).toEqual(configuredModel); expect(completionRequest.messages).toHaveLength(2); expect(completionRequest.messages[0]?.content).toContain('workspace language "it"'); expect(completionRequest.messages[0]?.content).toContain('{"results":['); expect(completionRequest.messages[1]?.content).toContain(`"targetId":"${column.id}"`); expect(completionRequest.messages[1]?.content).not.toMatch(/source rows|samples|example values/i); expect(start.body).not.toMatch(/test-provider-secret|Catalog metadata/); resolveCompletion(`\`\`\`json\n${JSON.stringify({ results: [{ targetId: column.id, outcome: "generated", description: "Data di nascita del paziente.", }], })}\n\`\`\``); const { response: completedResponse, run } = await waitForTerminalRun(app, start.json().id); expect(completedResponse.statusCode).toBe(200); expect(run).toMatchObject({ status: "completed", total: 1, processed: 1, generated: 1, nonGeneratable: 0, failed: 0, errorSummary: null, }); expect(run.startedAt).toEqual(expect.any(String)); expect(run.finishedAt).toEqual(expect.any(String)); const columns = await app.inject({ method: "GET", url: `/catalog/databases/${database.id}/tables/${table.id}/columns`, }); expect(columns.statusCode).toBe(200); expect(columns.json()).toEqual([ expect.objectContaining({ id: column.id, description: null, generatedDescription: "Data di nascita del paziente.", version: column.version + 1, }), ]); const events = await app.inject({ method: "GET", url: `/catalog/description-generation-runs/${run.id}/events-list?after=0`, }); expect(events.statusCode).toBe(200); expect(events.json().map((event: { sequence: number }) => event.sequence)).toEqual([1, 2, 3, 4]); expect(Object.keys(events.json()[0]).sort()).toEqual(["createdAt", "level", "message", "sequence"]); expect(events.json()).toEqual([ expect.objectContaining({ sequence: 1, level: "info", message: "Description generation queued." }), expect.objectContaining({ sequence: 2, level: "info", message: "Description generation started." }), expect.objectContaining({ sequence: 3, level: "info", message: `Generated description for Catalog Column ${column.id}.`, }), expect.objectContaining({ sequence: 4, level: "info", message: "Description generation completed." }), ]); expect(events.body).not.toMatch(/test-provider-secret|gpt-4\.1|openai\/gpt|Catalog metadata/); const laterEvents = await app.inject({ method: "GET", url: `/catalog/description-generation-runs/${run.id}/events-list?after=2`, }); expect(laterEvents.json().map((event: { sequence: number }) => event.sequence)).toEqual([3, 4]); } finally { resolveCompletion(JSON.stringify({ results: [{ targetId: column.id, outcome: "generated", description: "Data di nascita del paziente.", }], })); await app.close(); } }); test("keeps real source samples transient across the Fastify API and application logs", async () => { let selectedColumnId = ""; const sampleSecret = "FASTIFY_TRANSIENT_SAMPLE_9ca73e"; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => JSON.stringify({ results: [{ targetId: selectedColumnId, outcome: "generated", description: "Data di nascita del paziente.", }], })), }; const descriptionSourceSampler: DescriptionSourceSampler = { sample: vi.fn(async (_database, targets) => [{ targetId: targets[0]!.targetId, tableName: targets[0]!.tableName, rows: [{ fields: [{ name: targets[0]!.columnNames[0]!, value: sampleSecret }] }], representativeValues: [{ column: targets[0]!.columnNames[0]!, values: [sampleSecret], }], }]), }; const { app, repository, database, table, column } = await setup( modelCompleter, {}, "it", descriptionSourceSampler, ); selectedColumnId = column.id; const logSpies = [ vi.spyOn(app.log, "info"), vi.spyOn(app.log, "warn"), vi.spyOn(app.log, "error"), ]; try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }, }); expect(start.statusCode).toBe(202); await waitForTerminalRun(app, start.json().id); const completionRequest = vi.mocked(modelCompleter.complete).mock.calls[0]![0]; expect(JSON.stringify(completionRequest.messages)).toContain(sampleSecret); const apiResponses = await Promise.all([ app.inject({ method: "GET", url: `/catalog/description-generation-runs/${start.json().id}` }), app.inject({ method: "GET", url: `/catalog/description-generation-runs/${start.json().id}/events-list`, }), app.inject({ method: "GET", url: `/catalog/databases/${database.id}` }), app.inject({ method: "GET", url: `/catalog/databases/${database.id}/tables` }), app.inject({ method: "GET", url: `/catalog/databases/${database.id}/tables/${table.id}/columns`, }), ]); expect(apiResponses.every((response) => response.statusCode === 200)).toBe(true); expect(apiResponses.map((response) => response.body).join("\n")).not.toContain(sampleSecret); const persisted = JSON.stringify({ run: await repository.getDescriptionGenerationRun(start.json().id), events: await repository.listDescriptionGenerationEvents(start.json().id), database: await repository.get(database.id), table: await repository.getTable(database.id, table.id), column: await repository.getColumn(database.id, table.id, column.id), }); expect(persisted).not.toContain(sampleSecret); expect(JSON.stringify(logSpies.flatMap((spy) => spy.mock.calls))).not.toContain(sampleSecret); } finally { for (const spy of logSpies) spy.mockRestore(); await app.close(); } }); test("never exposes a protected source value to the model, persistence, logs, or browser APIs", async () => { let selectedColumnId = ""; const protectedValue = "PROTECTED_SOURCE_VALUE_8f4c2a"; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => JSON.stringify({ results: [{ targetId: selectedColumnId, outcome: "generated", description: "Data di nascita del paziente.", }], })), }; const descriptionSourceSampler: DescriptionSourceSampler = { sample: vi.fn(async (_database, targets) => [{ targetId: targets[0]!.targetId, tableName: targets[0]!.tableName, rows: [{ fields: [{ name: "birth_date", value: protectedValue }] }], representativeValues: [{ column: "birth_date", values: [protectedValue] }], }]), }; const { app, repository, database, table, column } = await setup( modelCompleter, {}, "it", descriptionSourceSampler, ); selectedColumnId = column.id; await repository.updateColumnMetadata( database.id, table.id, column.id, column.version, column.description, column.generatedDescription, true, ); const logSpies = [ vi.spyOn(app.log, "info"), vi.spyOn(app.log, "warn"), vi.spyOn(app.log, "error"), ]; try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }, }); expect(start.statusCode).toBe(202); await waitForTerminalRun(app, start.json().id); expect(descriptionSourceSampler.sample).toHaveBeenCalledWith( expect.anything(), [expect.objectContaining({ targetId: column.id, columnNames: [] })], expect.any(AbortSignal), ); const completionRequest = vi.mocked(modelCompleter.complete).mock.calls[0]![0]; const providerPayload = JSON.stringify(completionRequest.messages); expect(providerPayload).not.toContain(protectedValue); expect(providerPayload).toContain("1981-01-01"); const apiResponses = await Promise.all([ app.inject({ method: "GET", url: `/catalog/description-generation-runs/${start.json().id}` }), app.inject({ method: "GET", url: `/catalog/description-generation-runs/${start.json().id}/events-list`, }), app.inject({ method: "GET", url: `/catalog/databases/${database.id}` }), app.inject({ method: "GET", url: `/catalog/databases/${database.id}/tables` }), app.inject({ method: "GET", url: `/catalog/databases/${database.id}/tables/${table.id}/columns`, }), ]); expect(apiResponses.every((response) => response.statusCode === 200)).toBe(true); expect(apiResponses.map((response) => response.body).join("\n")).not.toContain(protectedValue); const persisted = JSON.stringify({ run: await repository.getDescriptionGenerationRun(start.json().id), events: await repository.listDescriptionGenerationEvents(start.json().id), database: await repository.get(database.id), table: await repository.getTable(database.id, table.id), column: await repository.getColumn(database.id, table.id, column.id), }); expect(persisted).not.toContain(protectedValue); expect(JSON.stringify(logSpies.flatMap((spy) => spy.mock.calls))).not.toContain(protectedValue); } finally { for (const spy of logSpies) spy.mockRestore(); await app.close(); } }); test("wires the production sampler to the same injected CatalogPostgresAccess instance", async () => { let selectedColumnId = ""; const sampleSecret = "PRODUCTION_WIRING_SAMPLE_f2986a"; const query = vi.fn(async (sql: string) => ({ rows: sql.startsWith("SELECT") ? [{ birth_date: sampleSecret }] : [], })); const end = vi.fn(async () => undefined); const catalogPostgresAccess: CatalogPostgresAccess = { connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient), }; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => JSON.stringify({ results: [{ targetId: selectedColumnId, outcome: "generated", description: "Data di nascita del paziente.", }], })), }; const { app, database, column } = await setup( modelCompleter, {}, "it", null, catalogPostgresAccess, ); selectedColumnId = column.id; try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }, }); expect(start.statusCode).toBe(202); expect((await waitForTerminalRun(app, start.json().id)).run.status).toBe("completed"); expect(catalogPostgresAccess.connect).toHaveBeenCalledWith( expect.objectContaining({ id: database.id }), expect.any(AbortSignal), ); expect(query.mock.calls.map(([sql]) => String(sql).split(" ")[0])).toEqual([ "BEGIN", "SELECT", "ROLLBACK", ]); expect(end).toHaveBeenCalledOnce(); const completionRequest = vi.mocked(modelCompleter.complete).mock.calls[0]![0]; expect(JSON.stringify(completionRequest.messages)).toContain(sampleSecret); } finally { await app.close(); } }); test("generates selected Catalog Columns in caller order through sequential batches of ten", async () => { const pending = Array.from({ length: 2 }, () => { let resolve!: (content: string) => void; const promise = new Promise((done) => { resolve = done; }); return { promise, resolve }; }); let active = 0; let maxActive = 0; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { const callIndex = (modelCompleter.complete as ReturnType).mock.calls.length - 1; active += 1; maxActive = Math.max(maxActive, active); try { return await pending[callIndex]!.promise; } finally { active -= 1; } }), }; const { app, repository, database, table } = await setup(modelCompleter); let orderedIds: string[] = []; const responseFor = (targetIds: readonly string[], offset: number) => JSON.stringify({ results: targetIds.map((targetId, index) => ({ targetId, outcome: "generated", description: `Generated ${offset + index + 1}`, })).reverse(), }); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: "Clinical patients" }], columns: Array.from({ length: 11 }, (_, index) => ({ tableName: "patients", name: `column_${String(index + 1).padStart(2, "0")}`, ordinalPosition: index + 1, dataType: "text", isNullable: index % 2 === 0, defaultExpression: null, primaryKeyPosition: index === 0 ? 1 : null, sourceComment: `Column ${index + 1}`, })), relationships: [], }); const columns = await repository.listColumns(database.id, table.id); orderedIds = columns.map((column) => column.id).reverse(); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: orderedIds, }, }); expect(start.statusCode).toBe(202); expect(start.json()).toMatchObject({ total: 11, scope: "selected_columns" }); expect(modelCompleter.complete).toHaveBeenCalledTimes(1); expect(maxActive).toBe(1); const firstRequest = (modelCompleter.complete as ReturnType).mock.calls[0]![0] as ModelCompletionRequest; const firstMetadata = JSON.parse(firstRequest.messages[1]!.content.split("\n").slice(1).join("\n")); expect(firstMetadata.targets.map((target: { targetId: string }) => target.targetId)).toEqual(orderedIds.slice(0, 10)); pending[0]!.resolve(responseFor(orderedIds.slice(0, 10), 0)); for (let attempt = 0; attempt < 100 && (modelCompleter.complete as ReturnType).mock.calls.length < 2; attempt += 1) { await new Promise((resolve) => setTimeout(resolve, 1)); } expect(modelCompleter.complete).toHaveBeenCalledTimes(2); expect(maxActive).toBe(1); const secondRequest = (modelCompleter.complete as ReturnType).mock.calls[1]![0] as ModelCompletionRequest; const secondMetadata = JSON.parse(secondRequest.messages[1]!.content.split("\n").slice(1).join("\n")); expect(secondMetadata.targets.map((target: { targetId: string }) => target.targetId)).toEqual(orderedIds.slice(10)); pending[1]!.resolve(responseFor(orderedIds.slice(10), 10)); const { run } = await waitForTerminalRun(app, start.json().id); expect(run).toMatchObject({ status: "completed", total: 11, processed: 11, generated: 11, nonGeneratable: 0, failed: 0, }); const updatedById = new Map( (await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]), ); orderedIds.forEach((targetId, index) => { expect(updatedById.get(targetId)).toMatchObject({ description: null, generatedDescription: `Generated ${index + 1}`, }); }); const events = await repository.listDescriptionGenerationEvents(run.id); expect(events.map((event) => event.sequence)).toEqual(Array.from({ length: 14 }, (_, index) => index + 1)); expect(events.slice(2, -1).map((event) => event.message)).toEqual( orderedIds.map((targetId) => `Generated description for Catalog Column ${targetId}.`), ); } finally { pending[0]!.resolve(responseFor(orderedIds.slice(0, 10), 0)); pending[1]!.resolve(responseFor(orderedIds.slice(10), 10)); await app.close(); } }); test("Stop cancels the current helper, retains prior writes, and prevents later batches", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => { const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")); const targetIds = context.targets.map((target: { targetId: string }) => target.targetId); if (vi.mocked(modelCompleter.complete).mock.calls.length === 1) { return JSON.stringify({ results: targetIds.map((targetId: string, index: number) => ({ targetId, outcome: "generated", description: `Completed before Stop ${index + 1}`, })), }); } return await new Promise((_resolve, reject) => { const cancel = () => reject(new ModelCompletionCancelledError()); if (request.signal.aborted) cancel(); else request.signal.addEventListener("abort", cancel, { once: true }); }); }), }; const { app, repository, database, table, operations } = await setup(modelCompleter); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: null }], columns: Array.from({ length: 21 }, (_, index) => ({ tableName: "patients", name: `column_${String(index + 1).padStart(2, "0")}`, ordinalPosition: index + 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, })), relationships: [], }); const columns = await repository.listColumns(database.id, table.id); const targetIds = columns.map((column) => column.id); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds, }, }); expect(start.statusCode).toBe(202); for (let attempt = 0; attempt < 100 && vi.mocked(modelCompleter.complete).mock.calls.length < 2; attempt += 1) { await new Promise((resolve) => setTimeout(resolve, 2)); } expect(modelCompleter.complete).toHaveBeenCalledTimes(2); const cancel = await app.inject({ method: "POST", url: `/catalog/description-generation-runs/${start.json().id}/cancel`, }); expect(cancel.statusCode).toBe(200); expect(cancel.json()).toMatchObject({ status: "cancelled", total: 21, processed: 10, generated: 10, nonGeneratable: 0, failed: 0, errorSummary: null, }); const requests = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => request); expect(requests).toHaveLength(2); expect(requests[0]!.signal).toBe(requests[1]!.signal); expect(requests[1]!.signal.aborted).toBe(true); const updated = new Map( (await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]), ); targetIds.slice(0, 10).forEach((targetId, index) => { expect(updated.get(targetId)?.generatedDescription).toBe(`Completed before Stop ${index + 1}`); }); targetIds.slice(10).forEach((targetId) => { expect(updated.get(targetId)?.generatedDescription).toBeNull(); }); const events = await repository.listDescriptionGenerationEvents(start.json().id); expect(events.map((event) => event.sequence)).toEqual( Array.from({ length: events.length }, (_, index) => index + 1), ); expect(events.at(-1)).toEqual(expect.objectContaining({ level: "warning", message: "Description generation cancelled.", })); const release = operations.reserve(database.id); release(); } finally { await app.close(); } }); test("Stop aborts source sampling before any model request", async () => { let samplingStarted!: () => void; const started = new Promise((resolve) => { samplingStarted = resolve; }); let samplingSignal: AbortSignal | undefined; const sourceSampler: DescriptionSourceSampler = { sample: vi.fn(async (_database, _targets, signal) => { samplingSignal = signal; samplingStarted(); return await new Promise((resolve, reject) => { const cancel = () => reject(new Error("sampling aborted")); if (signal.aborted) cancel(); else signal.addEventListener("abort", cancel, { once: true }); }); }), }; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("cancelled sampling must not call the model"); }), }; const { app, repository, database, column } = await setup( modelCompleter, {}, "it", sourceSampler, ); try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }, }); expect(start.statusCode).toBe(202); await started; const cancel = await app.inject({ method: "POST", url: `/catalog/description-generation-runs/${start.json().id}/cancel`, }); expect(cancel.statusCode).toBe(200); expect(cancel.json()).toMatchObject({ status: "cancelled", processed: 0 }); expect(samplingSignal?.aborted).toBe(true); expect(modelCompleter.complete).not.toHaveBeenCalled(); expect((await repository.listDescriptionGenerationEvents(start.json().id)).map( (event) => event.message, )).not.toContain( "Source samples unavailable for this batch; generation continued with catalog metadata only.", ); } finally { await app.close(); } }); test("an isolated exhausted technical batch failure allows completion with errors", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => { if (vi.mocked(modelCompleter.complete).mock.calls.length <= 2) { throw new ModelCompletionProviderError(); } const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")); return JSON.stringify({ results: context.targets.map((target: { targetId: string }) => ({ targetId: target.targetId, outcome: "generated", description: "Generated after an isolated technical failure.", })), }); }), }; const { app, repository, database, table } = await setup(modelCompleter); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: null }], columns: Array.from({ length: 11 }, (_, index) => ({ tableName: "patients", name: `column_${String(index + 1).padStart(2, "0")}`, ordinalPosition: index + 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, })), relationships: [], }); const targetIds = (await repository.listColumns(database.id, table.id)).map((column) => column.id); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds }, }); const { run } = await waitForTerminalRun(app, start.json().id); expect(run).toMatchObject({ status: "completed_with_errors", total: 11, processed: 11, generated: 1, nonGeneratable: 0, failed: 10, errorSummary: "Description generation completed with errors.", }); expect(modelCompleter.complete).toHaveBeenCalledTimes(3); const updated = new Map( (await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]), ); targetIds.slice(0, 10).forEach((targetId) => { expect(updated.get(targetId)?.generatedDescription).toBeNull(); }); expect(updated.get(targetIds[10]!)?.generatedDescription).toBe( "Generated after an isolated technical failure.", ); expect((await repository.listDescriptionGenerationEvents(run.id)).at(-1)).toEqual( expect.objectContaining({ level: "warning", message: "Description generation completed with errors.", }), ); } finally { await app.close(); } }); test("success resets the technical-failure streak and the third later failure stops the run", async () => { const sensitiveDiagnostic = "provider payload must remain redacted 7d41"; const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => { const call = vi.mocked(modelCompleter.complete).mock.calls.length; if ([1, 2, 4, 5, 6, 7, 8, 9].includes(call)) { throw Object.assign(new ModelCompletionProviderError(), { message: sensitiveDiagnostic }); } if (call > 9) throw new Error("a later batch must not start"); const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")); return JSON.stringify({ results: context.targets.map((target: { targetId: string }) => ({ targetId: target.targetId, outcome: "generated", description: "Successful reset batch.", })), }); }), }; const { app, repository, database, table } = await setup(modelCompleter); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: null }], columns: Array.from({ length: 61 }, (_, index) => ({ tableName: "patients", name: `column_${String(index + 1).padStart(2, "0")}`, ordinalPosition: index + 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, })), relationships: [], }); const targetIds = (await repository.listColumns(database.id, table.id)).map((column) => column.id); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds }, }); const { run } = await waitForTerminalRun(app, start.json().id); expect(run).toMatchObject({ status: "failed", total: 61, processed: 50, generated: 10, nonGeneratable: 0, failed: 40, errorSummary: "Description generation stopped after three consecutive technical batch failures.", }); expect(modelCompleter.complete).toHaveBeenCalledTimes(9); const requests = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => request); expect(new Set(requests.map((request) => request.signal)).size).toBe(1); expect(new Set(requests.map((request) => request.model.id))).toEqual(new Set([configuredModel.id])); const updated = new Map( (await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]), ); targetIds.slice(10, 20).forEach((targetId) => { expect(updated.get(targetId)?.generatedDescription).toBe("Successful reset batch."); }); expect(updated.get(targetIds[50]!)?.generatedDescription).toBeNull(); const events = await repository.listDescriptionGenerationEvents(run.id); expect(events.filter((event) => event.message.includes("model provider request failed"))).toHaveLength(8); expect(events.at(-1)).toEqual(expect.objectContaining({ level: "error", message: "Description generation stopped after three consecutive technical batch failures.", })); expect(JSON.stringify({ run, events })).not.toContain(sensitiveDiagnostic); } finally { await app.close(); } }); test.each(["queued", "running"] as const)( "backend startup marks a persisted %s Description Generation Run interrupted without resuming it", async (status) => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("startup must not resume generation"); }), }; const { app, repository, database } = await setup(modelCompleter); try { const created = await repository.createDescriptionGenerationRun( database.id, "missing", configuredModel.id, "it", 1, ); if (status === "running") { await repository.updateDescriptionGenerationRun(created.id, { status: "running", startedAt: "2026-08-28T08:00:00.000Z", }); } await app.ready(); expect(await repository.getDescriptionGenerationRun(created.id)).toMatchObject({ status: "interrupted", errorSummary: "Description generation was interrupted by backend restart.", finishedAt: expect.any(String), }); expect(await repository.listDescriptionGenerationEvents(created.id)).toEqual([ expect.objectContaining({ sequence: 1, level: "warning", message: "Description generation was interrupted by backend restart.", }), ]); expect(modelCompleter.complete).not.toHaveBeenCalled(); } finally { await app.close(); } }, ); test("Unlock is rejected while a local Description Generation worker is live", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => await new Promise((_resolve, reject) => { const cancel = () => reject(new ModelCompletionCancelledError()); if (request.signal.aborted) cancel(); else request.signal.addEventListener("abort", cancel, { once: true }); })), }; const { app, database, column } = await setup(modelCompleter); let runId = ""; try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }, }); runId = start.json().id; expect(modelCompleter.complete).toHaveBeenCalledOnce(); const unlock = await app.inject({ method: "POST", url: "/catalog/description-generation-runs/unlock", }); expect(unlock.statusCode).toBe(409); expect(unlock.json()).toEqual({ code: "description_generation_run_live", message: "A local Description Generation worker or helper is still running.", }); expect(vi.mocked(modelCompleter.complete).mock.calls[0]![0].signal.aborted).toBe(false); } finally { if (runId) { await app.inject({ method: "POST", url: `/catalog/description-generation-runs/${runId}/cancel`, }); } await app.close(); } }); test("Unlock is rejected while a Description Generation start is still launching its worker", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => await new Promise((_resolve, reject) => { const cancel = () => reject(new ModelCompletionCancelledError()); if (request.signal.aborted) cancel(); else request.signal.addEventListener("abort", cancel, { once: true }); })), }; const { app, repository, database, column } = await setup(modelCompleter); const appendEvent = repository.appendDescriptionGenerationEvent.bind(repository); let queuedEventEntered!: () => void; const queuedEventStarted = new Promise((resolve) => { queuedEventEntered = resolve; }); let releaseQueuedEvent!: () => void; const queuedEventGate = new Promise((resolve) => { releaseQueuedEvent = resolve; }); let blockQueuedEvent = true; vi.spyOn(repository, "appendDescriptionGenerationEvent").mockImplementation( async (runId, level, message) => { if (blockQueuedEvent && message === "Description generation queued.") { blockQueuedEvent = false; queuedEventEntered(); await queuedEventGate; } return await appendEvent(runId, level, message); }, ); let runId = ""; try { const startPromise = app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }, }); await queuedEventStarted; const unlock = await app.inject({ method: "POST", url: "/catalog/description-generation-runs/unlock", }); releaseQueuedEvent(); const start = await startPromise; runId = start.json().id; expect(start.statusCode).toBe(202); expect(unlock.statusCode).toBe(409); expect(unlock.json()).toEqual({ code: "description_generation_run_live", message: "A local Description Generation worker or helper is still running.", }); } finally { releaseQueuedEvent(); if (runId) { await app.inject({ method: "POST", url: `/catalog/description-generation-runs/${runId}/cancel`, }); } await app.close(); } }); test("Unlock interrupts a stale active run and releases its stale local catalog reservation", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("stale Unlock must not invoke the model"); }), }; const { app, repository, database, operations } = await setup(modelCompleter); try { await app.ready(); const stale = await repository.createDescriptionGenerationRun( database.id, "all", configuredModel.id, "it", 2, ); await repository.updateDescriptionGenerationRun(stale.id, { status: "running", startedAt: "2026-08-28T08:00:00.000Z", }); operations.reserve(database.id, "description_generation"); const unlock = await app.inject({ method: "POST", url: "/catalog/description-generation-runs/unlock", }); expect(unlock.statusCode).toBe(200); expect(unlock.json()).toMatchObject({ id: stale.id, databaseId: database.id, status: "interrupted", errorSummary: "Description generation was interrupted by Unlock.", finishedAt: expect.any(String), }); expect((await repository.listDescriptionGenerationEvents(stale.id)).at(-1)).toEqual( expect.objectContaining({ level: "warning", message: "Description generation was interrupted by Unlock.", }), ); const release = operations.reserve(database.id); release(); expect(modelCompleter.complete).not.toHaveBeenCalled(); } finally { await app.close(); } }); test("Description Generation history is newest-first, bounded, and exposes only the safe run shape", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("history must not invoke the model"); }), }; const { app, repository, database } = await setup(modelCompleter); try { const ids: string[] = []; for (const status of ["completed", "cancelled", "failed"] as const) { const run = await repository.createDescriptionGenerationRun( database.id, "missing", configuredModel.id, "it", 3, ); ids.push(run.id); await repository.updateDescriptionGenerationRun(run.id, { status, processed: status === "completed" ? 3 : 1, generated: status === "completed" ? 3 : 1, failed: status === "failed" ? 1 : 0, finishedAt: new Date().toISOString(), errorSummary: status === "failed" ? "Safe technical summary." : null, ...({ provider: "openai", providerPayload: "raw-provider-payload", apiKey: "history-secret-key", prompt: "private persisted prompt", sample: "private source sample", stack: "private stack trace", } as object), } as never); await new Promise((resolve) => setTimeout(resolve, 2)); } const response = await app.inject({ method: "GET", url: "/catalog/description-generation-runs?limit=2", }); expect(response.statusCode).toBe(200); expect(response.json().map((run: { id: string }) => run.id)).toEqual(ids.slice(1).reverse()); expect(Object.keys(response.json()[0]).sort()).toEqual([ "cacheReadTokens", "createdAt", "databaseId", "errorSummary", "failed", "finishedAt", "generated", "id", "inputTokens", "language", "modelId", "nonGeneratable", "outputTokens", "processed", "scope", "startedAt", "status", "total", "updatedAt", ]); expect(response.body).not.toMatch( /raw-provider-payload|history-secret-key|private persisted prompt|private source sample|private stack trace/, ); expect((await app.inject({ method: "GET", url: "/catalog/description-generation-runs?limit=0", })).statusCode).toBe(400); expect(modelCompleter.complete).not.toHaveBeenCalled(); } finally { await app.close(); } }); test("Description Generation SSE replays, subscribes, and reconnects in persisted sequence order", async () => { let resolveCompletion!: (content: string) => void; const completion = new Promise((resolve) => { resolveCompletion = resolve; }); const modelCompleter: ModelCompleter = { complete: vi.fn(async () => await completion), }; const { app, database, column } = await setup(modelCompleter); const controllers: AbortController[] = []; try { const baseUrl = await app.listen({ host: "127.0.0.1", port: 0 }); const accepted = await fetch(`${baseUrl}/catalog/databases/${database.id}/description-generation-runs`, { method: "POST", headers: { "content-type": "application/json" }, body: JSON.stringify({ modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }), }); expect(accepted.status).toBe(202); const runId = (await accepted.json() as { id: string }).id; const firstController = new AbortController(); controllers.push(firstController); const firstResponse = await fetch( `${baseUrl}/catalog/description-generation-runs/${runId}/events?after=0`, { signal: firstController.signal }, ); expect(firstResponse.status).toBe(200); expect(firstResponse.headers.get("content-type")).toBe("text/event-stream; charset=utf-8"); const firstReader = sseFrameReader(firstResponse); const firstLogs: Array<{ sequence: number; message: string }> = []; while (firstLogs.length < 2) { const frame = await firstReader.next(); if (frame.event === "log") firstLogs.push(JSON.parse(frame.data)); } expect(firstLogs.map((event) => event.sequence)).toEqual([1, 2]); resolveCompletion(JSON.stringify({ results: [{ targetId: column.id, outcome: "generated", description: "Descrizione ricevuta via subscription.", }], })); let terminalRun: { status: string } | undefined; while (firstLogs.length < 4 || terminalRun?.status !== "completed") { const frame = await firstReader.next(); if (frame.event === "log") firstLogs.push(JSON.parse(frame.data)); if (frame.event === "run") terminalRun = JSON.parse(frame.data); } expect(firstLogs.map((event) => event.sequence)).toEqual([1, 2, 3, 4]); expect(firstLogs.at(-1)?.message).toBe("Description generation completed."); expect(terminalRun).toMatchObject({ status: "completed" }); await firstReader.cancel(); firstController.abort(); const reconnectController = new AbortController(); controllers.push(reconnectController); const reconnectResponse = await fetch( `${baseUrl}/catalog/description-generation-runs/${runId}/events?after=2`, { signal: reconnectController.signal }, ); const reconnectReader = sseFrameReader(reconnectResponse); const replayed: Array<{ sequence: number }> = []; let replayedRun: { status: string } | undefined; while (replayed.length < 2 || replayedRun?.status !== "completed") { const frame = await reconnectReader.next(); if (frame.event === "log") replayed.push(JSON.parse(frame.data)); if (frame.event === "run") replayedRun = JSON.parse(frame.data); } expect(replayed.map((event) => event.sequence)).toEqual([3, 4]); expect(replayedRun).toMatchObject({ status: "completed" }); await reconnectReader.cancel(); reconnectController.abort(); } finally { controllers.forEach((controller) => controller.abort()); resolveCompletion(JSON.stringify({ results: [] })); await app.close(); } }, 10_000); test("SSE exposes a terminal run signal for every outcome that may leave partial catalog writes", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("terminal replay must not invoke the model"); }), }; const { app, repository, database } = await setup(modelCompleter); try { const baseUrl = await app.listen({ host: "127.0.0.1", port: 0 }); for (const status of [ "completed", "completed_with_errors", "cancelled", "failed", "interrupted", ] as const) { const created = await repository.createDescriptionGenerationRun( database.id, "missing", configuredModel.id, "it", 1, ); await repository.updateDescriptionGenerationRun(created.id, { status, processed: status === "cancelled" || status === "interrupted" ? 0 : 1, generated: status === "completed" ? 1 : 0, failed: status === "completed_with_errors" || status === "failed" ? 1 : 0, finishedAt: new Date().toISOString(), errorSummary: status === "completed" || status === "cancelled" ? null : `Safe ${status} summary.`, }); await repository.appendDescriptionGenerationEvent(created.id, "warning", `Terminal ${status}.`); const response = await fetch( `${baseUrl}/catalog/description-generation-runs/${created.id}/events?after=0`, ); const reader = sseFrameReader(response); let observed: { databaseId: string; status: string } | undefined; while (!observed) { const frame = await reader.next(); if (frame.event === "run") observed = JSON.parse(frame.data); } expect(observed).toMatchObject({ databaseId: database.id, status }); } expect(modelCompleter.complete).not.toHaveBeenCalled(); } finally { await app.close(); } }); test("emits one safe metadata-only warning for each batch whose optional sampling fails", async () => { const samplingFailureSecret = "BATCH_SAMPLE_FAILURE_DETAIL_d0139a"; const descriptionSourceSampler: DescriptionSourceSampler = { sample: vi.fn(async () => { throw new Error(samplingFailureSecret); }), }; const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => { const userContent = request.messages.find((message) => message.role === "user")!.content; const context = JSON.parse(userContent.slice(userContent.indexOf("\n") + 1)); return JSON.stringify({ results: context.targets.map((target: { targetId: string }) => ({ targetId: target.targetId, outcome: "generated", description: "Descrizione metadata-only.", })), }); }), }; const { app, repository, database, table } = await setup( modelCompleter, {}, "it", descriptionSourceSampler, ); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: null }], columns: Array.from({ length: 11 }, (_, index) => ({ tableName: "patients", name: `column_${index}`, ordinalPosition: index + 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, })), relationships: [], }); const columnIds = (await repository.listColumns(database.id, table.id)).map((column) => column.id); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: columnIds, }, }); expect(start.statusCode).toBe(202); expect((await waitForTerminalRun(app, start.json().id)).run.status).toBe("completed"); expect(descriptionSourceSampler.sample).toHaveBeenCalledTimes(2); const events = await repository.listDescriptionGenerationEvents(start.json().id); expect(events.filter((event) => event.level === "warning").map((event) => event.message)).toEqual([ "Source samples unavailable for this batch; generation continued with catalog metadata only.", "Source samples unavailable for this batch; generation continued with catalog metadata only.", ]); expect(JSON.stringify(events)).not.toContain(samplingFailureSecret); expect(vi.mocked(modelCompleter.complete).mock.calls.every(([request]) => ( !JSON.stringify(request.messages).includes(samplingFailureSecret) ))).toBe(true); } finally { await app.close(); } }); test("retains completed batch writes when a later batch response is malformed", async () => { let orderedIds: string[] = []; let callIndex = 0; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { const currentCall = callIndex; callIndex += 1; if (currentCall === 0) { return JSON.stringify({ results: orderedIds.slice(0, 10).map((targetId, index) => ({ targetId, outcome: "generated", description: `Retained ${index + 1}`, })), }); } return JSON.stringify({ results: [{ targetId: orderedIds[10], outcome: "generated", description: " ", }], }); }), }; const { app, repository, database, table } = await setup(modelCompleter); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: null }], columns: Array.from({ length: 11 }, (_, index) => ({ tableName: "patients", name: `retained_${String(index + 1).padStart(2, "0")}`, ordinalPosition: index + 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, })), relationships: [], }); const originalColumns = await repository.listColumns(database.id, table.id); orderedIds = originalColumns.map((column) => column.id); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: orderedIds, }, }); const { run } = await waitForTerminalRun(app, start.json().id); expect(modelCompleter.complete).toHaveBeenCalledTimes(3); expect(run).toMatchObject({ status: "completed_with_errors", total: 11, processed: 11, generated: 10, nonGeneratable: 0, failed: 1, errorSummary: "Description generation completed with errors.", }); const updatedById = new Map( (await repository.listColumns(database.id, table.id)).map((column) => [column.id, column]), ); orderedIds.slice(0, 10).forEach((targetId, index) => { expect(updatedById.get(targetId)).toMatchObject({ generatedDescription: `Retained ${index + 1}`, version: originalColumns[index]!.version + 1, }); }); expect(updatedById.get(orderedIds[10]!)).toMatchObject({ generatedDescription: null, version: originalColumns[10]!.version, }); const events = await repository.listDescriptionGenerationEvents(run.id); expect(events.slice(2, 12).map((event) => event.message)).toEqual( orderedIds.slice(0, 10).map((targetId) => `Generated description for Catalog Column ${targetId}.`), ); expect(events.find((event) => event.level === "warning" && event.message.includes("Retrying batch"))).toEqual( expect.objectContaining({ message: "The model response did not match the required schema. Retrying batch (attempt 2 of 2).", }), ); expect(events.find((event) => event.level === "error")).toEqual(expect.objectContaining({ level: "error", message: `The model response did not match the required schema. Affected Catalog Column target: ${orderedIds[10]}.`, })); } finally { await app.close(); } }); test("rejects duplicate selected target IDs before launching generation", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("must not be called"); }), }; const { app, database, column, operations } = await setup(modelCompleter); try { const response = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id, column.id], }, }); expect(response.statusCode).toBe(400); expect(response.json()).toEqual({ code: "description_generation_target_ids_duplicate", message: "Description generation target IDs must be unique.", }); expect(modelCompleter.complete).not.toHaveBeenCalled(); const release = operations.reserve(database.id); release(); } finally { await app.close(); } }); test("requires strict database-wide scopes and rejects zero eligible targets before creating a run", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("must not be called"); }), }; const { app, repository, database, column, operations } = await setup(modelCompleter); try { const unexpectedTargetIds = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "all", targetIds: [column.id] }, }); expect(unexpectedTargetIds.statusCode).toBe(400); expect(unexpectedTargetIds.json()).toEqual({ code: "description_generation_request_invalid", message: "Description generation request is invalid.", }); await repository.deleteDatabaseMetadata([database.id], "tables"); for (const scope of ["all", "missing"] as const) { const response = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope }, }); expect(response.statusCode).toBe(409); expect(response.json()).toEqual({ code: "description_generation_no_eligible_targets", message: scope === "all" ? "No Catalog Tables or Catalog Columns are available for description generation." : "No Catalog Tables or Catalog Columns have a missing Generated Description.", }); } expect(modelCompleter.complete).not.toHaveBeenCalled(); const release = operations.reserve(database.id); release(); } finally { await app.close(); } }); test("Generate All replaces every description in column-first homogeneous batches with refreshed table context", async () => { const requests: ModelCompletionRequest[] = []; const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => { requests.push(request); const metadata = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")); const kind = metadata.targets.every((target: Record) => "column" in target) ? "column" : metadata.targets.every((target: Record) => "columns" in target) ? "table" : "mixed"; return JSON.stringify({ results: metadata.targets.map((target: { targetId: string }) => ({ targetId: target.targetId, outcome: "generated", description: `Generated ${kind} ${target.targetId}`, })), }); }), }; const { app, repository, database } = await setup(modelCompleter); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: Array.from({ length: 11 }, (_, index) => ({ name: `table_${String(index + 1).padStart(2, "0")}`, sourceComment: `Table ${index + 1}`, })), columns: Array.from({ length: 11 }, (_, index) => ({ tableName: `table_${String(index + 1).padStart(2, "0")}`, name: `column_${String(index + 1).padStart(2, "0")}`, ordinalPosition: 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: `Column ${index + 1}`, })), relationships: [], }); const originalTables = await repository.listTables(database.id); const originalColumns = (await Promise.all( originalTables.map(async (table) => await repository.listColumns(database.id, table.id)), )).flat(); await repository.updateTableMetadata( database.id, originalTables[0]!.id, originalTables[0]!.version, originalTables[0]!.description, "Existing generated table description", ); await repository.updateColumnMetadata( database.id, originalTables[0]!.id, originalColumns[0]!.id, originalColumns[0]!.version, originalColumns[0]!.description, "Existing generated column description", ); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "all" }, }); expect(start.statusCode).toBe(202); expect(start.json()).toMatchObject({ scope: "all", total: 22 }); const { run } = await waitForTerminalRun(app, start.json().id); expect(run).toMatchObject({ scope: "all", status: "completed", total: 22, processed: 22, generated: 22, nonGeneratable: 0, failed: 0, }); expect(requests).toHaveLength(4); const requestMetadata = requests.map((request) => ({ system: request.messages[0]!.content, metadata: JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")), })); expect(requestMetadata.map(({ metadata }) => metadata.targets.length)).toEqual([10, 1, 10, 1]); expect(requestMetadata.slice(0, 2).every(({ system, metadata }) => ( system.includes("Catalog Column") && metadata.targets.every((target: Record) => "column" in target && !("columns" in target)) ))).toBe(true); expect(requestMetadata.slice(2).every(({ system, metadata }) => ( system.includes("Catalog Table") && metadata.targets.every((target: Record) => "columns" in target && !("column" in target)) ))).toBe(true); expect(requestMetadata.slice(0, 2).flatMap(({ metadata }) => ( metadata.targets.map((target: { targetId: string }) => target.targetId) ))).toEqual(originalColumns.map((column) => column.id)); expect(requestMetadata.slice(2).flatMap(({ metadata }) => ( metadata.targets.map((target: { targetId: string }) => target.targetId) ))).toEqual(originalTables.map((table) => table.id)); const columnByTable = new Map(originalColumns.map((column) => [column.tableId, column])); for (const target of requestMetadata.slice(2).flatMap(({ metadata }) => metadata.targets)) { const column = columnByTable.get(target.targetId)!; expect(target.columns[0].currentDescription).toBe(`Generated column ${column.id}`); } const updatedTables = await repository.listTables(database.id); const updatedColumns = (await Promise.all( updatedTables.map(async (table) => await repository.listColumns(database.id, table.id)), )).flat(); expect(updatedTables.map((table) => table.generatedDescription)).toEqual( updatedTables.map((table) => `Generated table ${table.id}`), ); expect(updatedColumns.map((column) => column.generatedDescription)).toEqual( updatedColumns.map((column) => `Generated column ${column.id}`), ); expect(updatedTables[0]!.generatedDescription).not.toBe("Existing generated table description"); expect(updatedColumns[0]!.generatedDescription).not.toBe("Existing generated column description"); const events = await repository.listDescriptionGenerationEvents(run.id); expect(events[0]!.message).toBe("Description generation queued (scope: all)."); expect(events[1]!.message).toBe("Description generation started (scope: all)."); expect(events.at(-1)!.message).toBe("Description generation completed (scope: all)."); } finally { await app.close(); } }); test("Generate Missing skips prior partial results and includes null, empty, and whitespace descriptions", async () => { let mode: "partial" | "missing" = "partial"; let partialCall = 0; const missingRequests: ModelCompletionRequest[] = []; const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => { const metadata = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")); if (mode === "partial") { partialCall += 1; if (partialCall === 2 || partialCall === 3) throw new ModelCompletionProviderError(); return JSON.stringify({ results: metadata.targets.map((target: { targetId: string }) => ({ targetId: target.targetId, outcome: "generated", description: `Retained partial ${target.targetId}`, })), }); } missingRequests.push(request); return JSON.stringify({ results: metadata.targets.map((target: { targetId: string }) => ({ targetId: target.targetId, outcome: "generated", description: `Recovered missing ${target.targetId}`, })), }); }), }; const { app, repository, database } = await setup(modelCompleter); try { const tableNames = ["table_01", "table_02", "table_03"]; await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: tableNames.map((name) => ({ name, sourceComment: null })), columns: Array.from({ length: 11 }, (_, index) => ({ tableName: index < 9 ? tableNames[0]! : tableNames[index - 8]!, name: `column_${String(index + 1).padStart(2, "0")}`, ordinalPosition: index < 9 ? index + 1 : 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, })), relationships: [], }); const tables = await repository.listTables(database.id); const columns = (await Promise.all( tables.map(async (table) => await repository.listColumns(database.id, table.id)), )).flat(); const seededTables = [ await repository.updateTableMetadata( database.id, tables[0]!.id, tables[0]!.version, null, "Existing valid table result", ), await repository.updateTableMetadata( database.id, tables[1]!.id, tables[1]!.version, null, " ", ), await repository.updateTableMetadata( database.id, tables[2]!.id, tables[2]!.version, null, "", ), ]; const partialStart = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: columns.map((column) => column.id), }, }); const { run: partialRun } = await waitForTerminalRun(app, partialStart.json().id); expect(partialRun).toMatchObject({ status: "completed_with_errors", total: 11, processed: 11, generated: 10, failed: 1, }); const afterPartial = (await Promise.all( tables.map(async (table) => await repository.listColumns(database.id, table.id)), )).flat(); expect(afterPartial.slice(0, 10).every((column) => ( column.generatedDescription === `Retained partial ${column.id}` ))).toBe(true); expect(afterPartial[10]!.generatedDescription).toBeNull(); mode = "missing"; const missingStart = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "missing" }, }); expect(missingStart.statusCode).toBe(202); expect(missingStart.json()).toMatchObject({ scope: "missing", total: 3 }); const { run: missingRun } = await waitForTerminalRun(app, missingStart.json().id); expect(missingRun).toMatchObject({ scope: "missing", status: "completed", total: 3, processed: 3, generated: 3, nonGeneratable: 0, failed: 0, }); expect(missingRequests).toHaveLength(2); const missingMetadata = missingRequests.map((request) => ( JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")) )); expect(missingMetadata[0]!.targets.map((target: { targetId: string }) => target.targetId)).toEqual([ afterPartial[10]!.id, ]); expect(missingMetadata[0]!.targets.every((target: Record) => "column" in target)).toBe(true); expect(missingMetadata[1]!.targets.map((target: { targetId: string }) => target.targetId)).toEqual([ tables[1]!.id, tables[2]!.id, ]); expect(missingMetadata[1]!.targets.every((target: Record) => "columns" in target)).toBe(true); const recoveredTableThree = missingMetadata[1]!.targets.find( (target: { targetId: string }) => target.targetId === tables[2]!.id, ); expect(recoveredTableThree.columns[0].currentDescription).toBe( `Recovered missing ${afterPartial[10]!.id}`, ); const finalColumns = (await Promise.all( tables.map(async (table) => await repository.listColumns(database.id, table.id)), )).flat(); expect(finalColumns.slice(0, 10)).toEqual(afterPartial.slice(0, 10)); expect(finalColumns[10]!.generatedDescription).toBe(`Recovered missing ${afterPartial[10]!.id}`); const finalTables = await repository.listTables(database.id); expect(finalTables[0]).toEqual(seededTables[0]); expect(finalTables[1]!.generatedDescription).toBe(`Recovered missing ${tables[1]!.id}`); expect(finalTables[2]!.generatedDescription).toBe(`Recovered missing ${tables[2]!.id}`); const events = await repository.listDescriptionGenerationEvents(missingRun.id); expect(events[0]!.message).toBe("Description generation queued (scope: missing)."); expect(events.at(-1)!.message).toBe("Description generation completed (scope: missing)."); const noMissingTargets = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "missing" }, }); expect(noMissingTargets.statusCode).toBe(409); expect(noMissingTargets.json()).toEqual({ code: "description_generation_no_eligible_targets", message: "No Catalog Tables or Catalog Columns have a missing Generated Description.", }); expect(missingRequests).toHaveLength(2); } finally { await app.close(); } }); test("generates selected Catalog Tables with structural column context and localized non-generatable text", async () => { let selectedTableId = ""; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => JSON.stringify({ results: [{ targetId: selectedTableId, outcome: "non_generatable" }], })), }; const { app, repository, database, table } = await setup(modelCompleter); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: "Clinical patients" }], columns: [ { tableName: "patients", name: "patient_id", ordinalPosition: 1, dataType: "uuid", isNullable: false, defaultExpression: "gen_random_uuid()", primaryKeyPosition: 1, sourceComment: "Patient identifier", }, { tableName: "patients", name: "status", ordinalPosition: 2, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: "Patient status", }, ], relationships: [], }); const currentTable = (await repository.getTable(database.id, table.id))!; const curatedTable = (await repository.updateTableMetadata( database.id, currentTable.id, currentTable.version, "Elenco curato dei pazienti.", null, ))!; const columns = await repository.listColumns(database.id, table.id); const patientId = columns.find((column) => column.name === "patient_id")!; const status = columns.find((column) => column.name === "status")!; await repository.updateColumnMetadata( database.id, table.id, patientId.id, patientId.version, "Identificativo curato.", "Identificativo generato.", ); await repository.updateColumnMetadata( database.id, table.id, status.id, status.version, "Stato curato.", null, ); selectedTableId = table.id; const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_tables", targetIds: [table.id], }, }); expect(start.statusCode).toBe(202); const { run } = await waitForTerminalRun(app, start.json().id); expect(run).toMatchObject({ scope: "selected_tables", language: "it", status: "completed", total: 1, processed: 1, generated: 0, nonGeneratable: 1, failed: 0, }); const request = (modelCompleter.complete as ReturnType).mock.calls[0]![0] as ModelCompletionRequest; expect(request.messages[0]!.content).toContain("Catalog Table"); expect(request.messages[0]!.content).toContain('workspace language "it"'); expect(request.messages[0]!.content).toContain("non_generatable"); expect(request.messages[0]!.content).toContain("untrusted"); const metadata = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")); expect(metadata.targets).toEqual([expect.objectContaining({ targetId: table.id, table: expect.objectContaining({ name: "patients", sourceComment: "Clinical patients", description: "Elenco curato dei pazienti.", generatedDescription: null, }), columns: [ expect.objectContaining({ name: "patient_id", ordinalPosition: 1, dataType: "uuid", isNullable: false, defaultExpression: "gen_random_uuid()", isPrimaryKey: true, currentDescription: "Identificativo generato.", }), expect.objectContaining({ name: "status", ordinalPosition: 2, dataType: "text", isNullable: true, currentDescription: "Stato curato.", }), ], })]); expect(await repository.getTable(database.id, table.id)).toMatchObject({ description: "Elenco curato dei pazienti.", generatedDescription: "Non generabile", version: curatedTable.version + 1, }); expect((await repository.listDescriptionGenerationEvents(run.id)).map((event) => event.message)).toEqual([ "Description generation queued.", "Description generation started.", `Stored non-generatable result for Catalog Table ${table.id}.`, "Description generation completed.", ]); } finally { await app.close(); } }); test("localizes valid non-generatable Catalog Column results in English", async () => { let selectedColumnId = ""; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => JSON.stringify({ results: [{ targetId: selectedColumnId, outcome: "non_generatable" }], })), }; const { app, repository, database, table, column } = await setup(modelCompleter, {}, "en"); selectedColumnId = column.id; try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id], }, }); const { run } = await waitForTerminalRun(app, start.json().id); expect(run).toMatchObject({ language: "en", status: "completed", processed: 1, generated: 0, nonGeneratable: 1, failed: 0, }); expect(await repository.getColumn(database.id, table.id, column.id)).toMatchObject({ description: null, generatedDescription: "Not generatable", version: column.version + 1, }); const request = (modelCompleter.complete as ReturnType).mock.calls[0]![0] as ModelCompletionRequest; expect(request.messages[0]!.content).toContain('workspace language "en"'); } finally { await app.close(); } }); test("retries invalid JSON once and completes the batch when the second response is valid", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => { if (vi.mocked(modelCompleter.complete).mock.calls.length === 1) { return { content: "not-json", usage: { input: 11, cacheRead: 3, output: 2 } }; } const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n")); return { content: JSON.stringify({ results: context.targets.map((target: { targetId: string }) => ({ targetId: target.targetId, outcome: "generated", description: "Generated after the application retry.", })), }), usage: { input: 7, cacheRead: 1, output: 5 }, }; }), }; const { app, repository, database, table, column } = await setup(modelCompleter); try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] }, }); const { run } = await waitForTerminalRun(app, start.json().id); expect(run).toMatchObject({ status: "completed", processed: 1, generated: 1, failed: 0, inputTokens: 18, cacheReadTokens: 4, outputTokens: 7, }); expect(modelCompleter.complete).toHaveBeenCalledTimes(2); expect(await repository.getColumn(database.id, table.id, column.id)).toMatchObject({ generatedDescription: "Generated after the application retry.", }); expect(await repository.listDescriptionGenerationEvents(run.id)).toEqual(expect.arrayContaining([ expect.objectContaining({ level: "warning", message: "The model response was not valid JSON. Retrying batch (attempt 2 of 2).", }), ])); } finally { await app.close(); } }); test("fails safely when the provider fails and redacts provider diagnostics", async () => { const sensitiveDiagnostic = "test-provider-secret private prompt raw provider payload"; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw Object.assign(new ModelCompletionProviderError(), { message: sensitiveDiagnostic }); }), }; const { app, database, table, column } = await setup(modelCompleter); try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] }, }); const { run } = await waitForTerminalRun(app, start.json().id); expect(modelCompleter.complete).toHaveBeenCalledTimes(2); expect(run).toMatchObject({ status: "completed_with_errors", processed: 1, generated: 0, nonGeneratable: 0, failed: 1, errorSummary: "Description generation completed with errors.", }); const columns = await app.inject({ method: "GET", url: `/catalog/databases/${database.id}/tables/${table.id}/columns`, }); expect(columns.json()).toEqual([ expect.objectContaining({ id: column.id, generatedDescription: null, version: column.version, }), ]); const events = await app.inject({ method: "GET", url: `/catalog/description-generation-runs/${run.id}/events-list`, }); expect(`${JSON.stringify(run)}${events.body}`).not.toContain(sensitiveDiagnostic); expect(events.json().find((event: { level: string }) => event.level === "error")).toEqual(expect.objectContaining({ level: "error", message: `The model provider request failed. Affected Catalog Column target: ${column.id}.`, })); } finally { await app.close(); } }); test.each([ ["duplicate mappings", (targetIds: readonly string[]) => ({ results: [ { targetId: targetIds[0], outcome: "generated", description: "First valid value" }, { targetId: targetIds[0], outcome: "non_generatable" }, ] }), "The model response was missing one or more requested targets."], ["unknown mappings", (targetIds: readonly string[]) => ({ results: [ { targetId: targetIds[0], outcome: "non_generatable" }, { targetId: "99999999-9999-4999-8999-999999999999", outcome: "generated", description: "Unknown target value", }, ] }), "The model response was missing one or more requested targets."], ["missing mappings", (targetIds: readonly string[]) => ({ results: [ { targetId: targetIds[0], outcome: "generated", description: "Only one result" }, ] }), "The model response was missing one or more requested targets."], ["malformed mappings", (targetIds: readonly string[]) => ({ results: [ { targetId: targetIds[0], outcome: "generated", description: "First valid value" }, { targetId: targetIds[1], outcome: "generated", description: " " }, ] }), "The model response did not match the required schema."], ] as const)("rejects %s without applying any result from the batch", async (_name, responseFor, failureMessage) => { let selectedColumnIds: string[] = []; const modelCompleter: ModelCompleter = { complete: vi.fn(async () => JSON.stringify(responseFor(selectedColumnIds))), }; const { app, repository, database, table } = await setup(modelCompleter); try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: null }], columns: [ { tableName: "patients", name: "first_column", ordinalPosition: 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, }, { tableName: "patients", name: "second_column", ordinalPosition: 2, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, }, ], relationships: [], }); const originalColumns = await repository.listColumns(database.id, table.id); selectedColumnIds = originalColumns.map((column) => column.id); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: selectedColumnIds }, }); const { run } = await waitForTerminalRun(app, start.json().id); expect(modelCompleter.complete).toHaveBeenCalledTimes(2); expect(run).toMatchObject({ status: "completed_with_errors", processed: 2, generated: 0, nonGeneratable: 0, failed: 2, errorSummary: "Description generation completed with errors.", }); expect(await repository.listColumns(database.id, table.id)).toEqual( originalColumns.map((column) => expect.objectContaining({ id: column.id, description: column.description, generatedDescription: null, version: column.version, })), ); expect((await repository.listDescriptionGenerationEvents(run.id)).find((event) => event.level === "error")).toEqual( expect.objectContaining({ level: "error", message: `${failureMessage} Affected Catalog Column targets: ${selectedColumnIds.join(", ")}.`, }), ); } finally { await app.close(); } }); test("returns an explicit global conflict while another run is active", async () => { let resolveCompletion!: (content: string) => void; const completion = new Promise((resolve) => { resolveCompletion = resolve; }); const modelCompleter = { complete: vi.fn(async () => await completion) }; const { app, database, column } = await setup(modelCompleter); let firstRunId = ""; try { const first = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] }, }); firstRunId = first.json().id; const second = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "missing" }, }); expect(second.statusCode).toBe(409); expect(second.json()).toEqual({ code: "description_generation_run_active", message: "A Description Generation Run is already active.", }); } finally { resolveCompletion(JSON.stringify({ targetId: column.id, outcome: "generated", description: "Descrizione valida.", })); if (firstRunId) await waitForTerminalRun(app, firstRunId); await app.close(); } }); test("rejects generation while another catalog operation owns the database", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("must not be called"); }), }; const { app, database, operations } = await setup(modelCompleter); const release = operations.reserve(database.id); try { const response = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "all" }, }); expect(response.statusCode).toBe(409); expect(response.json()).toEqual({ code: "database_operation_in_progress", message: "A database operation is already in progress.", }); expect(modelCompleter.complete).not.toHaveBeenCalled(); } finally { release(); await app.close(); } }); test("requires database.manage for every Description Generation route", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("must not be called"); }), }; const { app, database } = await setup(modelCompleter, { AUTH_MODE: "upstream" }); const headers = { "x-thoth-principal-issuer": "portal", "x-thoth-principal-subject": "catalog-reader", "x-thoth-is-admin": "0", }; try { const responses = await Promise.all([ app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, headers, payload: { modelId: configuredModel.id, scope: "all" }, }), app.inject({ method: "GET", url: "/catalog/description-generation-runs/99999999-9999-4999-8999-999999999999", headers, }), app.inject({ method: "GET", url: "/catalog/description-generation-runs/99999999-9999-4999-8999-999999999999/events-list", headers, }), app.inject({ method: "GET", url: "/catalog/description-generation-runs", headers, }), app.inject({ method: "POST", url: "/catalog/description-generation-runs/99999999-9999-4999-8999-999999999999/cancel", headers, }), app.inject({ method: "POST", url: "/catalog/description-generation-runs/unlock", headers, }), app.inject({ method: "GET", url: "/catalog/description-generation-runs/99999999-9999-4999-8999-999999999999/events", headers, }), app.inject({ method: "GET", url: "/catalog/sensitive-data-suggestion-runs", headers, }), app.inject({ method: "GET", url: "/catalog/sensitive-data-suggestion-runs/99999999-9999-4999-8999-999999999999", headers, }), app.inject({ method: "GET", url: "/catalog/sensitive-data-suggestion-runs/99999999-9999-4999-8999-999999999999/events-list", headers, }), ]); expect(responses.map((response) => response.statusCode)).toEqual([ 403, 403, 403, 403, 403, 403, 403, 403, 403, 403, ]); expect(modelCompleter.complete).not.toHaveBeenCalled(); } finally { await app.close(); } }); test("validates selected targets and resolves every requested target before launching", async () => { const modelCompleter: ModelCompleter = { complete: vi.fn(async () => { throw new Error("must not be called"); }), }; const { app, repository, database, column, operations } = await setup(modelCompleter); try { const foreignDatabase = await repository.create({ workspaceId: "other-workspace", engine: "postgres", databaseName: "other", schema: "public", binding: { transport: "postgres_direct", host: "other.internal", port: 5432, username: "reader" }, }); await repository.applySchemaSync(foreignDatabase.id, foreignDatabase.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "foreign_table", sourceComment: null }], columns: [{ tableName: "foreign_table", name: "foreign_column", ordinalPosition: 1, dataType: "text", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: null, }], relationships: [], }); const foreignTable = (await repository.listTables(foreignDatabase.id))[0]!; const foreignColumn = (await repository.listColumns(foreignDatabase.id, foreignTable.id))[0]!; const invalidPayloads = await Promise.all([ app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [] }, }), app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id, foreignColumn.id] }, }), ]); expect(invalidPayloads.map((response) => response.statusCode)).toEqual([400, 404]); expect(invalidPayloads[1]!.json()).toEqual({ code: "catalog_column_not_found", message: "One or more selected Catalog Columns were not found.", }); const unknownModel = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: "openai/unknown-model", scope: "selected_columns", targetIds: [column.id] }, }); expect(unknownModel.statusCode).toBe(409); expect(unknownModel.json().code).toBe("metadata_generation_model_unavailable"); const wrongDatabase = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [foreignColumn.id] }, }); expect(wrongDatabase.statusCode).toBe(404); expect(wrongDatabase.json().code).toBe("catalog_column_not_found"); expect(modelCompleter.complete).not.toHaveBeenCalled(); const release = operations.reserve(database.id); release(); } finally { await app.close(); } }); test("blocks table and column metadata edits while generation owns the database", async () => { let resolveCompletion!: (content: string) => void; const completion = new Promise((resolve) => { resolveCompletion = resolve; }); const modelCompleter = { complete: vi.fn(async () => await completion) }; const { app, database, table, column } = await setup(modelCompleter); let runId = ""; try { const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] }, }); runId = start.json().id; const responses = await Promise.all([ app.inject({ method: "PATCH", url: `/catalog/databases/${database.id}/tables/${table.id}`, payload: { version: table.version, description: "Concurrent table edit" }, }), app.inject({ method: "PATCH", url: `/catalog/databases/${database.id}/tables/${table.id}/columns/${column.id}`, payload: { version: column.version, description: "Concurrent column edit", generatedDescription: null, }, }), ]); expect(responses.map((response) => response.statusCode)).toEqual([409, 409]); expect(responses.map((response) => response.json().code)).toEqual([ "database_operation_in_progress", "database_operation_in_progress", ]); } finally { resolveCompletion(JSON.stringify({ targetId: column.id, outcome: "generated", description: "Descrizione valida.", })); if (runId) await waitForTerminalRun(app, runId); await app.close(); } }); test("bounds the metadata-only prompt before calling the selected model", async () => { let selectedColumnId = ""; const modelCompleter = { complete: vi.fn(async () => JSON.stringify({ targetId: selectedColumnId, outcome: "generated", description: "Descrizione valida.", })), }; const { app, repository, database } = await setup(modelCompleter); const omittedTail = "OMITTED_UNTRUSTED_METADATA_TAIL"; const oversizedMetadata = `Beginning ${"x".repeat(2_500)} ${omittedTail}`; try { await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: [{ name: "patients", sourceComment: oversizedMetadata }], columns: [{ tableName: "patients", name: "birth_date", ordinalPosition: 1, dataType: "date", isNullable: true, defaultExpression: null, primaryKeyPosition: null, sourceComment: oversizedMetadata, }], relationships: [], }); const table = (await repository.listTables(database.id))[0]!; const column = (await repository.listColumns(database.id, table.id))[0]!; selectedColumnId = column.id; const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_columns", targetIds: [column.id] }, }); await waitForTerminalRun(app, start.json().id); const request = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest; const serializedMessages = JSON.stringify(request.messages); expect(serializedMessages.length).toBeLessThan(10_000); expect(serializedMessages).toContain("Beginning"); expect(serializedMessages).not.toContain(omittedTail); expect(serializedMessages).not.toMatch(/sourceRows|sampleRows|sampleValues/i); } finally { await app.close(); } }); test("bounds aggregate structural columns and source samples below helper message limits", async () => { const requests: ModelCompletionRequest[] = []; const modelCompleter: ModelCompleter = { complete: vi.fn(async (request) => { requests.push(request); const userContent = request.messages.find((message) => message.role === "user")!.content; const context = JSON.parse(userContent.slice(userContent.indexOf("\n") + 1)); return JSON.stringify({ results: context.targets.map((target: { targetId: string }) => ({ targetId: target.targetId, outcome: "generated", description: "Descrizione bounded.", })), }); }), }; const hostileText = `FIRST_USEFUL_FACT_${'"\\\n'.repeat(900)}`; const omittedTail = "LAST_STRUCTURAL_COLUMN_MUST_BE_OMITTED"; const descriptionSourceSampler: DescriptionSourceSampler = { sample: vi.fn(async (_database, targets) => targets.map((target) => { const sampleColumns = target.columnNames.slice(0, 8); return { targetId: target.targetId, tableName: target.tableName, rows: Array.from({ length: 6 }, (_, rowIndex) => ({ fields: sampleColumns.map((name, columnIndex) => ({ name, value: `SAMPLE_${rowIndex}_${columnIndex}_${'"\\\n'.repeat(300)}`, })), })), representativeValues: sampleColumns.map((column, columnIndex) => ({ column, values: Array.from( { length: 7 }, (_, valueIndex) => `EXAMPLE_${columnIndex}_${valueIndex}_${'"\\\n'.repeat(300)}`, ), })), }; })), }; const { app, repository, database } = await setup( modelCompleter, {}, "it", descriptionSourceSampler, ); try { const tableInputs = Array.from({ length: 10 }, (_, tableIndex) => ({ name: `table_${String(tableIndex).padStart(2, "0")}`, sourceComment: tableIndex === 0 ? hostileText : `Table ${tableIndex}`, })); const columnInputs = tableInputs.flatMap((table, tableIndex) => ( Array.from({ length: 80 }, (_, columnIndex) => ({ tableName: table.name, name: tableIndex === 9 && columnIndex === 79 ? omittedTail : `column_${String(columnIndex).padStart(2, "0")}`, ordinalPosition: columnIndex + 1, dataType: `text_${'x'.repeat(300)}`, isNullable: true, defaultExpression: hostileText, primaryKeyPosition: null, sourceComment: columnIndex === 0 ? `COLUMN_FIRST_${hostileText}` : hostileText, })) )); await repository.applySchemaSync(database.id, database.version, "all", [], { schemaVersion: 1, capabilities: { tables: "available", columns: "available", relationships: "available" }, tables: tableInputs, columns: columnInputs, relationships: [], }); const tables = await repository.listTables(database.id); const start = await app.inject({ method: "POST", url: `/catalog/databases/${database.id}/description-generation-runs`, payload: { modelId: configuredModel.id, scope: "selected_tables", targetIds: tables.map((table) => table.id), }, }); expect(start.statusCode).toBe(202); expect((await waitForTerminalRun(app, start.json().id)).run.status).toBe("completed"); expect(requests).toHaveLength(1); const contentBytes = requests[0]!.messages.map((message) => Buffer.byteLength(message.content, "utf8")); expect(contentBytes.every((bytes) => bytes <= 64 * 1024)).toBe(true); expect(contentBytes.reduce((total, bytes) => total + bytes, 0)).toBeLessThanOrEqual(128 * 1024); const serializedMessages = JSON.stringify(requests[0]!.messages); expect(serializedMessages).toContain("FIRST_USEFUL_FACT"); expect(serializedMessages).toContain("COLUMN_FIRST"); expect(serializedMessages).toContain("sourceSample"); expect(serializedMessages).not.toContain(omittedTail); const userMessage = requests[0]!.messages.find((message) => message.role === "user")!.content; const context = JSON.parse(userMessage.slice(userMessage.indexOf("\n") + 1)); expect(context.targets.every((target: { columns?: unknown[] }) => ( (target.columns?.length ?? 0) <= 24 ))).toBe(true); for (const target of context.targets as Array<{ sourceSample?: { rows: Array<{ fields: Array<{ name: string; value: unknown }> }>; representativeValues: Array<{ column: string; values: unknown[] }>; }; }>) { if (!target.sourceSample) continue; expect(Buffer.byteLength(JSON.stringify(target.sourceSample), "utf8")).toBeLessThanOrEqual(8 * 1024); expect(target.sourceSample.rows.length).toBeLessThanOrEqual(5); expect(target.sourceSample.rows.every((row) => row.fields.length <= 4)).toBe(true); expect(target.sourceSample.representativeValues.every((examples) => ( examples.values.length <= 5 ))).toBe(true); for (const row of target.sourceSample.rows) { for (const field of row.fields) { expect(Buffer.byteLength(JSON.stringify(field.name).slice(1, -1), "utf8")).toBeLessThanOrEqual(128); if (typeof field.value === "string") { expect(Buffer.byteLength(JSON.stringify(field.value).slice(1, -1), "utf8")).toBeLessThanOrEqual(192); } } } } } finally { await app.close(); } });