feat: classify sensitive columns locally

This commit is contained in:
Codex
2026-09-03 02:11:13 +02:00
parent 7b87e95427
commit f114d0065a
57 changed files with 4038 additions and 1149 deletions
@@ -14,6 +14,7 @@ import {
type ModelCompletionRequest,
} from "../src/catalog/model-completer.js";
import { CatalogOperationCoordinator } from "../src/catalog/operation-coordinator.js";
import type { SensitivityValueSource } from "../src/catalog/sensitivity-classifier.js";
import type {
CatalogDatabaseClient,
CatalogPostgresAccess,
@@ -69,6 +70,16 @@ async function setup(
sample: vi.fn(async () => []),
},
catalogPostgresAccess?: CatalogPostgresAccess,
sensitivityValueSource: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume(request.columns.map((column) => ({
columnId: column.id,
value: "ordinary",
characterLength: 8,
})));
return { kind: "complete", observedRows: 1 };
}),
},
) {
const repository = new MemoryCatalogRepository();
const database = await repository.create({
@@ -115,10 +126,11 @@ async function setup(
catalogOperationCoordinator: operations,
metadataGenerationModels: models(),
modelCompleter,
sensitivityValueSource,
...(descriptionSourceSampler ? { descriptionSourceSampler } : {}),
...(catalogPostgresAccess ? { catalogPostgresAccess } : {}),
});
return { app, repository, database, table, column, operations };
return { app, repository, database, table, column, operations, sensitivityValueSource };
}
async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: string) {
@@ -136,22 +148,15 @@ async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: strin
throw new Error(`Description Generation Run ${runId} did not finish`);
}
test("suggests sensitive flags from structural metadata without persisting them", async () => {
const modelCompleter = {
complete: vi.fn(async () => JSON.stringify({
suggestions: [{ columnId: expect.any(String), sensitive: true }],
})),
};
test("assesses sensitive flags locally without persisting them or calling an LLM", async () => {
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
const { app, repository, database, table, column } = await setup(modelCompleter);
modelCompleter.complete.mockResolvedValueOnce(JSON.stringify({
suggestions: [{ columnId: column.id, sensitive: true }],
}));
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
payload: { scope: "all" },
});
expect(response.statusCode).toBe(200);
@@ -160,11 +165,14 @@ test("suggests sensitive flags from structural metadata without persisting them"
run: {
databaseId: database.id,
scope: "all",
modelId: configuredModel.id,
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
status: "completed",
total: 1,
suggestedSensitive: 1,
suggestedNonSensitive: 0,
unknown: 0,
errorSummary: null,
},
suggestions: [{
@@ -175,6 +183,8 @@ test("suggests sensitive flags from structural metadata without persisting them"
version: column.version,
currentSensitive: false,
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
}],
});
expect(await repository.getColumn(database.id, column.tableId, column.id))
@@ -207,49 +217,56 @@ test("suggests sensitive flags from structural metadata without persisting them"
runId: responseBody.run.id,
sequence: 1,
level: "info",
message: "Sensitive-field suggestion generation started.",
message: "Local sensitivity analysis started.",
},
{
runId: responseBody.run.id,
sequence: 2,
level: "info",
message: "Classified 1 of 1 columns.",
message: "Assessed 1 of 1 columns locally.",
},
{
runId: responseBody.run.id,
sequence: 3,
level: "info",
message: "Sensitive-field suggestion generation completed for 1 column.",
message: "Local sensitivity analysis completed for 1 column.",
},
]);
const request = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest;
const prompt = request.messages.map((message) => message.content).join("\n");
expect(prompt).toContain("patients");
expect(prompt).toContain("birth_date");
expect(prompt).toContain("date");
expect(prompt).not.toContain("Patient date of birth");
expect(prompt).not.toContain("test-provider-secret");
expect(modelCompleter.complete).not.toHaveBeenCalled();
} finally {
await app.close();
}
});
test("limits sensitive-data suggestions to the selected tables or columns", async () => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async (request) => {
const payload = JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
columns: Array<{ columnId: string; column: string }>;
};
return JSON.stringify({
suggestions: payload.columns.map((column) => ({
columnId: column.columnId,
sensitive: column.column.includes("name") || column.column.includes("note"),
})),
});
}),
};
const { app, repository, database } = await setup(modelCompleter);
test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => {
const controller = new AbortController();
controller.abort();
const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal);
const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") });
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { scope: "all" },
});
expect(response.statusCode).toBe(504);
expect(response.json()).toEqual({
code: "sensitivity_analysis_timeout",
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
});
expect(await repository.listSensitivityAnalysisRuns()).toEqual([]);
} finally {
timeout.mockRestore();
await app.close();
}
});
test("limits sensitivity analysis to the selected tables or columns", async () => {
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
const { app, repository, database, sensitivityValueSource } = await setup(modelCompleter);
await repository.applySchemaSync(database.id, database.version, "all", [], {
schemaVersion: 1,
capabilities: { tables: "available", columns: "available", relationships: "available" },
@@ -281,7 +298,6 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_tables",
targetIds: [visits.id, patients.id],
},
@@ -301,7 +317,6 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_columns",
targetIds: [clinicalNote.id, status.id],
},
@@ -313,44 +328,41 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
expect.objectContaining({ tableId: visits.id, columnId: clinicalNote.id, sensitive: true }),
]));
const prompts = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => (
JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
columns: Array<{ columnId: string }>;
}
));
expect(prompts[0]!.columns.map((column) => column.columnId).sort()).toEqual(
[...patientColumns, ...visitColumns].map((column) => column.id).sort(),
);
expect(prompts[0]!.columns.map((column) => column.columnId)).not.toContain(billingColumns[0]!.id);
expect(prompts[1]!.columns.map((column) => column.columnId).sort()).toEqual(
[status.id, clinicalNote.id].sort(),
const scannedColumnIds = vi.mocked(sensitivityValueSource.scanTable).mock.calls.flatMap(
([request]) => request.columns.map((column) => column.id),
);
expect(scannedColumnIds).toEqual([status.id, status.id]);
expect(scannedColumnIds).not.toContain(patientColumns.find(
(column) => column.name === "patient_name",
)!.id);
expect(scannedColumnIds).not.toContain(clinicalNote.id);
expect(scannedColumnIds).not.toContain(billingColumns[0]!.id);
expect(modelCompleter.complete).not.toHaveBeenCalled();
} finally {
await app.close();
}
});
test("explains invalid sensitive-data suggestion selections without calling the model", async () => {
test("explains invalid sensitivity-analysis selections without reading source values", async () => {
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
const { app, database, table } = await setup(modelCompleter);
const { app, database, table, sensitivityValueSource } = await setup(modelCompleter);
try {
const empty = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "selected_tables", targetIds: [] },
payload: { scope: "selected_tables", targetIds: [] },
});
expect(empty.statusCode).toBe(400);
expect(empty.json()).toEqual({
code: "sensitive_data_suggestion_request_invalid",
message: "Choose a database, one or more tables, or one or more columns to classify.",
message: "Choose a database, one or more tables, or one or more columns to assess.",
});
const duplicate = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_tables",
targetIds: [table.id, table.id],
},
@@ -365,7 +377,6 @@ test("explains invalid sensitive-data suggestion selections without calling the
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_tables",
targetIds: ["00000000-0000-4000-8000-000000000001"],
},
@@ -380,7 +391,6 @@ test("explains invalid sensitive-data suggestion selections without calling the
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: {
modelId: configuredModel.id,
scope: "selected_columns",
targetIds: ["00000000-0000-4000-8000-000000000002"],
},
@@ -391,205 +401,7 @@ test("explains invalid sensitive-data suggestion selections without calling the
message: "One or more selected Catalog Columns were not found in this database.",
});
expect(modelCompleter.complete).not.toHaveBeenCalled();
} finally {
await app.close();
}
});
test("batches sensitive-data suggestions for schemas larger than one helper message", async () => {
const maxHelperMessageBytes = 64 * 1024;
const seenColumnIds: string[] = [];
const modelCompleter: ModelCompleter = {
complete: vi.fn(async (request) => {
const userMessage = request.messages.find((message) => message.role === "user")!;
expect(Buffer.byteLength(userMessage.content, "utf8")).toBeLessThanOrEqual(maxHelperMessageBytes);
const payload = JSON.parse(userMessage.content) as {
columns: Array<{ columnId: string; column: string }>;
};
expect(payload.columns.length).toBeLessThanOrEqual(10);
seenColumnIds.push(...payload.columns.map((column) => column.columnId));
return JSON.stringify({
suggestions: payload.columns.map((column) => ({
columnId: column.columnId,
sensitive: column.column.endsWith("_private"),
})),
});
}),
};
const { app, repository, database } = await setup(modelCompleter);
const columnCount = 900;
await repository.applySchemaSync(database.id, database.version, "all", [], {
schemaVersion: 1,
capabilities: { tables: "available", columns: "available", relationships: "available" },
tables: [{ name: "wide_table", sourceComment: null }],
columns: Array.from({ length: columnCount }, (_, index) => ({
tableName: "wide_table",
name: `field_${index.toString().padStart(4, "0")}${index % 10 === 0 ? "_private" : ""}`,
ordinalPosition: index + 1,
dataType: "character varying(255)",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
sourceComment: null,
})),
relationships: [],
});
const wideTable = (await repository.listTables(database.id)).find((table) => table.name === "wide_table")!;
const expectedColumnIds = (await repository.listColumns(database.id, wideTable.id)).map((column) => column.id);
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
});
expect(response.statusCode).toBe(200);
const suggestions = response.json().suggestions as Array<{
columnName: string;
currentSensitive: boolean;
sensitive: boolean;
}>;
expect(suggestions).toHaveLength(columnCount);
expect(suggestions).toEqual(expect.arrayContaining([
expect.objectContaining({ columnName: "field_0000_private", currentSensitive: false, sensitive: true }),
expect.objectContaining({ columnName: "field_0001", currentSensitive: false, sensitive: false }),
]));
expect(vi.mocked(modelCompleter.complete).mock.calls.length).toBeGreaterThan(1);
expect(seenColumnIds.slice().sort()).toEqual(expectedColumnIds.slice().sort());
expect(new Set(seenColumnIds).size).toBe(columnCount);
} finally {
await app.close();
}
});
test("retries one invalid sensitive-data classification before returning the review draft", async () => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async () => "unused"),
};
const { app, database, column } = await setup(modelCompleter);
vi.mocked(modelCompleter.complete)
.mockResolvedValueOnce("not-json")
.mockResolvedValueOnce(JSON.stringify({
suggestions: [{ columnId: column.id, sensitive: true }],
}));
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
});
expect(response.statusCode).toBe(200);
expect(response.json().suggestions).toEqual([
expect.objectContaining({ columnId: column.id, sensitive: true }),
]);
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
} finally {
await app.close();
}
});
test.each(["malformed", "incomplete", "duplicate"] as const)(
"fails safely when sensitive-data suggestions are %s",
async (kind) => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async () => "unused"),
};
const { app, repository, database, column } = await setup(modelCompleter);
const rawResponse = kind === "malformed"
? "RAW_PROVIDER_RESPONSE_DO_NOT_EXPOSE_{"
: kind === "incomplete"
? JSON.stringify({ suggestions: [] })
: JSON.stringify({
suggestions: [
{ columnId: column.id, sensitive: true },
{ columnId: column.id, sensitive: true },
],
});
vi.mocked(modelCompleter.complete).mockResolvedValueOnce(rawResponse);
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
});
expect(response.statusCode).toBe(502);
expect(response.json()).toEqual({
code: "sensitive_data_suggestion_invalid_response",
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
});
expect(response.body).not.toContain(rawResponse);
expect(await repository.getColumn(database.id, column.tableId, column.id))
.toMatchObject({ sensitive: false });
} finally {
await app.close();
}
},
);
test("explains a sensitive-data suggestion provider failure without exposing provider details", async () => {
const modelCompleter: ModelCompleter = {
complete: vi.fn(async () => {
throw new ModelCompletionProviderError();
}),
};
const { app, repository, database, column } = await setup(modelCompleter);
try {
const response = await app.inject({
method: "POST",
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
payload: { modelId: configuredModel.id, scope: "all" },
});
expect(response.statusCode).toBe(502);
expect(response.json()).toEqual({
code: "sensitive_data_suggestion_provider_unavailable",
message: "The selected LLM service could not complete the request. No suggestions were applied.",
});
expect(response.body).not.toContain("model completion failed");
expect(await repository.getColumn(database.id, column.tableId, column.id))
.toMatchObject({ sensitive: false });
const history = await app.inject({
method: "GET",
url: "/catalog/sensitive-data-suggestion-runs",
});
expect(history.statusCode).toBe(200);
const [failedRun] = history.json();
expect(failedRun).toMatchObject({
databaseId: database.id,
status: "failed",
total: 1,
suggestedSensitive: 0,
suggestedNonSensitive: 0,
errorSummary: "Sensitive-field suggestion generation failed.",
});
const events = await app.inject({
method: "GET",
url: `/catalog/sensitive-data-suggestion-runs/${failedRun.id}/events-list`,
});
expect(events.statusCode).toBe(200);
expect(events.json()).toMatchObject([
{
runId: failedRun.id,
sequence: 1,
level: "info",
message: "Sensitive-field suggestion generation started.",
},
{
runId: failedRun.id,
sequence: 2,
level: "error",
message: "Sensitive-field suggestion generation failed.",
},
]);
expect(events.body).not.toContain("model completion failed");
expect(sensitivityValueSource.scanTable).not.toHaveBeenCalled();
} finally {
await app.close();
}
@@ -15,6 +15,7 @@ import { up as upSensitiveDataFlag } from "../src/catalog/migrations/006_sensiti
import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_sensitive_data_suggestion_runs.js";
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
import { up as upCanonicalModelIds } from "../src/catalog/migrations/010_canonical_model_ids.js";
import { up as upLocalSensitivityAnalysis } from "../src/catalog/migrations/011_local_sensitivity_analysis.js";
import { KyselyCatalogRepository, type CatalogDatabase } from "../src/catalog/repository.js";
import { loadConfig } from "../src/config.js";
import type { WorkspaceRegistry } from "../src/workspaces/registry.js";
@@ -52,6 +53,7 @@ test.skipIf(!dockerAvailable)("Fastify persists Description Generation success a
await upSensitiveSuggestionRuns(db);
await upAiTokenUsage(db);
await upCanonicalModelIds(db);
await upLocalSensitivityAnalysis(db);
const repository = new KyselyCatalogRepository(db);
const database = await repository.create({
workspaceId: "psd-clinical",
@@ -0,0 +1,131 @@
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { afterEach, expect, test, vi } from "vitest";
import { PythonLocalNerDetector } from "../src/catalog/local-ner-detector.js";
const roots: string[] = [];
afterEach(() => {
vi.unstubAllEnvs();
for (const root of roots.splice(0)) rmSync(root, { recursive: true, force: true });
});
test("keeps a CPU-only local worker warm and returns sanitized evidence", async () => {
vi.stubEnv("THT_MODEL_API_KEY", "must-not-reach-worker");
const root = mkdtempSync(join(tmpdir(), "thothii-local-ner-"));
roots.push(root);
const helper = join(root, "fake_ner_worker.py");
writeFileSync(helper, `
import json
import os
import pathlib
import sys
root = pathlib.Path.cwd()
root.joinpath("runtime.json").write_text(json.dumps({
"argv": sys.argv,
"cuda": os.environ.get("CUDA_VISIBLE_DEVICES"),
"hip": os.environ.get("HIP_VISIBLE_DEVICES"),
"offline": os.environ.get("HF_HUB_OFFLINE"),
"inherited_secret": os.environ.get("THT_MODEL_API_KEY"),
"pid": os.getpid(),
}), encoding="utf-8")
print(json.dumps({"ready": True}), flush=True)
for line in sys.stdin:
request = json.loads(line)
root.joinpath("request.json").write_text(json.dumps(request), encoding="utf-8")
print(json.dumps({
"id": request["id"],
"ok": True,
"evidence": [{
"columnId": request["candidates"][0]["columnId"],
"label": "person",
"confidence": 0.93,
}],
}), flush=True)
`, "utf8");
const detector = new PythonLocalNerDetector({
pythonExecutable: "python3",
workerScript: helper,
modelPath: join(root, "pinned-model"),
cwd: root,
threads: 2,
startupTimeoutMs: 5_000,
});
const candidate = {
columnId: "33333333-3333-4333-8333-333333333333",
text: "Dimesso Mario Rossi",
};
try {
expect(detector.isReady()).toBe(false);
await detector.warmup();
expect(detector.isReady()).toBe(true);
expect(existsSync(join(root, "request.json"))).toBe(false);
await expect(detector.detect(
[candidate],
new AbortController().signal,
Date.now() + 5_000,
)).resolves.toEqual([{
columnId: candidate.columnId,
label: "person",
confidence: 0.93,
}]);
const firstRuntime = JSON.parse(readFileSync(join(root, "runtime.json"), "utf8"));
expect(firstRuntime).toMatchObject({
cuda: "",
hip: "",
offline: "1",
inherited_secret: null,
});
expect(JSON.stringify(firstRuntime.argv)).not.toContain(candidate.text);
expect(JSON.parse(readFileSync(join(root, "request.json"), "utf8")).candidates).toEqual([candidate]);
await detector.detect([candidate], new AbortController().signal, Date.now() + 5_000);
const secondRuntime = JSON.parse(readFileSync(join(root, "runtime.json"), "utf8"));
expect(secondRuntime.pid).toBe(firstRuntime.pid);
} finally {
await detector.close();
}
});
test("bounds worker startup by the caller deadline", async () => {
const root = mkdtempSync(join(tmpdir(), "thothii-local-ner-deadline-"));
roots.push(root);
const helper = join(root, "slow_ner_worker.py");
writeFileSync(helper, `
import json
import sys
import time
time.sleep(2)
print(json.dumps({"ready": True}), flush=True)
for line in sys.stdin:
request = json.loads(line)
print(json.dumps({"id": request["id"], "ok": True, "evidence": []}), flush=True)
`, "utf8");
const detector = new PythonLocalNerDetector({
pythonExecutable: "python3",
workerScript: helper,
modelPath: join(root, "pinned-model"),
cwd: root,
startupTimeoutMs: 5_000,
});
const startedAt = Date.now();
try {
await expect(detector.detect(
[{
columnId: "33333333-3333-4333-8333-333333333333",
text: "Dimesso Mario Rossi",
}],
new AbortController().signal,
startedAt + 50,
)).rejects.toThrow("local NER is unavailable");
expect(Date.now() - startedAt).toBeLessThan(1_000);
} finally {
await detector.close();
}
});
@@ -16,6 +16,7 @@ import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_s
import { up as upLogicalRelationships } from "../src/catalog/migrations/008_catalog_logical_relationships.js";
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
import { up as upCanonicalModelIds } from "../src/catalog/migrations/010_canonical_model_ids.js";
import { up as upLocalSensitivityAnalysis } from "../src/catalog/migrations/011_local_sensitivity_analysis.js";
const dockerAvailable = spawnSync("docker", ["info"], { stdio: "ignore" }).status === 0;
@@ -47,12 +48,23 @@ test.skipIf(!dockerAvailable)("PostgreSQL migration enforces one database per wo
modelId: "openai-mini", language: "en", status: "completed", total: 1,
processed: 1, generated: 1,
}).execute();
const historicalSuggestionRunId = randomUUID();
await db.insertInto("sensitiveDataSuggestionRuns").values({
id: randomUUID(), databaseId: historicalDatabaseId, scope: "all",
id: historicalSuggestionRunId, databaseId: historicalDatabaseId, scope: "all",
modelId: "openai-mini", status: "completed", total: 1,
suggestedSensitive: 1,
}).execute();
await upCanonicalModelIds(db);
await upLocalSensitivityAnalysis(db);
await expect(db.selectFrom("sensitiveDataSuggestionRuns")
.select(["engine", "modelId", "policyVersion", "unknown"])
.where("id", "=", historicalSuggestionRunId)
.executeTakeFirstOrThrow()).resolves.toMatchObject({
engine: "llm",
modelId: "openai-mini",
policyVersion: null,
unknown: 0,
});
await expect(db.insertInto("descriptionGenerationRuns").values({
id: randomUUID(), databaseId: historicalDatabaseId, scope: "all",
modelId: "openai/gpt-5-mini", language: "en", status: "completed", total: 1,
@@ -418,6 +430,7 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
await upSensitiveSuggestionRuns(db);
await upAiTokenUsage(db);
await upCanonicalModelIds(db);
await upLocalSensitivityAnalysis(db);
const repository = new KyselyCatalogRepository(db);
const firstDatabase = await repository.create({
workspaceId: "generation-one",
@@ -586,10 +599,10 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
]);
expect(await repository.getActiveDescriptionGenerationRun()).toBeUndefined();
const suggestionRun = await repository.createSensitiveDataSuggestionRun(
const suggestionRun = await repository.createSensitivityAnalysisRun(
firstDatabase.id,
"selected_columns",
"openai/gpt-4.1-mini",
{ engine: "local", policyVersion: "sensitivity-v1" },
);
expect(suggestionRun).toMatchObject({
databaseId: firstDatabase.id,
@@ -597,43 +610,49 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
total: 0,
suggestedSensitive: 0,
suggestedNonSensitive: 0,
unknown: 0,
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
startedAt: expect.any(String),
});
await repository.appendSensitiveDataSuggestionEvent(
await repository.appendSensitivityAnalysisEvent(
suggestionRun.id,
"info",
"Sensitive-field suggestion generation started.",
);
await repository.appendSensitiveDataSuggestionEvent(
await repository.appendSensitivityAnalysisEvent(
suggestionRun.id,
"info",
"Sensitive-field suggestion generation completed for 2 columns.",
);
expect(await repository.updateSensitiveDataSuggestionRun(suggestionRun.id, {
expect(await repository.updateSensitivityAnalysisRun(suggestionRun.id, {
status: "completed",
total: 2,
suggestedSensitive: 1,
suggestedNonSensitive: 1,
suggestedNonSensitive: 0,
unknown: 1,
finishedAt: new Date().toISOString(),
})).toMatchObject({
status: "completed",
total: 2,
suggestedSensitive: 1,
suggestedNonSensitive: 1,
suggestedNonSensitive: 0,
unknown: 1,
});
expect(await repository.listSensitiveDataSuggestionEvents(suggestionRun.id, 1)).toEqual([
expect(await repository.listSensitivityAnalysisEvents(suggestionRun.id, 1)).toEqual([
expect.objectContaining({ sequence: 2, level: "info" }),
]);
expect((await repository.listSensitiveDataSuggestionRuns(1))[0]).toMatchObject({
expect((await repository.listSensitivityAnalysisRuns(1))[0]).toMatchObject({
id: suggestionRun.id,
});
const interruptedSuggestionRun = await repository.createSensitiveDataSuggestionRun(
const interruptedSuggestionRun = await repository.createSensitivityAnalysisRun(
secondDatabase.id,
"all",
"openai/gpt-4.1-mini",
{ engine: "local", policyVersion: "sensitivity-v1" },
);
expect(await repository.interruptActiveSensitiveDataSuggestionRuns(
expect(await repository.interruptActiveSensitivityAnalysisRuns(
"Sensitive-field suggestion generation was interrupted by backend restart.",
)).toEqual([
expect.objectContaining({
@@ -0,0 +1,99 @@
import { expect, test, vi } from "vitest";
import {
SensitivityAnalysisInterruptedError,
SensitivityAnalysisService,
} from "../src/catalog/sensitivity-analysis-service.js";
import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js";
import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js";
import type {
CatalogRepository,
SensitivityAnalysisRun,
WorkspaceDatabase,
} from "../src/catalog/types.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
workspaceId: "psd-clinical",
engine: "postgres",
databaseName: "warehouse",
schema: "public",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
connectionStatus: "reachable",
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
const running: SensitivityAnalysisRun = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
scope: "all",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
status: "running",
total: 0,
suggestedSensitive: 0,
suggestedNonSensitive: 0,
unknown: 0,
inputTokens: 0,
cacheReadTokens: 0,
outputTokens: 0,
createdAt: "2026-09-02T08:00:00Z",
startedAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
finishedAt: null,
errorSummary: null,
};
test("stops catalog selection when the request expires during a catalog read", async () => {
const controller = new AbortController();
const listTables = vi.fn();
const repository = {
get: vi.fn(async () => {
controller.abort();
return database;
}),
listTables,
} as unknown as CatalogRepository;
const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier;
const analysis = new SensitivityAnalysisService(repository, classifier);
await expect(analysis.analyze(
database.id,
"all",
[],
controller.signal,
)).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
expect(listTables).not.toHaveBeenCalled();
expect(classifier.assessTable).not.toHaveBeenCalled();
});
test("marks a created run interrupted if the request deadline expires during persistence", async () => {
const controller = new AbortController();
const update = vi.fn(async (_runId: string, changes: Partial<SensitivityAnalysisRun>) => ({
...running,
...changes,
}));
const repository = {
get: vi.fn(async () => database),
createSensitivityAnalysisRun: vi.fn(async () => {
controller.abort();
return running;
}),
updateSensitivityAnalysisRun: update,
appendSensitivityAnalysisEvent: vi.fn(async () => undefined),
} as unknown as CatalogRepository;
const analysis = { analyze: vi.fn() } as unknown as SensitivityAnalysisService;
const runner = new SensitivityAnalysisRunner(repository, analysis);
await expect(runner.run(database.id, "all", [], controller.signal))
.rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
expect(analysis.analyze).not.toHaveBeenCalled();
expect(update).toHaveBeenCalledWith(running.id, expect.objectContaining({
status: "interrupted",
total: 0,
unknown: 0,
errorSummary: "Local sensitivity analysis reached its time limit.",
}));
});
@@ -0,0 +1,450 @@
import { expect, test, vi } from "vitest";
import {
SensitivityClassifier,
type LocalNerDetector,
type SensitivityNerBudget,
type SensitivityTableScan,
type SensitivityValueSource,
} from "../src/catalog/sensitivity-classifier.js";
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
import { CatalogConnectorError } from "../src/catalog/types.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
workspaceId: "psd-clinical",
engine: "postgres",
databaseName: "warehouse",
schema: "public",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
connectionStatus: "reachable",
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
const table = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
name: "observations",
sourceComment: null,
description: null,
generatedDescription: null,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
} satisfies CatalogTable;
function column(overrides: Partial<CatalogColumn> = {}): CatalogColumn {
return {
id: "33333333-3333-4333-8333-333333333333",
tableId: table.id,
name: "note",
ordinalPosition: 1,
dataType: "character varying",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
isPrimaryKey: false,
isForeignKey: false,
foreignKeyCount: 0,
sourceComment: null,
description: null,
generatedDescription: null,
sensitive: false,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
...overrides,
};
}
function source(scan: SensitivityTableScan): SensitivityValueSource {
return { scanTable: vi.fn(async (_request, consume) => {
for (const batch of scan.batches) await consume(batch);
return scan.coverage;
}) };
}
test("one email hidden in a generically named column makes the whole column sensitive", async () => {
const target = column();
const values = source({
batches: [[
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
]],
coverage: { kind: "complete", observedRows: 2 },
});
const classifier = new SensitivityClassifier(values);
const [assessment] = await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
columnId: target.id,
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "content", ruleId: "pii.email" }],
});
});
test("one text value longer than 500 characters makes the whole column sensitive", async () => {
const target = column({ name: "comment" });
const values = source({
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
coverage: { kind: "sampled", observedRows: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "length", ruleId: "text.over_500_characters" }],
});
});
test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
const humanProtected = column({
id: "66666666-6666-4666-8666-666666666666",
name: "category",
sensitive: true,
});
const values = source({
batches: [[
{ columnId: benign.id, value: "active", characterLength: 6 },
{ columnId: empty.id, value: null, characterLength: null },
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
]],
coverage: { kind: "complete", observedRows: 1 },
});
const assessments = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [benign, empty, humanProtected] },
new AbortController().signal,
);
expect(assessments).toEqual([
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
expect.objectContaining({
columnId: empty.id,
assessment: "unknown",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
}),
expect.objectContaining({
columnId: humanProtected.id,
assessment: "non_sensitive",
proposedSensitive: false,
}),
]);
});
test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
const target = column({ sensitive: true });
const values = source({
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
coverage: { kind: "sampled", observedRows: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: true,
evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
});
});
test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
const unresolved = column();
const metadataMatch = column({
id: "44444444-4444-4444-8444-444444444444",
name: "codice_fiscale",
});
const values: SensitivityValueSource = {
scanTable: vi.fn(async () => {
throw new CatalogConnectorError("upstream detail must not escape");
}),
};
const assessments = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [unresolved, metadataMatch] },
new AbortController().signal,
);
expect(assessments).toEqual([
expect.objectContaining({
columnId: unresolved.id,
assessment: "unknown",
evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
}),
expect.objectContaining({
columnId: metadataMatch.id,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
}),
]);
});
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
const target = column({ name: "codice_fiscale" });
const values = source({
batches: [],
coverage: { kind: "complete", observedRows: 0 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
});
expect(values.scanTable).not.toHaveBeenCalled();
});
test.each([
["RSSMRA85T10A562S", "pii.italian_fiscal_code"],
["IT60 X054 2811 1010 0000 0123 456", "financial.iban"],
["4111 1111 1111 1111", "financial.payment_card"],
["SWIFT DEUTDEFF500", "financial.bic"],
["Partita IVA 00743110157", "pii.italian_vat"],
["Passaporto YA1234567", "pii.passport_number"],
["Carta d'identità CA12345AA", "pii.identity_card"],
["Patente di guida U11234567A", "pii.drivers_license_number"],
["Chiamare +39 347 123 4567", "pii.phone_number"],
["Client 192.168.1.5", "network.ip_address"],
["Device 00:1B:44:11:3A:B7", "network.mac_address"],
["https://example.org/profiles/mario", "network.url"],
["550e8400-e29b-41d4-a716-446655440000", "pii.uuid"],
["AWS key AKIAIOSFODNN7EXAMPLE", "credential.access_key"],
["Diagnosi: carcinoma mammario con metastasi ossee", "health.clinical_term"],
["-----BEGIN PRIVATE KEY----- secret -----END PRIVATE KEY-----", "credential.private_key"],
['{"profile":{"email":"not yet supplied"}}', "pii.json_sensitive_key"],
] as const)("recognizes validated sensitive content without relying on the column name: %s", async (
value,
ruleId,
) => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
coverage: { kind: "complete", observedRows: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "content", ruleId }],
});
});
test("does not make a malformed email decisive", async () => {
const target = column();
const [assessment] = await new SensitivityClassifier(source({
batches: [[{
columnId: target.id,
value: "contatto a@b..com non valido",
characterLength: 28,
}]],
coverage: { kind: "complete", observedRows: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
});
test("finds a valid email after a malformed candidate in the same value", async () => {
const target = column();
const [assessment] = await new SensitivityClassifier(source({
batches: [[{
columnId: target.id,
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
characterLength: 58,
}]],
coverage: { kind: "complete", observedRows: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
});
});
test("optional local NER evidence can make otherwise ambiguous Italian text sensitive", async () => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
};
const [assessment] = await new SensitivityClassifier(values, detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledWith(
[{ columnId: target.id, text: "Dimesso Mario Rossi" }],
expect.any(AbortSignal),
expect.any(Number),
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "ner", ruleId: "ner.entity", label: "person_name", confidence: 0.91 }],
});
});
test("does not wait for an optional NER worker that is still warming", async () => {
const target = column();
const detector: LocalNerDetector = {
isReady: () => false,
detect: vi.fn(async () => [{ columnId: target.id, label: "person", confidence: 0.99 }]),
};
const [assessment] = await new SensitivityClassifier(source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
}), detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).not.toHaveBeenCalled();
expect(assessment).toMatchObject({ assessment: "unknown" });
});
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
const columns = Array.from({ length: 17 }, (_, index) => column({
id: `00000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
name: `attribute_${index + 1}`,
ordinalPosition: index + 1,
}));
const observations = columns.flatMap((item, columnIndex) => Array.from(
{ length: 8 },
(_, valueIndex) => ({
columnId: item.id,
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
}),
));
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
await new SensitivityClassifier(source({
batches: [observations],
coverage: { kind: "complete", observedRows: 8 },
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
{ database, table, columns },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledTimes(2);
expect(vi.mocked(detector.detect).mock.calls.map(([candidates]) => candidates.length)).toEqual([
128,
8,
]);
});
test("limits default NER work to two candidates spread across a wide table", async () => {
const columns = Array.from({ length: 10 }, (_, index) => column({
id: `10000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
name: `attribute_${index + 1}`,
ordinalPosition: index + 1,
}));
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
await new SensitivityClassifier(source({
batches: [columns.flatMap((item, columnIndex) => [0, 1].map((valueIndex) => ({
columnId: item.id,
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
})))],
coverage: { kind: "complete", observedRows: 2 },
}), detector).assessTable(
{ database, table, columns },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledOnce();
const submitted = vi.mocked(detector.detect).mock.calls[0]![0];
expect(submitted).toHaveLength(2);
expect(new Set(submitted.map((candidate) => candidate.columnId)).size).toBe(2);
});
test("shares a bounded NER time allowance across tables in one analysis run", async () => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => {
await new Promise((resolve) => setTimeout(resolve, 20));
return [];
}),
};
const classifier = new SensitivityClassifier(values, detector);
const nerBudget: SensitivityNerBudget = { remainingMs: 1 };
await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
Date.now() + 1_000,
nerBudget,
);
await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
Date.now() + 1_000,
nerBudget,
);
expect(detector.detect).toHaveBeenCalledOnce();
expect(nerBudget.remainingMs).toBe(0);
});
test("uninterpretable binary content remains unknown after complete coverage", async () => {
const target = column({ dataType: "bytea" });
const values = source({
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
coverage: { kind: "complete", observedRows: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
});
});
@@ -0,0 +1,316 @@
import { expect, test, vi } from "vitest";
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js";
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
workspaceId: "psd-clinical",
engine: "postgres",
databaseName: "warehouse",
schema: 'clinical"data',
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
connectionStatus: "reachable",
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
const table = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
name: 'patient"facts',
sourceComment: null,
description: null,
generatedDescription: null,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
} satisfies CatalogTable;
function column(id: string, name: string): CatalogColumn {
return {
id,
tableId: table.id,
name,
ordinalPosition: 1,
dataType: "text",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
isPrimaryKey: false,
isForeignKey: false,
foreignKeyCount: 0,
sourceComment: null,
description: null,
generatedDescription: null,
sensitive: false,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
};
}
test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value');
const fullRows = Array.from({ length: 200 }, () => ({
__value_0: "ordinary",
__length_0: "8",
__value_1: null,
__length_1: null,
}));
const query = vi.fn(async (sql: string) => {
if (sql.includes("TABLESAMPLE")) {
return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] };
}
if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows };
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
let clockCalls = 0;
const values = new ConcreteSensitivityValueSource(access, undefined, {
now: () => clockCalls++ < 3 ? 1_000 : 6_100,
});
const consumed: unknown[] = [];
const coverage = await values.scanTable({
database,
table,
columns: [note, contact],
fullScanBudgetMs: 5_000,
deadline: 61_000,
}, (batch) => consumed.push(...batch), new AbortController().signal);
expect(coverage).toEqual({ kind: "sampled", observedRows: 201 });
expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 });
expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null });
expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 });
expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]);
expect(query.mock.calls.some(([sql]) => (
String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT")
))).toBe(true);
expect(query.mock.calls.some(([sql]) => String(sql) === (
"FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
))).toBe(true);
expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false);
expect(query.mock.calls.some(([sql]) => (
String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM')
))).toBe(true);
expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]);
expect(end).toHaveBeenCalledOnce();
});
test("reports complete coverage when the final full-scan page is short", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD")
? { rows: [{ __value_0: "ordinary", __length_0: 8 }] }
: { rows: [] });
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal);
expect(coverage).toEqual({ kind: "complete", observedRows: 1 });
expect(query.mock.calls.filter(([sql]) => (
String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
))).toHaveLength(1);
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "ordinary", characterLength: 8 },
]);
});
test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
let fullScanAttempts = 0;
const query = vi.fn(async (sql: string) => {
if (sql.includes("TABLESAMPLE")) {
return { rows: [{ __value_0: "sample", __length_0: 6 }] };
}
if (sql.startsWith("FETCH FORWARD")) {
fullScanAttempts += 1;
throw Object.assign(new Error("statement timeout"), { code: "57014" });
}
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal);
expect(fullScanAttempts).toBe(1);
expect(coverage).toEqual({ kind: "sampled", observedRows: 1 });
expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([
"SAVEPOINT sensitivity_full_scan",
"ROLLBACK TO SAVEPOINT sensitivity_full_scan",
]));
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "sample", characterLength: 6 },
]);
});
test("scans a REST run_query binding without using PostgreSQL-wire access", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
const release = vi.fn();
const secretStore = {
materialize: vi.fn(() => ({
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
release,
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
{ __value_0: "mario.rossi@example.it", __length_0: 23 },
]), { status: 200, headers: { "content-type": "application/json" } }));
vi.stubGlobal("fetch", fetchMock);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
};
const values = new ConcreteSensitivityValueSource(access, secretStore);
const restDatabase: WorkspaceDatabase = {
...database,
binding: {
transport: "rest_api",
baseUrl: "https://dwh.example.test/root/",
restPath: "/health",
restAuth: "x-api-key",
},
};
const note = column("33333333-3333-4333-8333-333333333333", "note");
const consume = vi.fn();
try {
await expect(values.scanTable({
database: restDatabase,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal)).resolves.toEqual({
kind: "complete",
observedRows: 1,
});
expect(access.connect).not.toHaveBeenCalled();
expect(fetchMock).toHaveBeenCalledWith(
"https://dwh.example.test/root/rpc/run_query",
expect.objectContaining({
method: "POST",
headers: { "content-type": "application/json", "x-api-key": "test-api-key" },
}),
);
const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body));
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0');
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 },
]);
expect(release).toHaveBeenCalledOnce();
} finally {
vi.unstubAllGlobals();
rmSync(root, { recursive: true, force: true });
}
});
test("keeps multi-request REST scans conservative without a source transaction", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
const secretStore = {
materialize: vi.fn(() => ({
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
release: vi.fn(),
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn()
.mockResolvedValueOnce(new Response(JSON.stringify([
{ __value_0: "ordinary", __length_0: 8 },
]), { status: 200 }))
.mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 }));
vi.stubGlobal("fetch", fetchMock);
const values = new ConcreteSensitivityValueSource({
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
}, secretStore, { batchRows: 1 });
const restDatabase: WorkspaceDatabase = {
...database,
binding: {
transport: "rest_api",
baseUrl: "https://dwh.example.test/root",
restPath: "/health",
restAuth: "x-api-key",
},
};
try {
await expect(values.scanTable({
database: restDatabase,
table,
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedRows: 1,
});
expect(fetchMock).toHaveBeenCalledTimes(2);
} finally {
vi.unstubAllGlobals();
rmSync(root, { recursive: true, force: true });
}
});
test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => {
const query = vi.fn(async () => ({ rows: [] }));
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const now = vi.fn()
.mockReturnValueOnce(1_000)
.mockReturnValue(61_000);
const values = new ConcreteSensitivityValueSource(access, undefined, { now });
await expect(values.scanTable({
database,
table,
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
fullScanBudgetMs: 5_000,
deadline: 60_000,
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedRows: 0,
});
expect(query).not.toHaveBeenCalled();
expect(end).toHaveBeenCalledOnce();
});
+29
View File
@@ -81,6 +81,35 @@ test("loadConfig keeps local development defaults", () => {
expect(loadConfig({}).dataRoot).toBeUndefined();
});
test("loadConfig keeps local NER disabled unless an absolute model path is configured", () => {
expect(loadConfig({}).sensitivityNer).toBeUndefined();
expect(loadConfig({
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
}).sensitivityNer?.pythonExecutable).toBe("/opt/sensitivity-ner/bin/python");
expect(loadConfig({
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
THT_SENSITIVITY_NER_PYTHON: "/opt/sensitivity-ner/bin/python",
THT_SENSITIVITY_NER_WORKER: "/app/backend/python/sensitivity_ner_worker.py",
THT_SENSITIVITY_NER_THREADS: "3",
}).sensitivityNer).toEqual({
modelPath: "/models/gliner2-pii",
pythonExecutable: "/opt/sensitivity-ner/bin/python",
workerScript: "/app/backend/python/sensitivity_ner_worker.py",
threads: 3,
});
});
test("loadConfig rejects ambiguous or unsafe local NER configuration", () => {
expect(() => loadConfig({ THT_SENSITIVITY_NER_MODEL_PATH: "fastino/model" }))
.toThrow("sensitivity NER model path configuration is invalid");
expect(() => loadConfig({
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
THT_SENSITIVITY_NER_THREADS: "0",
})).toThrow("sensitivity NER thread configuration is invalid");
expect(() => loadConfig({ THT_SENSITIVITY_NER_PYTHON: "/opt/ner/bin/python" }))
.toThrow("sensitivity NER settings require a model path");
});
test("loadConfig allows none and mock only outside production when auth.yaml is absent", () => {
const originalNodeEnvironment = process.env.NODE_ENV;
delete process.env.NODE_ENV;