feat: classify sensitive columns locally
This commit is contained in:
@@ -14,6 +14,7 @@ import {
|
||||
type ModelCompletionRequest,
|
||||
} from "../src/catalog/model-completer.js";
|
||||
import { CatalogOperationCoordinator } from "../src/catalog/operation-coordinator.js";
|
||||
import type { SensitivityValueSource } from "../src/catalog/sensitivity-classifier.js";
|
||||
import type {
|
||||
CatalogDatabaseClient,
|
||||
CatalogPostgresAccess,
|
||||
@@ -69,6 +70,16 @@ async function setup(
|
||||
sample: vi.fn(async () => []),
|
||||
},
|
||||
catalogPostgresAccess?: CatalogPostgresAccess,
|
||||
sensitivityValueSource: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async (request, consume) => {
|
||||
await consume(request.columns.map((column) => ({
|
||||
columnId: column.id,
|
||||
value: "ordinary",
|
||||
characterLength: 8,
|
||||
})));
|
||||
return { kind: "complete", observedRows: 1 };
|
||||
}),
|
||||
},
|
||||
) {
|
||||
const repository = new MemoryCatalogRepository();
|
||||
const database = await repository.create({
|
||||
@@ -115,10 +126,11 @@ async function setup(
|
||||
catalogOperationCoordinator: operations,
|
||||
metadataGenerationModels: models(),
|
||||
modelCompleter,
|
||||
sensitivityValueSource,
|
||||
...(descriptionSourceSampler ? { descriptionSourceSampler } : {}),
|
||||
...(catalogPostgresAccess ? { catalogPostgresAccess } : {}),
|
||||
});
|
||||
return { app, repository, database, table, column, operations };
|
||||
return { app, repository, database, table, column, operations, sensitivityValueSource };
|
||||
}
|
||||
|
||||
async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: string) {
|
||||
@@ -136,22 +148,15 @@ async function waitForTerminalRun(app: ReturnType<typeof buildApp>, runId: strin
|
||||
throw new Error(`Description Generation Run ${runId} did not finish`);
|
||||
}
|
||||
|
||||
test("suggests sensitive flags from structural metadata without persisting them", async () => {
|
||||
const modelCompleter = {
|
||||
complete: vi.fn(async () => JSON.stringify({
|
||||
suggestions: [{ columnId: expect.any(String), sensitive: true }],
|
||||
})),
|
||||
};
|
||||
test("assesses sensitive flags locally without persisting them or calling an LLM", async () => {
|
||||
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
|
||||
const { app, repository, database, table, column } = await setup(modelCompleter);
|
||||
modelCompleter.complete.mockResolvedValueOnce(JSON.stringify({
|
||||
suggestions: [{ columnId: column.id, sensitive: true }],
|
||||
}));
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
payload: { scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(200);
|
||||
@@ -160,11 +165,14 @@ test("suggests sensitive flags from structural metadata without persisting them"
|
||||
run: {
|
||||
databaseId: database.id,
|
||||
scope: "all",
|
||||
modelId: configuredModel.id,
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
status: "completed",
|
||||
total: 1,
|
||||
suggestedSensitive: 1,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 0,
|
||||
errorSummary: null,
|
||||
},
|
||||
suggestions: [{
|
||||
@@ -175,6 +183,8 @@ test("suggests sensitive flags from structural metadata without persisting them"
|
||||
version: column.version,
|
||||
currentSensitive: false,
|
||||
sensitive: true,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
|
||||
}],
|
||||
});
|
||||
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
||||
@@ -207,49 +217,56 @@ test("suggests sensitive flags from structural metadata without persisting them"
|
||||
runId: responseBody.run.id,
|
||||
sequence: 1,
|
||||
level: "info",
|
||||
message: "Sensitive-field suggestion generation started.",
|
||||
message: "Local sensitivity analysis started.",
|
||||
},
|
||||
{
|
||||
runId: responseBody.run.id,
|
||||
sequence: 2,
|
||||
level: "info",
|
||||
message: "Classified 1 of 1 columns.",
|
||||
message: "Assessed 1 of 1 columns locally.",
|
||||
},
|
||||
{
|
||||
runId: responseBody.run.id,
|
||||
sequence: 3,
|
||||
level: "info",
|
||||
message: "Sensitive-field suggestion generation completed for 1 column.",
|
||||
message: "Local sensitivity analysis completed for 1 column.",
|
||||
},
|
||||
]);
|
||||
|
||||
const request = modelCompleter.complete.mock.calls[0]![0] as ModelCompletionRequest;
|
||||
const prompt = request.messages.map((message) => message.content).join("\n");
|
||||
expect(prompt).toContain("patients");
|
||||
expect(prompt).toContain("birth_date");
|
||||
expect(prompt).toContain("date");
|
||||
expect(prompt).not.toContain("Patient date of birth");
|
||||
expect(prompt).not.toContain("test-provider-secret");
|
||||
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("limits sensitive-data suggestions to the selected tables or columns", async () => {
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async (request) => {
|
||||
const payload = JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
|
||||
columns: Array<{ columnId: string; column: string }>;
|
||||
};
|
||||
return JSON.stringify({
|
||||
suggestions: payload.columns.map((column) => ({
|
||||
columnId: column.columnId,
|
||||
sensitive: column.column.includes("name") || column.column.includes("note"),
|
||||
})),
|
||||
});
|
||||
}),
|
||||
};
|
||||
const { app, repository, database } = await setup(modelCompleter);
|
||||
test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal);
|
||||
const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") });
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(504);
|
||||
expect(response.json()).toEqual({
|
||||
code: "sensitivity_analysis_timeout",
|
||||
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
|
||||
});
|
||||
expect(await repository.listSensitivityAnalysisRuns()).toEqual([]);
|
||||
} finally {
|
||||
timeout.mockRestore();
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("limits sensitivity analysis to the selected tables or columns", async () => {
|
||||
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
|
||||
const { app, repository, database, sensitivityValueSource } = await setup(modelCompleter);
|
||||
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
||||
schemaVersion: 1,
|
||||
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
||||
@@ -281,7 +298,6 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_tables",
|
||||
targetIds: [visits.id, patients.id],
|
||||
},
|
||||
@@ -301,7 +317,6 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_columns",
|
||||
targetIds: [clinicalNote.id, status.id],
|
||||
},
|
||||
@@ -313,44 +328,41 @@ test("limits sensitive-data suggestions to the selected tables or columns", asyn
|
||||
expect.objectContaining({ tableId: visits.id, columnId: clinicalNote.id, sensitive: true }),
|
||||
]));
|
||||
|
||||
const prompts = vi.mocked(modelCompleter.complete).mock.calls.map(([request]) => (
|
||||
JSON.parse(request.messages.find((message) => message.role === "user")!.content) as {
|
||||
columns: Array<{ columnId: string }>;
|
||||
}
|
||||
));
|
||||
expect(prompts[0]!.columns.map((column) => column.columnId).sort()).toEqual(
|
||||
[...patientColumns, ...visitColumns].map((column) => column.id).sort(),
|
||||
);
|
||||
expect(prompts[0]!.columns.map((column) => column.columnId)).not.toContain(billingColumns[0]!.id);
|
||||
expect(prompts[1]!.columns.map((column) => column.columnId).sort()).toEqual(
|
||||
[status.id, clinicalNote.id].sort(),
|
||||
const scannedColumnIds = vi.mocked(sensitivityValueSource.scanTable).mock.calls.flatMap(
|
||||
([request]) => request.columns.map((column) => column.id),
|
||||
);
|
||||
expect(scannedColumnIds).toEqual([status.id, status.id]);
|
||||
expect(scannedColumnIds).not.toContain(patientColumns.find(
|
||||
(column) => column.name === "patient_name",
|
||||
)!.id);
|
||||
expect(scannedColumnIds).not.toContain(clinicalNote.id);
|
||||
expect(scannedColumnIds).not.toContain(billingColumns[0]!.id);
|
||||
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("explains invalid sensitive-data suggestion selections without calling the model", async () => {
|
||||
test("explains invalid sensitivity-analysis selections without reading source values", async () => {
|
||||
const modelCompleter: ModelCompleter = { complete: vi.fn(async () => "unused") };
|
||||
const { app, database, table } = await setup(modelCompleter);
|
||||
const { app, database, table, sensitivityValueSource } = await setup(modelCompleter);
|
||||
|
||||
try {
|
||||
const empty = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "selected_tables", targetIds: [] },
|
||||
payload: { scope: "selected_tables", targetIds: [] },
|
||||
});
|
||||
expect(empty.statusCode).toBe(400);
|
||||
expect(empty.json()).toEqual({
|
||||
code: "sensitive_data_suggestion_request_invalid",
|
||||
message: "Choose a database, one or more tables, or one or more columns to classify.",
|
||||
message: "Choose a database, one or more tables, or one or more columns to assess.",
|
||||
});
|
||||
|
||||
const duplicate = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_tables",
|
||||
targetIds: [table.id, table.id],
|
||||
},
|
||||
@@ -365,7 +377,6 @@ test("explains invalid sensitive-data suggestion selections without calling the
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_tables",
|
||||
targetIds: ["00000000-0000-4000-8000-000000000001"],
|
||||
},
|
||||
@@ -380,7 +391,6 @@ test("explains invalid sensitive-data suggestion selections without calling the
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: {
|
||||
modelId: configuredModel.id,
|
||||
scope: "selected_columns",
|
||||
targetIds: ["00000000-0000-4000-8000-000000000002"],
|
||||
},
|
||||
@@ -391,205 +401,7 @@ test("explains invalid sensitive-data suggestion selections without calling the
|
||||
message: "One or more selected Catalog Columns were not found in this database.",
|
||||
});
|
||||
expect(modelCompleter.complete).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("batches sensitive-data suggestions for schemas larger than one helper message", async () => {
|
||||
const maxHelperMessageBytes = 64 * 1024;
|
||||
const seenColumnIds: string[] = [];
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async (request) => {
|
||||
const userMessage = request.messages.find((message) => message.role === "user")!;
|
||||
expect(Buffer.byteLength(userMessage.content, "utf8")).toBeLessThanOrEqual(maxHelperMessageBytes);
|
||||
const payload = JSON.parse(userMessage.content) as {
|
||||
columns: Array<{ columnId: string; column: string }>;
|
||||
};
|
||||
expect(payload.columns.length).toBeLessThanOrEqual(10);
|
||||
seenColumnIds.push(...payload.columns.map((column) => column.columnId));
|
||||
return JSON.stringify({
|
||||
suggestions: payload.columns.map((column) => ({
|
||||
columnId: column.columnId,
|
||||
sensitive: column.column.endsWith("_private"),
|
||||
})),
|
||||
});
|
||||
}),
|
||||
};
|
||||
const { app, repository, database } = await setup(modelCompleter);
|
||||
const columnCount = 900;
|
||||
await repository.applySchemaSync(database.id, database.version, "all", [], {
|
||||
schemaVersion: 1,
|
||||
capabilities: { tables: "available", columns: "available", relationships: "available" },
|
||||
tables: [{ name: "wide_table", sourceComment: null }],
|
||||
columns: Array.from({ length: columnCount }, (_, index) => ({
|
||||
tableName: "wide_table",
|
||||
name: `field_${index.toString().padStart(4, "0")}${index % 10 === 0 ? "_private" : ""}`,
|
||||
ordinalPosition: index + 1,
|
||||
dataType: "character varying(255)",
|
||||
isNullable: true,
|
||||
defaultExpression: null,
|
||||
primaryKeyPosition: null,
|
||||
sourceComment: null,
|
||||
})),
|
||||
relationships: [],
|
||||
});
|
||||
const wideTable = (await repository.listTables(database.id)).find((table) => table.name === "wide_table")!;
|
||||
const expectedColumnIds = (await repository.listColumns(database.id, wideTable.id)).map((column) => column.id);
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(200);
|
||||
const suggestions = response.json().suggestions as Array<{
|
||||
columnName: string;
|
||||
currentSensitive: boolean;
|
||||
sensitive: boolean;
|
||||
}>;
|
||||
expect(suggestions).toHaveLength(columnCount);
|
||||
expect(suggestions).toEqual(expect.arrayContaining([
|
||||
expect.objectContaining({ columnName: "field_0000_private", currentSensitive: false, sensitive: true }),
|
||||
expect.objectContaining({ columnName: "field_0001", currentSensitive: false, sensitive: false }),
|
||||
]));
|
||||
expect(vi.mocked(modelCompleter.complete).mock.calls.length).toBeGreaterThan(1);
|
||||
expect(seenColumnIds.slice().sort()).toEqual(expectedColumnIds.slice().sort());
|
||||
expect(new Set(seenColumnIds).size).toBe(columnCount);
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("retries one invalid sensitive-data classification before returning the review draft", async () => {
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async () => "unused"),
|
||||
};
|
||||
const { app, database, column } = await setup(modelCompleter);
|
||||
vi.mocked(modelCompleter.complete)
|
||||
.mockResolvedValueOnce("not-json")
|
||||
.mockResolvedValueOnce(JSON.stringify({
|
||||
suggestions: [{ columnId: column.id, sensitive: true }],
|
||||
}));
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(200);
|
||||
expect(response.json().suggestions).toEqual([
|
||||
expect.objectContaining({ columnId: column.id, sensitive: true }),
|
||||
]);
|
||||
expect(modelCompleter.complete).toHaveBeenCalledTimes(2);
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
});
|
||||
|
||||
test.each(["malformed", "incomplete", "duplicate"] as const)(
|
||||
"fails safely when sensitive-data suggestions are %s",
|
||||
async (kind) => {
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async () => "unused"),
|
||||
};
|
||||
const { app, repository, database, column } = await setup(modelCompleter);
|
||||
const rawResponse = kind === "malformed"
|
||||
? "RAW_PROVIDER_RESPONSE_DO_NOT_EXPOSE_{"
|
||||
: kind === "incomplete"
|
||||
? JSON.stringify({ suggestions: [] })
|
||||
: JSON.stringify({
|
||||
suggestions: [
|
||||
{ columnId: column.id, sensitive: true },
|
||||
{ columnId: column.id, sensitive: true },
|
||||
],
|
||||
});
|
||||
vi.mocked(modelCompleter.complete).mockResolvedValueOnce(rawResponse);
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(502);
|
||||
expect(response.json()).toEqual({
|
||||
code: "sensitive_data_suggestion_invalid_response",
|
||||
message: "The LLM returned an incomplete or invalid classification. No suggestions were applied.",
|
||||
});
|
||||
expect(response.body).not.toContain(rawResponse);
|
||||
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
||||
.toMatchObject({ sensitive: false });
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
test("explains a sensitive-data suggestion provider failure without exposing provider details", async () => {
|
||||
const modelCompleter: ModelCompleter = {
|
||||
complete: vi.fn(async () => {
|
||||
throw new ModelCompletionProviderError();
|
||||
}),
|
||||
};
|
||||
const { app, repository, database, column } = await setup(modelCompleter);
|
||||
|
||||
try {
|
||||
const response = await app.inject({
|
||||
method: "POST",
|
||||
url: `/catalog/databases/${database.id}/sensitive-data-suggestions`,
|
||||
payload: { modelId: configuredModel.id, scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(502);
|
||||
expect(response.json()).toEqual({
|
||||
code: "sensitive_data_suggestion_provider_unavailable",
|
||||
message: "The selected LLM service could not complete the request. No suggestions were applied.",
|
||||
});
|
||||
expect(response.body).not.toContain("model completion failed");
|
||||
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
||||
.toMatchObject({ sensitive: false });
|
||||
|
||||
const history = await app.inject({
|
||||
method: "GET",
|
||||
url: "/catalog/sensitive-data-suggestion-runs",
|
||||
});
|
||||
expect(history.statusCode).toBe(200);
|
||||
const [failedRun] = history.json();
|
||||
expect(failedRun).toMatchObject({
|
||||
databaseId: database.id,
|
||||
status: "failed",
|
||||
total: 1,
|
||||
suggestedSensitive: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
errorSummary: "Sensitive-field suggestion generation failed.",
|
||||
});
|
||||
|
||||
const events = await app.inject({
|
||||
method: "GET",
|
||||
url: `/catalog/sensitive-data-suggestion-runs/${failedRun.id}/events-list`,
|
||||
});
|
||||
expect(events.statusCode).toBe(200);
|
||||
expect(events.json()).toMatchObject([
|
||||
{
|
||||
runId: failedRun.id,
|
||||
sequence: 1,
|
||||
level: "info",
|
||||
message: "Sensitive-field suggestion generation started.",
|
||||
},
|
||||
{
|
||||
runId: failedRun.id,
|
||||
sequence: 2,
|
||||
level: "error",
|
||||
message: "Sensitive-field suggestion generation failed.",
|
||||
},
|
||||
]);
|
||||
expect(events.body).not.toContain("model completion failed");
|
||||
expect(sensitivityValueSource.scanTable).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
await app.close();
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@ import { up as upSensitiveDataFlag } from "../src/catalog/migrations/006_sensiti
|
||||
import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_sensitive_data_suggestion_runs.js";
|
||||
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
|
||||
import { up as upCanonicalModelIds } from "../src/catalog/migrations/010_canonical_model_ids.js";
|
||||
import { up as upLocalSensitivityAnalysis } from "../src/catalog/migrations/011_local_sensitivity_analysis.js";
|
||||
import { KyselyCatalogRepository, type CatalogDatabase } from "../src/catalog/repository.js";
|
||||
import { loadConfig } from "../src/config.js";
|
||||
import type { WorkspaceRegistry } from "../src/workspaces/registry.js";
|
||||
@@ -52,6 +53,7 @@ test.skipIf(!dockerAvailable)("Fastify persists Description Generation success a
|
||||
await upSensitiveSuggestionRuns(db);
|
||||
await upAiTokenUsage(db);
|
||||
await upCanonicalModelIds(db);
|
||||
await upLocalSensitivityAnalysis(db);
|
||||
const repository = new KyselyCatalogRepository(db);
|
||||
const database = await repository.create({
|
||||
workspaceId: "psd-clinical",
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { afterEach, expect, test, vi } from "vitest";
|
||||
import { PythonLocalNerDetector } from "../src/catalog/local-ner-detector.js";
|
||||
|
||||
const roots: string[] = [];
|
||||
|
||||
afterEach(() => {
|
||||
vi.unstubAllEnvs();
|
||||
for (const root of roots.splice(0)) rmSync(root, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
test("keeps a CPU-only local worker warm and returns sanitized evidence", async () => {
|
||||
vi.stubEnv("THT_MODEL_API_KEY", "must-not-reach-worker");
|
||||
const root = mkdtempSync(join(tmpdir(), "thothii-local-ner-"));
|
||||
roots.push(root);
|
||||
const helper = join(root, "fake_ner_worker.py");
|
||||
writeFileSync(helper, `
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import sys
|
||||
|
||||
root = pathlib.Path.cwd()
|
||||
root.joinpath("runtime.json").write_text(json.dumps({
|
||||
"argv": sys.argv,
|
||||
"cuda": os.environ.get("CUDA_VISIBLE_DEVICES"),
|
||||
"hip": os.environ.get("HIP_VISIBLE_DEVICES"),
|
||||
"offline": os.environ.get("HF_HUB_OFFLINE"),
|
||||
"inherited_secret": os.environ.get("THT_MODEL_API_KEY"),
|
||||
"pid": os.getpid(),
|
||||
}), encoding="utf-8")
|
||||
print(json.dumps({"ready": True}), flush=True)
|
||||
for line in sys.stdin:
|
||||
request = json.loads(line)
|
||||
root.joinpath("request.json").write_text(json.dumps(request), encoding="utf-8")
|
||||
print(json.dumps({
|
||||
"id": request["id"],
|
||||
"ok": True,
|
||||
"evidence": [{
|
||||
"columnId": request["candidates"][0]["columnId"],
|
||||
"label": "person",
|
||||
"confidence": 0.93,
|
||||
}],
|
||||
}), flush=True)
|
||||
`, "utf8");
|
||||
const detector = new PythonLocalNerDetector({
|
||||
pythonExecutable: "python3",
|
||||
workerScript: helper,
|
||||
modelPath: join(root, "pinned-model"),
|
||||
cwd: root,
|
||||
threads: 2,
|
||||
startupTimeoutMs: 5_000,
|
||||
});
|
||||
const candidate = {
|
||||
columnId: "33333333-3333-4333-8333-333333333333",
|
||||
text: "Dimesso Mario Rossi",
|
||||
};
|
||||
|
||||
try {
|
||||
expect(detector.isReady()).toBe(false);
|
||||
await detector.warmup();
|
||||
expect(detector.isReady()).toBe(true);
|
||||
expect(existsSync(join(root, "request.json"))).toBe(false);
|
||||
|
||||
await expect(detector.detect(
|
||||
[candidate],
|
||||
new AbortController().signal,
|
||||
Date.now() + 5_000,
|
||||
)).resolves.toEqual([{
|
||||
columnId: candidate.columnId,
|
||||
label: "person",
|
||||
confidence: 0.93,
|
||||
}]);
|
||||
const firstRuntime = JSON.parse(readFileSync(join(root, "runtime.json"), "utf8"));
|
||||
expect(firstRuntime).toMatchObject({
|
||||
cuda: "",
|
||||
hip: "",
|
||||
offline: "1",
|
||||
inherited_secret: null,
|
||||
});
|
||||
expect(JSON.stringify(firstRuntime.argv)).not.toContain(candidate.text);
|
||||
expect(JSON.parse(readFileSync(join(root, "request.json"), "utf8")).candidates).toEqual([candidate]);
|
||||
|
||||
await detector.detect([candidate], new AbortController().signal, Date.now() + 5_000);
|
||||
const secondRuntime = JSON.parse(readFileSync(join(root, "runtime.json"), "utf8"));
|
||||
expect(secondRuntime.pid).toBe(firstRuntime.pid);
|
||||
} finally {
|
||||
await detector.close();
|
||||
}
|
||||
});
|
||||
|
||||
test("bounds worker startup by the caller deadline", async () => {
|
||||
const root = mkdtempSync(join(tmpdir(), "thothii-local-ner-deadline-"));
|
||||
roots.push(root);
|
||||
const helper = join(root, "slow_ner_worker.py");
|
||||
writeFileSync(helper, `
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
|
||||
time.sleep(2)
|
||||
print(json.dumps({"ready": True}), flush=True)
|
||||
for line in sys.stdin:
|
||||
request = json.loads(line)
|
||||
print(json.dumps({"id": request["id"], "ok": True, "evidence": []}), flush=True)
|
||||
`, "utf8");
|
||||
const detector = new PythonLocalNerDetector({
|
||||
pythonExecutable: "python3",
|
||||
workerScript: helper,
|
||||
modelPath: join(root, "pinned-model"),
|
||||
cwd: root,
|
||||
startupTimeoutMs: 5_000,
|
||||
});
|
||||
const startedAt = Date.now();
|
||||
|
||||
try {
|
||||
await expect(detector.detect(
|
||||
[{
|
||||
columnId: "33333333-3333-4333-8333-333333333333",
|
||||
text: "Dimesso Mario Rossi",
|
||||
}],
|
||||
new AbortController().signal,
|
||||
startedAt + 50,
|
||||
)).rejects.toThrow("local NER is unavailable");
|
||||
expect(Date.now() - startedAt).toBeLessThan(1_000);
|
||||
} finally {
|
||||
await detector.close();
|
||||
}
|
||||
});
|
||||
@@ -16,6 +16,7 @@ import { up as upSensitiveSuggestionRuns } from "../src/catalog/migrations/007_s
|
||||
import { up as upLogicalRelationships } from "../src/catalog/migrations/008_catalog_logical_relationships.js";
|
||||
import { up as upAiTokenUsage } from "../src/catalog/migrations/009_ai_token_usage.js";
|
||||
import { up as upCanonicalModelIds } from "../src/catalog/migrations/010_canonical_model_ids.js";
|
||||
import { up as upLocalSensitivityAnalysis } from "../src/catalog/migrations/011_local_sensitivity_analysis.js";
|
||||
|
||||
const dockerAvailable = spawnSync("docker", ["info"], { stdio: "ignore" }).status === 0;
|
||||
|
||||
@@ -47,12 +48,23 @@ test.skipIf(!dockerAvailable)("PostgreSQL migration enforces one database per wo
|
||||
modelId: "openai-mini", language: "en", status: "completed", total: 1,
|
||||
processed: 1, generated: 1,
|
||||
}).execute();
|
||||
const historicalSuggestionRunId = randomUUID();
|
||||
await db.insertInto("sensitiveDataSuggestionRuns").values({
|
||||
id: randomUUID(), databaseId: historicalDatabaseId, scope: "all",
|
||||
id: historicalSuggestionRunId, databaseId: historicalDatabaseId, scope: "all",
|
||||
modelId: "openai-mini", status: "completed", total: 1,
|
||||
suggestedSensitive: 1,
|
||||
}).execute();
|
||||
await upCanonicalModelIds(db);
|
||||
await upLocalSensitivityAnalysis(db);
|
||||
await expect(db.selectFrom("sensitiveDataSuggestionRuns")
|
||||
.select(["engine", "modelId", "policyVersion", "unknown"])
|
||||
.where("id", "=", historicalSuggestionRunId)
|
||||
.executeTakeFirstOrThrow()).resolves.toMatchObject({
|
||||
engine: "llm",
|
||||
modelId: "openai-mini",
|
||||
policyVersion: null,
|
||||
unknown: 0,
|
||||
});
|
||||
await expect(db.insertInto("descriptionGenerationRuns").values({
|
||||
id: randomUUID(), databaseId: historicalDatabaseId, scope: "all",
|
||||
modelId: "openai/gpt-5-mini", language: "en", status: "completed", total: 1,
|
||||
@@ -418,6 +430,7 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
|
||||
await upSensitiveSuggestionRuns(db);
|
||||
await upAiTokenUsage(db);
|
||||
await upCanonicalModelIds(db);
|
||||
await upLocalSensitivityAnalysis(db);
|
||||
const repository = new KyselyCatalogRepository(db);
|
||||
const firstDatabase = await repository.create({
|
||||
workspaceId: "generation-one",
|
||||
@@ -586,10 +599,10 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
|
||||
]);
|
||||
expect(await repository.getActiveDescriptionGenerationRun()).toBeUndefined();
|
||||
|
||||
const suggestionRun = await repository.createSensitiveDataSuggestionRun(
|
||||
const suggestionRun = await repository.createSensitivityAnalysisRun(
|
||||
firstDatabase.id,
|
||||
"selected_columns",
|
||||
"openai/gpt-4.1-mini",
|
||||
{ engine: "local", policyVersion: "sensitivity-v1" },
|
||||
);
|
||||
expect(suggestionRun).toMatchObject({
|
||||
databaseId: firstDatabase.id,
|
||||
@@ -597,43 +610,49 @@ test.skipIf(!dockerAvailable)("PostgreSQL repository persists description and se
|
||||
total: 0,
|
||||
suggestedSensitive: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 0,
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
startedAt: expect.any(String),
|
||||
});
|
||||
await repository.appendSensitiveDataSuggestionEvent(
|
||||
await repository.appendSensitivityAnalysisEvent(
|
||||
suggestionRun.id,
|
||||
"info",
|
||||
"Sensitive-field suggestion generation started.",
|
||||
);
|
||||
await repository.appendSensitiveDataSuggestionEvent(
|
||||
await repository.appendSensitivityAnalysisEvent(
|
||||
suggestionRun.id,
|
||||
"info",
|
||||
"Sensitive-field suggestion generation completed for 2 columns.",
|
||||
);
|
||||
expect(await repository.updateSensitiveDataSuggestionRun(suggestionRun.id, {
|
||||
expect(await repository.updateSensitivityAnalysisRun(suggestionRun.id, {
|
||||
status: "completed",
|
||||
total: 2,
|
||||
suggestedSensitive: 1,
|
||||
suggestedNonSensitive: 1,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 1,
|
||||
finishedAt: new Date().toISOString(),
|
||||
})).toMatchObject({
|
||||
status: "completed",
|
||||
total: 2,
|
||||
suggestedSensitive: 1,
|
||||
suggestedNonSensitive: 1,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 1,
|
||||
});
|
||||
expect(await repository.listSensitiveDataSuggestionEvents(suggestionRun.id, 1)).toEqual([
|
||||
expect(await repository.listSensitivityAnalysisEvents(suggestionRun.id, 1)).toEqual([
|
||||
expect.objectContaining({ sequence: 2, level: "info" }),
|
||||
]);
|
||||
expect((await repository.listSensitiveDataSuggestionRuns(1))[0]).toMatchObject({
|
||||
expect((await repository.listSensitivityAnalysisRuns(1))[0]).toMatchObject({
|
||||
id: suggestionRun.id,
|
||||
});
|
||||
|
||||
const interruptedSuggestionRun = await repository.createSensitiveDataSuggestionRun(
|
||||
const interruptedSuggestionRun = await repository.createSensitivityAnalysisRun(
|
||||
secondDatabase.id,
|
||||
"all",
|
||||
"openai/gpt-4.1-mini",
|
||||
{ engine: "local", policyVersion: "sensitivity-v1" },
|
||||
);
|
||||
expect(await repository.interruptActiveSensitiveDataSuggestionRuns(
|
||||
expect(await repository.interruptActiveSensitivityAnalysisRuns(
|
||||
"Sensitive-field suggestion generation was interrupted by backend restart.",
|
||||
)).toEqual([
|
||||
expect.objectContaining({
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
import { expect, test, vi } from "vitest";
|
||||
import {
|
||||
SensitivityAnalysisInterruptedError,
|
||||
SensitivityAnalysisService,
|
||||
} from "../src/catalog/sensitivity-analysis-service.js";
|
||||
import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js";
|
||||
import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js";
|
||||
import type {
|
||||
CatalogRepository,
|
||||
SensitivityAnalysisRun,
|
||||
WorkspaceDatabase,
|
||||
} from "../src/catalog/types.js";
|
||||
|
||||
const database = {
|
||||
id: "11111111-1111-4111-8111-111111111111",
|
||||
workspaceId: "psd-clinical",
|
||||
engine: "postgres",
|
||||
databaseName: "warehouse",
|
||||
schema: "public",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
connectionStatus: "reachable",
|
||||
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
|
||||
} satisfies WorkspaceDatabase;
|
||||
|
||||
const running: SensitivityAnalysisRun = {
|
||||
id: "22222222-2222-4222-8222-222222222222",
|
||||
databaseId: database.id,
|
||||
scope: "all",
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
status: "running",
|
||||
total: 0,
|
||||
suggestedSensitive: 0,
|
||||
suggestedNonSensitive: 0,
|
||||
unknown: 0,
|
||||
inputTokens: 0,
|
||||
cacheReadTokens: 0,
|
||||
outputTokens: 0,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
startedAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
finishedAt: null,
|
||||
errorSummary: null,
|
||||
};
|
||||
|
||||
test("stops catalog selection when the request expires during a catalog read", async () => {
|
||||
const controller = new AbortController();
|
||||
const listTables = vi.fn();
|
||||
const repository = {
|
||||
get: vi.fn(async () => {
|
||||
controller.abort();
|
||||
return database;
|
||||
}),
|
||||
listTables,
|
||||
} as unknown as CatalogRepository;
|
||||
const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier;
|
||||
const analysis = new SensitivityAnalysisService(repository, classifier);
|
||||
|
||||
await expect(analysis.analyze(
|
||||
database.id,
|
||||
"all",
|
||||
[],
|
||||
controller.signal,
|
||||
)).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
|
||||
expect(listTables).not.toHaveBeenCalled();
|
||||
expect(classifier.assessTable).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test("marks a created run interrupted if the request deadline expires during persistence", async () => {
|
||||
const controller = new AbortController();
|
||||
const update = vi.fn(async (_runId: string, changes: Partial<SensitivityAnalysisRun>) => ({
|
||||
...running,
|
||||
...changes,
|
||||
}));
|
||||
const repository = {
|
||||
get: vi.fn(async () => database),
|
||||
createSensitivityAnalysisRun: vi.fn(async () => {
|
||||
controller.abort();
|
||||
return running;
|
||||
}),
|
||||
updateSensitivityAnalysisRun: update,
|
||||
appendSensitivityAnalysisEvent: vi.fn(async () => undefined),
|
||||
} as unknown as CatalogRepository;
|
||||
const analysis = { analyze: vi.fn() } as unknown as SensitivityAnalysisService;
|
||||
const runner = new SensitivityAnalysisRunner(repository, analysis);
|
||||
|
||||
await expect(runner.run(database.id, "all", [], controller.signal))
|
||||
.rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
|
||||
expect(analysis.analyze).not.toHaveBeenCalled();
|
||||
expect(update).toHaveBeenCalledWith(running.id, expect.objectContaining({
|
||||
status: "interrupted",
|
||||
total: 0,
|
||||
unknown: 0,
|
||||
errorSummary: "Local sensitivity analysis reached its time limit.",
|
||||
}));
|
||||
});
|
||||
@@ -0,0 +1,450 @@
|
||||
import { expect, test, vi } from "vitest";
|
||||
import {
|
||||
SensitivityClassifier,
|
||||
type LocalNerDetector,
|
||||
type SensitivityNerBudget,
|
||||
type SensitivityTableScan,
|
||||
type SensitivityValueSource,
|
||||
} from "../src/catalog/sensitivity-classifier.js";
|
||||
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
|
||||
import { CatalogConnectorError } from "../src/catalog/types.js";
|
||||
|
||||
const database = {
|
||||
id: "11111111-1111-4111-8111-111111111111",
|
||||
workspaceId: "psd-clinical",
|
||||
engine: "postgres",
|
||||
databaseName: "warehouse",
|
||||
schema: "public",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
connectionStatus: "reachable",
|
||||
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
|
||||
} satisfies WorkspaceDatabase;
|
||||
|
||||
const table = {
|
||||
id: "22222222-2222-4222-8222-222222222222",
|
||||
databaseId: database.id,
|
||||
name: "observations",
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
} satisfies CatalogTable;
|
||||
|
||||
function column(overrides: Partial<CatalogColumn> = {}): CatalogColumn {
|
||||
return {
|
||||
id: "33333333-3333-4333-8333-333333333333",
|
||||
tableId: table.id,
|
||||
name: "note",
|
||||
ordinalPosition: 1,
|
||||
dataType: "character varying",
|
||||
isNullable: true,
|
||||
defaultExpression: null,
|
||||
primaryKeyPosition: null,
|
||||
isPrimaryKey: false,
|
||||
isForeignKey: false,
|
||||
foreignKeyCount: 0,
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
sensitive: false,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function source(scan: SensitivityTableScan): SensitivityValueSource {
|
||||
return { scanTable: vi.fn(async (_request, consume) => {
|
||||
for (const batch of scan.batches) await consume(batch);
|
||||
return scan.coverage;
|
||||
}) };
|
||||
}
|
||||
|
||||
test("one email hidden in a generically named column makes the whole column sensitive", async () => {
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[
|
||||
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
|
||||
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
|
||||
]],
|
||||
coverage: { kind: "complete", observedRows: 2 },
|
||||
});
|
||||
const classifier = new SensitivityClassifier(values);
|
||||
|
||||
const [assessment] = await classifier.assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
columnId: target.id,
|
||||
assessment: "sensitive",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("one text value longer than 500 characters makes the whole column sensitive", async () => {
|
||||
const target = column({ name: "comment" });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "length", ruleId: "text.over_500_characters" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
|
||||
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
|
||||
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
|
||||
const humanProtected = column({
|
||||
id: "66666666-6666-4666-8666-666666666666",
|
||||
name: "category",
|
||||
sensitive: true,
|
||||
});
|
||||
const values = source({
|
||||
batches: [[
|
||||
{ columnId: benign.id, value: "active", characterLength: 6 },
|
||||
{ columnId: empty.id, value: null, characterLength: null },
|
||||
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
|
||||
]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
});
|
||||
|
||||
const assessments = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [benign, empty, humanProtected] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessments).toEqual([
|
||||
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
|
||||
expect.objectContaining({
|
||||
columnId: empty.id,
|
||||
assessment: "unknown",
|
||||
proposedSensitive: false,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
|
||||
}),
|
||||
expect.objectContaining({
|
||||
columnId: humanProtected.id,
|
||||
assessment: "non_sensitive",
|
||||
proposedSensitive: false,
|
||||
}),
|
||||
]);
|
||||
});
|
||||
|
||||
test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
|
||||
const target = column({ sensitive: true });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "unknown",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
|
||||
const unresolved = column();
|
||||
const metadataMatch = column({
|
||||
id: "44444444-4444-4444-8444-444444444444",
|
||||
name: "codice_fiscale",
|
||||
});
|
||||
const values: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async () => {
|
||||
throw new CatalogConnectorError("upstream detail must not escape");
|
||||
}),
|
||||
};
|
||||
|
||||
const assessments = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [unresolved, metadataMatch] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessments).toEqual([
|
||||
expect.objectContaining({
|
||||
columnId: unresolved.id,
|
||||
assessment: "unknown",
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
|
||||
}),
|
||||
expect.objectContaining({
|
||||
columnId: metadataMatch.id,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
|
||||
}),
|
||||
]);
|
||||
});
|
||||
|
||||
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
|
||||
const target = column({ name: "codice_fiscale" });
|
||||
const values = source({
|
||||
batches: [],
|
||||
coverage: { kind: "complete", observedRows: 0 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
|
||||
});
|
||||
expect(values.scanTable).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test.each([
|
||||
["RSSMRA85T10A562S", "pii.italian_fiscal_code"],
|
||||
["IT60 X054 2811 1010 0000 0123 456", "financial.iban"],
|
||||
["4111 1111 1111 1111", "financial.payment_card"],
|
||||
["SWIFT DEUTDEFF500", "financial.bic"],
|
||||
["Partita IVA 00743110157", "pii.italian_vat"],
|
||||
["Passaporto YA1234567", "pii.passport_number"],
|
||||
["Carta d'identità CA12345AA", "pii.identity_card"],
|
||||
["Patente di guida U11234567A", "pii.drivers_license_number"],
|
||||
["Chiamare +39 347 123 4567", "pii.phone_number"],
|
||||
["Client 192.168.1.5", "network.ip_address"],
|
||||
["Device 00:1B:44:11:3A:B7", "network.mac_address"],
|
||||
["https://example.org/profiles/mario", "network.url"],
|
||||
["550e8400-e29b-41d4-a716-446655440000", "pii.uuid"],
|
||||
["AWS key AKIAIOSFODNN7EXAMPLE", "credential.access_key"],
|
||||
["Diagnosi: carcinoma mammario con metastasi ossee", "health.clinical_term"],
|
||||
["-----BEGIN PRIVATE KEY----- secret -----END PRIVATE KEY-----", "credential.private_key"],
|
||||
['{"profile":{"email":"not yet supplied"}}', "pii.json_sensitive_key"],
|
||||
] as const)("recognizes validated sensitive content without relying on the column name: %s", async (
|
||||
value,
|
||||
ruleId,
|
||||
) => {
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId }],
|
||||
});
|
||||
});
|
||||
|
||||
test("does not make a malformed email decisive", async () => {
|
||||
const target = column();
|
||||
const [assessment] = await new SensitivityClassifier(source({
|
||||
batches: [[{
|
||||
columnId: target.id,
|
||||
value: "contatto a@b..com non valido",
|
||||
characterLength: 28,
|
||||
}]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
})).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
|
||||
});
|
||||
|
||||
test("finds a valid email after a malformed candidate in the same value", async () => {
|
||||
const target = column();
|
||||
const [assessment] = await new SensitivityClassifier(source({
|
||||
batches: [[{
|
||||
columnId: target.id,
|
||||
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
|
||||
characterLength: 58,
|
||||
}]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
})).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("optional local NER evidence can make otherwise ambiguous Italian text sensitive", async () => {
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
});
|
||||
const detector: LocalNerDetector = {
|
||||
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
|
||||
};
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values, detector).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).toHaveBeenCalledWith(
|
||||
[{ columnId: target.id, text: "Dimesso Mario Rossi" }],
|
||||
expect.any(AbortSignal),
|
||||
expect.any(Number),
|
||||
);
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "ner", ruleId: "ner.entity", label: "person_name", confidence: 0.91 }],
|
||||
});
|
||||
});
|
||||
|
||||
test("does not wait for an optional NER worker that is still warming", async () => {
|
||||
const target = column();
|
||||
const detector: LocalNerDetector = {
|
||||
isReady: () => false,
|
||||
detect: vi.fn(async () => [{ columnId: target.id, label: "person", confidence: 0.99 }]),
|
||||
};
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
}), detector).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).not.toHaveBeenCalled();
|
||||
expect(assessment).toMatchObject({ assessment: "unknown" });
|
||||
});
|
||||
|
||||
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
|
||||
const columns = Array.from({ length: 17 }, (_, index) => column({
|
||||
id: `00000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
|
||||
name: `attribute_${index + 1}`,
|
||||
ordinalPosition: index + 1,
|
||||
}));
|
||||
const observations = columns.flatMap((item, columnIndex) => Array.from(
|
||||
{ length: 8 },
|
||||
(_, valueIndex) => ({
|
||||
columnId: item.id,
|
||||
value: `ordinary-${columnIndex}-${valueIndex}`,
|
||||
characterLength: 13,
|
||||
}),
|
||||
));
|
||||
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
|
||||
|
||||
await new SensitivityClassifier(source({
|
||||
batches: [observations],
|
||||
coverage: { kind: "complete", observedRows: 8 },
|
||||
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
|
||||
{ database, table, columns },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).toHaveBeenCalledTimes(2);
|
||||
expect(vi.mocked(detector.detect).mock.calls.map(([candidates]) => candidates.length)).toEqual([
|
||||
128,
|
||||
8,
|
||||
]);
|
||||
});
|
||||
|
||||
test("limits default NER work to two candidates spread across a wide table", async () => {
|
||||
const columns = Array.from({ length: 10 }, (_, index) => column({
|
||||
id: `10000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
|
||||
name: `attribute_${index + 1}`,
|
||||
ordinalPosition: index + 1,
|
||||
}));
|
||||
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
|
||||
|
||||
await new SensitivityClassifier(source({
|
||||
batches: [columns.flatMap((item, columnIndex) => [0, 1].map((valueIndex) => ({
|
||||
columnId: item.id,
|
||||
value: `ordinary-${columnIndex}-${valueIndex}`,
|
||||
characterLength: 13,
|
||||
})))],
|
||||
coverage: { kind: "complete", observedRows: 2 },
|
||||
}), detector).assessTable(
|
||||
{ database, table, columns },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).toHaveBeenCalledOnce();
|
||||
const submitted = vi.mocked(detector.detect).mock.calls[0]![0];
|
||||
expect(submitted).toHaveLength(2);
|
||||
expect(new Set(submitted.map((candidate) => candidate.columnId)).size).toBe(2);
|
||||
});
|
||||
|
||||
test("shares a bounded NER time allowance across tables in one analysis run", async () => {
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
});
|
||||
const detector: LocalNerDetector = {
|
||||
detect: vi.fn(async () => {
|
||||
await new Promise((resolve) => setTimeout(resolve, 20));
|
||||
return [];
|
||||
}),
|
||||
};
|
||||
const classifier = new SensitivityClassifier(values, detector);
|
||||
const nerBudget: SensitivityNerBudget = { remainingMs: 1 };
|
||||
|
||||
await classifier.assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
Date.now() + 1_000,
|
||||
nerBudget,
|
||||
);
|
||||
await classifier.assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
Date.now() + 1_000,
|
||||
nerBudget,
|
||||
);
|
||||
|
||||
expect(detector.detect).toHaveBeenCalledOnce();
|
||||
expect(nerBudget.remainingMs).toBe(0);
|
||||
});
|
||||
|
||||
test("uninterpretable binary content remains unknown after complete coverage", async () => {
|
||||
const target = column({ dataType: "bytea" });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "unknown",
|
||||
proposedSensitive: false,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,316 @@
|
||||
import { expect, test, vi } from "vitest";
|
||||
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js";
|
||||
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
|
||||
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
|
||||
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
|
||||
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
|
||||
|
||||
const database = {
|
||||
id: "11111111-1111-4111-8111-111111111111",
|
||||
workspaceId: "psd-clinical",
|
||||
engine: "postgres",
|
||||
databaseName: "warehouse",
|
||||
schema: 'clinical"data',
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
connectionStatus: "reachable",
|
||||
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
|
||||
} satisfies WorkspaceDatabase;
|
||||
|
||||
const table = {
|
||||
id: "22222222-2222-4222-8222-222222222222",
|
||||
databaseId: database.id,
|
||||
name: 'patient"facts',
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
} satisfies CatalogTable;
|
||||
|
||||
function column(id: string, name: string): CatalogColumn {
|
||||
return {
|
||||
id,
|
||||
tableId: table.id,
|
||||
name,
|
||||
ordinalPosition: 1,
|
||||
dataType: "text",
|
||||
isNullable: true,
|
||||
defaultExpression: null,
|
||||
primaryKeyPosition: null,
|
||||
isPrimaryKey: false,
|
||||
isForeignKey: false,
|
||||
foreignKeyCount: 0,
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
sensitive: false,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
};
|
||||
}
|
||||
|
||||
test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value');
|
||||
const fullRows = Array.from({ length: 200 }, () => ({
|
||||
__value_0: "ordinary",
|
||||
__length_0: "8",
|
||||
__value_1: null,
|
||||
__length_1: null,
|
||||
}));
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.includes("TABLESAMPLE")) {
|
||||
return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] };
|
||||
}
|
||||
if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows };
|
||||
return { rows: [] };
|
||||
});
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
let clockCalls = 0;
|
||||
const values = new ConcreteSensitivityValueSource(access, undefined, {
|
||||
now: () => clockCalls++ < 3 ? 1_000 : 6_100,
|
||||
});
|
||||
const consumed: unknown[] = [];
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note, contact],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: 61_000,
|
||||
}, (batch) => consumed.push(...batch), new AbortController().signal);
|
||||
|
||||
expect(coverage).toEqual({ kind: "sampled", observedRows: 201 });
|
||||
expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 });
|
||||
expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null });
|
||||
expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 });
|
||||
expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]);
|
||||
expect(query.mock.calls.some(([sql]) => (
|
||||
String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT")
|
||||
))).toBe(true);
|
||||
expect(query.mock.calls.some(([sql]) => String(sql) === (
|
||||
"FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
|
||||
))).toBe(true);
|
||||
expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false);
|
||||
expect(query.mock.calls.some(([sql]) => (
|
||||
String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM')
|
||||
))).toBe(true);
|
||||
expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]);
|
||||
expect(end).toHaveBeenCalledOnce();
|
||||
});
|
||||
|
||||
test("reports complete coverage when the final full-scan page is short", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD")
|
||||
? { rows: [{ __value_0: "ordinary", __length_0: 8 }] }
|
||||
: { rows: [] });
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
const values = new ConcreteSensitivityValueSource(access);
|
||||
const consume = vi.fn();
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal);
|
||||
|
||||
expect(coverage).toEqual({ kind: "complete", observedRows: 1 });
|
||||
expect(query.mock.calls.filter(([sql]) => (
|
||||
String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
|
||||
))).toHaveLength(1);
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "ordinary", characterLength: 8 },
|
||||
]);
|
||||
});
|
||||
|
||||
test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
let fullScanAttempts = 0;
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.includes("TABLESAMPLE")) {
|
||||
return { rows: [{ __value_0: "sample", __length_0: 6 }] };
|
||||
}
|
||||
if (sql.startsWith("FETCH FORWARD")) {
|
||||
fullScanAttempts += 1;
|
||||
throw Object.assign(new Error("statement timeout"), { code: "57014" });
|
||||
}
|
||||
return { rows: [] };
|
||||
});
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
const values = new ConcreteSensitivityValueSource(access);
|
||||
const consume = vi.fn();
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal);
|
||||
|
||||
expect(fullScanAttempts).toBe(1);
|
||||
expect(coverage).toEqual({ kind: "sampled", observedRows: 1 });
|
||||
expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([
|
||||
"SAVEPOINT sensitivity_full_scan",
|
||||
"ROLLBACK TO SAVEPOINT sensitivity_full_scan",
|
||||
]));
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "sample", characterLength: 6 },
|
||||
]);
|
||||
});
|
||||
|
||||
test("scans a REST run_query binding without using PostgreSQL-wire access", async () => {
|
||||
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-"));
|
||||
const credentialFile = join(root, "api-key");
|
||||
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
|
||||
const release = vi.fn();
|
||||
const secretStore = {
|
||||
materialize: vi.fn(() => ({
|
||||
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
|
||||
release,
|
||||
})),
|
||||
} as unknown as WorkspaceSecretStore;
|
||||
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
|
||||
{ __value_0: "mario.rossi@example.it", __length_0: 23 },
|
||||
]), { status: 200, headers: { "content-type": "application/json" } }));
|
||||
vi.stubGlobal("fetch", fetchMock);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
|
||||
};
|
||||
const values = new ConcreteSensitivityValueSource(access, secretStore);
|
||||
const restDatabase: WorkspaceDatabase = {
|
||||
...database,
|
||||
binding: {
|
||||
transport: "rest_api",
|
||||
baseUrl: "https://dwh.example.test/root/",
|
||||
restPath: "/health",
|
||||
restAuth: "x-api-key",
|
||||
},
|
||||
};
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const consume = vi.fn();
|
||||
|
||||
try {
|
||||
await expect(values.scanTable({
|
||||
database: restDatabase,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal)).resolves.toEqual({
|
||||
kind: "complete",
|
||||
observedRows: 1,
|
||||
});
|
||||
expect(access.connect).not.toHaveBeenCalled();
|
||||
expect(fetchMock).toHaveBeenCalledWith(
|
||||
"https://dwh.example.test/root/rpc/run_query",
|
||||
expect.objectContaining({
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json", "x-api-key": "test-api-key" },
|
||||
}),
|
||||
);
|
||||
const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body));
|
||||
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0');
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 },
|
||||
]);
|
||||
expect(release).toHaveBeenCalledOnce();
|
||||
} finally {
|
||||
vi.unstubAllGlobals();
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test("keeps multi-request REST scans conservative without a source transaction", async () => {
|
||||
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-"));
|
||||
const credentialFile = join(root, "api-key");
|
||||
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
|
||||
const secretStore = {
|
||||
materialize: vi.fn(() => ({
|
||||
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
|
||||
release: vi.fn(),
|
||||
})),
|
||||
} as unknown as WorkspaceSecretStore;
|
||||
const fetchMock = vi.fn()
|
||||
.mockResolvedValueOnce(new Response(JSON.stringify([
|
||||
{ __value_0: "ordinary", __length_0: 8 },
|
||||
]), { status: 200 }))
|
||||
.mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 }));
|
||||
vi.stubGlobal("fetch", fetchMock);
|
||||
const values = new ConcreteSensitivityValueSource({
|
||||
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
|
||||
}, secretStore, { batchRows: 1 });
|
||||
const restDatabase: WorkspaceDatabase = {
|
||||
...database,
|
||||
binding: {
|
||||
transport: "rest_api",
|
||||
baseUrl: "https://dwh.example.test/root",
|
||||
restPath: "/health",
|
||||
restAuth: "x-api-key",
|
||||
},
|
||||
};
|
||||
|
||||
try {
|
||||
await expect(values.scanTable({
|
||||
database: restDatabase,
|
||||
table,
|
||||
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
|
||||
kind: "sampled",
|
||||
observedRows: 1,
|
||||
});
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2);
|
||||
} finally {
|
||||
vi.unstubAllGlobals();
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => {
|
||||
const query = vi.fn(async () => ({ rows: [] }));
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
const now = vi.fn()
|
||||
.mockReturnValueOnce(1_000)
|
||||
.mockReturnValue(61_000);
|
||||
const values = new ConcreteSensitivityValueSource(access, undefined, { now });
|
||||
|
||||
await expect(values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: 60_000,
|
||||
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
|
||||
kind: "sampled",
|
||||
observedRows: 0,
|
||||
});
|
||||
expect(query).not.toHaveBeenCalled();
|
||||
expect(end).toHaveBeenCalledOnce();
|
||||
});
|
||||
@@ -81,6 +81,35 @@ test("loadConfig keeps local development defaults", () => {
|
||||
expect(loadConfig({}).dataRoot).toBeUndefined();
|
||||
});
|
||||
|
||||
test("loadConfig keeps local NER disabled unless an absolute model path is configured", () => {
|
||||
expect(loadConfig({}).sensitivityNer).toBeUndefined();
|
||||
expect(loadConfig({
|
||||
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
|
||||
}).sensitivityNer?.pythonExecutable).toBe("/opt/sensitivity-ner/bin/python");
|
||||
expect(loadConfig({
|
||||
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
|
||||
THT_SENSITIVITY_NER_PYTHON: "/opt/sensitivity-ner/bin/python",
|
||||
THT_SENSITIVITY_NER_WORKER: "/app/backend/python/sensitivity_ner_worker.py",
|
||||
THT_SENSITIVITY_NER_THREADS: "3",
|
||||
}).sensitivityNer).toEqual({
|
||||
modelPath: "/models/gliner2-pii",
|
||||
pythonExecutable: "/opt/sensitivity-ner/bin/python",
|
||||
workerScript: "/app/backend/python/sensitivity_ner_worker.py",
|
||||
threads: 3,
|
||||
});
|
||||
});
|
||||
|
||||
test("loadConfig rejects ambiguous or unsafe local NER configuration", () => {
|
||||
expect(() => loadConfig({ THT_SENSITIVITY_NER_MODEL_PATH: "fastino/model" }))
|
||||
.toThrow("sensitivity NER model path configuration is invalid");
|
||||
expect(() => loadConfig({
|
||||
THT_SENSITIVITY_NER_MODEL_PATH: "/models/gliner2-pii",
|
||||
THT_SENSITIVITY_NER_THREADS: "0",
|
||||
})).toThrow("sensitivity NER thread configuration is invalid");
|
||||
expect(() => loadConfig({ THT_SENSITIVITY_NER_PYTHON: "/opt/ner/bin/python" }))
|
||||
.toThrow("sensitivity NER settings require a model path");
|
||||
});
|
||||
|
||||
test("loadConfig allows none and mock only outside production when auth.yaml is absent", () => {
|
||||
const originalNodeEnvironment = process.env.NODE_ENV;
|
||||
delete process.env.NODE_ENV;
|
||||
|
||||
Reference in New Issue
Block a user