Files
ThothII/backend/test/catalog-sensitivity-classifier.test.ts
T

707 lines
23 KiB
TypeScript

import { expect, test, vi } from "vitest";
import {
SensitivityClassifier,
type LocalNerDetector,
type SensitivityNerBudget,
type SensitivityTableScan,
type SensitivityValueSource,
} from "../src/catalog/sensitivity-classifier.js";
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
import { CatalogConnectorError } from "../src/catalog/types.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
workspaceId: "psd-clinical",
engine: "postgres",
databaseName: "warehouse",
schema: "public",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
connectionStatus: "reachable",
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
const table = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
name: "observations",
sourceComment: null,
description: null,
generatedDescription: null,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
} satisfies CatalogTable;
function column(overrides: Partial<CatalogColumn> = {}): CatalogColumn {
return {
id: "33333333-3333-4333-8333-333333333333",
tableId: table.id,
name: "note",
ordinalPosition: 1,
dataType: "character varying",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
isPrimaryKey: false,
isForeignKey: false,
foreignKeyCount: 0,
sourceComment: null,
description: null,
generatedDescription: null,
sensitive: false,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
...overrides,
};
}
function source(scan: SensitivityTableScan): SensitivityValueSource {
return { scanTable: vi.fn(async (_request, consume) => {
for (const batch of scan.batches) await consume(batch);
return scan.coverage;
}) };
}
test("reports source-scan activity before a long table scan completes", async () => {
let releaseScan!: () => void;
const scanGate = new Promise<void>((resolve) => {
releaseScan = resolve;
});
const scanTable = vi.fn(async () => {
await scanGate;
return { kind: "complete" as const, observedValues: 0 };
});
const activity = vi.fn();
const analysis = new SensitivityClassifier({ scanTable }).assess(
[{ database, table, columns: [column()] }],
new AbortController().signal,
undefined,
activity,
);
await vi.waitFor(() => expect(scanTable).toHaveBeenCalledOnce());
releaseScan();
await analysis;
expect(activity).toHaveBeenCalledWith(
"Scanning source data: pass 1 of 3, table batch 1 of 1.",
);
});
test("one email hidden in a generically named column makes the whole column sensitive", async () => {
const target = column();
const values = source({
batches: [[
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
]],
coverage: { kind: "complete", observedValues: 2 },
});
const classifier = new SensitivityClassifier(values);
const [assessment] = await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
columnId: target.id,
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "content", ruleId: "pii.email" }],
});
});
test("one text value longer than 500 characters makes the whole column sensitive", async () => {
const target = column({ name: "comment" });
const values = source({
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "length", ruleId: "text.over_500_characters" }],
});
});
test("scans every table at 300 before advancing to 1,000 and 3,000 values", async () => {
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
const first = column({ name: "status" });
const second = column({
id: "88888888-8888-4888-8888-888888888888",
tableId: otherTable.id,
name: "comment",
});
const calls: string[] = [];
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
calls.push(`${request.table.name}:${request.valuesPerColumn}:${request.sampleOffset}`);
await consume(request.columns.map((item) => ({
columnId: item.id,
value: "ordinary",
characterLength: 8,
})));
return { kind: "sampled", observedValues: request.columns.length };
}),
};
await new SensitivityClassifier(values).assess([
{ database, table, columns: [first] },
{ database, table: otherTable, columns: [second] },
], new AbortController().signal);
expect(calls).toEqual([
"observations:300:0",
"events:300:0",
"observations:700:300",
"events:700:300",
"observations:2000:1000",
"events:2000:1000",
]);
});
test("runs at most two table scans concurrently", async () => {
const targets = Array.from({ length: 3 }, (_, index) => {
const targetTable = {
...table,
id: `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
name: `table_${index + 1}`,
};
return {
database,
table: targetTable,
columns: [column({
id: `10000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
tableId: targetTable.id,
name: `attribute_${index + 1}`,
})],
};
});
let active = 0;
let maximum = 0;
const values: SensitivityValueSource = {
scanTable: vi.fn(async () => {
active += 1;
maximum = Math.max(maximum, active);
await Promise.resolve();
active -= 1;
return { kind: "complete", observedValues: 0 };
}),
};
await new SensitivityClassifier(values).assess(targets, new AbortController().signal);
expect(maximum).toBe(2);
expect(values.scanTable).toHaveBeenCalledTimes(3);
});
test("aborts a peer table scan when another concurrent source scan fails", async () => {
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
const first = column({ name: "status" });
const second = column({
id: "88888888-8888-4888-8888-888888888888",
tableId: otherTable.id,
name: "comment",
});
let peerSignal: AbortSignal | undefined;
const failure = new CatalogConnectorError("source unavailable");
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, _consume, scanSignal) => {
if (request.table.id === table.id) {
await Promise.resolve();
throw failure;
}
peerSignal = scanSignal;
return await new Promise((_resolve, reject) => {
scanSignal.addEventListener("abort", () => reject(scanSignal.reason), { once: true });
});
}),
};
await expect(new SensitivityClassifier(values).assess([
{ database, table, columns: [first] },
{ database, table: otherTable, columns: [second] },
], new AbortController().signal)).rejects.toBe(failure);
expect(peerSignal?.aborted).toBe(true);
});
test("stops sampling a column as soon as one value is sensitive", async () => {
const target = column();
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume([{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 }]);
return { kind: "sampled", observedValues: 1 };
}),
};
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(values.scanTable).toHaveBeenCalledOnce();
expect(assessment).toMatchObject({ assessment: "sensitive", proposedSensitive: true });
});
test("stops non-text columns after the 1,000-value stage", async () => {
const target = column({ dataType: "integer", name: "sequence_number" });
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume([{ columnId: target.id, value: "42", characterLength: 2 }]);
return { kind: "sampled", observedValues: 1 };
}),
};
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(vi.mocked(values.scanTable).mock.calls.map(([request]) => request.valuesPerColumn))
.toEqual([300, 700]);
expect(assessment).toMatchObject({
assessment: "non_sensitive",
proposedSensitive: false,
coverage: "sampled",
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_1000" }],
});
});
test("complete coverage classifies benign and empty columns as non-sensitive", async () => {
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
const humanProtected = column({
id: "66666666-6666-4666-8666-666666666666",
name: "category",
sensitive: true,
});
const values = source({
batches: [[
{ columnId: benign.id, value: "active", characterLength: 6 },
{ columnId: empty.id, value: null, characterLength: null },
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
]],
coverage: { kind: "complete", observedValues: 1 },
});
const assessments = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [benign, empty, humanProtected] },
new AbortController().signal,
);
expect(assessments).toEqual([
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
expect.objectContaining({
columnId: empty.id,
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
}),
expect.objectContaining({
columnId: humanProtected.id,
assessment: "non_sensitive",
proposedSensitive: false,
}),
]);
});
test("sampled coverage without a match proposes non-sensitive independently of the current flag", async () => {
const target = column({ sensitive: true });
const values = source({
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
test("an unavailable source fails the analysis instead of producing unknown decisions", async () => {
const unresolved = column();
const metadataMatch = column({
id: "44444444-4444-4444-8444-444444444444",
name: "codice_fiscale",
});
const values: SensitivityValueSource = {
scanTable: vi.fn(async () => {
throw new CatalogConnectorError("upstream detail must not escape");
}),
};
await expect(new SensitivityClassifier(values).assessTable(
{ database, table, columns: [unresolved, metadataMatch] },
new AbortController().signal,
)).rejects.toBeInstanceOf(CatalogConnectorError);
});
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
const target = column({ name: "codice_fiscale" });
const values = source({
batches: [],
coverage: { kind: "complete", observedValues: 0 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
});
expect(values.scanTable).not.toHaveBeenCalled();
});
test("excludes bigint primary keys from content analysis as non-informative identifiers", async () => {
const target = column({
name: "id",
dataType: "bigint",
primaryKeyPosition: 1,
isPrimaryKey: true,
});
const values = source({
batches: [[{
columnId: target.id,
value: "3471234567",
characterLength: 10,
}]],
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{
kind: "type",
ruleId: "type.bigint_primary_key_non_informative",
label: "non-informative bigint primary key",
}],
coverage: "metadata",
});
expect(values.scanTable).not.toHaveBeenCalled();
});
test("infers an undeclared bigint column named pk as a non-informative primary-key identifier", async () => {
const target = column({
name: "pk",
dataType: "bigint",
primaryKeyPosition: null,
isPrimaryKey: false,
});
const values = source({
batches: [[{
columnId: target.id,
value: "3471234567",
characterLength: 10,
}]],
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{
kind: "metadata",
ruleId: "metadata.bigint_pk_identifier_non_informative",
label: "non-informative conventional bigint primary-key identifier",
}],
coverage: "metadata",
});
expect(values.scanTable).not.toHaveBeenCalled();
});
test("still inspects phone-like values in bigint columns that are not primary keys", async () => {
const target = column({ name: "id", dataType: "bigint" });
const values = source({
batches: [[{
columnId: target.id,
value: "3471234567",
characterLength: 10,
}]],
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "content", ruleId: "pii.phone_number" }],
});
expect(values.scanTable).toHaveBeenCalledOnce();
});
test.each([
["RSSMRA85T10A562S", "pii.italian_fiscal_code"],
["IT60 X054 2811 1010 0000 0123 456", "financial.iban"],
["4111 1111 1111 1111", "financial.payment_card"],
["SWIFT DEUTDEFF500", "financial.bic"],
["Partita IVA 00743110157", "pii.italian_vat"],
["Passaporto YA1234567", "pii.passport_number"],
["Carta d'identità CA12345AA", "pii.identity_card"],
["Patente di guida U11234567A", "pii.drivers_license_number"],
["Chiamare +39 347 123 4567", "pii.phone_number"],
["Client 192.168.1.5", "network.ip_address"],
["Device 00:1B:44:11:3A:B7", "network.mac_address"],
["https://example.org/profiles/mario", "network.url"],
["550e8400-e29b-41d4-a716-446655440000", "pii.uuid"],
["AWS key AKIAIOSFODNN7EXAMPLE", "credential.access_key"],
["Diagnosi: carcinoma mammario con metastasi ossee", "health.clinical_term"],
["-----BEGIN PRIVATE KEY----- secret -----END PRIVATE KEY-----", "credential.private_key"],
['{"profile":{"email":"not yet supplied"}}', "pii.json_sensitive_key"],
] as const)("recognizes validated sensitive content without relying on the column name: %s", async (
value,
ruleId,
) => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "content", ruleId }],
});
});
test("does not make a malformed email decisive", async () => {
const target = column();
const [assessment] = await new SensitivityClassifier(source({
batches: [[{
columnId: target.id,
value: "contatto a@b..com non valido",
characterLength: 28,
}]],
coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "non_sensitive",
evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
});
});
test("finds a valid email after a malformed candidate in the same value", async () => {
const target = column();
const [assessment] = await new SensitivityClassifier(source({
batches: [[{
columnId: target.id,
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
characterLength: 58,
}]],
coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
});
});
test("optional local NER evidence can make otherwise ambiguous Italian text sensitive", async () => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
};
const [assessment] = await new SensitivityClassifier(values, detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledWith(
[{ columnId: target.id, text: "Dimesso Mario Rossi" }],
expect.any(AbortSignal),
expect.any(Number),
);
expect(assessment).toMatchObject({
assessment: "sensitive",
evidence: [{ kind: "ner", ruleId: "ner.entity", label: "person_name", confidence: 0.91 }],
});
});
test("does not wait for an optional NER worker that is still warming", async () => {
const target = column();
const detector: LocalNerDetector = {
isReady: () => false,
detect: vi.fn(async () => [{ columnId: target.id, label: "person", confidence: 0.99 }]),
};
const [assessment] = await new SensitivityClassifier(source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedValues: 1 },
}), detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).not.toHaveBeenCalled();
expect(assessment).toMatchObject({
assessment: "non_sensitive",
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
const columns = Array.from({ length: 17 }, (_, index) => column({
id: `00000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
name: `attribute_${index + 1}`,
ordinalPosition: index + 1,
}));
const observations = columns.flatMap((item, columnIndex) => Array.from(
{ length: 8 },
(_, valueIndex) => ({
columnId: item.id,
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
}),
));
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
await new SensitivityClassifier(source({
batches: [observations],
coverage: { kind: "complete", observedValues: 8 },
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
{ database, table, columns },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledTimes(2);
expect(vi.mocked(detector.detect).mock.calls.map(([candidates]) => candidates.length)).toEqual([
128,
8,
]);
});
test("limits default NER work to two candidates spread across a wide table", async () => {
const columns = Array.from({ length: 10 }, (_, index) => column({
id: `10000000-0000-4000-8000-${(index + 1).toString(16).padStart(12, "0")}`,
name: `attribute_${index + 1}`,
ordinalPosition: index + 1,
}));
const detector: LocalNerDetector = { detect: vi.fn(async () => []) };
await new SensitivityClassifier(source({
batches: [columns.flatMap((item, columnIndex) => [0, 1].map((valueIndex) => ({
columnId: item.id,
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
})))],
coverage: { kind: "complete", observedValues: 2 },
}), detector).assessTable(
{ database, table, columns },
new AbortController().signal,
);
expect(detector.detect).toHaveBeenCalledOnce();
const submitted = vi.mocked(detector.detect).mock.calls[0]![0];
expect(submitted).toHaveLength(2);
expect(new Set(submitted.map((candidate) => candidate.columnId)).size).toBe(2);
});
test("shares a bounded NER time allowance across tables in one analysis run", async () => {
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => {
await new Promise((resolve) => setTimeout(resolve, 20));
return [];
}),
};
const classifier = new SensitivityClassifier(values, detector);
const nerBudget: SensitivityNerBudget = { remainingMs: 1 };
await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
Date.now() + 1_000,
nerBudget,
);
await classifier.assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
Date.now() + 1_000,
nerBudget,
);
expect(detector.detect).toHaveBeenCalledOnce();
expect(nerBudget.remainingMs).toBe(0);
});
test("uninterpretable binary content is protected conservatively without scanning", async () => {
const target = column({ dataType: "bytea" });
const values = source({
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "type", ruleId: "type.binary_uninspectable" }],
});
expect(values.scanTable).not.toHaveBeenCalled();
});