feat: sample sensitive columns progressively

This commit is contained in:
Codex
2026-09-03 10:25:05 +02:00
parent f114d0065a
commit 8e778b9edb
24 changed files with 1001 additions and 581 deletions
@@ -77,7 +77,7 @@ async function setup(
value: "ordinary",
characterLength: 8,
})));
return { kind: "complete", observedRows: 1 };
return { kind: "complete", observedValues: 1 };
}),
},
) {
@@ -167,7 +167,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
scope: "all",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
policyVersion: "sensitivity-v2",
status: "completed",
total: 1,
suggestedSensitive: 1,
@@ -185,6 +185,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
coverage: "metadata",
}],
});
expect(await repository.getColumn(database.id, column.tableId, column.id))
@@ -239,10 +240,8 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
}
});
test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => {
const controller = new AbortController();
controller.abort();
const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal);
test("does not impose a global HTTP deadline on sensitivity analysis", async () => {
const timeout = vi.spyOn(AbortSignal, "timeout");
const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") });
try {
@@ -252,12 +251,9 @@ test("stops sensitivity analysis at the HTTP deadline without creating a review"
payload: { scope: "all" },
});
expect(response.statusCode).toBe(504);
expect(response.json()).toEqual({
code: "sensitivity_analysis_timeout",
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
});
expect(await repository.listSensitivityAnalysisRuns()).toEqual([]);
expect(response.statusCode).toBe(200);
expect(timeout).not.toHaveBeenCalled();
expect(await repository.listSensitivityAnalysisRuns()).toHaveLength(1);
} finally {
timeout.mockRestore();
await app.close();
@@ -6,7 +6,9 @@ import {
import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js";
import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js";
import type {
CatalogColumn,
CatalogRepository,
CatalogTable,
SensitivityAnalysisRun,
WorkspaceDatabase,
} from "../src/catalog/types.js";
@@ -24,13 +26,54 @@ const database = {
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
function catalogTable(id: string, name: string): CatalogTable {
return {
id,
databaseId: database.id,
name,
sourceComment: null,
description: null,
generatedDescription: null,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
};
}
function catalogColumn(id: string, tableId: string, name: string): CatalogColumn {
return {
id,
tableId,
name,
ordinalPosition: 1,
dataType: "text",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
isPrimaryKey: false,
isForeignKey: false,
foreignKeyCount: 0,
sourceComment: null,
description: null,
generatedDescription: null,
sensitive: false,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
};
}
const running: SensitivityAnalysisRun = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
scope: "all",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
policyVersion: "sensitivity-v2",
status: "running",
total: 0,
suggestedSensitive: 0,
@@ -56,7 +99,7 @@ test("stops catalog selection when the request expires during a catalog read", a
}),
listTables,
} as unknown as CatalogRepository;
const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier;
const classifier = { assess: vi.fn() } as unknown as SensitivityClassifier;
const analysis = new SensitivityAnalysisService(repository, classifier);
await expect(analysis.analyze(
@@ -66,7 +109,71 @@ test("stops catalog selection when the request expires during a catalog read", a
controller.signal,
)).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
expect(listTables).not.toHaveBeenCalled();
expect(classifier.assessTable).not.toHaveBeenCalled();
expect(classifier.assess).not.toHaveBeenCalled();
});
test("classifies all selected tables in one breadth-first run and reports coverage", async () => {
const firstTable = catalogTable("33333333-3333-4333-8333-333333333333", "patients");
const secondTable = catalogTable("44444444-4444-4444-8444-444444444444", "encounters");
const firstColumn = catalogColumn(
"55555555-5555-4555-8555-555555555555",
firstTable.id,
"status",
);
const secondColumn = catalogColumn(
"66666666-6666-4666-8666-666666666666",
secondTable.id,
"note",
);
const repository = {
get: vi.fn(async () => database),
listTables: vi.fn(async () => [firstTable, secondTable]),
listColumns: vi.fn(async (_databaseId: string, tableId: string) => (
tableId === firstTable.id ? [firstColumn] : [secondColumn]
)),
} as unknown as CatalogRepository;
const assess = vi.fn(async () => [
{
columnId: firstColumn.id,
assessment: "non_sensitive" as const,
proposedSensitive: false,
evidence: [{ kind: "coverage" as const, ruleId: "coverage.sampled_1000" }],
observedValues: 1_000,
coverage: "sampled" as const,
},
{
columnId: secondColumn.id,
assessment: "sensitive" as const,
proposedSensitive: true,
evidence: [{ kind: "content" as const, ruleId: "pii.email" }],
observedValues: 12,
coverage: "sampled" as const,
},
]);
const classifier = { assess } as unknown as SensitivityClassifier;
const onPrepared = vi.fn();
const onProgress = vi.fn();
const suggestions = await new SensitivityAnalysisService(repository, classifier).analyze(
database.id,
"all",
[],
new AbortController().signal,
onPrepared,
onProgress,
);
expect(assess).toHaveBeenCalledOnce();
expect(assess.mock.calls[0]![0]).toEqual([
{ database, table: firstTable, columns: [firstColumn] },
{ database, table: secondTable, columns: [secondColumn] },
]);
expect(onPrepared).toHaveBeenCalledWith(2);
expect(onProgress.mock.calls.map(([processed]) => processed)).toEqual([1, 2]);
expect(suggestions).toEqual([
expect.objectContaining({ columnId: firstColumn.id, sensitive: false, coverage: "sampled" }),
expect.objectContaining({ columnId: secondColumn.id, sensitive: true, coverage: "sampled" }),
]);
});
test("marks a created run interrupted if the request deadline expires during persistence", async () => {
@@ -94,6 +201,6 @@ test("marks a created run interrupted if the request deadline expires during per
status: "interrupted",
total: 0,
unknown: 0,
errorSummary: "Local sensitivity analysis reached its time limit.",
errorSummary: "Local sensitivity analysis was interrupted before completion.",
}));
});
@@ -76,7 +76,7 @@ test("one email hidden in a generically named column makes the whole column sens
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
]],
coverage: { kind: "complete", observedRows: 2 },
coverage: { kind: "complete", observedValues: 2 },
});
const classifier = new SensitivityClassifier(values);
@@ -97,7 +97,7 @@ test("one text value longer than 500 characters makes the whole column sensitive
const target = column({ name: "comment" });
const values = source({
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -112,7 +112,151 @@ test("one text value longer than 500 characters makes the whole column sensitive
});
});
test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
test("scans every table at 300 before advancing to 1,000 and 3,000 values", async () => {
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
const first = column({ name: "status" });
const second = column({
id: "88888888-8888-4888-8888-888888888888",
tableId: otherTable.id,
name: "comment",
});
const calls: string[] = [];
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
calls.push(`${request.table.name}:${request.valuesPerColumn}:${request.sampleOffset}`);
await consume(request.columns.map((item) => ({
columnId: item.id,
value: "ordinary",
characterLength: 8,
})));
return { kind: "sampled", observedValues: request.columns.length };
}),
};
await new SensitivityClassifier(values).assess([
{ database, table, columns: [first] },
{ database, table: otherTable, columns: [second] },
], new AbortController().signal);
expect(calls).toEqual([
"observations:300:0",
"events:300:0",
"observations:700:300",
"events:700:300",
"observations:2000:1000",
"events:2000:1000",
]);
});
test("runs at most two table scans concurrently", async () => {
const targets = Array.from({ length: 3 }, (_, index) => {
const targetTable = {
...table,
id: `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
name: `table_${index + 1}`,
};
return {
database,
table: targetTable,
columns: [column({
id: `10000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
tableId: targetTable.id,
name: `attribute_${index + 1}`,
})],
};
});
let active = 0;
let maximum = 0;
const values: SensitivityValueSource = {
scanTable: vi.fn(async () => {
active += 1;
maximum = Math.max(maximum, active);
await Promise.resolve();
active -= 1;
return { kind: "complete", observedValues: 0 };
}),
};
await new SensitivityClassifier(values).assess(targets, new AbortController().signal);
expect(maximum).toBe(2);
expect(values.scanTable).toHaveBeenCalledTimes(3);
});
test("aborts a peer table scan when another concurrent source scan fails", async () => {
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
const first = column({ name: "status" });
const second = column({
id: "88888888-8888-4888-8888-888888888888",
tableId: otherTable.id,
name: "comment",
});
let peerSignal: AbortSignal | undefined;
const failure = new CatalogConnectorError("source unavailable");
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, _consume, scanSignal) => {
if (request.table.id === table.id) {
await Promise.resolve();
throw failure;
}
peerSignal = scanSignal;
return await new Promise((_resolve, reject) => {
scanSignal.addEventListener("abort", () => reject(scanSignal.reason), { once: true });
});
}),
};
await expect(new SensitivityClassifier(values).assess([
{ database, table, columns: [first] },
{ database, table: otherTable, columns: [second] },
], new AbortController().signal)).rejects.toBe(failure);
expect(peerSignal?.aborted).toBe(true);
});
test("stops sampling a column as soon as one value is sensitive", async () => {
const target = column();
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume([{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 }]);
return { kind: "sampled", observedValues: 1 };
}),
};
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(values.scanTable).toHaveBeenCalledOnce();
expect(assessment).toMatchObject({ assessment: "sensitive", proposedSensitive: true });
});
test("stops non-text columns after the 1,000-value stage", async () => {
const target = column({ dataType: "integer", name: "sequence_number" });
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume([{ columnId: target.id, value: "42", characterLength: 2 }]);
return { kind: "sampled", observedValues: 1 };
}),
};
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(vi.mocked(values.scanTable).mock.calls.map(([request]) => request.valuesPerColumn))
.toEqual([300, 700]);
expect(assessment).toMatchObject({
assessment: "non_sensitive",
proposedSensitive: false,
coverage: "sampled",
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_1000" }],
});
});
test("complete coverage classifies benign and empty columns as non-sensitive", async () => {
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
const humanProtected = column({
@@ -126,7 +270,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
{ columnId: empty.id, value: null, characterLength: null },
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const assessments = await new SensitivityClassifier(values).assessTable(
@@ -138,7 +282,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
expect.objectContaining({
columnId: empty.id,
assessment: "unknown",
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
}),
@@ -150,11 +294,11 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
]);
});
test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
test("sampled coverage without a match proposes non-sensitive independently of the current flag", async () => {
const target = column({ sensitive: true });
const values = source({
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -163,13 +307,13 @@ test("sampled coverage without a match is unknown and preserves the current huma
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: true,
evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
test("an unavailable source fails the analysis instead of producing unknown decisions", async () => {
const unresolved = column();
const metadataMatch = column({
id: "44444444-4444-4444-8444-444444444444",
@@ -181,30 +325,17 @@ test("an unavailable source produces sanitized unknown evidence without losing m
}),
};
const assessments = await new SensitivityClassifier(values).assessTable(
await expect(new SensitivityClassifier(values).assessTable(
{ database, table, columns: [unresolved, metadataMatch] },
new AbortController().signal,
);
expect(assessments).toEqual([
expect.objectContaining({
columnId: unresolved.id,
assessment: "unknown",
evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
}),
expect.objectContaining({
columnId: metadataMatch.id,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
}),
]);
)).rejects.toBeInstanceOf(CatalogConnectorError);
});
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
const target = column({ name: "codice_fiscale" });
const values = source({
batches: [],
coverage: { kind: "complete", observedRows: 0 },
coverage: { kind: "complete", observedValues: 0 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -245,7 +376,7 @@ test.each([
const target = column();
const values = source({
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -267,13 +398,16 @@ test("does not make a malformed email decisive", async () => {
value: "contatto a@b..com non valido",
characterLength: 28,
}]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
expect(assessment).toMatchObject({
assessment: "non_sensitive",
evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
});
});
test("finds a valid email after a malformed candidate in the same value", async () => {
@@ -284,7 +418,7 @@ test("finds a valid email after a malformed candidate in the same value", async
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
characterLength: 58,
}]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
@@ -300,7 +434,7 @@ test("optional local NER evidence can make otherwise ambiguous Italian text sens
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
@@ -331,14 +465,17 @@ test("does not wait for an optional NER worker that is still warming", async ()
const [assessment] = await new SensitivityClassifier(source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
}), detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).not.toHaveBeenCalled();
expect(assessment).toMatchObject({ assessment: "unknown" });
expect(assessment).toMatchObject({
assessment: "non_sensitive",
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
@@ -359,7 +496,7 @@ test("bounds each optional NER request when an installation raises the per-table
await new SensitivityClassifier(source({
batches: [observations],
coverage: { kind: "complete", observedRows: 8 },
coverage: { kind: "complete", observedValues: 8 },
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
{ database, table, columns },
new AbortController().signal,
@@ -386,7 +523,7 @@ test("limits default NER work to two candidates spread across a wide table", asy
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
})))],
coverage: { kind: "complete", observedRows: 2 },
coverage: { kind: "complete", observedValues: 2 },
}), detector).assessTable(
{ database, table, columns },
new AbortController().signal,
@@ -402,7 +539,7 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => {
@@ -430,11 +567,11 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
expect(nerBudget.remainingMs).toBe(0);
});
test("uninterpretable binary content remains unknown after complete coverage", async () => {
test("uninterpretable binary content is protected conservatively without scanning", async () => {
const target = column({ dataType: "bytea" });
const values = source({
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -443,8 +580,9 @@ test("uninterpretable binary content remains unknown after complete coverage", a
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "type", ruleId: "type.binary_uninspectable" }],
});
expect(values.scanTable).not.toHaveBeenCalled();
});
@@ -1,12 +1,17 @@
import { expect, test, vi } from "vitest";
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { expect, test, vi } from "vitest";
import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js";
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
import {
CatalogConnectorError,
type CatalogColumn,
type CatalogTable,
type WorkspaceDatabase,
} from "../src/catalog/types.js";
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
@@ -60,128 +65,145 @@ function column(id: string, name: string): CatalogColumn {
};
}
test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => {
function request(columns: readonly CatalogColumn[], overrides: Record<string, unknown> = {}) {
return {
database,
table,
columns,
valuesPerColumn: 300,
sampleOffset: 0,
sampleSeed: 37,
queryTimeoutMs: 5_000,
fullScanThreshold: 1_000,
...overrides,
};
}
test("uses bounded read-only PostgreSQL sampling for tables above 1,000 rows", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value');
const fullRows = Array.from({ length: 200 }, () => ({
__value_0: "ordinary",
__length_0: "8",
__value_1: null,
__length_1: null,
}));
const query = vi.fn(async (sql: string) => {
if (sql.includes("TABLESAMPLE")) {
return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] };
if (sql.startsWith("SELECT 1 AS __present")) {
return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) };
}
if (sql.startsWith("WITH sampled")) {
return { rows: [
{ __column_index: 0, __value: "ordinary", __length: "8" },
{ __column_index: 1, __value: "mario.rossi@example.it", __length: 23 },
] };
}
if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows };
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
let clockCalls = 0;
const values = new ConcreteSensitivityValueSource(access, undefined, {
now: () => clockCalls++ < 3 ? 1_000 : 6_100,
});
const consumed: unknown[] = [];
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note, contact],
fullScanBudgetMs: 5_000,
deadline: 61_000,
}, (batch) => consumed.push(...batch), new AbortController().signal);
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note, contact]),
consume,
new AbortController().signal,
)).resolves.toEqual({ kind: "sampled", observedValues: 2 });
expect(coverage).toEqual({ kind: "sampled", observedRows: 201 });
expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 });
expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null });
expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 });
expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]);
expect(query.mock.calls.some(([sql]) => (
String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT")
))).toBe(true);
expect(query.mock.calls.some(([sql]) => String(sql) === (
"FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
))).toBe(true);
expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false);
expect(query.mock.calls.some(([sql]) => (
String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM')
))).toBe(true);
expect(query).toHaveBeenCalledWith("SELECT set_config('statement_timeout', $1, true)", ["5000ms"]);
const sampleSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled"));
expect(sampleSql).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)');
expect(sampleSql).toContain("REPEATABLE (37)");
expect(sampleSql).toContain("LIMIT 3000 OFFSET 0");
expect(sampleSql).toContain("CROSS JOIN LATERAL");
expect(sampleSql).toContain("WHERE __rank <= 300");
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "ordinary", characterLength: 8 },
{ columnId: contact.id, value: "mario.rossi@example.it", characterLength: 23 },
]);
expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]);
expect(end).toHaveBeenCalledOnce();
});
test("reports complete coverage when the final full-scan page is short", async () => {
test("fully scans a table when the 1,001-row probe proves it is small", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD")
? { rows: [{ __value_0: "ordinary", __length_0: 8 }] }
: { rows: [] });
const end = vi.fn(async () => undefined);
const query = vi.fn(async (sql: string) => {
if (sql.startsWith("SELECT 1 AS __present")) return { rows: [{ __present: 1 }] };
if (sql.startsWith("WITH sampled")) {
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
}
return { rows: [] };
});
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal);
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note]),
consume,
new AbortController().signal,
)).resolves.toEqual({ kind: "complete", observedValues: 1 });
expect(coverage).toEqual({ kind: "complete", observedRows: 1 });
expect(query.mock.calls.filter(([sql]) => (
String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
))).toHaveLength(1);
const valueSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled"));
expect(valueSql).not.toContain("TABLESAMPLE");
expect(valueSql).toContain("WHERE __rank <= 1000");
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "ordinary", characterLength: 8 },
]);
});
test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => {
test("falls back to sampling when the small-table probe reaches its query timeout", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
let fullScanAttempts = 0;
const query = vi.fn(async (sql: string) => {
if (sql.includes("TABLESAMPLE")) {
return { rows: [{ __value_0: "sample", __length_0: 6 }] };
}
if (sql.startsWith("FETCH FORWARD")) {
fullScanAttempts += 1;
if (sql.startsWith("SELECT 1 AS __present")) {
throw Object.assign(new Error("statement timeout"), { code: "57014" });
}
if (sql.startsWith("WITH sampled")) {
return { rows: [{ __column_index: 0, __value: "sample", __length: 6 }] };
}
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal);
expect(fullScanAttempts).toBe(1);
expect(coverage).toEqual({ kind: "sampled", observedRows: 1 });
expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([
"SAVEPOINT sensitivity_full_scan",
"ROLLBACK TO SAVEPOINT sensitivity_full_scan",
]));
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "sample", characterLength: 6 },
]);
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note]),
consume,
new AbortController().signal,
)).resolves.toEqual({ kind: "sampled", observedValues: 1 });
expect(query.mock.calls.map(([sql]) => String(sql))).toContain(
"ROLLBACK TO SAVEPOINT sensitivity_scan_1",
);
});
test("scans a REST run_query binding without using PostgreSQL-wire access", async () => {
test("limits each source query to at most 25 columns", async () => {
const columns = Array.from({ length: 26 }, (_, index) => column(
`00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
`attribute_${index + 1}`,
));
const query = vi.fn(async (sql: string) => {
if (sql.startsWith("SELECT 1 AS __present")) {
return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) };
}
if (sql.startsWith("WITH sampled")) {
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
}
return { rows: [] };
});
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
await new ConcreteSensitivityValueSource(access).scanTable(
request(columns),
vi.fn(),
new AbortController().signal,
);
expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2);
});
test("scans a REST run_query binding without PostgreSQL-wire access", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
@@ -193,7 +215,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
{ __value_0: "mario.rossi@example.it", __length_0: 23 },
{ __column_index: 0, __value: "mario.rossi@example.it", __length: 23 },
]), { status: 200, headers: { "content-type": "application/json" } }));
vi.stubGlobal("fetch", fetchMock);
const access: CatalogPostgresAccess = {
@@ -213,15 +235,12 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
const consume = vi.fn();
try {
await expect(values.scanTable({
await expect(values.scanTable(request([note], {
database: restDatabase,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal)).resolves.toEqual({
kind: "complete",
observedRows: 1,
fullScanThreshold: undefined,
}), consume, new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedValues: 1,
});
expect(access.connect).not.toHaveBeenCalled();
expect(fetchMock).toHaveBeenCalledWith(
@@ -232,7 +251,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
}),
);
const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body));
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0');
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)');
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 },
]);
@@ -243,74 +262,44 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
}
});
test("keeps multi-request REST scans conservative without a source transaction", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
const secretStore = {
materialize: vi.fn(() => ({
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
release: vi.fn(),
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn()
.mockResolvedValueOnce(new Response(JSON.stringify([
{ __value_0: "ordinary", __length_0: 8 },
]), { status: 200 }))
.mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 }));
vi.stubGlobal("fetch", fetchMock);
const values = new ConcreteSensitivityValueSource({
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
}, secretStore, { batchRows: 1 });
const restDatabase: WorkspaceDatabase = {
...database,
binding: {
transport: "rest_api",
baseUrl: "https://dwh.example.test/root",
restPath: "/health",
restAuth: "x-api-key",
},
};
try {
await expect(values.scanTable({
database: restDatabase,
table,
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedRows: 1,
});
expect(fetchMock).toHaveBeenCalledTimes(2);
} finally {
vi.unstubAllGlobals();
rmSync(root, { recursive: true, force: true });
}
});
test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => {
const query = vi.fn(async () => ({ rows: [] }));
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const now = vi.fn()
.mockReturnValueOnce(1_000)
.mockReturnValue(61_000);
const values = new ConcreteSensitivityValueSource(access, undefined, { now });
await expect(values.scanTable({
database,
table,
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
fullScanBudgetMs: 5_000,
deadline: 60_000,
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedRows: 0,
test("falls back to a sequential bounded sample when randomized sampling times out", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const query = vi.fn(async (sql: string) => {
if (sql.startsWith("WITH sampled") && sql.includes("TABLESAMPLE")) {
throw Object.assign(new Error("raw source detail"), { code: "57014" });
}
if (sql.startsWith("WITH sampled")) {
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
}
return { rows: [] };
});
expect(query).not.toHaveBeenCalled();
expect(end).toHaveBeenCalledOnce();
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note], { fullScanThreshold: undefined }),
vi.fn(),
new AbortController().signal,
)).resolves.toEqual({ kind: "sampled", observedValues: 1 });
expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2);
});
test("fails explicitly when both randomized and sequential sample queries time out", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const query = vi.fn(async (sql: string) => {
if (sql.startsWith("WITH sampled")) {
throw Object.assign(new Error("raw source detail"), { code: "57014" });
}
return { rows: [] };
});
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note], { fullScanThreshold: undefined }),
vi.fn(),
new AbortController().signal,
)).rejects.toEqual(new CatalogConnectorError("Sensitivity sample query timed out"));
});