feat: sample sensitive columns progressively

This commit is contained in:
Codex
2026-09-03 10:25:05 +02:00
parent f114d0065a
commit 8e778b9edb
24 changed files with 1001 additions and 581 deletions
@@ -76,7 +76,7 @@ test("one email hidden in a generically named column makes the whole column sens
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
]],
coverage: { kind: "complete", observedRows: 2 },
coverage: { kind: "complete", observedValues: 2 },
});
const classifier = new SensitivityClassifier(values);
@@ -97,7 +97,7 @@ test("one text value longer than 500 characters makes the whole column sensitive
const target = column({ name: "comment" });
const values = source({
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -112,7 +112,151 @@ test("one text value longer than 500 characters makes the whole column sensitive
});
});
test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
test("scans every table at 300 before advancing to 1,000 and 3,000 values", async () => {
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
const first = column({ name: "status" });
const second = column({
id: "88888888-8888-4888-8888-888888888888",
tableId: otherTable.id,
name: "comment",
});
const calls: string[] = [];
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
calls.push(`${request.table.name}:${request.valuesPerColumn}:${request.sampleOffset}`);
await consume(request.columns.map((item) => ({
columnId: item.id,
value: "ordinary",
characterLength: 8,
})));
return { kind: "sampled", observedValues: request.columns.length };
}),
};
await new SensitivityClassifier(values).assess([
{ database, table, columns: [first] },
{ database, table: otherTable, columns: [second] },
], new AbortController().signal);
expect(calls).toEqual([
"observations:300:0",
"events:300:0",
"observations:700:300",
"events:700:300",
"observations:2000:1000",
"events:2000:1000",
]);
});
test("runs at most two table scans concurrently", async () => {
const targets = Array.from({ length: 3 }, (_, index) => {
const targetTable = {
...table,
id: `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
name: `table_${index + 1}`,
};
return {
database,
table: targetTable,
columns: [column({
id: `10000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
tableId: targetTable.id,
name: `attribute_${index + 1}`,
})],
};
});
let active = 0;
let maximum = 0;
const values: SensitivityValueSource = {
scanTable: vi.fn(async () => {
active += 1;
maximum = Math.max(maximum, active);
await Promise.resolve();
active -= 1;
return { kind: "complete", observedValues: 0 };
}),
};
await new SensitivityClassifier(values).assess(targets, new AbortController().signal);
expect(maximum).toBe(2);
expect(values.scanTable).toHaveBeenCalledTimes(3);
});
test("aborts a peer table scan when another concurrent source scan fails", async () => {
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
const first = column({ name: "status" });
const second = column({
id: "88888888-8888-4888-8888-888888888888",
tableId: otherTable.id,
name: "comment",
});
let peerSignal: AbortSignal | undefined;
const failure = new CatalogConnectorError("source unavailable");
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, _consume, scanSignal) => {
if (request.table.id === table.id) {
await Promise.resolve();
throw failure;
}
peerSignal = scanSignal;
return await new Promise((_resolve, reject) => {
scanSignal.addEventListener("abort", () => reject(scanSignal.reason), { once: true });
});
}),
};
await expect(new SensitivityClassifier(values).assess([
{ database, table, columns: [first] },
{ database, table: otherTable, columns: [second] },
], new AbortController().signal)).rejects.toBe(failure);
expect(peerSignal?.aborted).toBe(true);
});
test("stops sampling a column as soon as one value is sensitive", async () => {
const target = column();
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume([{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 }]);
return { kind: "sampled", observedValues: 1 };
}),
};
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(values.scanTable).toHaveBeenCalledOnce();
expect(assessment).toMatchObject({ assessment: "sensitive", proposedSensitive: true });
});
test("stops non-text columns after the 1,000-value stage", async () => {
const target = column({ dataType: "integer", name: "sequence_number" });
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume([{ columnId: target.id, value: "42", characterLength: 2 }]);
return { kind: "sampled", observedValues: 1 };
}),
};
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(vi.mocked(values.scanTable).mock.calls.map(([request]) => request.valuesPerColumn))
.toEqual([300, 700]);
expect(assessment).toMatchObject({
assessment: "non_sensitive",
proposedSensitive: false,
coverage: "sampled",
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_1000" }],
});
});
test("complete coverage classifies benign and empty columns as non-sensitive", async () => {
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
const humanProtected = column({
@@ -126,7 +270,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
{ columnId: empty.id, value: null, characterLength: null },
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const assessments = await new SensitivityClassifier(values).assessTable(
@@ -138,7 +282,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
expect.objectContaining({
columnId: empty.id,
assessment: "unknown",
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
}),
@@ -150,11 +294,11 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
]);
});
test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
test("sampled coverage without a match proposes non-sensitive independently of the current flag", async () => {
const target = column({ sensitive: true });
const values = source({
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -163,13 +307,13 @@ test("sampled coverage without a match is unknown and preserves the current huma
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: true,
evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
test("an unavailable source fails the analysis instead of producing unknown decisions", async () => {
const unresolved = column();
const metadataMatch = column({
id: "44444444-4444-4444-8444-444444444444",
@@ -181,30 +325,17 @@ test("an unavailable source produces sanitized unknown evidence without losing m
}),
};
const assessments = await new SensitivityClassifier(values).assessTable(
await expect(new SensitivityClassifier(values).assessTable(
{ database, table, columns: [unresolved, metadataMatch] },
new AbortController().signal,
);
expect(assessments).toEqual([
expect.objectContaining({
columnId: unresolved.id,
assessment: "unknown",
evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
}),
expect.objectContaining({
columnId: metadataMatch.id,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
}),
]);
)).rejects.toBeInstanceOf(CatalogConnectorError);
});
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
const target = column({ name: "codice_fiscale" });
const values = source({
batches: [],
coverage: { kind: "complete", observedRows: 0 },
coverage: { kind: "complete", observedValues: 0 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -245,7 +376,7 @@ test.each([
const target = column();
const values = source({
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -267,13 +398,16 @@ test("does not make a malformed email decisive", async () => {
value: "contatto a@b..com non valido",
characterLength: 28,
}]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
expect(assessment).toMatchObject({
assessment: "non_sensitive",
evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
});
});
test("finds a valid email after a malformed candidate in the same value", async () => {
@@ -284,7 +418,7 @@ test("finds a valid email after a malformed candidate in the same value", async
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
characterLength: 58,
}]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
@@ -300,7 +434,7 @@ test("optional local NER evidence can make otherwise ambiguous Italian text sens
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
@@ -331,14 +465,17 @@ test("does not wait for an optional NER worker that is still warming", async ()
const [assessment] = await new SensitivityClassifier(source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
}), detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).not.toHaveBeenCalled();
expect(assessment).toMatchObject({ assessment: "unknown" });
expect(assessment).toMatchObject({
assessment: "non_sensitive",
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
@@ -359,7 +496,7 @@ test("bounds each optional NER request when an installation raises the per-table
await new SensitivityClassifier(source({
batches: [observations],
coverage: { kind: "complete", observedRows: 8 },
coverage: { kind: "complete", observedValues: 8 },
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
{ database, table, columns },
new AbortController().signal,
@@ -386,7 +523,7 @@ test("limits default NER work to two candidates spread across a wide table", asy
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
})))],
coverage: { kind: "complete", observedRows: 2 },
coverage: { kind: "complete", observedValues: 2 },
}), detector).assessTable(
{ database, table, columns },
new AbortController().signal,
@@ -402,7 +539,7 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => {
@@ -430,11 +567,11 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
expect(nerBudget.remainingMs).toBe(0);
});
test("uninterpretable binary content remains unknown after complete coverage", async () => {
test("uninterpretable binary content is protected conservatively without scanning", async () => {
const target = column({ dataType: "bytea" });
const values = source({
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -443,8 +580,9 @@ test("uninterpretable binary content remains unknown after complete coverage", a
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "type", ruleId: "type.binary_uninspectable" }],
});
expect(values.scanTable).not.toHaveBeenCalled();
});