feat: sample sensitive columns progressively

This commit is contained in:
Codex
2026-09-03 10:25:05 +02:00
parent f114d0065a
commit 8e778b9edb
24 changed files with 1001 additions and 581 deletions
+13 -12
View File
@@ -98,16 +98,17 @@ column. The KPI strip reads installation-wide or selected-database aggregates fr
description history, and sensitive-field review/history use the production APIs in right-side
drawers rather than prototype fixtures; closing a history drawer does not stop its background run.
Sensitive-field review is now driven by the versioned local `sensitivity-v1` policy, not by a
Sensitive-field review is now driven by the versioned local `sensitivity-v2` policy, not by a
catalog model. The backend reads selected source tables through read-only, database-specific
adapters and makes every `sensitive | non_sensitive | unknown` decision in the TypeScript
`SensitivityClassifier`. A single validated match protects the column; a full scan is limited to
five seconds per table before sampling and the whole request to sixty seconds. Draft assessments
remain transient until an administrator explicitly saves them. Optional GLiNER2 evidence is
CPU-only, offline, opt-in, and never replaces the deterministic decision point; see
`docs/operations/sensitivity-analysis.md`. The aggregate PSD shadow comparison kept NER disabled by
default because its extra findings did not offset the coverage lost to inference within the global
deadline; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`.
adapters and makes every `sensitive | non_sensitive` draft decision in the TypeScript
`SensitivityClassifier`. A single validated match protects the column. Tables up to 1,000 rows are
fully scanned; larger tables use breadth-first 300, 1,000, and text-only 3,000-value targets, with a
five-second limit per source query and no global request deadline. Source failures fail the run
instead of yielding `unknown`; coverage remains visible separately from the proposal. Draft
assessments remain transient until an administrator explicitly saves them. Optional GLiNER2
evidence is CPU-only, offline, opt-in, and never replaces the deterministic decision point; see
`docs/operations/sensitivity-analysis.md`. The earlier v1 PSD shadow comparison kept NER disabled by
default; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`. A v2 PSD benchmark is still due.
Physical membership, source
comments, column types/default/nullability/PK positions, and constraint-level ordered FK pairs are
@@ -169,9 +170,9 @@ AI Description Generation uses the catalog's human-owned Sensitive Data Flag. Th
`false`, including for newly synchronized columns. An administrator may request a local sensitivity
analysis for one selected database, selected tables, or selected columns. One deterministic
TypeScript classifier combines metadata, bounded source-content rules, and optional CPU-only NER;
no generative model decides the result. Its `sensitive`, `non_sensitive`, or `unknown` assessments
remain an unsaved draft until the human reviews and saves any chosen flag changes, including a
downgrade to non-sensitive.
no generative model decides the result. Its `sensitive` or `non_sensitive` assessments remain an
unsaved draft until the human reviews and saves any chosen flag changes, including a downgrade to
non-sensitive. Coverage is reported separately; interrupted history may count unprocessed columns.
Each started analysis records a separate Sensitivity Analysis Run with aggregate counters and safe
ordered events. This operational history never stores per-column assessments, source values,
matched spans, prompts, or free-form diagnostics; reloading still discards an unsaved review draft.
@@ -14,7 +14,7 @@ import type {
} from "./types.js";
const interruptedMessage = "Local sensitivity analysis was interrupted by backend restart.";
const deadlineMessage = "Local sensitivity analysis reached its time limit.";
const interruptedDuringRunMessage = "Local sensitivity analysis was interrupted before completion.";
const failedMessage = "Local sensitivity analysis failed.";
function ensureActive(signal: AbortSignal): void {
@@ -98,14 +98,12 @@ export class SensitivityAnalysisRunner {
const suggestedNonSensitive = batch.filter(
(suggestion) => suggestion.assessment === "non_sensitive",
).length;
const unknown = batch.filter((suggestion) => suggestion.assessment === "unknown").length;
const current = await this.repository.getSensitivityAnalysisRun(started.id);
ensureActive(signal);
if (!current) throw new Error("Sensitivity Analysis Run disappeared");
const progress = await this.repository.updateSensitivityAnalysisRun(started.id, {
suggestedSensitive: current.suggestedSensitive + suggestedSensitive,
suggestedNonSensitive: current.suggestedNonSensitive + suggestedNonSensitive,
unknown: current.unknown + unknown,
});
if (!progress) throw new Error("Sensitivity Analysis Run disappeared");
processedSensitive += suggestedSensitive;
@@ -126,7 +124,6 @@ export class SensitivityAnalysisRunner {
const suggestedNonSensitive = suggestions.filter(
(suggestion) => suggestion.assessment === "non_sensitive",
).length;
const unknown = suggestions.filter((suggestion) => suggestion.assessment === "unknown").length;
await this.repository.appendSensitivityAnalysisEvent(
started.id,
"info",
@@ -140,7 +137,7 @@ export class SensitivityAnalysisRunner {
total: suggestions.length,
suggestedSensitive,
suggestedNonSensitive,
unknown,
unknown: 0,
finishedAt: new Date().toISOString(),
errorSummary: null,
});
@@ -149,7 +146,7 @@ export class SensitivityAnalysisRunner {
return { suggestions, run: completed };
} catch (error) {
const interrupted = signal.aborted || error instanceof SensitivityAnalysisInterruptedError;
const message = interrupted ? deadlineMessage : failedMessage;
const message = interrupted ? interruptedDuringRunMessage : failedMessage;
await this.repository.updateSensitivityAnalysisRun(started.id, {
status: interrupted ? "interrupted" : "failed",
...(interrupted ? {
@@ -12,7 +12,7 @@ import type {
} from "./types.js";
export type { SensitivityAnalysisScope } from "./types.js";
export const SENSITIVITY_POLICY_VERSION = "sensitivity-v1";
export const SENSITIVITY_POLICY_VERSION = "sensitivity-v2";
interface SelectedColumn {
table: CatalogTable;
@@ -30,6 +30,7 @@ export interface SensitivityReviewItem {
assessment: SensitivityColumnAssessment["assessment"];
evidence: readonly SensitivityEvidence[];
observedValues: number;
coverage: SensitivityColumnAssessment["coverage"];
}
export class SensitivityAnalysisTargetNotFoundError extends Error {
@@ -48,7 +49,7 @@ export class SensitivityAnalysisDuplicateTargetIdsError extends Error {
export class SensitivityAnalysisInterruptedError extends Error {
constructor() {
super("sensitivity analysis deadline exceeded");
super("sensitivity analysis interrupted");
this.name = "SensitivityAnalysisInterruptedError";
}
}
@@ -69,7 +70,7 @@ export class SensitivityAnalysisService {
constructor(
private readonly repository: CatalogRepository,
private readonly classifier: SensitivityClassifier,
private readonly options: { runBudgetMs?: number; nerBudgetMs?: number; now?: () => number } = {},
private readonly options: { nerBudgetMs?: number } = {},
) {}
private async selectColumns(
@@ -116,8 +117,6 @@ export class SensitivityAnalysisService {
onPrepared?: (total: number) => void | Promise<void>,
onProgress?: (processed: number, suggestions: readonly SensitivityReviewItem[]) => void | Promise<void>,
): Promise<readonly SensitivityReviewItem[]> {
const now = this.options.now ?? Date.now;
const deadline = now() + (this.options.runBudgetMs ?? 60_000);
const configuredNerBudget = this.options.nerBudgetMs ?? 10_000;
const nerBudget: SensitivityNerBudget = {
remainingMs: Number.isFinite(configuredNerBudget) && configuredNerBudget >= 0
@@ -137,20 +136,23 @@ export class SensitivityAnalysisService {
items.push(item);
byTable.set(item.table.id, items);
}
const suggestions: SensitivityReviewItem[] = [];
for (const items of byTable.values()) {
ensureActive(signal);
const tableTargets = [...byTable.values()].map((items) => {
const first = items[0]!;
const assessments = await this.classifier.assessTable({
return {
database,
table: first.table,
columns: items.map(({ column }) => column),
}, signal, deadline, nerBudget);
};
});
const assessments = await this.classifier.assess(tableTargets, signal, nerBudget);
ensureActive(signal);
const assessmentById = new Map(assessments.map((assessment) => [
assessment.columnId,
assessment,
]));
const suggestions: SensitivityReviewItem[] = [];
for (const items of byTable.values()) {
ensureActive(signal);
const batch = items.map(({ table, column }) => {
const assessment = assessmentById.get(column.id)!;
return {
@@ -164,6 +166,7 @@ export class SensitivityAnalysisService {
assessment: assessment.assessment,
evidence: assessment.evidence,
observedValues: assessment.observedValues,
coverage: assessment.coverage,
};
});
suggestions.push(...batch);
+158 -101
View File
@@ -1,11 +1,11 @@
import { CatalogConnectorError, type CatalogColumn, type CatalogTable, type WorkspaceDatabase } from "./types.js";
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "./types.js";
import { findPhoneNumbersInText } from "libphonenumber-js/max";
import validator from "validator";
export type SensitivityAssessment = "sensitive" | "non_sensitive" | "unknown";
export type SensitivityAssessment = "sensitive" | "non_sensitive";
export interface SensitivityEvidence {
kind: "metadata" | "content" | "length" | "ner" | "coverage";
kind: "metadata" | "content" | "length" | "ner" | "coverage" | "type";
ruleId: string;
label?: string;
confidence?: number;
@@ -18,8 +18,8 @@ export interface SensitivityValueObservation {
}
export interface SensitivityScanCoverage {
kind: "complete" | "sampled" | "unavailable";
observedRows: number;
kind: "complete" | "sampled";
observedValues: number;
}
export interface SensitivityTableScan {
@@ -31,8 +31,11 @@ export interface SensitivityScanRequest {
database: WorkspaceDatabase;
table: CatalogTable;
columns: readonly CatalogColumn[];
fullScanBudgetMs: number;
deadline: number;
valuesPerColumn: number;
sampleOffset: number;
sampleSeed: number;
queryTimeoutMs: number;
fullScanThreshold?: number;
}
export interface SensitivityValueSource {
@@ -76,6 +79,7 @@ export interface SensitivityColumnAssessment {
proposedSensitive: boolean;
evidence: readonly SensitivityEvidence[];
observedValues: number;
coverage: "metadata" | "complete" | "sampled" | "no_values";
}
export interface SensitivityTableTarget {
@@ -94,7 +98,15 @@ const CREDENTIAL_NAME = /(?:^|_)(?:api_key|credential|password|passwd|private_ke
const HEALTH_NAME = /(?:^|_)(?:anamnesi|clinical|diagnos(?:i|is)|health|medical|patient|patologia|therapy|terapia)(?:_|$)/u;
const CLINICAL_TERM = /(?:^|[^\p{L}])(?:allergi[ae]|anamnesi|carcinoma|chemioterapia|diabete|diagnos[ei]|epatite|farmac[io]|gravidanza|hiv|metastasi|neoplasia|patologia|radioterapia|referto|terapia|tumore)(?:$|[^\p{L}])/iu;
const UNSUPPORTED_BINARY_TYPE = /(?:^|\s)(?:binary|blob|bytea|image|varbinary)(?:\s|$|\()/iu;
const DEEP_TEXT_TYPE = /(?:^|\s)(?:char|character|citext|clob|json|jsonb|nchar|nvarchar|string|text|varchar|xml)(?:\s|$|\()/iu;
const MAX_NER_CANDIDATES_PER_REQUEST = 128;
const MAX_CONCURRENT_TABLE_SCANS = 2;
export const SENSITIVITY_SAMPLE_PHASES = [
{ targetValuesPerColumn: 300, additionalValuesPerColumn: 300, sampleSeed: 37, deepTextOnly: false },
{ targetValuesPerColumn: 1_000, additionalValuesPerColumn: 700, sampleSeed: 73, deepTextOnly: false },
{ targetValuesPerColumn: 3_000, additionalValuesPerColumn: 2_000, sampleSeed: 109, deepTextOnly: true },
] as const;
function normalizedName(value: string): string {
return value.normalize("NFKD")
@@ -284,14 +296,22 @@ function contentEvidence(value: string): SensitivityEvidence | undefined {
return undefined;
}
interface ColumnState {
column: CatalogColumn;
evidence: SensitivityEvidence[];
observedValues: number;
nerCandidates: string[];
coverage: "metadata" | "complete" | "sampled" | "no_values";
sampledTarget: number;
}
/** Sole decision module for local column-level sensitivity assessments. */
export class SensitivityClassifier {
constructor(
private readonly values: SensitivityValueSource,
private readonly detector?: LocalNerDetector,
private readonly options: {
fullScanBudgetMs?: number;
runBudgetMs?: number;
queryTimeoutMs?: number;
nerConfidenceThreshold?: number;
maxNerValuesPerColumn?: number;
maxNerCandidatesPerTable?: number;
@@ -299,95 +319,131 @@ export class SensitivityClassifier {
} = {},
) {}
async assessTable(
target: SensitivityTableTarget,
async assess(
targets: readonly SensitivityTableTarget[],
signal: AbortSignal,
runDeadline?: number,
nerBudget?: SensitivityNerBudget,
sharedNerBudget?: SensitivityNerBudget,
): Promise<readonly SensitivityColumnAssessment[]> {
const now = this.options.now ?? Date.now;
const deadline = runDeadline ?? now() + (this.options.runBudgetMs ?? 60_000);
const evidence = new Map(target.columns.map((column) => {
const match = metadataEvidence(column);
return [column.id, match ? [match] : [] as SensitivityEvidence[]];
}));
const observed = new Map(target.columns.map((column) => [column.id, 0]));
const nerCandidates = new Map(target.columns.map((column) => [column.id, [] as string[]]));
const maxNerValuesPerColumn = boundedCount(this.options.maxNerValuesPerColumn, 8, 8);
const unsupported = new Set(target.columns
.filter((column) => UNSUPPORTED_BINARY_TYPE.test(column.dataType))
.map((column) => column.id));
const scannableColumns = target.columns.filter((column) => (
!unsupported.has(column.id) && evidence.get(column.id)!.length === 0
));
let coverage: SensitivityScanCoverage = { kind: "unavailable", observedRows: 0 };
if (scannableColumns.length > 0 && now() < deadline) {
try {
coverage = await this.values.scanTable({
...target,
columns: scannableColumns,
fullScanBudgetMs: this.options.fullScanBudgetMs ?? 5_000,
deadline,
}, (batch) => {
for (const item of batch) {
if (!evidence.has(item.columnId) || item.value === null) continue;
observed.set(item.columnId, (observed.get(item.columnId) ?? 0) + 1);
const matches = evidence.get(item.columnId)!;
if (matches.length === 0 && (item.characterLength ?? item.value.length) > 500) {
matches.push({ kind: "length", ruleId: "text.over_500_characters" });
} else if (matches.length === 0) {
const match = contentEvidence(item.value);
if (match) matches.push(match);
else {
const candidates = nerCandidates.get(item.columnId)!;
if (candidates.length < maxNerValuesPerColumn && !candidates.includes(item.value)) {
candidates.push(item.value);
}
}
}
}
}, signal);
} catch (error) {
if (!(error instanceof CatalogConnectorError)) throw error;
const states = new Map<string, ColumnState>();
for (const target of targets) {
for (const column of target.columns) {
const metadataMatch = metadataEvidence(column);
const binary = UNSUPPORTED_BINARY_TYPE.test(column.dataType);
states.set(column.id, {
column,
evidence: metadataMatch
? [metadataMatch]
: binary
? [{ kind: "type", ruleId: "type.binary_uninspectable" }]
: [],
observedValues: 0,
nerCandidates: [],
coverage: metadataMatch || binary ? "metadata" : "no_values",
sampledTarget: 0,
});
}
}
const completeTables = new Set<string>();
for (const [phaseIndex, phase] of SENSITIVITY_SAMPLE_PHASES.entries()) {
for (let offset = 0; offset < targets.length; offset += MAX_CONCURRENT_TABLE_SCANS) {
signal.throwIfAborted();
const peerController = new AbortController();
const scanSignal = AbortSignal.any([signal, peerController.signal]);
try {
await Promise.all(targets.slice(offset, offset + MAX_CONCURRENT_TABLE_SCANS).map(async (target) => {
if (completeTables.has(target.table.id)) return;
const columns = target.columns.filter((column) => {
const state = states.get(column.id)!;
return state.evidence.length === 0
&& (!phase.deepTextOnly || DEEP_TEXT_TYPE.test(column.dataType));
});
if (columns.length === 0) return;
const coverage = await this.values.scanTable({
...target,
columns,
valuesPerColumn: phase.additionalValuesPerColumn,
sampleOffset: phase.targetValuesPerColumn - phase.additionalValuesPerColumn,
sampleSeed: phase.sampleSeed,
queryTimeoutMs: this.options.queryTimeoutMs ?? 5_000,
...(phaseIndex === 0 ? { fullScanThreshold: 1_000 } : {}),
}, (batch) => {
for (const item of batch) {
if (item.value === null) continue;
const state = states.get(item.columnId);
if (!state || state.evidence.length > 0) continue;
state.observedValues += 1;
if ((item.characterLength ?? item.value.length) > 500) {
state.evidence.push({ kind: "length", ruleId: "text.over_500_characters" });
continue;
}
const match = contentEvidence(item.value);
if (match) {
state.evidence.push(match);
continue;
}
if (state.nerCandidates.length < maxNerValuesPerColumn
&& !state.nerCandidates.includes(item.value)) {
state.nerCandidates.push(item.value);
}
}
}, scanSignal);
for (const column of columns) {
const state = states.get(column.id)!;
state.sampledTarget = Math.max(state.sampledTarget, phase.targetValuesPerColumn);
state.coverage = coverage.kind === "complete"
? "complete"
: state.observedValues === 0 ? "no_values" : "sampled";
}
if (coverage.kind === "complete") completeTables.add(target.table.id);
}));
} catch (error) {
peerController.abort(error);
throw error;
}
}
}
const nerBudget = sharedNerBudget ?? { remainingMs: 10_000 };
if (this.detector && (this.detector.isReady?.() ?? true) && !signal.aborted
&& now() < deadline && (nerBudget?.remainingMs ?? 1) > 0) {
const candidates: LocalNerCandidate[] = [];
&& nerBudget.remainingMs > 0) {
const maxCandidates = boundedCount(this.options.maxNerCandidatesPerTable, 2, 1_024);
const threshold = this.options.nerConfidenceThreshold ?? 0.8;
for (const target of targets) {
signal.throwIfAborted();
if (nerBudget.remainingMs <= 0) break;
const candidates: LocalNerCandidate[] = [];
candidateSelection: for (let valueIndex = 0; valueIndex < maxNerValuesPerColumn; valueIndex += 1) {
for (const column of target.columns) {
if (evidence.get(column.id)!.length > 0) continue;
const text = nerCandidates.get(column.id)![valueIndex];
const state = states.get(column.id)!;
if (state.evidence.length > 0) continue;
const text = state.nerCandidates[valueIndex];
if (text === undefined) continue;
candidates.push({ columnId: column.id, text });
if (candidates.length >= maxCandidates) break candidateSelection;
}
}
if (candidates.length > 0) {
const threshold = this.options.nerConfidenceThreshold ?? 0.8;
const nerStartedAt = now();
const allowedNerMs = nerBudget
? Math.max(0, nerBudget.remainingMs)
: Math.max(0, deadline - nerStartedAt);
const nerDeadline = Math.min(deadline, nerStartedAt + allowedNerMs);
if (candidates.length === 0) continue;
const startedAt = now();
const deadline = startedAt + nerBudget.remainingMs;
try {
for (let offset = 0; offset < candidates.length; offset += MAX_NER_CANDIDATES_PER_REQUEST) {
if (signal.aborted || now() >= nerDeadline) break;
if (signal.aborted || now() >= deadline) break;
try {
const detected = await this.detector.detect(
candidates.slice(offset, offset + MAX_NER_CANDIDATES_PER_REQUEST),
signal,
nerDeadline,
deadline,
);
for (const item of detected) {
const matches = evidence.get(item.columnId);
if (!matches || matches.length > 0 || !Number.isFinite(item.confidence)
const state = states.get(item.columnId);
if (!state || state.evidence.length > 0 || !Number.isFinite(item.confidence)
|| item.confidence < threshold || item.confidence > 1) continue;
const label = normalizedName(item.label).slice(0, 80);
if (!label) continue;
matches.push({
state.evidence.push({
kind: "ner",
ruleId: "ner.entity",
label,
@@ -400,42 +456,43 @@ export class SensitivityClassifier {
}
}
} finally {
if (nerBudget) {
const elapsedMs = Math.max(1, now() - nerStartedAt);
nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - elapsedMs);
}
nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - Math.max(1, now() - startedAt));
}
}
}
return target.columns.map((column) => {
const matches = evidence.get(column.id)!;
const count = observed.get(column.id) ?? 0;
const assessment: SensitivityAssessment = matches.length > 0
? "sensitive"
: unsupported.has(column.id) || count === 0 || coverage.kind !== "complete"
? "unknown"
: "non_sensitive";
return targets.flatMap((target) => target.columns.map((column) => {
const state = states.get(column.id)!;
const sensitive = state.evidence.length > 0;
const coverage = state.observedValues === 0 && !sensitive ? "no_values" : state.coverage;
const coverageEvidence: SensitivityEvidence[] = sensitive
? state.evidence
: [{
kind: "coverage",
ruleId: coverage === "complete"
? "coverage.complete"
: coverage === "no_values"
? "coverage.no_values"
: `coverage.sampled_${state.sampledTarget}`,
}];
return {
columnId: column.id,
assessment,
proposedSensitive: assessment === "unknown" ? column.sensitive : assessment === "sensitive",
evidence: matches.length > 0
? matches
: assessment === "unknown"
? [{
kind: "coverage",
ruleId: unsupported.has(column.id)
? "coverage.unsupported_type"
: coverage.kind === "unavailable"
? "coverage.unavailable"
: count === 0
? "coverage.no_values"
: "coverage.incomplete",
}]
: [],
observedValues: count,
assessment: sensitive ? "sensitive" : "non_sensitive",
proposedSensitive: sensitive,
evidence: coverageEvidence,
observedValues: state.observedValues,
coverage,
};
});
}));
}
/** Convenience for focused callers and rule-level tests. Production orchestration uses assess(). */
async assessTable(
target: SensitivityTableTarget,
signal: AbortSignal,
_retiredRunDeadline?: number,
nerBudget?: SensitivityNerBudget,
): Promise<readonly SensitivityColumnAssessment[]> {
return await this.assess([target], signal, nerBudget);
}
}
+12 -6
View File
@@ -5,7 +5,10 @@ import { WorkspaceSecretStore } from "../workspaces/secret-store.js";
import { PythonLocalNerDetector } from "./local-ner-detector.js";
import { ConcreteCatalogPostgresAccess } from "./postgres-access.js";
import { createCatalogRepository } from "./repository.js";
import { SensitivityAnalysisService } from "./sensitivity-analysis-service.js";
import {
SENSITIVITY_POLICY_VERSION,
SensitivityAnalysisService,
} from "./sensitivity-analysis-service.js";
import { SensitivityClassifier } from "./sensitivity-classifier.js";
import { ConcreteSensitivityValueSource } from "./sensitivity-value-source.js";
@@ -60,23 +63,26 @@ async function main(): Promise<void> {
const suggestions = await new SensitivityAnalysisService(
repository,
new SensitivityClassifier(source, detector),
).analyze(database.id, "all", [], AbortSignal.timeout(65_000));
const assessments = { sensitive: 0, nonSensitive: 0, unknown: 0 };
).analyze(database.id, "all", [], new AbortController().signal);
const assessments = { sensitive: 0, nonSensitive: 0 };
const coverage = { metadata: 0, complete: 0, sampled: 0, noValues: 0 };
const rules = new Map<string, number>();
for (const suggestion of suggestions) {
if (suggestion.assessment === "sensitive") assessments.sensitive += 1;
else if (suggestion.assessment === "non_sensitive") assessments.nonSensitive += 1;
else assessments.unknown += 1;
else assessments.nonSensitive += 1;
if (suggestion.coverage === "no_values") coverage.noValues += 1;
else coverage[suggestion.coverage] += 1;
for (const evidence of suggestion.evidence) {
rules.set(evidence.ruleId, (rules.get(evidence.ruleId) ?? 0) + 1);
}
}
process.stdout.write(`${JSON.stringify({
ok: true,
policyVersion: "sensitivity-v1",
policyVersion: SENSITIVITY_POLICY_VERSION,
nerEnabled: detector !== undefined,
total: suggestions.length,
assessments,
coverage,
rules: Object.fromEntries([...rules].sort(([left], [right]) => left.localeCompare(right))),
elapsedMs: Date.now() - startedAt,
})}\n`);
+159 -127
View File
@@ -8,82 +8,117 @@ import type {
SensitivityValueObservation,
SensitivityValueSource,
} from "./sensitivity-classifier.js";
import { CatalogConnectorError } from "./types.js";
import { CatalogConnectorError, type CatalogColumn } from "./types.js";
const MAX_VALUE_CHARACTERS = 501;
const DEFAULT_BATCH_ROWS = 200;
const DEFAULT_SAMPLE_ROWS = 200;
const MAX_COLUMNS_PER_QUERY = 25;
const SAMPLE_OVERSCAN_FACTOR = 10;
function quoteIdentifier(identifier: string): string {
return `"${identifier.replaceAll('"', '""')}"`;
}
function projections(request: SensitivityScanRequest): string {
return request.columns.flatMap((column, index) => {
function chunks<T>(items: readonly T[], size: number): T[][] {
const result: T[][] = [];
for (let offset = 0; offset < items.length; offset += size) {
result.push(items.slice(offset, offset + size));
}
return result;
}
function tableReference(request: SensitivityScanRequest): string {
return `${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`;
}
function samplePercentage(valuesPerColumn: number): number {
if (valuesPerColumn <= 300) return 30;
if (valuesPerColumn <= 700) return 70;
return 100;
}
function flatValueQuery(
request: SensitivityScanRequest,
columns: readonly CatalogColumn[],
options: { complete: boolean; randomized: boolean },
): string {
const projections = columns.map((column) => quoteIdentifier(column.name)).join(", ");
const perColumnLimit = options.complete
? request.fullScanThreshold ?? request.valuesPerColumn
: request.valuesPerColumn;
const rowLimit = Math.max(perColumnLimit, perColumnLimit * SAMPLE_OVERSCAN_FACTOR);
const sample = options.complete
? `SELECT ${projections} FROM ${tableReference(request)}`
: [
`SELECT ${projections} FROM ${tableReference(request)}`,
...(options.randomized
? [`TABLESAMPLE SYSTEM (${samplePercentage(request.valuesPerColumn)}) REPEATABLE (${request.sampleSeed})`]
: []),
`LIMIT ${rowLimit} OFFSET ${request.sampleOffset}`,
].join(" ");
const values = columns.map((column, index) => {
const identifier = quoteIdentifier(column.name);
return [
`LEFT((${identifier})::text, ${MAX_VALUE_CHARACTERS}) AS "__value_${index}"`,
`CASE WHEN ${identifier} IS NULL THEN NULL ELSE char_length((${identifier})::text) END AS "__length_${index}"`,
];
`(${index}, LEFT((sampled.${identifier})::text, ${MAX_VALUE_CHARACTERS}),`,
`CASE WHEN sampled.${identifier} IS NULL THEN NULL`,
`ELSE char_length((sampled.${identifier})::text) END)`,
].join(" ");
}).join(", ");
return [
`WITH sampled AS MATERIALIZED (${sample}),`,
"ranked AS (",
"SELECT value.__column_index, value.__value, value.__length,",
"row_number() OVER (PARTITION BY value.__column_index) AS __rank",
"FROM sampled",
`CROSS JOIN LATERAL (VALUES ${values}) AS value(__column_index, __value, __length)`,
"WHERE value.__value IS NOT NULL",
")",
"SELECT __column_index, __value, __length FROM ranked",
`WHERE __rank <= ${perColumnLimit}`,
].join(" ");
}
function observations(
request: SensitivityScanRequest,
columns: readonly CatalogColumn[],
rows: readonly Record<string, unknown>[],
): SensitivityValueObservation[] {
return rows.flatMap((row) => request.columns.map((column, index) => {
const sourceValue = row[`__value_${index}`];
const sourceLength = row[`__length_${index}`];
const value = sourceValue === null || sourceValue === undefined ? null : String(sourceValue);
const parsedLength = sourceLength === null || sourceLength === undefined
return rows.flatMap((row) => {
const index = Number(row.__column_index);
const column = Number.isSafeInteger(index) && index >= 0 ? columns[index] : undefined;
if (!column || row.__value === null || row.__value === undefined) return [];
const value = String(row.__value);
const parsedLength = row.__length === null || row.__length === undefined
? null
: Number(sourceLength);
return {
: Number(row.__length);
return [{
columnId: column.id,
value,
characterLength: parsedLength !== null && Number.isSafeInteger(parsedLength) && parsedLength >= 0
? parsedLength
: value?.length ?? null,
};
}));
: value.length,
}];
});
}
function cancelled(error: unknown): boolean {
return Boolean(error && typeof error === "object" && "code" in error && error.code === "57014");
}
interface SensitivityValueSourceOptions {
now?: () => number;
batchRows?: number;
sampleRows?: number;
}
/**
* PostgreSQL value adapter. It owns bounded read mechanics and emits normalized values, never a
* sensitivity decision.
* Database-specific sampling adapter. Policy stays in SensitivityClassifier; this module only
* produces bounded, normalized non-null observations without persisting or logging values.
*/
export class ConcreteSensitivityValueSource implements SensitivityValueSource {
private readonly now: () => number;
private readonly batchRows: number;
private readonly sampleRows: number;
constructor(
private readonly access: CatalogPostgresAccess,
private readonly secretStore?: Pick<WorkspaceSecretStore, "materialize">,
options: SensitivityValueSourceOptions = {},
) {
this.now = options.now ?? Date.now;
this.batchRows = options.batchRows ?? DEFAULT_BATCH_ROWS;
this.sampleRows = options.sampleRows ?? DEFAULT_SAMPLE_ROWS;
}
) {}
async scanTable(
request: SensitivityScanRequest,
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
signal: AbortSignal,
): Promise<SensitivityScanCoverage> {
if (request.columns.length === 0) return { kind: "unavailable", observedRows: 0 };
if (request.columns.length === 0) return { kind: "complete", observedValues: 0 };
if (request.database.binding.transport === "rest_api") {
return await this.scanRest(request, consume, signal);
}
@@ -97,69 +132,63 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
): Promise<SensitivityScanCoverage> {
const client = await this.access.connect(request.database, signal);
let transactionOpen = false;
const startedAt = this.now();
const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
let observedRows = 0;
let cursorOpen = false;
let savepointSequence = 0;
let observedValues = 0;
try {
if (signal.aborted || this.now() >= request.deadline) {
return { kind: "sampled", observedRows: 0 };
}
signal.throwIfAborted();
await client.query("BEGIN TRANSACTION READ ONLY", []);
transactionOpen = true;
await client.query("SELECT set_config('statement_timeout', $1, true)", [
`${Math.max(1, Math.floor(fullDeadline - startedAt))}ms`,
`${Math.max(1, Math.floor(request.queryTimeoutMs))}ms`,
]);
await client.query("SAVEPOINT sensitivity_full_scan", []);
const cursor = [
"DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR",
`SELECT ${projections(request)}`,
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
].join(" ");
await client.query(cursor, []);
cursorOpen = true;
while (!signal.aborted && this.now() < fullDeadline) {
let rows: Array<Record<string, unknown>>;
const boundedQuery = async (sql: string): Promise<Array<Record<string, unknown>> | undefined> => {
signal.throwIfAborted();
savepointSequence += 1;
const savepoint = `sensitivity_scan_${savepointSequence}`;
await client.query(`SAVEPOINT ${savepoint}`, []);
try {
await client.query("SELECT set_config('statement_timeout', $1, true)", [
`${Math.max(1, Math.floor(fullDeadline - this.now()))}ms`,
]);
rows = (await client.query(
`FETCH FORWARD ${this.batchRows} FROM sensitivity_full_scan_cursor`,
[],
)).rows;
return (await client.query(sql, [])).rows;
} catch (error) {
if (!cancelled(error)) throw error;
await client.query("ROLLBACK TO SAVEPOINT sensitivity_full_scan", []);
cursorOpen = false;
break;
await client.query(`ROLLBACK TO SAVEPOINT ${savepoint}`, []);
return undefined;
} finally {
await client.query(`RELEASE SAVEPOINT ${savepoint}`, []).catch(() => undefined);
}
if (rows.length > 0) {
observedRows += rows.length;
await consume(observations(request, rows));
};
let complete = false;
if (request.fullScanThreshold !== undefined) {
const probe = await boundedQuery(
`SELECT 1 AS __present FROM ${tableReference(request)} LIMIT ${request.fullScanThreshold + 1}`,
);
complete = probe !== undefined && probe.length <= request.fullScanThreshold;
}
if (rows.length < this.batchRows) {
return { kind: "complete", observedRows };
for (const columnChunk of chunks(request.columns, MAX_COLUMNS_PER_QUERY)) {
signal.throwIfAborted();
let rows = await boundedQuery(flatValueQuery(request, columnChunk, {
complete,
randomized: !complete,
}));
if (rows === undefined && complete) {
complete = false;
rows = await boundedQuery(flatValueQuery(request, columnChunk, {
complete: false,
randomized: true,
}));
}
if (!complete && (rows === undefined || rows.length === 0)) {
rows = await boundedQuery(flatValueQuery(request, columnChunk, {
complete: false,
randomized: false,
}));
}
if (signal.aborted || this.now() >= request.deadline) {
return { kind: "sampled", observedRows };
if (rows === undefined) throw new CatalogConnectorError("Sensitivity sample query timed out");
const batch = observations(columnChunk, rows);
observedValues += batch.length;
if (batch.length > 0) await consume(batch);
}
if (cursorOpen) await client.query("CLOSE sensitivity_full_scan_cursor", []);
await client.query("RELEASE SAVEPOINT sensitivity_full_scan", []);
await client.query("SELECT set_config('statement_timeout', $1, true)", [
`${Math.max(1, Math.floor(request.deadline - this.now()))}ms`,
]);
const sampleSql = [
`SELECT ${projections(request)}`,
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
"TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
"LIMIT $1",
].join(" ");
const sampledRows = (await client.query(sampleSql, [this.sampleRows])).rows;
observedRows += sampledRows.length;
if (sampledRows.length > 0) await consume(observations(request, sampledRows));
return { kind: "sampled", observedRows };
return { kind: complete ? "complete" : "sampled", observedValues };
} catch (error) {
if (error instanceof CatalogConnectorError) throw error;
throw new CatalogConnectorError("Sensitivity source scan failed");
@@ -180,9 +209,7 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
request.database.workspaceId,
auth === "none" ? [] : [CATALOG_SECRET_IDS.apiKey],
);
const startedAt = this.now();
const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
let observedRows = 0;
let observedValues = 0;
try {
const headers: Record<string, string> = { "content-type": "application/json" };
if (auth !== "none") {
@@ -194,15 +221,14 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
}
const baseUrl = request.database.binding.baseUrl?.replace(/\/+$/u, "");
if (!baseUrl) throw new CatalogConnectorError("Database binding is incomplete");
const runQuery = async (sql: string, deadline: number): Promise<Array<Record<string, unknown>>> => {
const runQuery = async (sql: string): Promise<Array<Record<string, unknown>> | undefined> => {
const timeout = AbortSignal.timeout(Math.max(1, Math.floor(request.queryTimeoutMs)));
try {
const response = await fetch(`${baseUrl}/rpc/run_query`, {
method: "POST",
headers,
body: JSON.stringify({ query_text: sql }),
signal: AbortSignal.any([
signal,
AbortSignal.timeout(Math.max(1, Math.floor(deadline - this.now()))),
]),
signal: AbortSignal.any([signal, timeout]),
});
if (!response.ok) throw new CatalogConnectorError("REST sensitivity source scan failed");
const body: unknown = await response.json();
@@ -211,44 +237,50 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
throw new CatalogConnectorError("REST sensitivity source response is invalid");
}
return body as Array<Record<string, unknown>>;
} catch (error) {
if (signal.aborted) throw error;
if (timeout.aborted) return undefined;
throw error;
}
};
let offset = 0;
const baseSelect = [
`SELECT ${projections(request)}`,
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
].join(" ");
while (!signal.aborted) {
let rows: Array<Record<string, unknown>>;
try {
rows = await runQuery(
`${baseSelect} LIMIT ${this.batchRows} OFFSET ${offset}`,
fullDeadline,
let complete = false;
if (request.fullScanThreshold !== undefined) {
const probe = await runQuery(
`SELECT 1 AS __present FROM ${tableReference(request)} LIMIT ${request.fullScanThreshold + 1}`,
);
} catch (error) {
if (signal.aborted || this.now() < fullDeadline) throw error;
break;
complete = probe !== undefined && probe.length <= request.fullScanThreshold;
}
observedRows += rows.length;
if (rows.length > 0) await consume(observations(request, rows));
if (rows.length < this.batchRows) {
return { kind: offset === 0 ? "complete" : "sampled", observedRows };
let requestCount = request.fullScanThreshold === undefined ? 0 : 1;
for (const columnChunk of chunks(request.columns, MAX_COLUMNS_PER_QUERY)) {
signal.throwIfAborted();
let rows = await runQuery(flatValueQuery(request, columnChunk, {
complete,
randomized: !complete,
}));
requestCount += 1;
if (rows === undefined && complete) {
complete = false;
rows = await runQuery(flatValueQuery(request, columnChunk, {
complete: false,
randomized: true,
}));
requestCount += 1;
}
offset += rows.length;
if (this.now() >= fullDeadline) break;
if (!complete && (rows === undefined || rows.length === 0)) {
rows = await runQuery(flatValueQuery(request, columnChunk, {
complete: false,
randomized: false,
}));
requestCount += 1;
}
if (signal.aborted || this.now() >= request.deadline) {
return { kind: "sampled", observedRows };
if (rows === undefined) throw new CatalogConnectorError("REST sensitivity sample query timed out");
const batch = observations(columnChunk, rows);
observedValues += batch.length;
if (batch.length > 0) await consume(batch);
}
const sampleSql = [
baseSelect,
"TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
`LIMIT ${this.sampleRows}`,
].join(" ");
const sampledRows = await runQuery(sampleSql, request.deadline);
observedRows += sampledRows.length;
if (sampledRows.length > 0) await consume(observations(request, sampledRows));
return { kind: "sampled", observedRows };
// Multiple HTTP requests cannot share a source snapshot, so only one-request reads are complete.
return { kind: complete && requestCount === 1 ? "complete" : "sampled", observedValues };
} catch (error) {
if (error instanceof CatalogConnectorError) throw error;
throw new CatalogConnectorError("REST sensitivity source scan failed");
@@ -261,9 +261,9 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
});
}
if (error instanceof SensitivityAnalysisInterruptedError) {
return reply.code(504).send({
code: "sensitivity_analysis_timeout",
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
return reply.code(499).send({
code: "sensitivity_analysis_interrupted",
message: "Sensitivity analysis was interrupted before completion. No assessments were applied.",
});
}
if (error instanceof CatalogConnectorError) {
@@ -284,32 +284,6 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
});
}
function untilAborted<T>(operation: Promise<T>, signal: AbortSignal): Promise<T> {
if (signal.aborted) {
void operation.catch(() => undefined);
return Promise.reject(new SensitivityAnalysisInterruptedError());
}
return new Promise<T>((resolve, reject) => {
const abort = () => reject(new SensitivityAnalysisInterruptedError());
signal.addEventListener("abort", abort, { once: true });
if (signal.aborted) {
void operation.catch(() => undefined);
abort();
return;
}
operation.then(
(value) => {
signal.removeEventListener("abort", abort);
resolve(value);
},
(error: unknown) => {
signal.removeEventListener("abort", abort);
reject(error);
},
);
});
}
function safeSuggestionHistoryError(reply: FastifyReply, error: unknown) {
if (error instanceof CatalogUnavailableError) {
return reply.code(503).send({
@@ -342,13 +316,22 @@ export function catalogDescriptionGenerationRoutes(
try {
const databaseId = idSchema.parse((request.params as { databaseId?: unknown }).databaseId);
const input = suggestionSchema.parse(request.body);
const signal = AbortSignal.timeout(60_000);
const result = await untilAborted(deps.sensitivityAnalysisRunner.run(
const controller = new AbortController();
const abort = () => controller.abort();
request.raw.once("aborted", abort);
reply.raw.once("close", abort);
let result;
try {
result = await deps.sensitivityAnalysisRunner.run(
databaseId,
input.scope,
"targetIds" in input ? input.targetIds : [],
signal,
), signal);
controller.signal,
);
} finally {
request.raw.off("aborted", abort);
reply.raw.off("close", abort);
}
return {
suggestions: result.suggestions,
run: publicSensitivityAnalysisRun(result.run),
@@ -77,7 +77,7 @@ async function setup(
value: "ordinary",
characterLength: 8,
})));
return { kind: "complete", observedRows: 1 };
return { kind: "complete", observedValues: 1 };
}),
},
) {
@@ -167,7 +167,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
scope: "all",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
policyVersion: "sensitivity-v2",
status: "completed",
total: 1,
suggestedSensitive: 1,
@@ -185,6 +185,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
coverage: "metadata",
}],
});
expect(await repository.getColumn(database.id, column.tableId, column.id))
@@ -239,10 +240,8 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
}
});
test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => {
const controller = new AbortController();
controller.abort();
const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal);
test("does not impose a global HTTP deadline on sensitivity analysis", async () => {
const timeout = vi.spyOn(AbortSignal, "timeout");
const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") });
try {
@@ -252,12 +251,9 @@ test("stops sensitivity analysis at the HTTP deadline without creating a review"
payload: { scope: "all" },
});
expect(response.statusCode).toBe(504);
expect(response.json()).toEqual({
code: "sensitivity_analysis_timeout",
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
});
expect(await repository.listSensitivityAnalysisRuns()).toEqual([]);
expect(response.statusCode).toBe(200);
expect(timeout).not.toHaveBeenCalled();
expect(await repository.listSensitivityAnalysisRuns()).toHaveLength(1);
} finally {
timeout.mockRestore();
await app.close();
@@ -6,7 +6,9 @@ import {
import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js";
import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js";
import type {
CatalogColumn,
CatalogRepository,
CatalogTable,
SensitivityAnalysisRun,
WorkspaceDatabase,
} from "../src/catalog/types.js";
@@ -24,13 +26,54 @@ const database = {
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
function catalogTable(id: string, name: string): CatalogTable {
return {
id,
databaseId: database.id,
name,
sourceComment: null,
description: null,
generatedDescription: null,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
};
}
function catalogColumn(id: string, tableId: string, name: string): CatalogColumn {
return {
id,
tableId,
name,
ordinalPosition: 1,
dataType: "text",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
isPrimaryKey: false,
isForeignKey: false,
foreignKeyCount: 0,
sourceComment: null,
description: null,
generatedDescription: null,
sensitive: false,
lastSyncedDatabaseVersion: 1,
lastSyncedAt: "2026-09-02T08:00:00Z",
version: 1,
createdAt: "2026-09-02T08:00:00Z",
updatedAt: "2026-09-02T08:00:00Z",
};
}
const running: SensitivityAnalysisRun = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
scope: "all",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
policyVersion: "sensitivity-v2",
status: "running",
total: 0,
suggestedSensitive: 0,
@@ -56,7 +99,7 @@ test("stops catalog selection when the request expires during a catalog read", a
}),
listTables,
} as unknown as CatalogRepository;
const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier;
const classifier = { assess: vi.fn() } as unknown as SensitivityClassifier;
const analysis = new SensitivityAnalysisService(repository, classifier);
await expect(analysis.analyze(
@@ -66,7 +109,71 @@ test("stops catalog selection when the request expires during a catalog read", a
controller.signal,
)).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
expect(listTables).not.toHaveBeenCalled();
expect(classifier.assessTable).not.toHaveBeenCalled();
expect(classifier.assess).not.toHaveBeenCalled();
});
test("classifies all selected tables in one breadth-first run and reports coverage", async () => {
const firstTable = catalogTable("33333333-3333-4333-8333-333333333333", "patients");
const secondTable = catalogTable("44444444-4444-4444-8444-444444444444", "encounters");
const firstColumn = catalogColumn(
"55555555-5555-4555-8555-555555555555",
firstTable.id,
"status",
);
const secondColumn = catalogColumn(
"66666666-6666-4666-8666-666666666666",
secondTable.id,
"note",
);
const repository = {
get: vi.fn(async () => database),
listTables: vi.fn(async () => [firstTable, secondTable]),
listColumns: vi.fn(async (_databaseId: string, tableId: string) => (
tableId === firstTable.id ? [firstColumn] : [secondColumn]
)),
} as unknown as CatalogRepository;
const assess = vi.fn(async () => [
{
columnId: firstColumn.id,
assessment: "non_sensitive" as const,
proposedSensitive: false,
evidence: [{ kind: "coverage" as const, ruleId: "coverage.sampled_1000" }],
observedValues: 1_000,
coverage: "sampled" as const,
},
{
columnId: secondColumn.id,
assessment: "sensitive" as const,
proposedSensitive: true,
evidence: [{ kind: "content" as const, ruleId: "pii.email" }],
observedValues: 12,
coverage: "sampled" as const,
},
]);
const classifier = { assess } as unknown as SensitivityClassifier;
const onPrepared = vi.fn();
const onProgress = vi.fn();
const suggestions = await new SensitivityAnalysisService(repository, classifier).analyze(
database.id,
"all",
[],
new AbortController().signal,
onPrepared,
onProgress,
);
expect(assess).toHaveBeenCalledOnce();
expect(assess.mock.calls[0]![0]).toEqual([
{ database, table: firstTable, columns: [firstColumn] },
{ database, table: secondTable, columns: [secondColumn] },
]);
expect(onPrepared).toHaveBeenCalledWith(2);
expect(onProgress.mock.calls.map(([processed]) => processed)).toEqual([1, 2]);
expect(suggestions).toEqual([
expect.objectContaining({ columnId: firstColumn.id, sensitive: false, coverage: "sampled" }),
expect.objectContaining({ columnId: secondColumn.id, sensitive: true, coverage: "sampled" }),
]);
});
test("marks a created run interrupted if the request deadline expires during persistence", async () => {
@@ -94,6 +201,6 @@ test("marks a created run interrupted if the request deadline expires during per
status: "interrupted",
total: 0,
unknown: 0,
errorSummary: "Local sensitivity analysis reached its time limit.",
errorSummary: "Local sensitivity analysis was interrupted before completion.",
}));
});
@@ -76,7 +76,7 @@ test("one email hidden in a generically named column makes the whole column sens
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
]],
coverage: { kind: "complete", observedRows: 2 },
coverage: { kind: "complete", observedValues: 2 },
});
const classifier = new SensitivityClassifier(values);
@@ -97,7 +97,7 @@ test("one text value longer than 500 characters makes the whole column sensitive
const target = column({ name: "comment" });
const values = source({
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -112,7 +112,151 @@ test("one text value longer than 500 characters makes the whole column sensitive
});
});
test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
test("scans every table at 300 before advancing to 1,000 and 3,000 values", async () => {
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
const first = column({ name: "status" });
const second = column({
id: "88888888-8888-4888-8888-888888888888",
tableId: otherTable.id,
name: "comment",
});
const calls: string[] = [];
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
calls.push(`${request.table.name}:${request.valuesPerColumn}:${request.sampleOffset}`);
await consume(request.columns.map((item) => ({
columnId: item.id,
value: "ordinary",
characterLength: 8,
})));
return { kind: "sampled", observedValues: request.columns.length };
}),
};
await new SensitivityClassifier(values).assess([
{ database, table, columns: [first] },
{ database, table: otherTable, columns: [second] },
], new AbortController().signal);
expect(calls).toEqual([
"observations:300:0",
"events:300:0",
"observations:700:300",
"events:700:300",
"observations:2000:1000",
"events:2000:1000",
]);
});
test("runs at most two table scans concurrently", async () => {
const targets = Array.from({ length: 3 }, (_, index) => {
const targetTable = {
...table,
id: `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
name: `table_${index + 1}`,
};
return {
database,
table: targetTable,
columns: [column({
id: `10000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
tableId: targetTable.id,
name: `attribute_${index + 1}`,
})],
};
});
let active = 0;
let maximum = 0;
const values: SensitivityValueSource = {
scanTable: vi.fn(async () => {
active += 1;
maximum = Math.max(maximum, active);
await Promise.resolve();
active -= 1;
return { kind: "complete", observedValues: 0 };
}),
};
await new SensitivityClassifier(values).assess(targets, new AbortController().signal);
expect(maximum).toBe(2);
expect(values.scanTable).toHaveBeenCalledTimes(3);
});
test("aborts a peer table scan when another concurrent source scan fails", async () => {
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
const first = column({ name: "status" });
const second = column({
id: "88888888-8888-4888-8888-888888888888",
tableId: otherTable.id,
name: "comment",
});
let peerSignal: AbortSignal | undefined;
const failure = new CatalogConnectorError("source unavailable");
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, _consume, scanSignal) => {
if (request.table.id === table.id) {
await Promise.resolve();
throw failure;
}
peerSignal = scanSignal;
return await new Promise((_resolve, reject) => {
scanSignal.addEventListener("abort", () => reject(scanSignal.reason), { once: true });
});
}),
};
await expect(new SensitivityClassifier(values).assess([
{ database, table, columns: [first] },
{ database, table: otherTable, columns: [second] },
], new AbortController().signal)).rejects.toBe(failure);
expect(peerSignal?.aborted).toBe(true);
});
test("stops sampling a column as soon as one value is sensitive", async () => {
const target = column();
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume([{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 }]);
return { kind: "sampled", observedValues: 1 };
}),
};
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(values.scanTable).toHaveBeenCalledOnce();
expect(assessment).toMatchObject({ assessment: "sensitive", proposedSensitive: true });
});
test("stops non-text columns after the 1,000-value stage", async () => {
const target = column({ dataType: "integer", name: "sequence_number" });
const values: SensitivityValueSource = {
scanTable: vi.fn(async (request, consume) => {
await consume([{ columnId: target.id, value: "42", characterLength: 2 }]);
return { kind: "sampled", observedValues: 1 };
}),
};
const [assessment] = await new SensitivityClassifier(values).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(vi.mocked(values.scanTable).mock.calls.map(([request]) => request.valuesPerColumn))
.toEqual([300, 700]);
expect(assessment).toMatchObject({
assessment: "non_sensitive",
proposedSensitive: false,
coverage: "sampled",
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_1000" }],
});
});
test("complete coverage classifies benign and empty columns as non-sensitive", async () => {
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
const humanProtected = column({
@@ -126,7 +270,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
{ columnId: empty.id, value: null, characterLength: null },
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const assessments = await new SensitivityClassifier(values).assessTable(
@@ -138,7 +282,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
expect.objectContaining({
columnId: empty.id,
assessment: "unknown",
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
}),
@@ -150,11 +294,11 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
]);
});
test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
test("sampled coverage without a match proposes non-sensitive independently of the current flag", async () => {
const target = column({ sensitive: true });
const values = source({
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -163,13 +307,13 @@ test("sampled coverage without a match is unknown and preserves the current huma
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: true,
evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
test("an unavailable source fails the analysis instead of producing unknown decisions", async () => {
const unresolved = column();
const metadataMatch = column({
id: "44444444-4444-4444-8444-444444444444",
@@ -181,30 +325,17 @@ test("an unavailable source produces sanitized unknown evidence without losing m
}),
};
const assessments = await new SensitivityClassifier(values).assessTable(
await expect(new SensitivityClassifier(values).assessTable(
{ database, table, columns: [unresolved, metadataMatch] },
new AbortController().signal,
);
expect(assessments).toEqual([
expect.objectContaining({
columnId: unresolved.id,
assessment: "unknown",
evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
}),
expect.objectContaining({
columnId: metadataMatch.id,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
}),
]);
)).rejects.toBeInstanceOf(CatalogConnectorError);
});
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
const target = column({ name: "codice_fiscale" });
const values = source({
batches: [],
coverage: { kind: "complete", observedRows: 0 },
coverage: { kind: "complete", observedValues: 0 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -245,7 +376,7 @@ test.each([
const target = column();
const values = source({
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -267,13 +398,16 @@ test("does not make a malformed email decisive", async () => {
value: "contatto a@b..com non valido",
characterLength: 28,
}]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
expect(assessment).toMatchObject({
assessment: "non_sensitive",
evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
});
});
test("finds a valid email after a malformed candidate in the same value", async () => {
@@ -284,7 +418,7 @@ test("finds a valid email after a malformed candidate in the same value", async
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
characterLength: 58,
}]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
@@ -300,7 +434,7 @@ test("optional local NER evidence can make otherwise ambiguous Italian text sens
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
@@ -331,14 +465,17 @@ test("does not wait for an optional NER worker that is still warming", async ()
const [assessment] = await new SensitivityClassifier(source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
}), detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).not.toHaveBeenCalled();
expect(assessment).toMatchObject({ assessment: "unknown" });
expect(assessment).toMatchObject({
assessment: "non_sensitive",
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
@@ -359,7 +496,7 @@ test("bounds each optional NER request when an installation raises the per-table
await new SensitivityClassifier(source({
batches: [observations],
coverage: { kind: "complete", observedRows: 8 },
coverage: { kind: "complete", observedValues: 8 },
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
{ database, table, columns },
new AbortController().signal,
@@ -386,7 +523,7 @@ test("limits default NER work to two candidates spread across a wide table", asy
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
})))],
coverage: { kind: "complete", observedRows: 2 },
coverage: { kind: "complete", observedValues: 2 },
}), detector).assessTable(
{ database, table, columns },
new AbortController().signal,
@@ -402,7 +539,7 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
coverage: { kind: "sampled", observedRows: 1 },
coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => {
@@ -430,11 +567,11 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
expect(nerBudget.remainingMs).toBe(0);
});
test("uninterpretable binary content remains unknown after complete coverage", async () => {
test("uninterpretable binary content is protected conservatively without scanning", async () => {
const target = column({ dataType: "bytea" });
const values = source({
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
coverage: { kind: "complete", observedRows: 1 },
coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -443,8 +580,9 @@ test("uninterpretable binary content remains unknown after complete coverage", a
);
expect(assessment).toMatchObject({
assessment: "unknown",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
assessment: "sensitive",
proposedSensitive: true,
evidence: [{ kind: "type", ruleId: "type.binary_uninspectable" }],
});
expect(values.scanTable).not.toHaveBeenCalled();
});
@@ -1,12 +1,17 @@
import { expect, test, vi } from "vitest";
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { expect, test, vi } from "vitest";
import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js";
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
import {
CatalogConnectorError,
type CatalogColumn,
type CatalogTable,
type WorkspaceDatabase,
} from "../src/catalog/types.js";
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
@@ -60,128 +65,145 @@ function column(id: string, name: string): CatalogColumn {
};
}
test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => {
function request(columns: readonly CatalogColumn[], overrides: Record<string, unknown> = {}) {
return {
database,
table,
columns,
valuesPerColumn: 300,
sampleOffset: 0,
sampleSeed: 37,
queryTimeoutMs: 5_000,
fullScanThreshold: 1_000,
...overrides,
};
}
test("uses bounded read-only PostgreSQL sampling for tables above 1,000 rows", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value');
const fullRows = Array.from({ length: 200 }, () => ({
__value_0: "ordinary",
__length_0: "8",
__value_1: null,
__length_1: null,
}));
const query = vi.fn(async (sql: string) => {
if (sql.includes("TABLESAMPLE")) {
return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] };
if (sql.startsWith("SELECT 1 AS __present")) {
return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) };
}
if (sql.startsWith("WITH sampled")) {
return { rows: [
{ __column_index: 0, __value: "ordinary", __length: "8" },
{ __column_index: 1, __value: "mario.rossi@example.it", __length: 23 },
] };
}
if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows };
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
let clockCalls = 0;
const values = new ConcreteSensitivityValueSource(access, undefined, {
now: () => clockCalls++ < 3 ? 1_000 : 6_100,
});
const consumed: unknown[] = [];
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note, contact],
fullScanBudgetMs: 5_000,
deadline: 61_000,
}, (batch) => consumed.push(...batch), new AbortController().signal);
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note, contact]),
consume,
new AbortController().signal,
)).resolves.toEqual({ kind: "sampled", observedValues: 2 });
expect(coverage).toEqual({ kind: "sampled", observedRows: 201 });
expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 });
expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null });
expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 });
expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]);
expect(query.mock.calls.some(([sql]) => (
String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT")
))).toBe(true);
expect(query.mock.calls.some(([sql]) => String(sql) === (
"FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
))).toBe(true);
expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false);
expect(query.mock.calls.some(([sql]) => (
String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM')
))).toBe(true);
expect(query).toHaveBeenCalledWith("SELECT set_config('statement_timeout', $1, true)", ["5000ms"]);
const sampleSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled"));
expect(sampleSql).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)');
expect(sampleSql).toContain("REPEATABLE (37)");
expect(sampleSql).toContain("LIMIT 3000 OFFSET 0");
expect(sampleSql).toContain("CROSS JOIN LATERAL");
expect(sampleSql).toContain("WHERE __rank <= 300");
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "ordinary", characterLength: 8 },
{ columnId: contact.id, value: "mario.rossi@example.it", characterLength: 23 },
]);
expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]);
expect(end).toHaveBeenCalledOnce();
});
test("reports complete coverage when the final full-scan page is short", async () => {
test("fully scans a table when the 1,001-row probe proves it is small", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD")
? { rows: [{ __value_0: "ordinary", __length_0: 8 }] }
: { rows: [] });
const end = vi.fn(async () => undefined);
const query = vi.fn(async (sql: string) => {
if (sql.startsWith("SELECT 1 AS __present")) return { rows: [{ __present: 1 }] };
if (sql.startsWith("WITH sampled")) {
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
}
return { rows: [] };
});
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal);
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note]),
consume,
new AbortController().signal,
)).resolves.toEqual({ kind: "complete", observedValues: 1 });
expect(coverage).toEqual({ kind: "complete", observedRows: 1 });
expect(query.mock.calls.filter(([sql]) => (
String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
))).toHaveLength(1);
const valueSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled"));
expect(valueSql).not.toContain("TABLESAMPLE");
expect(valueSql).toContain("WHERE __rank <= 1000");
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "ordinary", characterLength: 8 },
]);
});
test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => {
test("falls back to sampling when the small-table probe reaches its query timeout", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
let fullScanAttempts = 0;
const query = vi.fn(async (sql: string) => {
if (sql.includes("TABLESAMPLE")) {
return { rows: [{ __value_0: "sample", __length_0: 6 }] };
}
if (sql.startsWith("FETCH FORWARD")) {
fullScanAttempts += 1;
if (sql.startsWith("SELECT 1 AS __present")) {
throw Object.assign(new Error("statement timeout"), { code: "57014" });
}
if (sql.startsWith("WITH sampled")) {
return { rows: [{ __column_index: 0, __value: "sample", __length: 6 }] };
}
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
const coverage = await values.scanTable({
database,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal);
expect(fullScanAttempts).toBe(1);
expect(coverage).toEqual({ kind: "sampled", observedRows: 1 });
expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([
"SAVEPOINT sensitivity_full_scan",
"ROLLBACK TO SAVEPOINT sensitivity_full_scan",
]));
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "sample", characterLength: 6 },
]);
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note]),
consume,
new AbortController().signal,
)).resolves.toEqual({ kind: "sampled", observedValues: 1 });
expect(query.mock.calls.map(([sql]) => String(sql))).toContain(
"ROLLBACK TO SAVEPOINT sensitivity_scan_1",
);
});
test("scans a REST run_query binding without using PostgreSQL-wire access", async () => {
test("limits each source query to at most 25 columns", async () => {
const columns = Array.from({ length: 26 }, (_, index) => column(
`00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
`attribute_${index + 1}`,
));
const query = vi.fn(async (sql: string) => {
if (sql.startsWith("SELECT 1 AS __present")) {
return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) };
}
if (sql.startsWith("WITH sampled")) {
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
}
return { rows: [] };
});
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
await new ConcreteSensitivityValueSource(access).scanTable(
request(columns),
vi.fn(),
new AbortController().signal,
);
expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2);
});
test("scans a REST run_query binding without PostgreSQL-wire access", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
@@ -193,7 +215,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
{ __value_0: "mario.rossi@example.it", __length_0: 23 },
{ __column_index: 0, __value: "mario.rossi@example.it", __length: 23 },
]), { status: 200, headers: { "content-type": "application/json" } }));
vi.stubGlobal("fetch", fetchMock);
const access: CatalogPostgresAccess = {
@@ -213,15 +235,12 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
const consume = vi.fn();
try {
await expect(values.scanTable({
await expect(values.scanTable(request([note], {
database: restDatabase,
table,
columns: [note],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, consume, new AbortController().signal)).resolves.toEqual({
kind: "complete",
observedRows: 1,
fullScanThreshold: undefined,
}), consume, new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedValues: 1,
});
expect(access.connect).not.toHaveBeenCalled();
expect(fetchMock).toHaveBeenCalledWith(
@@ -232,7 +251,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
}),
);
const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body));
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0');
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)');
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 },
]);
@@ -243,74 +262,44 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
}
});
test("keeps multi-request REST scans conservative without a source transaction", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
const secretStore = {
materialize: vi.fn(() => ({
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
release: vi.fn(),
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn()
.mockResolvedValueOnce(new Response(JSON.stringify([
{ __value_0: "ordinary", __length_0: 8 },
]), { status: 200 }))
.mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 }));
vi.stubGlobal("fetch", fetchMock);
const values = new ConcreteSensitivityValueSource({
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
}, secretStore, { batchRows: 1 });
const restDatabase: WorkspaceDatabase = {
...database,
binding: {
transport: "rest_api",
baseUrl: "https://dwh.example.test/root",
restPath: "/health",
restAuth: "x-api-key",
},
};
try {
await expect(values.scanTable({
database: restDatabase,
table,
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
fullScanBudgetMs: 5_000,
deadline: Date.now() + 60_000,
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedRows: 1,
});
expect(fetchMock).toHaveBeenCalledTimes(2);
} finally {
vi.unstubAllGlobals();
rmSync(root, { recursive: true, force: true });
test("falls back to a sequential bounded sample when randomized sampling times out", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const query = vi.fn(async (sql: string) => {
if (sql.startsWith("WITH sampled") && sql.includes("TABLESAMPLE")) {
throw Object.assign(new Error("raw source detail"), { code: "57014" });
}
});
test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => {
const query = vi.fn(async () => ({ rows: [] }));
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
const now = vi.fn()
.mockReturnValueOnce(1_000)
.mockReturnValue(61_000);
const values = new ConcreteSensitivityValueSource(access, undefined, { now });
await expect(values.scanTable({
database,
table,
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
fullScanBudgetMs: 5_000,
deadline: 60_000,
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
kind: "sampled",
observedRows: 0,
if (sql.startsWith("WITH sampled")) {
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
}
return { rows: [] };
});
expect(query).not.toHaveBeenCalled();
expect(end).toHaveBeenCalledOnce();
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note], { fullScanThreshold: undefined }),
vi.fn(),
new AbortController().signal,
)).resolves.toEqual({ kind: "sampled", observedValues: 1 });
expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2);
});
test("fails explicitly when both randomized and sequential sample queries time out", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const query = vi.fn(async (sql: string) => {
if (sql.startsWith("WITH sampled")) {
throw Object.assign(new Error("raw source detail"), { code: "57014" });
}
return { rows: [] };
});
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
await expect(new ConcreteSensitivityValueSource(access).scanTable(
request([note], { fullScanThreshold: undefined }),
vi.fn(),
new AbortController().signal,
)).rejects.toEqual(new CatalogConnectorError("Sensitivity sample query timed out"));
});
@@ -1,5 +1,5 @@
---
status: accepted
status: superseded by ADR-0015
---
# Assess sensitive columns locally from source content
@@ -0,0 +1,33 @@
---
status: accepted
---
# Use progressive sampling for sensitive columns
The `sensitivity-v1` wall-clock policy produced too many `unknown` assessments: a global
sixty-second deadline coupled the outcome of one column to table order, source latency, and optional
NER cost. Those outcomes were not useful for description-generation gating, because they did not
provide a usable draft Sensitive Data Flag.
`sensitivity-v2` bounds database effort by inspected values rather than by one global clock. Tables
proven to contain at most 1,000 rows are fully scanned. Larger tables are processed breadth-first in
three passes: 300 values per unresolved column, 700 additional values to reach 1,000, then 2,000
additional values to reach 3,000 for unresolved text, JSON, and XML columns. One positive rule or
NER finding is enough to stop later work for that column. At most two tables are scanned
concurrently. Source queries contain at most 25 columns and each has a five-second statement timeout.
An empty or timed-out randomized sample gets one sequential bounded retry; two timeouts fail the run.
A completed v2 analysis returns only `sensitive` or `non_sensitive`. Sampled no-match, empty, and
all-null columns are proposed as `non_sensitive`, with coverage reported independently so the human
reviewer can judge the strength of the proposal. Binary or otherwise uninspectable column types are
proposed as `sensitive`. A source failure fails the analysis and returns no review; it is not
converted into `unknown`. The administrator can still set either final value.
The HTTP operation has no global analysis deadline. It is canceled when the client disconnects or
the backend restarts. Historic and interrupted run records retain the database field named
`unknown` for compatibility, where it counts unprocessed columns rather than a v2 assessment.
This decision supersedes ADR-0014 only for scan effort, coverage semantics, and the assessment
domain. ADR-0014 remains authoritative for the single local TypeScript decision point, the absence
of generative LLMs, transient human-reviewed drafts, the 500-character rule, and optional CPU-only
NER evidence.
+9 -8
View File
@@ -107,16 +107,17 @@ explicitly unlocked; it never resumes automatically.
Sensitivity analysis is a synchronous administrative request and does not use the installation
model catalog. Database-specific adapters stream bounded normalized values from read-only source
connections; the TypeScript `SensitivityClassifier` is the single decision point for
`sensitive | non_sensitive | unknown`. Deterministic rules run first. A complete scan is attempted
for at most five seconds per table, then the adapter samples within the sixty-second request budget.
An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved short text, but it
cannot make or persist the decision itself.
`sensitive | non_sensitive`. Deterministic rules run first. Tables up to 1,000 rows are fully
scanned; larger tables use breadth-first targets of 300, 1,000, and 3,000 values, with the last pass
limited to text-like columns. Source queries have five-second limits, but the request has no global
analysis deadline. An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved
short text, but it cannot make or persist the decision itself.
Each attempt has its own durable run and ordered sanitized events, separate from Description
Generation because its lifecycle and counters differ. The run records the local policy version,
coverage aggregates, and sanitized rule identifiers. Proposed flags, source values, NER spans, and
worker diagnostics remain transient. Only an explicit administrator save changes the human-owned
Sensitive Data Flag.
Generation because its lifecycle and counters differ. The run records the local policy version and
aggregate decision counts. Coverage, rule identifiers, proposed flags, source values, NER spans,
and worker diagnostics remain transient. Only an explicit administrator save changes the
human-owned Sensitive Data Flag.
## Main backend classes
+34 -18
View File
@@ -7,7 +7,7 @@ may set either value, including overriding a `sensitive` proposal.
## Default policy
`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v1`
`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v2`
policy combines:
- normalized column-name rules for direct identifiers, credentials, and health data;
@@ -18,22 +18,35 @@ policy combines:
- a conservative length rule: any observed textual value longer than 500 characters makes the
entire column sensitive.
One decisive value is enough to classify the column as `sensitive`. A complete scan with no match
may classify it as `non_sensitive`. Empty, all-null, binary/uninspectable, interrupted, and sampled
no-match columns are `unknown`; an `unknown` draft preserves the current human flag.
One decisive value is enough to classify the column as `sensitive` and removes it from subsequent
passes. Binary or otherwise uninspectable column types are also proposed as `sensitive`, because
their contents cannot be cleared by the textual rules. A completed analysis has only two draft
outcomes: `sensitive` and `non_sensitive`. Empty or all-null columns are `non_sensitive` with
`no_values` coverage; a sampled column with no match is `non_sensitive` with explicit sampled
coverage. The administrator remains free to reverse either proposal before saving it.
Source reads are database-specific, but decisions are database-independent. PostgreSQL direct and
REST `run_query` adapters project at most 501 characters per value, use only `SELECT`, and never
persist source values. A full scan gets five seconds per table. If it cannot finish, the adapter uses
a bounded repeatable sample within the sixty-second request deadline. PostgreSQL-wire reads run in a
read-only transaction and always end with rollback. A REST scan can claim complete coverage only
when it finishes in one request; multi-request pagination has no shared source transaction and is
therefore conservatively reported as sampled.
persist source values. Tables proven to contain at most 1,000 rows are fully scanned. Larger tables
are processed breadth-first so every table gets the cheapest pass before any table gets a deeper
one:
The HTTP operation stops waiting at sixty seconds. The same expiring signal is checked before and
after catalog selection, source access, progress writes, and every table. If it expires after a run
has been created, that run is finalized as `interrupted` and all not-decisively-processed columns
are counted as `unknown`; no review payload is returned from the timed-out request.
1. inspect up to 300 non-null values per unresolved column;
2. inspect up to 700 additional values, reaching a 1,000-value target;
3. for unresolved text, JSON, and XML columns only, inspect up to 2,000 additional values, reaching
a 3,000-value target.
At most two tables are scanned concurrently, and the database adapter groups at most 25 columns in
one source query. Each probe or value query has a five-second statement timeout; PostgreSQL-wire
reads run in a read-only transaction and always end with rollback. Sampling is bounded and
repeatable for a policy version. If a randomized sample is empty or reaches its query timeout, the
adapter tries one sequential bounded sample; if that also times out, the source error fails the run
and returns no review instead of manufacturing `unknown` decisions.
There is no global sixty-second analysis deadline. Work is bounded by sample counts, per-query
timeouts, and early column exits. The operation is interrupted only when its request connection is
aborted or the backend restarts. Historical or interrupted run counters named `unknown` represent
columns that were not processed; `unknown` is not a `sensitivity-v2` column assessment.
History stores only the policy version, aggregate outcomes, timestamps, and fixed operational
events. Sanitized rule IDs are returned in the transient review and shadow report, not persisted.
@@ -116,13 +129,16 @@ Enabling NER by default requires all of these gates:
1. the pinned artifact and `MODEL_SHA256SUMS` are archived with the installation inventory;
2. the Python dependency/license inventory contains only redistribution-compatible licenses;
3. the CPU benchmark stays within the configured deadlines and does not use a GPU;
3. the CPU benchmark stays within the configured NER allowance and does not use a GPU;
4. the labeled Italian evaluation meets thresholds approved by the product owner.
If a gate fails, leave NER disabled. The deterministic policy remains available and unresolved
columns remain `unknown` rather than being sent to an internal or external LLM.
If a gate fails, leave NER disabled. The deterministic policy remains available and produces the
binary draft from its scan coverage; no content is sent to an internal or external LLM.
The first aggregate PSD shadow comparison is recorded in
[`2026-09-02-psd-sensitivity-shadow.md`](../reports/2026-09-02-psd-sensitivity-shadow.md). On the
local CPU runner, NER found additional entities but reduced total coverage inside the 60-second
deadline, so the accepted setting remains disabled by default.
local CPU runner, NER found additional entities but reduced total coverage under the superseded
global deadline, so the accepted setting remains disabled by default pending a new v2 benchmark.
The deterministic progressive PSD run is recorded in
[`2026-09-03-psd-progressive-sensitivity-shadow.md`](../reports/2026-09-03-psd-progressive-sensitivity-shadow.md):
it assessed all 2,275 columns with zero `unknown` decisions and left NER disabled.
@@ -0,0 +1,49 @@
# PSD progressive sensitivity shadow evaluation
Date: 2026-09-03
This report records an aggregate, non-mutating evaluation of `sensitivity-v2` against the PSD
workspace. The configured connector accessed the source data warehouse with its read-only role.
The shadow command did not create an analysis run, update local catalog metadata, or save Sensitive
Data Flags. No database, table, column, source value, or matched span was emitted.
The final post-fix run used the local Docker CPU environment, without NER. It inspected all 2,275
catalog columns through the progressive 300, 1,000, and text-only 3,000-value policy.
| Sensitive | Non-sensitive | Unknown decisions | Analysis time |
| ---: | ---: | ---: | ---: |
| 343 | 1,932 | 0 | 50,082 ms |
Coverage was reported independently from the decision:
| Metadata decision | Complete scan | Sampled | No observed values |
| ---: | ---: | ---: | ---: |
| 39 | 0 | 2,128 | 108 |
The sampled no-match population comprised 1,337 columns ending after the 1,000-value target and
487 text-like columns ending after the 3,000-value target. Positive findings were:
| Rule | Columns |
| --- | ---: |
| `pii.phone_number` | 202 |
| `text.over_500_characters` | 52 |
| `metadata.health` | 34 |
| `health.clinical_term` | 25 |
| `pii.italian_vat` | 8 |
| `pii.email` | 7 |
| `metadata.direct_identifier` | 5 |
| `pii.uuid` | 5 |
| `financial.payment_card` | 3 |
| `pii.italian_fiscal_code` | 2 |
Three successful v2 diagnostic runs produced the same decisions and aggregate rule counts. Their
times ranged from 49,253 to 130,818 ms, showing that source load still affects latency even though
it no longer changes the outcome through a global deadline. An earlier run exposed an intermittent
randomized-query timeout. The adapter now retries that case once with a sequential bounded query;
a regression test covers the fallback, while two consecutive timeouts still fail the whole analysis
instead of creating `unknown` decisions.
This is a coverage and operational benchmark, not a precision/recall acceptance test. In
particular, the 202 phone-number findings and every other rule family still require human review or
a separately approved labeled corpus before their false-positive rate can be measured. NER remains
disabled by default.
+3 -2
View File
@@ -134,14 +134,15 @@ export interface SensitivityReviewItem {
version: number;
currentSensitive: boolean;
sensitive: boolean;
assessment: "sensitive" | "non_sensitive" | "unknown";
assessment: "sensitive" | "non_sensitive";
evidence: Array<{
kind: "metadata" | "content" | "length" | "ner" | "coverage";
kind: "metadata" | "content" | "length" | "ner" | "coverage" | "type";
ruleId: string;
label?: string;
confidence?: number;
}>;
observedValues: number;
coverage: "metadata" | "complete" | "sampled" | "no_values";
}
export interface SensitivityAnalysisResult {
+1
View File
@@ -149,6 +149,7 @@ test.each([
["catalog_table_not_found", "One or more selected catalog tables were not found."],
["sensitivity_source_unavailable", "The database content could not be read for sensitivity analysis. No assessments were applied."],
["sensitivity_analysis_timeout", "Sensitivity analysis reached its time limit. No assessments were applied."],
["sensitivity_analysis_interrupted", "Sensitivity analysis was interrupted before completion. No assessments were applied."],
["sensitive_data_suggestion_history_request_invalid", "Sensitivity analysis history parameters are invalid."],
["sensitive_data_suggestion_history_failed", "Sensitivity analysis history could not be loaded."],
["sensitive_data_suggestion_run_not_found", "The sensitivity analysis run was not found."],
+2
View File
@@ -28,6 +28,7 @@ const safeErrorCodes = new Set([
"sensitive_data_suggestion_no_columns",
"sensitivity_source_unavailable",
"sensitivity_analysis_timeout",
"sensitivity_analysis_interrupted",
"sensitive_data_suggestion_failed",
"sensitive_data_suggestion_history_request_invalid",
"sensitive_data_suggestion_history_failed",
@@ -94,6 +95,7 @@ const localCodeMessages: Record<string, string> = {
sensitive_data_suggestion_no_columns: "The selected scope contains no catalog columns to assess.",
sensitivity_source_unavailable: "The database content could not be read for sensitivity analysis. No assessments were applied.",
sensitivity_analysis_timeout: "Sensitivity analysis reached its time limit. No assessments were applied.",
sensitivity_analysis_interrupted: "Sensitivity analysis was interrupted before completion. No assessments were applied.",
sensitive_data_suggestion_failed: "Sensitivity analysis failed before review. No changes were applied.",
sensitive_data_suggestion_history_request_invalid: "Sensitivity analysis history parameters are invalid.",
sensitive_data_suggestion_history_failed: "Sensitivity analysis history could not be loaded.",
@@ -167,7 +167,7 @@ function makeSensitivityAnalysisRun(
databaseId: "11111111-1111-4111-8111-111111111111",
engine: "local",
modelId: null,
policyVersion: "sensitivity-v1",
policyVersion: "sensitivity-v2",
scope: "selected_columns",
status: "completed",
total: 2,
@@ -2081,6 +2081,7 @@ test("requests database-level sensitivity analysis for the only selected databas
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
coverage: "sampled",
}],
});
}),
@@ -2162,6 +2163,7 @@ test("requests sensitivity analysis only for selected tables", async () => {
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
observedValues: 0,
coverage: "metadata",
}],
});
}),
@@ -2236,6 +2238,7 @@ test("allows a human downgrade and saves only explicit sensitivity changes", asy
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
coverage: "sampled",
},
{
columnId: nameColumn.id,
@@ -2246,8 +2249,9 @@ test("allows a human downgrade and saves only explicit sensitivity changes", asy
currentSensitive: true,
sensitive: false,
assessment: "non_sensitive",
evidence: [],
evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
observedValues: 2,
coverage: "complete",
},
],
});
@@ -21,6 +21,7 @@ const suggestions: SensitivityReviewItem[] = [
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
coverage: "sampled",
},
{
columnId: "22222222-2222-4222-8222-222222222222",
@@ -33,6 +34,7 @@ const suggestions: SensitivityReviewItem[] = [
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
observedValues: 0,
coverage: "metadata",
},
];
@@ -183,7 +183,7 @@ export function SensitiveDataReviewDrawer({
? suggestion.evidence.map((item) => item.label
? `${item.ruleId} (${item.label}${item.confidence === undefined ? "" : ` ${Math.round(item.confidence * 100)}%`})`
: item.ruleId).join(", ")
: "no sensitive match"}. Observed values: {suggestion.observedValues}.
: "no sensitive match"}. Coverage: {suggestion.coverage.replace("_", " ")}. Observed values: {suggestion.observedValues}.
</p>
</div>
<span className={`rounded px-2 py-0.5 text-[11px] font-semibold ${changedFromCurrent ? "bg-amber-500/12 text-amber-800 dark:text-amber-300" : "bg-muted text-muted-foreground"}`}>
@@ -145,7 +145,7 @@ export function SensitivityAnalysisHistoryDrawer({
["total", "Total columns"],
["suggestedSensitive", "Sensitive"],
["suggestedNonSensitive", "Not sensitive"],
["unknown", "Unknown"],
["unknown", "Unprocessed"],
] as const;
return (
<FleetLedgerDrawer
+2
View File
@@ -89,9 +89,11 @@ nav:
- 0012 Effective relationship authority: adr/0012-use-the-catalog-as-the-logical-relationship-authority.md
- 0013 Installation model catalog: adr/0013-use-one-installation-model-catalog-with-runtime-projections.md
- 0014 Local sensitive-column assessment: adr/0014-assess-sensitive-columns-locally-from-source-content.md
- 0015 Progressive sensitive-column sampling: adr/0015-use-progressive-sampling-for-sensitive-columns.md
- AI catalog description acceptance: testing/2026-08-29-ai-catalog-description-generation-acceptance.md
- Sensitivity NER license inventory: reports/2026-09-02-sensitivity-ner-license-inventory.md
- PSD sensitivity shadow evaluation: reports/2026-09-02-psd-sensitivity-shadow.md
- PSD progressive sensitivity shadow evaluation: reports/2026-09-03-psd-progressive-sensitivity-shadow.md
- Design records:
- Metadata catalog design: plans/2026-08-26-metadata-catalog-from-thothai.md
- Description generation design: plans/2026-08-28-ai-catalog-description-generation.md