diff --git a/PROJECT_STATE.md b/PROJECT_STATE.md
index 462c628d..658da211 100644
--- a/PROJECT_STATE.md
+++ b/PROJECT_STATE.md
@@ -98,16 +98,17 @@ column. The KPI strip reads installation-wide or selected-database aggregates fr
description history, and sensitive-field review/history use the production APIs in right-side
drawers rather than prototype fixtures; closing a history drawer does not stop its background run.
-Sensitive-field review is now driven by the versioned local `sensitivity-v1` policy, not by a
+Sensitive-field review is now driven by the versioned local `sensitivity-v2` policy, not by a
catalog model. The backend reads selected source tables through read-only, database-specific
-adapters and makes every `sensitive | non_sensitive | unknown` decision in the TypeScript
-`SensitivityClassifier`. A single validated match protects the column; a full scan is limited to
-five seconds per table before sampling and the whole request to sixty seconds. Draft assessments
-remain transient until an administrator explicitly saves them. Optional GLiNER2 evidence is
-CPU-only, offline, opt-in, and never replaces the deterministic decision point; see
-`docs/operations/sensitivity-analysis.md`. The aggregate PSD shadow comparison kept NER disabled by
-default because its extra findings did not offset the coverage lost to inference within the global
-deadline; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`.
+adapters and makes every `sensitive | non_sensitive` draft decision in the TypeScript
+`SensitivityClassifier`. A single validated match protects the column. Tables up to 1,000 rows are
+fully scanned; larger tables use breadth-first 300, 1,000, and text-only 3,000-value targets, with a
+five-second limit per source query and no global request deadline. Source failures fail the run
+instead of yielding `unknown`; coverage remains visible separately from the proposal. Draft
+assessments remain transient until an administrator explicitly saves them. Optional GLiNER2
+evidence is CPU-only, offline, opt-in, and never replaces the deterministic decision point; see
+`docs/operations/sensitivity-analysis.md`. The earlier v1 PSD shadow comparison kept NER disabled by
+default; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`. A v2 PSD benchmark is still due.
Physical membership, source
comments, column types/default/nullability/PK positions, and constraint-level ordered FK pairs are
@@ -169,9 +170,9 @@ AI Description Generation uses the catalog's human-owned Sensitive Data Flag. Th
`false`, including for newly synchronized columns. An administrator may request a local sensitivity
analysis for one selected database, selected tables, or selected columns. One deterministic
TypeScript classifier combines metadata, bounded source-content rules, and optional CPU-only NER;
-no generative model decides the result. Its `sensitive`, `non_sensitive`, or `unknown` assessments
-remain an unsaved draft until the human reviews and saves any chosen flag changes, including a
-downgrade to non-sensitive.
+no generative model decides the result. Its `sensitive` or `non_sensitive` assessments remain an
+unsaved draft until the human reviews and saves any chosen flag changes, including a downgrade to
+non-sensitive. Coverage is reported separately; interrupted history may count unprocessed columns.
Each started analysis records a separate Sensitivity Analysis Run with aggregate counters and safe
ordered events. This operational history never stores per-column assessments, source values,
matched spans, prompts, or free-form diagnostics; reloading still discards an unsaved review draft.
diff --git a/backend/src/catalog/sensitivity-analysis-runner.ts b/backend/src/catalog/sensitivity-analysis-runner.ts
index 0543d1b2..23f10af9 100644
--- a/backend/src/catalog/sensitivity-analysis-runner.ts
+++ b/backend/src/catalog/sensitivity-analysis-runner.ts
@@ -14,7 +14,7 @@ import type {
} from "./types.js";
const interruptedMessage = "Local sensitivity analysis was interrupted by backend restart.";
-const deadlineMessage = "Local sensitivity analysis reached its time limit.";
+const interruptedDuringRunMessage = "Local sensitivity analysis was interrupted before completion.";
const failedMessage = "Local sensitivity analysis failed.";
function ensureActive(signal: AbortSignal): void {
@@ -98,14 +98,12 @@ export class SensitivityAnalysisRunner {
const suggestedNonSensitive = batch.filter(
(suggestion) => suggestion.assessment === "non_sensitive",
).length;
- const unknown = batch.filter((suggestion) => suggestion.assessment === "unknown").length;
const current = await this.repository.getSensitivityAnalysisRun(started.id);
ensureActive(signal);
if (!current) throw new Error("Sensitivity Analysis Run disappeared");
const progress = await this.repository.updateSensitivityAnalysisRun(started.id, {
suggestedSensitive: current.suggestedSensitive + suggestedSensitive,
suggestedNonSensitive: current.suggestedNonSensitive + suggestedNonSensitive,
- unknown: current.unknown + unknown,
});
if (!progress) throw new Error("Sensitivity Analysis Run disappeared");
processedSensitive += suggestedSensitive;
@@ -126,7 +124,6 @@ export class SensitivityAnalysisRunner {
const suggestedNonSensitive = suggestions.filter(
(suggestion) => suggestion.assessment === "non_sensitive",
).length;
- const unknown = suggestions.filter((suggestion) => suggestion.assessment === "unknown").length;
await this.repository.appendSensitivityAnalysisEvent(
started.id,
"info",
@@ -140,7 +137,7 @@ export class SensitivityAnalysisRunner {
total: suggestions.length,
suggestedSensitive,
suggestedNonSensitive,
- unknown,
+ unknown: 0,
finishedAt: new Date().toISOString(),
errorSummary: null,
});
@@ -149,7 +146,7 @@ export class SensitivityAnalysisRunner {
return { suggestions, run: completed };
} catch (error) {
const interrupted = signal.aborted || error instanceof SensitivityAnalysisInterruptedError;
- const message = interrupted ? deadlineMessage : failedMessage;
+ const message = interrupted ? interruptedDuringRunMessage : failedMessage;
await this.repository.updateSensitivityAnalysisRun(started.id, {
status: interrupted ? "interrupted" : "failed",
...(interrupted ? {
diff --git a/backend/src/catalog/sensitivity-analysis-service.ts b/backend/src/catalog/sensitivity-analysis-service.ts
index 9f9034c4..95c779d1 100644
--- a/backend/src/catalog/sensitivity-analysis-service.ts
+++ b/backend/src/catalog/sensitivity-analysis-service.ts
@@ -12,7 +12,7 @@ import type {
} from "./types.js";
export type { SensitivityAnalysisScope } from "./types.js";
-export const SENSITIVITY_POLICY_VERSION = "sensitivity-v1";
+export const SENSITIVITY_POLICY_VERSION = "sensitivity-v2";
interface SelectedColumn {
table: CatalogTable;
@@ -30,6 +30,7 @@ export interface SensitivityReviewItem {
assessment: SensitivityColumnAssessment["assessment"];
evidence: readonly SensitivityEvidence[];
observedValues: number;
+ coverage: SensitivityColumnAssessment["coverage"];
}
export class SensitivityAnalysisTargetNotFoundError extends Error {
@@ -48,7 +49,7 @@ export class SensitivityAnalysisDuplicateTargetIdsError extends Error {
export class SensitivityAnalysisInterruptedError extends Error {
constructor() {
- super("sensitivity analysis deadline exceeded");
+ super("sensitivity analysis interrupted");
this.name = "SensitivityAnalysisInterruptedError";
}
}
@@ -69,7 +70,7 @@ export class SensitivityAnalysisService {
constructor(
private readonly repository: CatalogRepository,
private readonly classifier: SensitivityClassifier,
- private readonly options: { runBudgetMs?: number; nerBudgetMs?: number; now?: () => number } = {},
+ private readonly options: { nerBudgetMs?: number } = {},
) {}
private async selectColumns(
@@ -116,8 +117,6 @@ export class SensitivityAnalysisService {
onPrepared?: (total: number) => void | Promise,
onProgress?: (processed: number, suggestions: readonly SensitivityReviewItem[]) => void | Promise,
): Promise {
- const now = this.options.now ?? Date.now;
- const deadline = now() + (this.options.runBudgetMs ?? 60_000);
const configuredNerBudget = this.options.nerBudgetMs ?? 10_000;
const nerBudget: SensitivityNerBudget = {
remainingMs: Number.isFinite(configuredNerBudget) && configuredNerBudget >= 0
@@ -137,20 +136,23 @@ export class SensitivityAnalysisService {
items.push(item);
byTable.set(item.table.id, items);
}
- const suggestions: SensitivityReviewItem[] = [];
- for (const items of byTable.values()) {
- ensureActive(signal);
+ const tableTargets = [...byTable.values()].map((items) => {
const first = items[0]!;
- const assessments = await this.classifier.assessTable({
+ return {
database,
table: first.table,
columns: items.map(({ column }) => column),
- }, signal, deadline, nerBudget);
+ };
+ });
+ const assessments = await this.classifier.assess(tableTargets, signal, nerBudget);
+ ensureActive(signal);
+ const assessmentById = new Map(assessments.map((assessment) => [
+ assessment.columnId,
+ assessment,
+ ]));
+ const suggestions: SensitivityReviewItem[] = [];
+ for (const items of byTable.values()) {
ensureActive(signal);
- const assessmentById = new Map(assessments.map((assessment) => [
- assessment.columnId,
- assessment,
- ]));
const batch = items.map(({ table, column }) => {
const assessment = assessmentById.get(column.id)!;
return {
@@ -164,6 +166,7 @@ export class SensitivityAnalysisService {
assessment: assessment.assessment,
evidence: assessment.evidence,
observedValues: assessment.observedValues,
+ coverage: assessment.coverage,
};
});
suggestions.push(...batch);
diff --git a/backend/src/catalog/sensitivity-classifier.ts b/backend/src/catalog/sensitivity-classifier.ts
index 80e044df..70383680 100644
--- a/backend/src/catalog/sensitivity-classifier.ts
+++ b/backend/src/catalog/sensitivity-classifier.ts
@@ -1,11 +1,11 @@
-import { CatalogConnectorError, type CatalogColumn, type CatalogTable, type WorkspaceDatabase } from "./types.js";
+import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "./types.js";
import { findPhoneNumbersInText } from "libphonenumber-js/max";
import validator from "validator";
-export type SensitivityAssessment = "sensitive" | "non_sensitive" | "unknown";
+export type SensitivityAssessment = "sensitive" | "non_sensitive";
export interface SensitivityEvidence {
- kind: "metadata" | "content" | "length" | "ner" | "coverage";
+ kind: "metadata" | "content" | "length" | "ner" | "coverage" | "type";
ruleId: string;
label?: string;
confidence?: number;
@@ -18,8 +18,8 @@ export interface SensitivityValueObservation {
}
export interface SensitivityScanCoverage {
- kind: "complete" | "sampled" | "unavailable";
- observedRows: number;
+ kind: "complete" | "sampled";
+ observedValues: number;
}
export interface SensitivityTableScan {
@@ -31,8 +31,11 @@ export interface SensitivityScanRequest {
database: WorkspaceDatabase;
table: CatalogTable;
columns: readonly CatalogColumn[];
- fullScanBudgetMs: number;
- deadline: number;
+ valuesPerColumn: number;
+ sampleOffset: number;
+ sampleSeed: number;
+ queryTimeoutMs: number;
+ fullScanThreshold?: number;
}
export interface SensitivityValueSource {
@@ -76,6 +79,7 @@ export interface SensitivityColumnAssessment {
proposedSensitive: boolean;
evidence: readonly SensitivityEvidence[];
observedValues: number;
+ coverage: "metadata" | "complete" | "sampled" | "no_values";
}
export interface SensitivityTableTarget {
@@ -94,7 +98,15 @@ const CREDENTIAL_NAME = /(?:^|_)(?:api_key|credential|password|passwd|private_ke
const HEALTH_NAME = /(?:^|_)(?:anamnesi|clinical|diagnos(?:i|is)|health|medical|patient|patologia|therapy|terapia)(?:_|$)/u;
const CLINICAL_TERM = /(?:^|[^\p{L}])(?:allergi[ae]|anamnesi|carcinoma|chemioterapia|diabete|diagnos[ei]|epatite|farmac[io]|gravidanza|hiv|metastasi|neoplasia|patologia|radioterapia|referto|terapia|tumore)(?:$|[^\p{L}])/iu;
const UNSUPPORTED_BINARY_TYPE = /(?:^|\s)(?:binary|blob|bytea|image|varbinary)(?:\s|$|\()/iu;
+const DEEP_TEXT_TYPE = /(?:^|\s)(?:char|character|citext|clob|json|jsonb|nchar|nvarchar|string|text|varchar|xml)(?:\s|$|\()/iu;
const MAX_NER_CANDIDATES_PER_REQUEST = 128;
+const MAX_CONCURRENT_TABLE_SCANS = 2;
+
+export const SENSITIVITY_SAMPLE_PHASES = [
+ { targetValuesPerColumn: 300, additionalValuesPerColumn: 300, sampleSeed: 37, deepTextOnly: false },
+ { targetValuesPerColumn: 1_000, additionalValuesPerColumn: 700, sampleSeed: 73, deepTextOnly: false },
+ { targetValuesPerColumn: 3_000, additionalValuesPerColumn: 2_000, sampleSeed: 109, deepTextOnly: true },
+] as const;
function normalizedName(value: string): string {
return value.normalize("NFKD")
@@ -284,14 +296,22 @@ function contentEvidence(value: string): SensitivityEvidence | undefined {
return undefined;
}
+interface ColumnState {
+ column: CatalogColumn;
+ evidence: SensitivityEvidence[];
+ observedValues: number;
+ nerCandidates: string[];
+ coverage: "metadata" | "complete" | "sampled" | "no_values";
+ sampledTarget: number;
+}
+
/** Sole decision module for local column-level sensitivity assessments. */
export class SensitivityClassifier {
constructor(
private readonly values: SensitivityValueSource,
private readonly detector?: LocalNerDetector,
private readonly options: {
- fullScanBudgetMs?: number;
- runBudgetMs?: number;
+ queryTimeoutMs?: number;
nerConfidenceThreshold?: number;
maxNerValuesPerColumn?: number;
maxNerCandidatesPerTable?: number;
@@ -299,95 +319,131 @@ export class SensitivityClassifier {
} = {},
) {}
- async assessTable(
- target: SensitivityTableTarget,
+ async assess(
+ targets: readonly SensitivityTableTarget[],
signal: AbortSignal,
- runDeadline?: number,
- nerBudget?: SensitivityNerBudget,
+ sharedNerBudget?: SensitivityNerBudget,
): Promise {
const now = this.options.now ?? Date.now;
- const deadline = runDeadline ?? now() + (this.options.runBudgetMs ?? 60_000);
- const evidence = new Map(target.columns.map((column) => {
- const match = metadataEvidence(column);
- return [column.id, match ? [match] : [] as SensitivityEvidence[]];
- }));
- const observed = new Map(target.columns.map((column) => [column.id, 0]));
- const nerCandidates = new Map(target.columns.map((column) => [column.id, [] as string[]]));
const maxNerValuesPerColumn = boundedCount(this.options.maxNerValuesPerColumn, 8, 8);
- const unsupported = new Set(target.columns
- .filter((column) => UNSUPPORTED_BINARY_TYPE.test(column.dataType))
- .map((column) => column.id));
- const scannableColumns = target.columns.filter((column) => (
- !unsupported.has(column.id) && evidence.get(column.id)!.length === 0
- ));
- let coverage: SensitivityScanCoverage = { kind: "unavailable", observedRows: 0 };
- if (scannableColumns.length > 0 && now() < deadline) {
- try {
- coverage = await this.values.scanTable({
- ...target,
- columns: scannableColumns,
- fullScanBudgetMs: this.options.fullScanBudgetMs ?? 5_000,
- deadline,
- }, (batch) => {
- for (const item of batch) {
- if (!evidence.has(item.columnId) || item.value === null) continue;
- observed.set(item.columnId, (observed.get(item.columnId) ?? 0) + 1);
- const matches = evidence.get(item.columnId)!;
- if (matches.length === 0 && (item.characterLength ?? item.value.length) > 500) {
- matches.push({ kind: "length", ruleId: "text.over_500_characters" });
- } else if (matches.length === 0) {
- const match = contentEvidence(item.value);
- if (match) matches.push(match);
- else {
- const candidates = nerCandidates.get(item.columnId)!;
- if (candidates.length < maxNerValuesPerColumn && !candidates.includes(item.value)) {
- candidates.push(item.value);
- }
- }
- }
- }
- }, signal);
- } catch (error) {
- if (!(error instanceof CatalogConnectorError)) throw error;
+ const states = new Map();
+ for (const target of targets) {
+ for (const column of target.columns) {
+ const metadataMatch = metadataEvidence(column);
+ const binary = UNSUPPORTED_BINARY_TYPE.test(column.dataType);
+ states.set(column.id, {
+ column,
+ evidence: metadataMatch
+ ? [metadataMatch]
+ : binary
+ ? [{ kind: "type", ruleId: "type.binary_uninspectable" }]
+ : [],
+ observedValues: 0,
+ nerCandidates: [],
+ coverage: metadataMatch || binary ? "metadata" : "no_values",
+ sampledTarget: 0,
+ });
}
}
- if (this.detector && (this.detector.isReady?.() ?? true) && !signal.aborted
- && now() < deadline && (nerBudget?.remainingMs ?? 1) > 0) {
- const candidates: LocalNerCandidate[] = [];
- const maxCandidates = boundedCount(this.options.maxNerCandidatesPerTable, 2, 1_024);
- candidateSelection: for (let valueIndex = 0; valueIndex < maxNerValuesPerColumn; valueIndex += 1) {
- for (const column of target.columns) {
- if (evidence.get(column.id)!.length > 0) continue;
- const text = nerCandidates.get(column.id)![valueIndex];
- if (text === undefined) continue;
- candidates.push({ columnId: column.id, text });
- if (candidates.length >= maxCandidates) break candidateSelection;
+ const completeTables = new Set();
+ for (const [phaseIndex, phase] of SENSITIVITY_SAMPLE_PHASES.entries()) {
+ for (let offset = 0; offset < targets.length; offset += MAX_CONCURRENT_TABLE_SCANS) {
+ signal.throwIfAborted();
+ const peerController = new AbortController();
+ const scanSignal = AbortSignal.any([signal, peerController.signal]);
+ try {
+ await Promise.all(targets.slice(offset, offset + MAX_CONCURRENT_TABLE_SCANS).map(async (target) => {
+ if (completeTables.has(target.table.id)) return;
+ const columns = target.columns.filter((column) => {
+ const state = states.get(column.id)!;
+ return state.evidence.length === 0
+ && (!phase.deepTextOnly || DEEP_TEXT_TYPE.test(column.dataType));
+ });
+ if (columns.length === 0) return;
+ const coverage = await this.values.scanTable({
+ ...target,
+ columns,
+ valuesPerColumn: phase.additionalValuesPerColumn,
+ sampleOffset: phase.targetValuesPerColumn - phase.additionalValuesPerColumn,
+ sampleSeed: phase.sampleSeed,
+ queryTimeoutMs: this.options.queryTimeoutMs ?? 5_000,
+ ...(phaseIndex === 0 ? { fullScanThreshold: 1_000 } : {}),
+ }, (batch) => {
+ for (const item of batch) {
+ if (item.value === null) continue;
+ const state = states.get(item.columnId);
+ if (!state || state.evidence.length > 0) continue;
+ state.observedValues += 1;
+ if ((item.characterLength ?? item.value.length) > 500) {
+ state.evidence.push({ kind: "length", ruleId: "text.over_500_characters" });
+ continue;
+ }
+ const match = contentEvidence(item.value);
+ if (match) {
+ state.evidence.push(match);
+ continue;
+ }
+ if (state.nerCandidates.length < maxNerValuesPerColumn
+ && !state.nerCandidates.includes(item.value)) {
+ state.nerCandidates.push(item.value);
+ }
+ }
+ }, scanSignal);
+ for (const column of columns) {
+ const state = states.get(column.id)!;
+ state.sampledTarget = Math.max(state.sampledTarget, phase.targetValuesPerColumn);
+ state.coverage = coverage.kind === "complete"
+ ? "complete"
+ : state.observedValues === 0 ? "no_values" : "sampled";
+ }
+ if (coverage.kind === "complete") completeTables.add(target.table.id);
+ }));
+ } catch (error) {
+ peerController.abort(error);
+ throw error;
}
}
- if (candidates.length > 0) {
- const threshold = this.options.nerConfidenceThreshold ?? 0.8;
- const nerStartedAt = now();
- const allowedNerMs = nerBudget
- ? Math.max(0, nerBudget.remainingMs)
- : Math.max(0, deadline - nerStartedAt);
- const nerDeadline = Math.min(deadline, nerStartedAt + allowedNerMs);
+ }
+
+ const nerBudget = sharedNerBudget ?? { remainingMs: 10_000 };
+ if (this.detector && (this.detector.isReady?.() ?? true) && !signal.aborted
+ && nerBudget.remainingMs > 0) {
+ const maxCandidates = boundedCount(this.options.maxNerCandidatesPerTable, 2, 1_024);
+ const threshold = this.options.nerConfidenceThreshold ?? 0.8;
+ for (const target of targets) {
+ signal.throwIfAborted();
+ if (nerBudget.remainingMs <= 0) break;
+ const candidates: LocalNerCandidate[] = [];
+ candidateSelection: for (let valueIndex = 0; valueIndex < maxNerValuesPerColumn; valueIndex += 1) {
+ for (const column of target.columns) {
+ const state = states.get(column.id)!;
+ if (state.evidence.length > 0) continue;
+ const text = state.nerCandidates[valueIndex];
+ if (text === undefined) continue;
+ candidates.push({ columnId: column.id, text });
+ if (candidates.length >= maxCandidates) break candidateSelection;
+ }
+ }
+ if (candidates.length === 0) continue;
+ const startedAt = now();
+ const deadline = startedAt + nerBudget.remainingMs;
try {
for (let offset = 0; offset < candidates.length; offset += MAX_NER_CANDIDATES_PER_REQUEST) {
- if (signal.aborted || now() >= nerDeadline) break;
+ if (signal.aborted || now() >= deadline) break;
try {
const detected = await this.detector.detect(
candidates.slice(offset, offset + MAX_NER_CANDIDATES_PER_REQUEST),
signal,
- nerDeadline,
+ deadline,
);
for (const item of detected) {
- const matches = evidence.get(item.columnId);
- if (!matches || matches.length > 0 || !Number.isFinite(item.confidence)
+ const state = states.get(item.columnId);
+ if (!state || state.evidence.length > 0 || !Number.isFinite(item.confidence)
|| item.confidence < threshold || item.confidence > 1) continue;
const label = normalizedName(item.label).slice(0, 80);
if (!label) continue;
- matches.push({
+ state.evidence.push({
kind: "ner",
ruleId: "ner.entity",
label,
@@ -400,42 +456,43 @@ export class SensitivityClassifier {
}
}
} finally {
- if (nerBudget) {
- const elapsedMs = Math.max(1, now() - nerStartedAt);
- nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - elapsedMs);
- }
+ nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - Math.max(1, now() - startedAt));
}
}
}
- return target.columns.map((column) => {
- const matches = evidence.get(column.id)!;
- const count = observed.get(column.id) ?? 0;
- const assessment: SensitivityAssessment = matches.length > 0
- ? "sensitive"
- : unsupported.has(column.id) || count === 0 || coverage.kind !== "complete"
- ? "unknown"
- : "non_sensitive";
+ return targets.flatMap((target) => target.columns.map((column) => {
+ const state = states.get(column.id)!;
+ const sensitive = state.evidence.length > 0;
+ const coverage = state.observedValues === 0 && !sensitive ? "no_values" : state.coverage;
+ const coverageEvidence: SensitivityEvidence[] = sensitive
+ ? state.evidence
+ : [{
+ kind: "coverage",
+ ruleId: coverage === "complete"
+ ? "coverage.complete"
+ : coverage === "no_values"
+ ? "coverage.no_values"
+ : `coverage.sampled_${state.sampledTarget}`,
+ }];
return {
columnId: column.id,
- assessment,
- proposedSensitive: assessment === "unknown" ? column.sensitive : assessment === "sensitive",
- evidence: matches.length > 0
- ? matches
- : assessment === "unknown"
- ? [{
- kind: "coverage",
- ruleId: unsupported.has(column.id)
- ? "coverage.unsupported_type"
- : coverage.kind === "unavailable"
- ? "coverage.unavailable"
- : count === 0
- ? "coverage.no_values"
- : "coverage.incomplete",
- }]
- : [],
- observedValues: count,
+ assessment: sensitive ? "sensitive" : "non_sensitive",
+ proposedSensitive: sensitive,
+ evidence: coverageEvidence,
+ observedValues: state.observedValues,
+ coverage,
};
- });
+ }));
+ }
+
+ /** Convenience for focused callers and rule-level tests. Production orchestration uses assess(). */
+ async assessTable(
+ target: SensitivityTableTarget,
+ signal: AbortSignal,
+ _retiredRunDeadline?: number,
+ nerBudget?: SensitivityNerBudget,
+ ): Promise {
+ return await this.assess([target], signal, nerBudget);
}
}
diff --git a/backend/src/catalog/sensitivity-shadow.ts b/backend/src/catalog/sensitivity-shadow.ts
index e7e3b18c..aeb34101 100644
--- a/backend/src/catalog/sensitivity-shadow.ts
+++ b/backend/src/catalog/sensitivity-shadow.ts
@@ -5,7 +5,10 @@ import { WorkspaceSecretStore } from "../workspaces/secret-store.js";
import { PythonLocalNerDetector } from "./local-ner-detector.js";
import { ConcreteCatalogPostgresAccess } from "./postgres-access.js";
import { createCatalogRepository } from "./repository.js";
-import { SensitivityAnalysisService } from "./sensitivity-analysis-service.js";
+import {
+ SENSITIVITY_POLICY_VERSION,
+ SensitivityAnalysisService,
+} from "./sensitivity-analysis-service.js";
import { SensitivityClassifier } from "./sensitivity-classifier.js";
import { ConcreteSensitivityValueSource } from "./sensitivity-value-source.js";
@@ -60,23 +63,26 @@ async function main(): Promise {
const suggestions = await new SensitivityAnalysisService(
repository,
new SensitivityClassifier(source, detector),
- ).analyze(database.id, "all", [], AbortSignal.timeout(65_000));
- const assessments = { sensitive: 0, nonSensitive: 0, unknown: 0 };
+ ).analyze(database.id, "all", [], new AbortController().signal);
+ const assessments = { sensitive: 0, nonSensitive: 0 };
+ const coverage = { metadata: 0, complete: 0, sampled: 0, noValues: 0 };
const rules = new Map();
for (const suggestion of suggestions) {
if (suggestion.assessment === "sensitive") assessments.sensitive += 1;
- else if (suggestion.assessment === "non_sensitive") assessments.nonSensitive += 1;
- else assessments.unknown += 1;
+ else assessments.nonSensitive += 1;
+ if (suggestion.coverage === "no_values") coverage.noValues += 1;
+ else coverage[suggestion.coverage] += 1;
for (const evidence of suggestion.evidence) {
rules.set(evidence.ruleId, (rules.get(evidence.ruleId) ?? 0) + 1);
}
}
process.stdout.write(`${JSON.stringify({
ok: true,
- policyVersion: "sensitivity-v1",
+ policyVersion: SENSITIVITY_POLICY_VERSION,
nerEnabled: detector !== undefined,
total: suggestions.length,
assessments,
+ coverage,
rules: Object.fromEntries([...rules].sort(([left], [right]) => left.localeCompare(right))),
elapsedMs: Date.now() - startedAt,
})}\n`);
diff --git a/backend/src/catalog/sensitivity-value-source.ts b/backend/src/catalog/sensitivity-value-source.ts
index d06d665b..24b73fac 100644
--- a/backend/src/catalog/sensitivity-value-source.ts
+++ b/backend/src/catalog/sensitivity-value-source.ts
@@ -8,82 +8,117 @@ import type {
SensitivityValueObservation,
SensitivityValueSource,
} from "./sensitivity-classifier.js";
-import { CatalogConnectorError } from "./types.js";
+import { CatalogConnectorError, type CatalogColumn } from "./types.js";
const MAX_VALUE_CHARACTERS = 501;
-const DEFAULT_BATCH_ROWS = 200;
-const DEFAULT_SAMPLE_ROWS = 200;
+const MAX_COLUMNS_PER_QUERY = 25;
+const SAMPLE_OVERSCAN_FACTOR = 10;
function quoteIdentifier(identifier: string): string {
return `"${identifier.replaceAll('"', '""')}"`;
}
-function projections(request: SensitivityScanRequest): string {
- return request.columns.flatMap((column, index) => {
+function chunks(items: readonly T[], size: number): T[][] {
+ const result: T[][] = [];
+ for (let offset = 0; offset < items.length; offset += size) {
+ result.push(items.slice(offset, offset + size));
+ }
+ return result;
+}
+
+function tableReference(request: SensitivityScanRequest): string {
+ return `${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`;
+}
+
+function samplePercentage(valuesPerColumn: number): number {
+ if (valuesPerColumn <= 300) return 30;
+ if (valuesPerColumn <= 700) return 70;
+ return 100;
+}
+
+function flatValueQuery(
+ request: SensitivityScanRequest,
+ columns: readonly CatalogColumn[],
+ options: { complete: boolean; randomized: boolean },
+): string {
+ const projections = columns.map((column) => quoteIdentifier(column.name)).join(", ");
+ const perColumnLimit = options.complete
+ ? request.fullScanThreshold ?? request.valuesPerColumn
+ : request.valuesPerColumn;
+ const rowLimit = Math.max(perColumnLimit, perColumnLimit * SAMPLE_OVERSCAN_FACTOR);
+ const sample = options.complete
+ ? `SELECT ${projections} FROM ${tableReference(request)}`
+ : [
+ `SELECT ${projections} FROM ${tableReference(request)}`,
+ ...(options.randomized
+ ? [`TABLESAMPLE SYSTEM (${samplePercentage(request.valuesPerColumn)}) REPEATABLE (${request.sampleSeed})`]
+ : []),
+ `LIMIT ${rowLimit} OFFSET ${request.sampleOffset}`,
+ ].join(" ");
+ const values = columns.map((column, index) => {
const identifier = quoteIdentifier(column.name);
return [
- `LEFT((${identifier})::text, ${MAX_VALUE_CHARACTERS}) AS "__value_${index}"`,
- `CASE WHEN ${identifier} IS NULL THEN NULL ELSE char_length((${identifier})::text) END AS "__length_${index}"`,
- ];
+ `(${index}, LEFT((sampled.${identifier})::text, ${MAX_VALUE_CHARACTERS}),`,
+ `CASE WHEN sampled.${identifier} IS NULL THEN NULL`,
+ `ELSE char_length((sampled.${identifier})::text) END)`,
+ ].join(" ");
}).join(", ");
+ return [
+ `WITH sampled AS MATERIALIZED (${sample}),`,
+ "ranked AS (",
+ "SELECT value.__column_index, value.__value, value.__length,",
+ "row_number() OVER (PARTITION BY value.__column_index) AS __rank",
+ "FROM sampled",
+ `CROSS JOIN LATERAL (VALUES ${values}) AS value(__column_index, __value, __length)`,
+ "WHERE value.__value IS NOT NULL",
+ ")",
+ "SELECT __column_index, __value, __length FROM ranked",
+ `WHERE __rank <= ${perColumnLimit}`,
+ ].join(" ");
}
function observations(
- request: SensitivityScanRequest,
+ columns: readonly CatalogColumn[],
rows: readonly Record[],
): SensitivityValueObservation[] {
- return rows.flatMap((row) => request.columns.map((column, index) => {
- const sourceValue = row[`__value_${index}`];
- const sourceLength = row[`__length_${index}`];
- const value = sourceValue === null || sourceValue === undefined ? null : String(sourceValue);
- const parsedLength = sourceLength === null || sourceLength === undefined
+ return rows.flatMap((row) => {
+ const index = Number(row.__column_index);
+ const column = Number.isSafeInteger(index) && index >= 0 ? columns[index] : undefined;
+ if (!column || row.__value === null || row.__value === undefined) return [];
+ const value = String(row.__value);
+ const parsedLength = row.__length === null || row.__length === undefined
? null
- : Number(sourceLength);
- return {
+ : Number(row.__length);
+ return [{
columnId: column.id,
value,
characterLength: parsedLength !== null && Number.isSafeInteger(parsedLength) && parsedLength >= 0
? parsedLength
- : value?.length ?? null,
- };
- }));
+ : value.length,
+ }];
+ });
}
function cancelled(error: unknown): boolean {
return Boolean(error && typeof error === "object" && "code" in error && error.code === "57014");
}
-interface SensitivityValueSourceOptions {
- now?: () => number;
- batchRows?: number;
- sampleRows?: number;
-}
-
/**
- * PostgreSQL value adapter. It owns bounded read mechanics and emits normalized values, never a
- * sensitivity decision.
+ * Database-specific sampling adapter. Policy stays in SensitivityClassifier; this module only
+ * produces bounded, normalized non-null observations without persisting or logging values.
*/
export class ConcreteSensitivityValueSource implements SensitivityValueSource {
- private readonly now: () => number;
- private readonly batchRows: number;
- private readonly sampleRows: number;
-
constructor(
private readonly access: CatalogPostgresAccess,
private readonly secretStore?: Pick,
- options: SensitivityValueSourceOptions = {},
- ) {
- this.now = options.now ?? Date.now;
- this.batchRows = options.batchRows ?? DEFAULT_BATCH_ROWS;
- this.sampleRows = options.sampleRows ?? DEFAULT_SAMPLE_ROWS;
- }
+ ) {}
async scanTable(
request: SensitivityScanRequest,
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise,
signal: AbortSignal,
): Promise {
- if (request.columns.length === 0) return { kind: "unavailable", observedRows: 0 };
+ if (request.columns.length === 0) return { kind: "complete", observedValues: 0 };
if (request.database.binding.transport === "rest_api") {
return await this.scanRest(request, consume, signal);
}
@@ -97,69 +132,63 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
): Promise {
const client = await this.access.connect(request.database, signal);
let transactionOpen = false;
- const startedAt = this.now();
- const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
- let observedRows = 0;
- let cursorOpen = false;
+ let savepointSequence = 0;
+ let observedValues = 0;
try {
- if (signal.aborted || this.now() >= request.deadline) {
- return { kind: "sampled", observedRows: 0 };
- }
+ signal.throwIfAborted();
await client.query("BEGIN TRANSACTION READ ONLY", []);
transactionOpen = true;
await client.query("SELECT set_config('statement_timeout', $1, true)", [
- `${Math.max(1, Math.floor(fullDeadline - startedAt))}ms`,
+ `${Math.max(1, Math.floor(request.queryTimeoutMs))}ms`,
]);
- await client.query("SAVEPOINT sensitivity_full_scan", []);
- const cursor = [
- "DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR",
- `SELECT ${projections(request)}`,
- `FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
- ].join(" ");
- await client.query(cursor, []);
- cursorOpen = true;
- while (!signal.aborted && this.now() < fullDeadline) {
- let rows: Array>;
+ const boundedQuery = async (sql: string): Promise> | undefined> => {
+ signal.throwIfAborted();
+ savepointSequence += 1;
+ const savepoint = `sensitivity_scan_${savepointSequence}`;
+ await client.query(`SAVEPOINT ${savepoint}`, []);
try {
- await client.query("SELECT set_config('statement_timeout', $1, true)", [
- `${Math.max(1, Math.floor(fullDeadline - this.now()))}ms`,
- ]);
- rows = (await client.query(
- `FETCH FORWARD ${this.batchRows} FROM sensitivity_full_scan_cursor`,
- [],
- )).rows;
+ return (await client.query(sql, [])).rows;
} catch (error) {
if (!cancelled(error)) throw error;
- await client.query("ROLLBACK TO SAVEPOINT sensitivity_full_scan", []);
- cursorOpen = false;
- break;
- }
- if (rows.length > 0) {
- observedRows += rows.length;
- await consume(observations(request, rows));
- }
- if (rows.length < this.batchRows) {
- return { kind: "complete", observedRows };
+ await client.query(`ROLLBACK TO SAVEPOINT ${savepoint}`, []);
+ return undefined;
+ } finally {
+ await client.query(`RELEASE SAVEPOINT ${savepoint}`, []).catch(() => undefined);
}
+ };
+
+ let complete = false;
+ if (request.fullScanThreshold !== undefined) {
+ const probe = await boundedQuery(
+ `SELECT 1 AS __present FROM ${tableReference(request)} LIMIT ${request.fullScanThreshold + 1}`,
+ );
+ complete = probe !== undefined && probe.length <= request.fullScanThreshold;
}
- if (signal.aborted || this.now() >= request.deadline) {
- return { kind: "sampled", observedRows };
+ for (const columnChunk of chunks(request.columns, MAX_COLUMNS_PER_QUERY)) {
+ signal.throwIfAborted();
+ let rows = await boundedQuery(flatValueQuery(request, columnChunk, {
+ complete,
+ randomized: !complete,
+ }));
+ if (rows === undefined && complete) {
+ complete = false;
+ rows = await boundedQuery(flatValueQuery(request, columnChunk, {
+ complete: false,
+ randomized: true,
+ }));
+ }
+ if (!complete && (rows === undefined || rows.length === 0)) {
+ rows = await boundedQuery(flatValueQuery(request, columnChunk, {
+ complete: false,
+ randomized: false,
+ }));
+ }
+ if (rows === undefined) throw new CatalogConnectorError("Sensitivity sample query timed out");
+ const batch = observations(columnChunk, rows);
+ observedValues += batch.length;
+ if (batch.length > 0) await consume(batch);
}
- if (cursorOpen) await client.query("CLOSE sensitivity_full_scan_cursor", []);
- await client.query("RELEASE SAVEPOINT sensitivity_full_scan", []);
- await client.query("SELECT set_config('statement_timeout', $1, true)", [
- `${Math.max(1, Math.floor(request.deadline - this.now()))}ms`,
- ]);
- const sampleSql = [
- `SELECT ${projections(request)}`,
- `FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
- "TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
- "LIMIT $1",
- ].join(" ");
- const sampledRows = (await client.query(sampleSql, [this.sampleRows])).rows;
- observedRows += sampledRows.length;
- if (sampledRows.length > 0) await consume(observations(request, sampledRows));
- return { kind: "sampled", observedRows };
+ return { kind: complete ? "complete" : "sampled", observedValues };
} catch (error) {
if (error instanceof CatalogConnectorError) throw error;
throw new CatalogConnectorError("Sensitivity source scan failed");
@@ -180,9 +209,7 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
request.database.workspaceId,
auth === "none" ? [] : [CATALOG_SECRET_IDS.apiKey],
);
- const startedAt = this.now();
- const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
- let observedRows = 0;
+ let observedValues = 0;
try {
const headers: Record = { "content-type": "application/json" };
if (auth !== "none") {
@@ -194,61 +221,66 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
}
const baseUrl = request.database.binding.baseUrl?.replace(/\/+$/u, "");
if (!baseUrl) throw new CatalogConnectorError("Database binding is incomplete");
- const runQuery = async (sql: string, deadline: number): Promise>> => {
- const response = await fetch(`${baseUrl}/rpc/run_query`, {
- method: "POST",
- headers,
- body: JSON.stringify({ query_text: sql }),
- signal: AbortSignal.any([
- signal,
- AbortSignal.timeout(Math.max(1, Math.floor(deadline - this.now()))),
- ]),
- });
- if (!response.ok) throw new CatalogConnectorError("REST sensitivity source scan failed");
- const body: unknown = await response.json();
- if (!Array.isArray(body)
- || body.some((row) => !row || typeof row !== "object" || Array.isArray(row))) {
- throw new CatalogConnectorError("REST sensitivity source response is invalid");
+ const runQuery = async (sql: string): Promise> | undefined> => {
+ const timeout = AbortSignal.timeout(Math.max(1, Math.floor(request.queryTimeoutMs)));
+ try {
+ const response = await fetch(`${baseUrl}/rpc/run_query`, {
+ method: "POST",
+ headers,
+ body: JSON.stringify({ query_text: sql }),
+ signal: AbortSignal.any([signal, timeout]),
+ });
+ if (!response.ok) throw new CatalogConnectorError("REST sensitivity source scan failed");
+ const body: unknown = await response.json();
+ if (!Array.isArray(body)
+ || body.some((row) => !row || typeof row !== "object" || Array.isArray(row))) {
+ throw new CatalogConnectorError("REST sensitivity source response is invalid");
+ }
+ return body as Array>;
+ } catch (error) {
+ if (signal.aborted) throw error;
+ if (timeout.aborted) return undefined;
+ throw error;
}
- return body as Array>;
};
- let offset = 0;
- const baseSelect = [
- `SELECT ${projections(request)}`,
- `FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
- ].join(" ");
- while (!signal.aborted) {
- let rows: Array>;
- try {
- rows = await runQuery(
- `${baseSelect} LIMIT ${this.batchRows} OFFSET ${offset}`,
- fullDeadline,
- );
- } catch (error) {
- if (signal.aborted || this.now() < fullDeadline) throw error;
- break;
- }
- observedRows += rows.length;
- if (rows.length > 0) await consume(observations(request, rows));
- if (rows.length < this.batchRows) {
- return { kind: offset === 0 ? "complete" : "sampled", observedRows };
- }
- offset += rows.length;
- if (this.now() >= fullDeadline) break;
+ let complete = false;
+ if (request.fullScanThreshold !== undefined) {
+ const probe = await runQuery(
+ `SELECT 1 AS __present FROM ${tableReference(request)} LIMIT ${request.fullScanThreshold + 1}`,
+ );
+ complete = probe !== undefined && probe.length <= request.fullScanThreshold;
}
- if (signal.aborted || this.now() >= request.deadline) {
- return { kind: "sampled", observedRows };
+ let requestCount = request.fullScanThreshold === undefined ? 0 : 1;
+ for (const columnChunk of chunks(request.columns, MAX_COLUMNS_PER_QUERY)) {
+ signal.throwIfAborted();
+ let rows = await runQuery(flatValueQuery(request, columnChunk, {
+ complete,
+ randomized: !complete,
+ }));
+ requestCount += 1;
+ if (rows === undefined && complete) {
+ complete = false;
+ rows = await runQuery(flatValueQuery(request, columnChunk, {
+ complete: false,
+ randomized: true,
+ }));
+ requestCount += 1;
+ }
+ if (!complete && (rows === undefined || rows.length === 0)) {
+ rows = await runQuery(flatValueQuery(request, columnChunk, {
+ complete: false,
+ randomized: false,
+ }));
+ requestCount += 1;
+ }
+ if (rows === undefined) throw new CatalogConnectorError("REST sensitivity sample query timed out");
+ const batch = observations(columnChunk, rows);
+ observedValues += batch.length;
+ if (batch.length > 0) await consume(batch);
}
- const sampleSql = [
- baseSelect,
- "TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
- `LIMIT ${this.sampleRows}`,
- ].join(" ");
- const sampledRows = await runQuery(sampleSql, request.deadline);
- observedRows += sampledRows.length;
- if (sampledRows.length > 0) await consume(observations(request, sampledRows));
- return { kind: "sampled", observedRows };
+ // Multiple HTTP requests cannot share a source snapshot, so only one-request reads are complete.
+ return { kind: complete && requestCount === 1 ? "complete" : "sampled", observedValues };
} catch (error) {
if (error instanceof CatalogConnectorError) throw error;
throw new CatalogConnectorError("REST sensitivity source scan failed");
diff --git a/backend/src/routes/catalog-description-generation.ts b/backend/src/routes/catalog-description-generation.ts
index 82af82bf..8103ab44 100644
--- a/backend/src/routes/catalog-description-generation.ts
+++ b/backend/src/routes/catalog-description-generation.ts
@@ -261,9 +261,9 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
});
}
if (error instanceof SensitivityAnalysisInterruptedError) {
- return reply.code(504).send({
- code: "sensitivity_analysis_timeout",
- message: "Sensitivity analysis reached its time limit. No assessments were applied.",
+ return reply.code(499).send({
+ code: "sensitivity_analysis_interrupted",
+ message: "Sensitivity analysis was interrupted before completion. No assessments were applied.",
});
}
if (error instanceof CatalogConnectorError) {
@@ -284,32 +284,6 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
});
}
-function untilAborted(operation: Promise, signal: AbortSignal): Promise {
- if (signal.aborted) {
- void operation.catch(() => undefined);
- return Promise.reject(new SensitivityAnalysisInterruptedError());
- }
- return new Promise((resolve, reject) => {
- const abort = () => reject(new SensitivityAnalysisInterruptedError());
- signal.addEventListener("abort", abort, { once: true });
- if (signal.aborted) {
- void operation.catch(() => undefined);
- abort();
- return;
- }
- operation.then(
- (value) => {
- signal.removeEventListener("abort", abort);
- resolve(value);
- },
- (error: unknown) => {
- signal.removeEventListener("abort", abort);
- reject(error);
- },
- );
- });
-}
-
function safeSuggestionHistoryError(reply: FastifyReply, error: unknown) {
if (error instanceof CatalogUnavailableError) {
return reply.code(503).send({
@@ -342,13 +316,22 @@ export function catalogDescriptionGenerationRoutes(
try {
const databaseId = idSchema.parse((request.params as { databaseId?: unknown }).databaseId);
const input = suggestionSchema.parse(request.body);
- const signal = AbortSignal.timeout(60_000);
- const result = await untilAborted(deps.sensitivityAnalysisRunner.run(
- databaseId,
- input.scope,
- "targetIds" in input ? input.targetIds : [],
- signal,
- ), signal);
+ const controller = new AbortController();
+ const abort = () => controller.abort();
+ request.raw.once("aborted", abort);
+ reply.raw.once("close", abort);
+ let result;
+ try {
+ result = await deps.sensitivityAnalysisRunner.run(
+ databaseId,
+ input.scope,
+ "targetIds" in input ? input.targetIds : [],
+ controller.signal,
+ );
+ } finally {
+ request.raw.off("aborted", abort);
+ reply.raw.off("close", abort);
+ }
return {
suggestions: result.suggestions,
run: publicSensitivityAnalysisRun(result.run),
diff --git a/backend/test/catalog-description-generation-routes.test.ts b/backend/test/catalog-description-generation-routes.test.ts
index 9bed0500..ff8b395e 100644
--- a/backend/test/catalog-description-generation-routes.test.ts
+++ b/backend/test/catalog-description-generation-routes.test.ts
@@ -77,7 +77,7 @@ async function setup(
value: "ordinary",
characterLength: 8,
})));
- return { kind: "complete", observedRows: 1 };
+ return { kind: "complete", observedValues: 1 };
}),
},
) {
@@ -167,7 +167,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
scope: "all",
engine: "local",
modelId: null,
- policyVersion: "sensitivity-v1",
+ policyVersion: "sensitivity-v2",
status: "completed",
total: 1,
suggestedSensitive: 1,
@@ -185,6 +185,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
sensitive: true,
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
+ coverage: "metadata",
}],
});
expect(await repository.getColumn(database.id, column.tableId, column.id))
@@ -239,10 +240,8 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
}
});
-test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => {
- const controller = new AbortController();
- controller.abort();
- const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal);
+test("does not impose a global HTTP deadline on sensitivity analysis", async () => {
+ const timeout = vi.spyOn(AbortSignal, "timeout");
const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") });
try {
@@ -252,12 +251,9 @@ test("stops sensitivity analysis at the HTTP deadline without creating a review"
payload: { scope: "all" },
});
- expect(response.statusCode).toBe(504);
- expect(response.json()).toEqual({
- code: "sensitivity_analysis_timeout",
- message: "Sensitivity analysis reached its time limit. No assessments were applied.",
- });
- expect(await repository.listSensitivityAnalysisRuns()).toEqual([]);
+ expect(response.statusCode).toBe(200);
+ expect(timeout).not.toHaveBeenCalled();
+ expect(await repository.listSensitivityAnalysisRuns()).toHaveLength(1);
} finally {
timeout.mockRestore();
await app.close();
diff --git a/backend/test/catalog-sensitivity-analysis.test.ts b/backend/test/catalog-sensitivity-analysis.test.ts
index 033dc079..f441f486 100644
--- a/backend/test/catalog-sensitivity-analysis.test.ts
+++ b/backend/test/catalog-sensitivity-analysis.test.ts
@@ -6,7 +6,9 @@ import {
import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js";
import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js";
import type {
+ CatalogColumn,
CatalogRepository,
+ CatalogTable,
SensitivityAnalysisRun,
WorkspaceDatabase,
} from "../src/catalog/types.js";
@@ -24,13 +26,54 @@ const database = {
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
} satisfies WorkspaceDatabase;
+function catalogTable(id: string, name: string): CatalogTable {
+ return {
+ id,
+ databaseId: database.id,
+ name,
+ sourceComment: null,
+ description: null,
+ generatedDescription: null,
+ lastSyncedDatabaseVersion: 1,
+ lastSyncedAt: "2026-09-02T08:00:00Z",
+ version: 1,
+ createdAt: "2026-09-02T08:00:00Z",
+ updatedAt: "2026-09-02T08:00:00Z",
+ };
+}
+
+function catalogColumn(id: string, tableId: string, name: string): CatalogColumn {
+ return {
+ id,
+ tableId,
+ name,
+ ordinalPosition: 1,
+ dataType: "text",
+ isNullable: true,
+ defaultExpression: null,
+ primaryKeyPosition: null,
+ isPrimaryKey: false,
+ isForeignKey: false,
+ foreignKeyCount: 0,
+ sourceComment: null,
+ description: null,
+ generatedDescription: null,
+ sensitive: false,
+ lastSyncedDatabaseVersion: 1,
+ lastSyncedAt: "2026-09-02T08:00:00Z",
+ version: 1,
+ createdAt: "2026-09-02T08:00:00Z",
+ updatedAt: "2026-09-02T08:00:00Z",
+ };
+}
+
const running: SensitivityAnalysisRun = {
id: "22222222-2222-4222-8222-222222222222",
databaseId: database.id,
scope: "all",
engine: "local",
modelId: null,
- policyVersion: "sensitivity-v1",
+ policyVersion: "sensitivity-v2",
status: "running",
total: 0,
suggestedSensitive: 0,
@@ -56,7 +99,7 @@ test("stops catalog selection when the request expires during a catalog read", a
}),
listTables,
} as unknown as CatalogRepository;
- const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier;
+ const classifier = { assess: vi.fn() } as unknown as SensitivityClassifier;
const analysis = new SensitivityAnalysisService(repository, classifier);
await expect(analysis.analyze(
@@ -66,7 +109,71 @@ test("stops catalog selection when the request expires during a catalog read", a
controller.signal,
)).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
expect(listTables).not.toHaveBeenCalled();
- expect(classifier.assessTable).not.toHaveBeenCalled();
+ expect(classifier.assess).not.toHaveBeenCalled();
+});
+
+test("classifies all selected tables in one breadth-first run and reports coverage", async () => {
+ const firstTable = catalogTable("33333333-3333-4333-8333-333333333333", "patients");
+ const secondTable = catalogTable("44444444-4444-4444-8444-444444444444", "encounters");
+ const firstColumn = catalogColumn(
+ "55555555-5555-4555-8555-555555555555",
+ firstTable.id,
+ "status",
+ );
+ const secondColumn = catalogColumn(
+ "66666666-6666-4666-8666-666666666666",
+ secondTable.id,
+ "note",
+ );
+ const repository = {
+ get: vi.fn(async () => database),
+ listTables: vi.fn(async () => [firstTable, secondTable]),
+ listColumns: vi.fn(async (_databaseId: string, tableId: string) => (
+ tableId === firstTable.id ? [firstColumn] : [secondColumn]
+ )),
+ } as unknown as CatalogRepository;
+ const assess = vi.fn(async () => [
+ {
+ columnId: firstColumn.id,
+ assessment: "non_sensitive" as const,
+ proposedSensitive: false,
+ evidence: [{ kind: "coverage" as const, ruleId: "coverage.sampled_1000" }],
+ observedValues: 1_000,
+ coverage: "sampled" as const,
+ },
+ {
+ columnId: secondColumn.id,
+ assessment: "sensitive" as const,
+ proposedSensitive: true,
+ evidence: [{ kind: "content" as const, ruleId: "pii.email" }],
+ observedValues: 12,
+ coverage: "sampled" as const,
+ },
+ ]);
+ const classifier = { assess } as unknown as SensitivityClassifier;
+ const onPrepared = vi.fn();
+ const onProgress = vi.fn();
+
+ const suggestions = await new SensitivityAnalysisService(repository, classifier).analyze(
+ database.id,
+ "all",
+ [],
+ new AbortController().signal,
+ onPrepared,
+ onProgress,
+ );
+
+ expect(assess).toHaveBeenCalledOnce();
+ expect(assess.mock.calls[0]![0]).toEqual([
+ { database, table: firstTable, columns: [firstColumn] },
+ { database, table: secondTable, columns: [secondColumn] },
+ ]);
+ expect(onPrepared).toHaveBeenCalledWith(2);
+ expect(onProgress.mock.calls.map(([processed]) => processed)).toEqual([1, 2]);
+ expect(suggestions).toEqual([
+ expect.objectContaining({ columnId: firstColumn.id, sensitive: false, coverage: "sampled" }),
+ expect.objectContaining({ columnId: secondColumn.id, sensitive: true, coverage: "sampled" }),
+ ]);
});
test("marks a created run interrupted if the request deadline expires during persistence", async () => {
@@ -94,6 +201,6 @@ test("marks a created run interrupted if the request deadline expires during per
status: "interrupted",
total: 0,
unknown: 0,
- errorSummary: "Local sensitivity analysis reached its time limit.",
+ errorSummary: "Local sensitivity analysis was interrupted before completion.",
}));
});
diff --git a/backend/test/catalog-sensitivity-classifier.test.ts b/backend/test/catalog-sensitivity-classifier.test.ts
index 15dcf7be..3692ab75 100644
--- a/backend/test/catalog-sensitivity-classifier.test.ts
+++ b/backend/test/catalog-sensitivity-classifier.test.ts
@@ -76,7 +76,7 @@ test("one email hidden in a generically named column makes the whole column sens
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
]],
- coverage: { kind: "complete", observedRows: 2 },
+ coverage: { kind: "complete", observedValues: 2 },
});
const classifier = new SensitivityClassifier(values);
@@ -97,7 +97,7 @@ test("one text value longer than 500 characters makes the whole column sensitive
const target = column({ name: "comment" });
const values = source({
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
- coverage: { kind: "sampled", observedRows: 1 },
+ coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -112,7 +112,151 @@ test("one text value longer than 500 characters makes the whole column sensitive
});
});
-test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
+test("scans every table at 300 before advancing to 1,000 and 3,000 values", async () => {
+ const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
+ const first = column({ name: "status" });
+ const second = column({
+ id: "88888888-8888-4888-8888-888888888888",
+ tableId: otherTable.id,
+ name: "comment",
+ });
+ const calls: string[] = [];
+ const values: SensitivityValueSource = {
+ scanTable: vi.fn(async (request, consume) => {
+ calls.push(`${request.table.name}:${request.valuesPerColumn}:${request.sampleOffset}`);
+ await consume(request.columns.map((item) => ({
+ columnId: item.id,
+ value: "ordinary",
+ characterLength: 8,
+ })));
+ return { kind: "sampled", observedValues: request.columns.length };
+ }),
+ };
+
+ await new SensitivityClassifier(values).assess([
+ { database, table, columns: [first] },
+ { database, table: otherTable, columns: [second] },
+ ], new AbortController().signal);
+
+ expect(calls).toEqual([
+ "observations:300:0",
+ "events:300:0",
+ "observations:700:300",
+ "events:700:300",
+ "observations:2000:1000",
+ "events:2000:1000",
+ ]);
+});
+
+test("runs at most two table scans concurrently", async () => {
+ const targets = Array.from({ length: 3 }, (_, index) => {
+ const targetTable = {
+ ...table,
+ id: `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
+ name: `table_${index + 1}`,
+ };
+ return {
+ database,
+ table: targetTable,
+ columns: [column({
+ id: `10000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
+ tableId: targetTable.id,
+ name: `attribute_${index + 1}`,
+ })],
+ };
+ });
+ let active = 0;
+ let maximum = 0;
+ const values: SensitivityValueSource = {
+ scanTable: vi.fn(async () => {
+ active += 1;
+ maximum = Math.max(maximum, active);
+ await Promise.resolve();
+ active -= 1;
+ return { kind: "complete", observedValues: 0 };
+ }),
+ };
+
+ await new SensitivityClassifier(values).assess(targets, new AbortController().signal);
+
+ expect(maximum).toBe(2);
+ expect(values.scanTable).toHaveBeenCalledTimes(3);
+});
+
+test("aborts a peer table scan when another concurrent source scan fails", async () => {
+ const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
+ const first = column({ name: "status" });
+ const second = column({
+ id: "88888888-8888-4888-8888-888888888888",
+ tableId: otherTable.id,
+ name: "comment",
+ });
+ let peerSignal: AbortSignal | undefined;
+ const failure = new CatalogConnectorError("source unavailable");
+ const values: SensitivityValueSource = {
+ scanTable: vi.fn(async (request, _consume, scanSignal) => {
+ if (request.table.id === table.id) {
+ await Promise.resolve();
+ throw failure;
+ }
+ peerSignal = scanSignal;
+ return await new Promise((_resolve, reject) => {
+ scanSignal.addEventListener("abort", () => reject(scanSignal.reason), { once: true });
+ });
+ }),
+ };
+
+ await expect(new SensitivityClassifier(values).assess([
+ { database, table, columns: [first] },
+ { database, table: otherTable, columns: [second] },
+ ], new AbortController().signal)).rejects.toBe(failure);
+
+ expect(peerSignal?.aborted).toBe(true);
+});
+
+test("stops sampling a column as soon as one value is sensitive", async () => {
+ const target = column();
+ const values: SensitivityValueSource = {
+ scanTable: vi.fn(async (request, consume) => {
+ await consume([{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 }]);
+ return { kind: "sampled", observedValues: 1 };
+ }),
+ };
+
+ const [assessment] = await new SensitivityClassifier(values).assessTable(
+ { database, table, columns: [target] },
+ new AbortController().signal,
+ );
+
+ expect(values.scanTable).toHaveBeenCalledOnce();
+ expect(assessment).toMatchObject({ assessment: "sensitive", proposedSensitive: true });
+});
+
+test("stops non-text columns after the 1,000-value stage", async () => {
+ const target = column({ dataType: "integer", name: "sequence_number" });
+ const values: SensitivityValueSource = {
+ scanTable: vi.fn(async (request, consume) => {
+ await consume([{ columnId: target.id, value: "42", characterLength: 2 }]);
+ return { kind: "sampled", observedValues: 1 };
+ }),
+ };
+
+ const [assessment] = await new SensitivityClassifier(values).assessTable(
+ { database, table, columns: [target] },
+ new AbortController().signal,
+ );
+
+ expect(vi.mocked(values.scanTable).mock.calls.map(([request]) => request.valuesPerColumn))
+ .toEqual([300, 700]);
+ expect(assessment).toMatchObject({
+ assessment: "non_sensitive",
+ proposedSensitive: false,
+ coverage: "sampled",
+ evidence: [{ kind: "coverage", ruleId: "coverage.sampled_1000" }],
+ });
+});
+
+test("complete coverage classifies benign and empty columns as non-sensitive", async () => {
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
const humanProtected = column({
@@ -126,7 +270,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
{ columnId: empty.id, value: null, characterLength: null },
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
]],
- coverage: { kind: "complete", observedRows: 1 },
+ coverage: { kind: "complete", observedValues: 1 },
});
const assessments = await new SensitivityClassifier(values).assessTable(
@@ -138,7 +282,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
expect.objectContaining({
columnId: empty.id,
- assessment: "unknown",
+ assessment: "non_sensitive",
proposedSensitive: false,
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
}),
@@ -150,11 +294,11 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
]);
});
-test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
+test("sampled coverage without a match proposes non-sensitive independently of the current flag", async () => {
const target = column({ sensitive: true });
const values = source({
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
- coverage: { kind: "sampled", observedRows: 1 },
+ coverage: { kind: "sampled", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -163,13 +307,13 @@ test("sampled coverage without a match is unknown and preserves the current huma
);
expect(assessment).toMatchObject({
- assessment: "unknown",
- proposedSensitive: true,
- evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
+ assessment: "non_sensitive",
+ proposedSensitive: false,
+ evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
});
});
-test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
+test("an unavailable source fails the analysis instead of producing unknown decisions", async () => {
const unresolved = column();
const metadataMatch = column({
id: "44444444-4444-4444-8444-444444444444",
@@ -181,30 +325,17 @@ test("an unavailable source produces sanitized unknown evidence without losing m
}),
};
- const assessments = await new SensitivityClassifier(values).assessTable(
+ await expect(new SensitivityClassifier(values).assessTable(
{ database, table, columns: [unresolved, metadataMatch] },
new AbortController().signal,
- );
-
- expect(assessments).toEqual([
- expect.objectContaining({
- columnId: unresolved.id,
- assessment: "unknown",
- evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
- }),
- expect.objectContaining({
- columnId: metadataMatch.id,
- assessment: "sensitive",
- evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
- }),
- ]);
+ )).rejects.toBeInstanceOf(CatalogConnectorError);
});
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
const target = column({ name: "codice_fiscale" });
const values = source({
batches: [],
- coverage: { kind: "complete", observedRows: 0 },
+ coverage: { kind: "complete", observedValues: 0 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -245,7 +376,7 @@ test.each([
const target = column();
const values = source({
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
- coverage: { kind: "complete", observedRows: 1 },
+ coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -267,13 +398,16 @@ test("does not make a malformed email decisive", async () => {
value: "contatto a@b..com non valido",
characterLength: 28,
}]],
- coverage: { kind: "complete", observedRows: 1 },
+ coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
- expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
+ expect(assessment).toMatchObject({
+ assessment: "non_sensitive",
+ evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
+ });
});
test("finds a valid email after a malformed candidate in the same value", async () => {
@@ -284,7 +418,7 @@ test("finds a valid email after a malformed candidate in the same value", async
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
characterLength: 58,
}]],
- coverage: { kind: "complete", observedRows: 1 },
+ coverage: { kind: "complete", observedValues: 1 },
})).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
@@ -300,7 +434,7 @@ test("optional local NER evidence can make otherwise ambiguous Italian text sens
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
- coverage: { kind: "sampled", observedRows: 1 },
+ coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
@@ -331,14 +465,17 @@ test("does not wait for an optional NER worker that is still warming", async ()
const [assessment] = await new SensitivityClassifier(source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
- coverage: { kind: "sampled", observedRows: 1 },
+ coverage: { kind: "sampled", observedValues: 1 },
}), detector).assessTable(
{ database, table, columns: [target] },
new AbortController().signal,
);
expect(detector.detect).not.toHaveBeenCalled();
- expect(assessment).toMatchObject({ assessment: "unknown" });
+ expect(assessment).toMatchObject({
+ assessment: "non_sensitive",
+ evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
+ });
});
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
@@ -359,7 +496,7 @@ test("bounds each optional NER request when an installation raises the per-table
await new SensitivityClassifier(source({
batches: [observations],
- coverage: { kind: "complete", observedRows: 8 },
+ coverage: { kind: "complete", observedValues: 8 },
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
{ database, table, columns },
new AbortController().signal,
@@ -386,7 +523,7 @@ test("limits default NER work to two candidates spread across a wide table", asy
value: `ordinary-${columnIndex}-${valueIndex}`,
characterLength: 13,
})))],
- coverage: { kind: "complete", observedRows: 2 },
+ coverage: { kind: "complete", observedValues: 2 },
}), detector).assessTable(
{ database, table, columns },
new AbortController().signal,
@@ -402,7 +539,7 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
const target = column();
const values = source({
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
- coverage: { kind: "sampled", observedRows: 1 },
+ coverage: { kind: "sampled", observedValues: 1 },
});
const detector: LocalNerDetector = {
detect: vi.fn(async () => {
@@ -430,11 +567,11 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
expect(nerBudget.remainingMs).toBe(0);
});
-test("uninterpretable binary content remains unknown after complete coverage", async () => {
+test("uninterpretable binary content is protected conservatively without scanning", async () => {
const target = column({ dataType: "bytea" });
const values = source({
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
- coverage: { kind: "complete", observedRows: 1 },
+ coverage: { kind: "complete", observedValues: 1 },
});
const [assessment] = await new SensitivityClassifier(values).assessTable(
@@ -443,8 +580,9 @@ test("uninterpretable binary content remains unknown after complete coverage", a
);
expect(assessment).toMatchObject({
- assessment: "unknown",
- proposedSensitive: false,
- evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
+ assessment: "sensitive",
+ proposedSensitive: true,
+ evidence: [{ kind: "type", ruleId: "type.binary_uninspectable" }],
});
+ expect(values.scanTable).not.toHaveBeenCalled();
});
diff --git a/backend/test/catalog-sensitivity-value-source.test.ts b/backend/test/catalog-sensitivity-value-source.test.ts
index 269942a5..660d9ec6 100644
--- a/backend/test/catalog-sensitivity-value-source.test.ts
+++ b/backend/test/catalog-sensitivity-value-source.test.ts
@@ -1,12 +1,17 @@
-import { expect, test, vi } from "vitest";
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
+import { expect, test, vi } from "vitest";
import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js";
-import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
-import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
-import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
+import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
+import {
+ CatalogConnectorError,
+ type CatalogColumn,
+ type CatalogTable,
+ type WorkspaceDatabase,
+} from "../src/catalog/types.js";
+import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
const database = {
id: "11111111-1111-4111-8111-111111111111",
@@ -60,128 +65,145 @@ function column(id: string, name: string): CatalogColumn {
};
}
-test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => {
+function request(columns: readonly CatalogColumn[], overrides: Record = {}) {
+ return {
+ database,
+ table,
+ columns,
+ valuesPerColumn: 300,
+ sampleOffset: 0,
+ sampleSeed: 37,
+ queryTimeoutMs: 5_000,
+ fullScanThreshold: 1_000,
+ ...overrides,
+ };
+}
+
+test("uses bounded read-only PostgreSQL sampling for tables above 1,000 rows", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value');
- const fullRows = Array.from({ length: 200 }, () => ({
- __value_0: "ordinary",
- __length_0: "8",
- __value_1: null,
- __length_1: null,
- }));
const query = vi.fn(async (sql: string) => {
- if (sql.includes("TABLESAMPLE")) {
- return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] };
+ if (sql.startsWith("SELECT 1 AS __present")) {
+ return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) };
+ }
+ if (sql.startsWith("WITH sampled")) {
+ return { rows: [
+ { __column_index: 0, __value: "ordinary", __length: "8" },
+ { __column_index: 1, __value: "mario.rossi@example.it", __length: 23 },
+ ] };
}
- if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows };
return { rows: [] };
});
const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
};
- let clockCalls = 0;
- const values = new ConcreteSensitivityValueSource(access, undefined, {
- now: () => clockCalls++ < 3 ? 1_000 : 6_100,
- });
- const consumed: unknown[] = [];
+ const consume = vi.fn();
- const coverage = await values.scanTable({
- database,
- table,
- columns: [note, contact],
- fullScanBudgetMs: 5_000,
- deadline: 61_000,
- }, (batch) => consumed.push(...batch), new AbortController().signal);
+ await expect(new ConcreteSensitivityValueSource(access).scanTable(
+ request([note, contact]),
+ consume,
+ new AbortController().signal,
+ )).resolves.toEqual({ kind: "sampled", observedValues: 2 });
- expect(coverage).toEqual({ kind: "sampled", observedRows: 201 });
- expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 });
- expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null });
- expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 });
expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]);
- expect(query.mock.calls.some(([sql]) => (
- String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT")
- ))).toBe(true);
- expect(query.mock.calls.some(([sql]) => String(sql) === (
- "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
- ))).toBe(true);
- expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false);
- expect(query.mock.calls.some(([sql]) => (
- String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM')
- ))).toBe(true);
+ expect(query).toHaveBeenCalledWith("SELECT set_config('statement_timeout', $1, true)", ["5000ms"]);
+ const sampleSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled"));
+ expect(sampleSql).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)');
+ expect(sampleSql).toContain("REPEATABLE (37)");
+ expect(sampleSql).toContain("LIMIT 3000 OFFSET 0");
+ expect(sampleSql).toContain("CROSS JOIN LATERAL");
+ expect(sampleSql).toContain("WHERE __rank <= 300");
+ expect(consume).toHaveBeenCalledWith([
+ { columnId: note.id, value: "ordinary", characterLength: 8 },
+ { columnId: contact.id, value: "mario.rossi@example.it", characterLength: 23 },
+ ]);
expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]);
expect(end).toHaveBeenCalledOnce();
});
-test("reports complete coverage when the final full-scan page is short", async () => {
+test("fully scans a table when the 1,001-row probe proves it is small", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
- const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD")
- ? { rows: [{ __value_0: "ordinary", __length_0: 8 }] }
- : { rows: [] });
- const end = vi.fn(async () => undefined);
+ const query = vi.fn(async (sql: string) => {
+ if (sql.startsWith("SELECT 1 AS __present")) return { rows: [{ __present: 1 }] };
+ if (sql.startsWith("WITH sampled")) {
+ return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
+ }
+ return { rows: [] };
+ });
const access: CatalogPostgresAccess = {
- connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
+ connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
- const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
- const coverage = await values.scanTable({
- database,
- table,
- columns: [note],
- fullScanBudgetMs: 5_000,
- deadline: Date.now() + 60_000,
- }, consume, new AbortController().signal);
+ await expect(new ConcreteSensitivityValueSource(access).scanTable(
+ request([note]),
+ consume,
+ new AbortController().signal,
+ )).resolves.toEqual({ kind: "complete", observedValues: 1 });
- expect(coverage).toEqual({ kind: "complete", observedRows: 1 });
- expect(query.mock.calls.filter(([sql]) => (
- String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
- ))).toHaveLength(1);
+ const valueSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled"));
+ expect(valueSql).not.toContain("TABLESAMPLE");
+ expect(valueSql).toContain("WHERE __rank <= 1000");
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "ordinary", characterLength: 8 },
]);
});
-test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => {
+test("falls back to sampling when the small-table probe reaches its query timeout", async () => {
const note = column("33333333-3333-4333-8333-333333333333", "note");
- let fullScanAttempts = 0;
const query = vi.fn(async (sql: string) => {
- if (sql.includes("TABLESAMPLE")) {
- return { rows: [{ __value_0: "sample", __length_0: 6 }] };
- }
- if (sql.startsWith("FETCH FORWARD")) {
- fullScanAttempts += 1;
+ if (sql.startsWith("SELECT 1 AS __present")) {
throw Object.assign(new Error("statement timeout"), { code: "57014" });
}
+ if (sql.startsWith("WITH sampled")) {
+ return { rows: [{ __column_index: 0, __value: "sample", __length: 6 }] };
+ }
return { rows: [] };
});
- const end = vi.fn(async () => undefined);
const access: CatalogPostgresAccess = {
- connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
+ connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
};
- const values = new ConcreteSensitivityValueSource(access);
const consume = vi.fn();
- const coverage = await values.scanTable({
- database,
- table,
- columns: [note],
- fullScanBudgetMs: 5_000,
- deadline: Date.now() + 60_000,
- }, consume, new AbortController().signal);
-
- expect(fullScanAttempts).toBe(1);
- expect(coverage).toEqual({ kind: "sampled", observedRows: 1 });
- expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([
- "SAVEPOINT sensitivity_full_scan",
- "ROLLBACK TO SAVEPOINT sensitivity_full_scan",
- ]));
- expect(consume).toHaveBeenCalledWith([
- { columnId: note.id, value: "sample", characterLength: 6 },
- ]);
+ await expect(new ConcreteSensitivityValueSource(access).scanTable(
+ request([note]),
+ consume,
+ new AbortController().signal,
+ )).resolves.toEqual({ kind: "sampled", observedValues: 1 });
+ expect(query.mock.calls.map(([sql]) => String(sql))).toContain(
+ "ROLLBACK TO SAVEPOINT sensitivity_scan_1",
+ );
});
-test("scans a REST run_query binding without using PostgreSQL-wire access", async () => {
+test("limits each source query to at most 25 columns", async () => {
+ const columns = Array.from({ length: 26 }, (_, index) => column(
+ `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
+ `attribute_${index + 1}`,
+ ));
+ const query = vi.fn(async (sql: string) => {
+ if (sql.startsWith("SELECT 1 AS __present")) {
+ return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) };
+ }
+ if (sql.startsWith("WITH sampled")) {
+ return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
+ }
+ return { rows: [] };
+ });
+ const access: CatalogPostgresAccess = {
+ connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
+ };
+
+ await new ConcreteSensitivityValueSource(access).scanTable(
+ request(columns),
+ vi.fn(),
+ new AbortController().signal,
+ );
+
+ expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2);
+});
+
+test("scans a REST run_query binding without PostgreSQL-wire access", async () => {
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-"));
const credentialFile = join(root, "api-key");
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
@@ -193,7 +215,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
})),
} as unknown as WorkspaceSecretStore;
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
- { __value_0: "mario.rossi@example.it", __length_0: 23 },
+ { __column_index: 0, __value: "mario.rossi@example.it", __length: 23 },
]), { status: 200, headers: { "content-type": "application/json" } }));
vi.stubGlobal("fetch", fetchMock);
const access: CatalogPostgresAccess = {
@@ -213,15 +235,12 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
const consume = vi.fn();
try {
- await expect(values.scanTable({
+ await expect(values.scanTable(request([note], {
database: restDatabase,
- table,
- columns: [note],
- fullScanBudgetMs: 5_000,
- deadline: Date.now() + 60_000,
- }, consume, new AbortController().signal)).resolves.toEqual({
- kind: "complete",
- observedRows: 1,
+ fullScanThreshold: undefined,
+ }), consume, new AbortController().signal)).resolves.toEqual({
+ kind: "sampled",
+ observedValues: 1,
});
expect(access.connect).not.toHaveBeenCalled();
expect(fetchMock).toHaveBeenCalledWith(
@@ -232,7 +251,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
}),
);
const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body));
- expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0');
+ expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)');
expect(consume).toHaveBeenCalledWith([
{ columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 },
]);
@@ -243,74 +262,44 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
}
});
-test("keeps multi-request REST scans conservative without a source transaction", async () => {
- const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-"));
- const credentialFile = join(root, "api-key");
- writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
- const secretStore = {
- materialize: vi.fn(() => ({
- files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
- release: vi.fn(),
- })),
- } as unknown as WorkspaceSecretStore;
- const fetchMock = vi.fn()
- .mockResolvedValueOnce(new Response(JSON.stringify([
- { __value_0: "ordinary", __length_0: 8 },
- ]), { status: 200 }))
- .mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 }));
- vi.stubGlobal("fetch", fetchMock);
- const values = new ConcreteSensitivityValueSource({
- connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
- }, secretStore, { batchRows: 1 });
- const restDatabase: WorkspaceDatabase = {
- ...database,
- binding: {
- transport: "rest_api",
- baseUrl: "https://dwh.example.test/root",
- restPath: "/health",
- restAuth: "x-api-key",
- },
- };
-
- try {
- await expect(values.scanTable({
- database: restDatabase,
- table,
- columns: [column("33333333-3333-4333-8333-333333333333", "note")],
- fullScanBudgetMs: 5_000,
- deadline: Date.now() + 60_000,
- }, vi.fn(), new AbortController().signal)).resolves.toEqual({
- kind: "sampled",
- observedRows: 1,
- });
- expect(fetchMock).toHaveBeenCalledTimes(2);
- } finally {
- vi.unstubAllGlobals();
- rmSync(root, { recursive: true, force: true });
- }
-});
-
-test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => {
- const query = vi.fn(async () => ({ rows: [] }));
- const end = vi.fn(async () => undefined);
- const access: CatalogPostgresAccess = {
- connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
- };
- const now = vi.fn()
- .mockReturnValueOnce(1_000)
- .mockReturnValue(61_000);
- const values = new ConcreteSensitivityValueSource(access, undefined, { now });
-
- await expect(values.scanTable({
- database,
- table,
- columns: [column("33333333-3333-4333-8333-333333333333", "note")],
- fullScanBudgetMs: 5_000,
- deadline: 60_000,
- }, vi.fn(), new AbortController().signal)).resolves.toEqual({
- kind: "sampled",
- observedRows: 0,
+test("falls back to a sequential bounded sample when randomized sampling times out", async () => {
+ const note = column("33333333-3333-4333-8333-333333333333", "note");
+ const query = vi.fn(async (sql: string) => {
+ if (sql.startsWith("WITH sampled") && sql.includes("TABLESAMPLE")) {
+ throw Object.assign(new Error("raw source detail"), { code: "57014" });
+ }
+ if (sql.startsWith("WITH sampled")) {
+ return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
+ }
+ return { rows: [] };
});
- expect(query).not.toHaveBeenCalled();
- expect(end).toHaveBeenCalledOnce();
+ const access: CatalogPostgresAccess = {
+ connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
+ };
+
+ await expect(new ConcreteSensitivityValueSource(access).scanTable(
+ request([note], { fullScanThreshold: undefined }),
+ vi.fn(),
+ new AbortController().signal,
+ )).resolves.toEqual({ kind: "sampled", observedValues: 1 });
+ expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2);
+});
+
+test("fails explicitly when both randomized and sequential sample queries time out", async () => {
+ const note = column("33333333-3333-4333-8333-333333333333", "note");
+ const query = vi.fn(async (sql: string) => {
+ if (sql.startsWith("WITH sampled")) {
+ throw Object.assign(new Error("raw source detail"), { code: "57014" });
+ }
+ return { rows: [] };
+ });
+ const access: CatalogPostgresAccess = {
+ connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
+ };
+
+ await expect(new ConcreteSensitivityValueSource(access).scanTable(
+ request([note], { fullScanThreshold: undefined }),
+ vi.fn(),
+ new AbortController().signal,
+ )).rejects.toEqual(new CatalogConnectorError("Sensitivity sample query timed out"));
});
diff --git a/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md b/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md
index 4b971547..f6b60e84 100644
--- a/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md
+++ b/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md
@@ -1,5 +1,5 @@
---
-status: accepted
+status: superseded by ADR-0015
---
# Assess sensitive columns locally from source content
diff --git a/docs/adr/0015-use-progressive-sampling-for-sensitive-columns.md b/docs/adr/0015-use-progressive-sampling-for-sensitive-columns.md
new file mode 100644
index 00000000..68f8d806
--- /dev/null
+++ b/docs/adr/0015-use-progressive-sampling-for-sensitive-columns.md
@@ -0,0 +1,33 @@
+---
+status: accepted
+---
+
+# Use progressive sampling for sensitive columns
+
+The `sensitivity-v1` wall-clock policy produced too many `unknown` assessments: a global
+sixty-second deadline coupled the outcome of one column to table order, source latency, and optional
+NER cost. Those outcomes were not useful for description-generation gating, because they did not
+provide a usable draft Sensitive Data Flag.
+
+`sensitivity-v2` bounds database effort by inspected values rather than by one global clock. Tables
+proven to contain at most 1,000 rows are fully scanned. Larger tables are processed breadth-first in
+three passes: 300 values per unresolved column, 700 additional values to reach 1,000, then 2,000
+additional values to reach 3,000 for unresolved text, JSON, and XML columns. One positive rule or
+NER finding is enough to stop later work for that column. At most two tables are scanned
+concurrently. Source queries contain at most 25 columns and each has a five-second statement timeout.
+An empty or timed-out randomized sample gets one sequential bounded retry; two timeouts fail the run.
+
+A completed v2 analysis returns only `sensitive` or `non_sensitive`. Sampled no-match, empty, and
+all-null columns are proposed as `non_sensitive`, with coverage reported independently so the human
+reviewer can judge the strength of the proposal. Binary or otherwise uninspectable column types are
+proposed as `sensitive`. A source failure fails the analysis and returns no review; it is not
+converted into `unknown`. The administrator can still set either final value.
+
+The HTTP operation has no global analysis deadline. It is canceled when the client disconnects or
+the backend restarts. Historic and interrupted run records retain the database field named
+`unknown` for compatibility, where it counts unprocessed columns rather than a v2 assessment.
+
+This decision supersedes ADR-0014 only for scan effort, coverage semantics, and the assessment
+domain. ADR-0014 remains authoritative for the single local TypeScript decision point, the absence
+of generative LLMs, transient human-reviewed drafts, the 500-character rule, and optional CPU-only
+NER evidence.
diff --git a/docs/architecture/components.md b/docs/architecture/components.md
index b028d0b3..653b5d11 100644
--- a/docs/architecture/components.md
+++ b/docs/architecture/components.md
@@ -107,16 +107,17 @@ explicitly unlocked; it never resumes automatically.
Sensitivity analysis is a synchronous administrative request and does not use the installation
model catalog. Database-specific adapters stream bounded normalized values from read-only source
connections; the TypeScript `SensitivityClassifier` is the single decision point for
-`sensitive | non_sensitive | unknown`. Deterministic rules run first. A complete scan is attempted
-for at most five seconds per table, then the adapter samples within the sixty-second request budget.
-An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved short text, but it
-cannot make or persist the decision itself.
+`sensitive | non_sensitive`. Deterministic rules run first. Tables up to 1,000 rows are fully
+scanned; larger tables use breadth-first targets of 300, 1,000, and 3,000 values, with the last pass
+limited to text-like columns. Source queries have five-second limits, but the request has no global
+analysis deadline. An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved
+short text, but it cannot make or persist the decision itself.
Each attempt has its own durable run and ordered sanitized events, separate from Description
-Generation because its lifecycle and counters differ. The run records the local policy version,
-coverage aggregates, and sanitized rule identifiers. Proposed flags, source values, NER spans, and
-worker diagnostics remain transient. Only an explicit administrator save changes the human-owned
-Sensitive Data Flag.
+Generation because its lifecycle and counters differ. The run records the local policy version and
+aggregate decision counts. Coverage, rule identifiers, proposed flags, source values, NER spans,
+and worker diagnostics remain transient. Only an explicit administrator save changes the
+human-owned Sensitive Data Flag.
## Main backend classes
diff --git a/docs/operations/sensitivity-analysis.md b/docs/operations/sensitivity-analysis.md
index 422b3573..3c6a95f1 100644
--- a/docs/operations/sensitivity-analysis.md
+++ b/docs/operations/sensitivity-analysis.md
@@ -7,7 +7,7 @@ may set either value, including overriding a `sensitive` proposal.
## Default policy
-`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v1`
+`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v2`
policy combines:
- normalized column-name rules for direct identifiers, credentials, and health data;
@@ -18,22 +18,35 @@ policy combines:
- a conservative length rule: any observed textual value longer than 500 characters makes the
entire column sensitive.
-One decisive value is enough to classify the column as `sensitive`. A complete scan with no match
-may classify it as `non_sensitive`. Empty, all-null, binary/uninspectable, interrupted, and sampled
-no-match columns are `unknown`; an `unknown` draft preserves the current human flag.
+One decisive value is enough to classify the column as `sensitive` and removes it from subsequent
+passes. Binary or otherwise uninspectable column types are also proposed as `sensitive`, because
+their contents cannot be cleared by the textual rules. A completed analysis has only two draft
+outcomes: `sensitive` and `non_sensitive`. Empty or all-null columns are `non_sensitive` with
+`no_values` coverage; a sampled column with no match is `non_sensitive` with explicit sampled
+coverage. The administrator remains free to reverse either proposal before saving it.
Source reads are database-specific, but decisions are database-independent. PostgreSQL direct and
REST `run_query` adapters project at most 501 characters per value, use only `SELECT`, and never
-persist source values. A full scan gets five seconds per table. If it cannot finish, the adapter uses
-a bounded repeatable sample within the sixty-second request deadline. PostgreSQL-wire reads run in a
-read-only transaction and always end with rollback. A REST scan can claim complete coverage only
-when it finishes in one request; multi-request pagination has no shared source transaction and is
-therefore conservatively reported as sampled.
+persist source values. Tables proven to contain at most 1,000 rows are fully scanned. Larger tables
+are processed breadth-first so every table gets the cheapest pass before any table gets a deeper
+one:
-The HTTP operation stops waiting at sixty seconds. The same expiring signal is checked before and
-after catalog selection, source access, progress writes, and every table. If it expires after a run
-has been created, that run is finalized as `interrupted` and all not-decisively-processed columns
-are counted as `unknown`; no review payload is returned from the timed-out request.
+1. inspect up to 300 non-null values per unresolved column;
+2. inspect up to 700 additional values, reaching a 1,000-value target;
+3. for unresolved text, JSON, and XML columns only, inspect up to 2,000 additional values, reaching
+ a 3,000-value target.
+
+At most two tables are scanned concurrently, and the database adapter groups at most 25 columns in
+one source query. Each probe or value query has a five-second statement timeout; PostgreSQL-wire
+reads run in a read-only transaction and always end with rollback. Sampling is bounded and
+repeatable for a policy version. If a randomized sample is empty or reaches its query timeout, the
+adapter tries one sequential bounded sample; if that also times out, the source error fails the run
+and returns no review instead of manufacturing `unknown` decisions.
+
+There is no global sixty-second analysis deadline. Work is bounded by sample counts, per-query
+timeouts, and early column exits. The operation is interrupted only when its request connection is
+aborted or the backend restarts. Historical or interrupted run counters named `unknown` represent
+columns that were not processed; `unknown` is not a `sensitivity-v2` column assessment.
History stores only the policy version, aggregate outcomes, timestamps, and fixed operational
events. Sanitized rule IDs are returned in the transient review and shadow report, not persisted.
@@ -116,13 +129,16 @@ Enabling NER by default requires all of these gates:
1. the pinned artifact and `MODEL_SHA256SUMS` are archived with the installation inventory;
2. the Python dependency/license inventory contains only redistribution-compatible licenses;
-3. the CPU benchmark stays within the configured deadlines and does not use a GPU;
+3. the CPU benchmark stays within the configured NER allowance and does not use a GPU;
4. the labeled Italian evaluation meets thresholds approved by the product owner.
-If a gate fails, leave NER disabled. The deterministic policy remains available and unresolved
-columns remain `unknown` rather than being sent to an internal or external LLM.
+If a gate fails, leave NER disabled. The deterministic policy remains available and produces the
+binary draft from its scan coverage; no content is sent to an internal or external LLM.
The first aggregate PSD shadow comparison is recorded in
[`2026-09-02-psd-sensitivity-shadow.md`](../reports/2026-09-02-psd-sensitivity-shadow.md). On the
-local CPU runner, NER found additional entities but reduced total coverage inside the 60-second
-deadline, so the accepted setting remains disabled by default.
+local CPU runner, NER found additional entities but reduced total coverage under the superseded
+global deadline, so the accepted setting remains disabled by default pending a new v2 benchmark.
+The deterministic progressive PSD run is recorded in
+[`2026-09-03-psd-progressive-sensitivity-shadow.md`](../reports/2026-09-03-psd-progressive-sensitivity-shadow.md):
+it assessed all 2,275 columns with zero `unknown` decisions and left NER disabled.
diff --git a/docs/reports/2026-09-03-psd-progressive-sensitivity-shadow.md b/docs/reports/2026-09-03-psd-progressive-sensitivity-shadow.md
new file mode 100644
index 00000000..34890527
--- /dev/null
+++ b/docs/reports/2026-09-03-psd-progressive-sensitivity-shadow.md
@@ -0,0 +1,49 @@
+# PSD progressive sensitivity shadow evaluation
+
+Date: 2026-09-03
+
+This report records an aggregate, non-mutating evaluation of `sensitivity-v2` against the PSD
+workspace. The configured connector accessed the source data warehouse with its read-only role.
+The shadow command did not create an analysis run, update local catalog metadata, or save Sensitive
+Data Flags. No database, table, column, source value, or matched span was emitted.
+
+The final post-fix run used the local Docker CPU environment, without NER. It inspected all 2,275
+catalog columns through the progressive 300, 1,000, and text-only 3,000-value policy.
+
+| Sensitive | Non-sensitive | Unknown decisions | Analysis time |
+| ---: | ---: | ---: | ---: |
+| 343 | 1,932 | 0 | 50,082 ms |
+
+Coverage was reported independently from the decision:
+
+| Metadata decision | Complete scan | Sampled | No observed values |
+| ---: | ---: | ---: | ---: |
+| 39 | 0 | 2,128 | 108 |
+
+The sampled no-match population comprised 1,337 columns ending after the 1,000-value target and
+487 text-like columns ending after the 3,000-value target. Positive findings were:
+
+| Rule | Columns |
+| --- | ---: |
+| `pii.phone_number` | 202 |
+| `text.over_500_characters` | 52 |
+| `metadata.health` | 34 |
+| `health.clinical_term` | 25 |
+| `pii.italian_vat` | 8 |
+| `pii.email` | 7 |
+| `metadata.direct_identifier` | 5 |
+| `pii.uuid` | 5 |
+| `financial.payment_card` | 3 |
+| `pii.italian_fiscal_code` | 2 |
+
+Three successful v2 diagnostic runs produced the same decisions and aggregate rule counts. Their
+times ranged from 49,253 to 130,818 ms, showing that source load still affects latency even though
+it no longer changes the outcome through a global deadline. An earlier run exposed an intermittent
+randomized-query timeout. The adapter now retries that case once with a sequential bounded query;
+a regression test covers the fallback, while two consecutive timeouts still fail the whole analysis
+instead of creating `unknown` decisions.
+
+This is a coverage and operational benchmark, not a precision/recall acceptance test. In
+particular, the 202 phone-number findings and every other rule family still require human review or
+a separately approved labeled corpus before their false-positive rate can be measured. NER remains
+disabled by default.
diff --git a/frontend/src/api/catalog-databases.ts b/frontend/src/api/catalog-databases.ts
index 97e773fa..21da4be8 100644
--- a/frontend/src/api/catalog-databases.ts
+++ b/frontend/src/api/catalog-databases.ts
@@ -134,14 +134,15 @@ export interface SensitivityReviewItem {
version: number;
currentSensitive: boolean;
sensitive: boolean;
- assessment: "sensitive" | "non_sensitive" | "unknown";
+ assessment: "sensitive" | "non_sensitive";
evidence: Array<{
- kind: "metadata" | "content" | "length" | "ner" | "coverage";
+ kind: "metadata" | "content" | "length" | "ner" | "coverage" | "type";
ruleId: string;
label?: string;
confidence?: number;
}>;
observedValues: number;
+ coverage: "metadata" | "complete" | "sampled" | "no_values";
}
export interface SensitivityAnalysisResult {
diff --git a/frontend/src/api/client.test.ts b/frontend/src/api/client.test.ts
index 7c5e1e3c..2113b3ce 100644
--- a/frontend/src/api/client.test.ts
+++ b/frontend/src/api/client.test.ts
@@ -149,6 +149,7 @@ test.each([
["catalog_table_not_found", "One or more selected catalog tables were not found."],
["sensitivity_source_unavailable", "The database content could not be read for sensitivity analysis. No assessments were applied."],
["sensitivity_analysis_timeout", "Sensitivity analysis reached its time limit. No assessments were applied."],
+ ["sensitivity_analysis_interrupted", "Sensitivity analysis was interrupted before completion. No assessments were applied."],
["sensitive_data_suggestion_history_request_invalid", "Sensitivity analysis history parameters are invalid."],
["sensitive_data_suggestion_history_failed", "Sensitivity analysis history could not be loaded."],
["sensitive_data_suggestion_run_not_found", "The sensitivity analysis run was not found."],
diff --git a/frontend/src/api/client.ts b/frontend/src/api/client.ts
index 47bac65b..6bd3dc2b 100644
--- a/frontend/src/api/client.ts
+++ b/frontend/src/api/client.ts
@@ -28,6 +28,7 @@ const safeErrorCodes = new Set([
"sensitive_data_suggestion_no_columns",
"sensitivity_source_unavailable",
"sensitivity_analysis_timeout",
+ "sensitivity_analysis_interrupted",
"sensitive_data_suggestion_failed",
"sensitive_data_suggestion_history_request_invalid",
"sensitive_data_suggestion_history_failed",
@@ -94,6 +95,7 @@ const localCodeMessages: Record = {
sensitive_data_suggestion_no_columns: "The selected scope contains no catalog columns to assess.",
sensitivity_source_unavailable: "The database content could not be read for sensitivity analysis. No assessments were applied.",
sensitivity_analysis_timeout: "Sensitivity analysis reached its time limit. No assessments were applied.",
+ sensitivity_analysis_interrupted: "Sensitivity analysis was interrupted before completion. No assessments were applied.",
sensitive_data_suggestion_failed: "Sensitivity analysis failed before review. No changes were applied.",
sensitive_data_suggestion_history_request_invalid: "Sensitivity analysis history parameters are invalid.",
sensitive_data_suggestion_history_failed: "Sensitivity analysis history could not be loaded.",
diff --git a/frontend/src/shell/DatabaseManagementPage.test.tsx b/frontend/src/shell/DatabaseManagementPage.test.tsx
index 8e0269ad..ef07fb37 100644
--- a/frontend/src/shell/DatabaseManagementPage.test.tsx
+++ b/frontend/src/shell/DatabaseManagementPage.test.tsx
@@ -167,7 +167,7 @@ function makeSensitivityAnalysisRun(
databaseId: "11111111-1111-4111-8111-111111111111",
engine: "local",
modelId: null,
- policyVersion: "sensitivity-v1",
+ policyVersion: "sensitivity-v2",
scope: "selected_columns",
status: "completed",
total: 2,
@@ -2081,6 +2081,7 @@ test("requests database-level sensitivity analysis for the only selected databas
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
+ coverage: "sampled",
}],
});
}),
@@ -2162,6 +2163,7 @@ test("requests sensitivity analysis only for selected tables", async () => {
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
observedValues: 0,
+ coverage: "metadata",
}],
});
}),
@@ -2236,6 +2238,7 @@ test("allows a human downgrade and saves only explicit sensitivity changes", asy
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
+ coverage: "sampled",
},
{
columnId: nameColumn.id,
@@ -2246,8 +2249,9 @@ test("allows a human downgrade and saves only explicit sensitivity changes", asy
currentSensitive: true,
sensitive: false,
assessment: "non_sensitive",
- evidence: [],
+ evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
observedValues: 2,
+ coverage: "complete",
},
],
});
diff --git a/frontend/src/shell/database-management/SensitiveDataReviewDrawer.test.tsx b/frontend/src/shell/database-management/SensitiveDataReviewDrawer.test.tsx
index d09e6448..0f32a9db 100644
--- a/frontend/src/shell/database-management/SensitiveDataReviewDrawer.test.tsx
+++ b/frontend/src/shell/database-management/SensitiveDataReviewDrawer.test.tsx
@@ -21,6 +21,7 @@ const suggestions: SensitivityReviewItem[] = [
assessment: "sensitive",
evidence: [{ kind: "content", ruleId: "pii.email" }],
observedValues: 1,
+ coverage: "sampled",
},
{
columnId: "22222222-2222-4222-8222-222222222222",
@@ -33,6 +34,7 @@ const suggestions: SensitivityReviewItem[] = [
assessment: "sensitive",
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
observedValues: 0,
+ coverage: "metadata",
},
];
diff --git a/frontend/src/shell/database-management/SensitiveDataReviewDrawer.tsx b/frontend/src/shell/database-management/SensitiveDataReviewDrawer.tsx
index ca437018..77508e99 100644
--- a/frontend/src/shell/database-management/SensitiveDataReviewDrawer.tsx
+++ b/frontend/src/shell/database-management/SensitiveDataReviewDrawer.tsx
@@ -183,7 +183,7 @@ export function SensitiveDataReviewDrawer({
? suggestion.evidence.map((item) => item.label
? `${item.ruleId} (${item.label}${item.confidence === undefined ? "" : ` ${Math.round(item.confidence * 100)}%`})`
: item.ruleId).join(", ")
- : "no sensitive match"}. Observed values: {suggestion.observedValues}.
+ : "no sensitive match"}. Coverage: {suggestion.coverage.replace("_", " ")}. Observed values: {suggestion.observedValues}.
diff --git a/frontend/src/shell/database-management/SensitivityAnalysisHistoryDrawer.tsx b/frontend/src/shell/database-management/SensitivityAnalysisHistoryDrawer.tsx
index 31437235..b5e89b58 100644
--- a/frontend/src/shell/database-management/SensitivityAnalysisHistoryDrawer.tsx
+++ b/frontend/src/shell/database-management/SensitivityAnalysisHistoryDrawer.tsx
@@ -145,7 +145,7 @@ export function SensitivityAnalysisHistoryDrawer({
["total", "Total columns"],
["suggestedSensitive", "Sensitive"],
["suggestedNonSensitive", "Not sensitive"],
- ["unknown", "Unknown"],
+ ["unknown", "Unprocessed"],
] as const;
return (