diff --git a/PROJECT_STATE.md b/PROJECT_STATE.md index 462c628d..658da211 100644 --- a/PROJECT_STATE.md +++ b/PROJECT_STATE.md @@ -98,16 +98,17 @@ column. The KPI strip reads installation-wide or selected-database aggregates fr description history, and sensitive-field review/history use the production APIs in right-side drawers rather than prototype fixtures; closing a history drawer does not stop its background run. -Sensitive-field review is now driven by the versioned local `sensitivity-v1` policy, not by a +Sensitive-field review is now driven by the versioned local `sensitivity-v2` policy, not by a catalog model. The backend reads selected source tables through read-only, database-specific -adapters and makes every `sensitive | non_sensitive | unknown` decision in the TypeScript -`SensitivityClassifier`. A single validated match protects the column; a full scan is limited to -five seconds per table before sampling and the whole request to sixty seconds. Draft assessments -remain transient until an administrator explicitly saves them. Optional GLiNER2 evidence is -CPU-only, offline, opt-in, and never replaces the deterministic decision point; see -`docs/operations/sensitivity-analysis.md`. The aggregate PSD shadow comparison kept NER disabled by -default because its extra findings did not offset the coverage lost to inference within the global -deadline; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`. +adapters and makes every `sensitive | non_sensitive` draft decision in the TypeScript +`SensitivityClassifier`. A single validated match protects the column. Tables up to 1,000 rows are +fully scanned; larger tables use breadth-first 300, 1,000, and text-only 3,000-value targets, with a +five-second limit per source query and no global request deadline. Source failures fail the run +instead of yielding `unknown`; coverage remains visible separately from the proposal. Draft +assessments remain transient until an administrator explicitly saves them. Optional GLiNER2 +evidence is CPU-only, offline, opt-in, and never replaces the deterministic decision point; see +`docs/operations/sensitivity-analysis.md`. The earlier v1 PSD shadow comparison kept NER disabled by +default; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`. A v2 PSD benchmark is still due. Physical membership, source comments, column types/default/nullability/PK positions, and constraint-level ordered FK pairs are @@ -169,9 +170,9 @@ AI Description Generation uses the catalog's human-owned Sensitive Data Flag. Th `false`, including for newly synchronized columns. An administrator may request a local sensitivity analysis for one selected database, selected tables, or selected columns. One deterministic TypeScript classifier combines metadata, bounded source-content rules, and optional CPU-only NER; -no generative model decides the result. Its `sensitive`, `non_sensitive`, or `unknown` assessments -remain an unsaved draft until the human reviews and saves any chosen flag changes, including a -downgrade to non-sensitive. +no generative model decides the result. Its `sensitive` or `non_sensitive` assessments remain an +unsaved draft until the human reviews and saves any chosen flag changes, including a downgrade to +non-sensitive. Coverage is reported separately; interrupted history may count unprocessed columns. Each started analysis records a separate Sensitivity Analysis Run with aggregate counters and safe ordered events. This operational history never stores per-column assessments, source values, matched spans, prompts, or free-form diagnostics; reloading still discards an unsaved review draft. diff --git a/backend/src/catalog/sensitivity-analysis-runner.ts b/backend/src/catalog/sensitivity-analysis-runner.ts index 0543d1b2..23f10af9 100644 --- a/backend/src/catalog/sensitivity-analysis-runner.ts +++ b/backend/src/catalog/sensitivity-analysis-runner.ts @@ -14,7 +14,7 @@ import type { } from "./types.js"; const interruptedMessage = "Local sensitivity analysis was interrupted by backend restart."; -const deadlineMessage = "Local sensitivity analysis reached its time limit."; +const interruptedDuringRunMessage = "Local sensitivity analysis was interrupted before completion."; const failedMessage = "Local sensitivity analysis failed."; function ensureActive(signal: AbortSignal): void { @@ -98,14 +98,12 @@ export class SensitivityAnalysisRunner { const suggestedNonSensitive = batch.filter( (suggestion) => suggestion.assessment === "non_sensitive", ).length; - const unknown = batch.filter((suggestion) => suggestion.assessment === "unknown").length; const current = await this.repository.getSensitivityAnalysisRun(started.id); ensureActive(signal); if (!current) throw new Error("Sensitivity Analysis Run disappeared"); const progress = await this.repository.updateSensitivityAnalysisRun(started.id, { suggestedSensitive: current.suggestedSensitive + suggestedSensitive, suggestedNonSensitive: current.suggestedNonSensitive + suggestedNonSensitive, - unknown: current.unknown + unknown, }); if (!progress) throw new Error("Sensitivity Analysis Run disappeared"); processedSensitive += suggestedSensitive; @@ -126,7 +124,6 @@ export class SensitivityAnalysisRunner { const suggestedNonSensitive = suggestions.filter( (suggestion) => suggestion.assessment === "non_sensitive", ).length; - const unknown = suggestions.filter((suggestion) => suggestion.assessment === "unknown").length; await this.repository.appendSensitivityAnalysisEvent( started.id, "info", @@ -140,7 +137,7 @@ export class SensitivityAnalysisRunner { total: suggestions.length, suggestedSensitive, suggestedNonSensitive, - unknown, + unknown: 0, finishedAt: new Date().toISOString(), errorSummary: null, }); @@ -149,7 +146,7 @@ export class SensitivityAnalysisRunner { return { suggestions, run: completed }; } catch (error) { const interrupted = signal.aborted || error instanceof SensitivityAnalysisInterruptedError; - const message = interrupted ? deadlineMessage : failedMessage; + const message = interrupted ? interruptedDuringRunMessage : failedMessage; await this.repository.updateSensitivityAnalysisRun(started.id, { status: interrupted ? "interrupted" : "failed", ...(interrupted ? { diff --git a/backend/src/catalog/sensitivity-analysis-service.ts b/backend/src/catalog/sensitivity-analysis-service.ts index 9f9034c4..95c779d1 100644 --- a/backend/src/catalog/sensitivity-analysis-service.ts +++ b/backend/src/catalog/sensitivity-analysis-service.ts @@ -12,7 +12,7 @@ import type { } from "./types.js"; export type { SensitivityAnalysisScope } from "./types.js"; -export const SENSITIVITY_POLICY_VERSION = "sensitivity-v1"; +export const SENSITIVITY_POLICY_VERSION = "sensitivity-v2"; interface SelectedColumn { table: CatalogTable; @@ -30,6 +30,7 @@ export interface SensitivityReviewItem { assessment: SensitivityColumnAssessment["assessment"]; evidence: readonly SensitivityEvidence[]; observedValues: number; + coverage: SensitivityColumnAssessment["coverage"]; } export class SensitivityAnalysisTargetNotFoundError extends Error { @@ -48,7 +49,7 @@ export class SensitivityAnalysisDuplicateTargetIdsError extends Error { export class SensitivityAnalysisInterruptedError extends Error { constructor() { - super("sensitivity analysis deadline exceeded"); + super("sensitivity analysis interrupted"); this.name = "SensitivityAnalysisInterruptedError"; } } @@ -69,7 +70,7 @@ export class SensitivityAnalysisService { constructor( private readonly repository: CatalogRepository, private readonly classifier: SensitivityClassifier, - private readonly options: { runBudgetMs?: number; nerBudgetMs?: number; now?: () => number } = {}, + private readonly options: { nerBudgetMs?: number } = {}, ) {} private async selectColumns( @@ -116,8 +117,6 @@ export class SensitivityAnalysisService { onPrepared?: (total: number) => void | Promise, onProgress?: (processed: number, suggestions: readonly SensitivityReviewItem[]) => void | Promise, ): Promise { - const now = this.options.now ?? Date.now; - const deadline = now() + (this.options.runBudgetMs ?? 60_000); const configuredNerBudget = this.options.nerBudgetMs ?? 10_000; const nerBudget: SensitivityNerBudget = { remainingMs: Number.isFinite(configuredNerBudget) && configuredNerBudget >= 0 @@ -137,20 +136,23 @@ export class SensitivityAnalysisService { items.push(item); byTable.set(item.table.id, items); } - const suggestions: SensitivityReviewItem[] = []; - for (const items of byTable.values()) { - ensureActive(signal); + const tableTargets = [...byTable.values()].map((items) => { const first = items[0]!; - const assessments = await this.classifier.assessTable({ + return { database, table: first.table, columns: items.map(({ column }) => column), - }, signal, deadline, nerBudget); + }; + }); + const assessments = await this.classifier.assess(tableTargets, signal, nerBudget); + ensureActive(signal); + const assessmentById = new Map(assessments.map((assessment) => [ + assessment.columnId, + assessment, + ])); + const suggestions: SensitivityReviewItem[] = []; + for (const items of byTable.values()) { ensureActive(signal); - const assessmentById = new Map(assessments.map((assessment) => [ - assessment.columnId, - assessment, - ])); const batch = items.map(({ table, column }) => { const assessment = assessmentById.get(column.id)!; return { @@ -164,6 +166,7 @@ export class SensitivityAnalysisService { assessment: assessment.assessment, evidence: assessment.evidence, observedValues: assessment.observedValues, + coverage: assessment.coverage, }; }); suggestions.push(...batch); diff --git a/backend/src/catalog/sensitivity-classifier.ts b/backend/src/catalog/sensitivity-classifier.ts index 80e044df..70383680 100644 --- a/backend/src/catalog/sensitivity-classifier.ts +++ b/backend/src/catalog/sensitivity-classifier.ts @@ -1,11 +1,11 @@ -import { CatalogConnectorError, type CatalogColumn, type CatalogTable, type WorkspaceDatabase } from "./types.js"; +import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "./types.js"; import { findPhoneNumbersInText } from "libphonenumber-js/max"; import validator from "validator"; -export type SensitivityAssessment = "sensitive" | "non_sensitive" | "unknown"; +export type SensitivityAssessment = "sensitive" | "non_sensitive"; export interface SensitivityEvidence { - kind: "metadata" | "content" | "length" | "ner" | "coverage"; + kind: "metadata" | "content" | "length" | "ner" | "coverage" | "type"; ruleId: string; label?: string; confidence?: number; @@ -18,8 +18,8 @@ export interface SensitivityValueObservation { } export interface SensitivityScanCoverage { - kind: "complete" | "sampled" | "unavailable"; - observedRows: number; + kind: "complete" | "sampled"; + observedValues: number; } export interface SensitivityTableScan { @@ -31,8 +31,11 @@ export interface SensitivityScanRequest { database: WorkspaceDatabase; table: CatalogTable; columns: readonly CatalogColumn[]; - fullScanBudgetMs: number; - deadline: number; + valuesPerColumn: number; + sampleOffset: number; + sampleSeed: number; + queryTimeoutMs: number; + fullScanThreshold?: number; } export interface SensitivityValueSource { @@ -76,6 +79,7 @@ export interface SensitivityColumnAssessment { proposedSensitive: boolean; evidence: readonly SensitivityEvidence[]; observedValues: number; + coverage: "metadata" | "complete" | "sampled" | "no_values"; } export interface SensitivityTableTarget { @@ -94,7 +98,15 @@ const CREDENTIAL_NAME = /(?:^|_)(?:api_key|credential|password|passwd|private_ke const HEALTH_NAME = /(?:^|_)(?:anamnesi|clinical|diagnos(?:i|is)|health|medical|patient|patologia|therapy|terapia)(?:_|$)/u; const CLINICAL_TERM = /(?:^|[^\p{L}])(?:allergi[ae]|anamnesi|carcinoma|chemioterapia|diabete|diagnos[ei]|epatite|farmac[io]|gravidanza|hiv|metastasi|neoplasia|patologia|radioterapia|referto|terapia|tumore)(?:$|[^\p{L}])/iu; const UNSUPPORTED_BINARY_TYPE = /(?:^|\s)(?:binary|blob|bytea|image|varbinary)(?:\s|$|\()/iu; +const DEEP_TEXT_TYPE = /(?:^|\s)(?:char|character|citext|clob|json|jsonb|nchar|nvarchar|string|text|varchar|xml)(?:\s|$|\()/iu; const MAX_NER_CANDIDATES_PER_REQUEST = 128; +const MAX_CONCURRENT_TABLE_SCANS = 2; + +export const SENSITIVITY_SAMPLE_PHASES = [ + { targetValuesPerColumn: 300, additionalValuesPerColumn: 300, sampleSeed: 37, deepTextOnly: false }, + { targetValuesPerColumn: 1_000, additionalValuesPerColumn: 700, sampleSeed: 73, deepTextOnly: false }, + { targetValuesPerColumn: 3_000, additionalValuesPerColumn: 2_000, sampleSeed: 109, deepTextOnly: true }, +] as const; function normalizedName(value: string): string { return value.normalize("NFKD") @@ -284,14 +296,22 @@ function contentEvidence(value: string): SensitivityEvidence | undefined { return undefined; } +interface ColumnState { + column: CatalogColumn; + evidence: SensitivityEvidence[]; + observedValues: number; + nerCandidates: string[]; + coverage: "metadata" | "complete" | "sampled" | "no_values"; + sampledTarget: number; +} + /** Sole decision module for local column-level sensitivity assessments. */ export class SensitivityClassifier { constructor( private readonly values: SensitivityValueSource, private readonly detector?: LocalNerDetector, private readonly options: { - fullScanBudgetMs?: number; - runBudgetMs?: number; + queryTimeoutMs?: number; nerConfidenceThreshold?: number; maxNerValuesPerColumn?: number; maxNerCandidatesPerTable?: number; @@ -299,95 +319,131 @@ export class SensitivityClassifier { } = {}, ) {} - async assessTable( - target: SensitivityTableTarget, + async assess( + targets: readonly SensitivityTableTarget[], signal: AbortSignal, - runDeadline?: number, - nerBudget?: SensitivityNerBudget, + sharedNerBudget?: SensitivityNerBudget, ): Promise { const now = this.options.now ?? Date.now; - const deadline = runDeadline ?? now() + (this.options.runBudgetMs ?? 60_000); - const evidence = new Map(target.columns.map((column) => { - const match = metadataEvidence(column); - return [column.id, match ? [match] : [] as SensitivityEvidence[]]; - })); - const observed = new Map(target.columns.map((column) => [column.id, 0])); - const nerCandidates = new Map(target.columns.map((column) => [column.id, [] as string[]])); const maxNerValuesPerColumn = boundedCount(this.options.maxNerValuesPerColumn, 8, 8); - const unsupported = new Set(target.columns - .filter((column) => UNSUPPORTED_BINARY_TYPE.test(column.dataType)) - .map((column) => column.id)); - const scannableColumns = target.columns.filter((column) => ( - !unsupported.has(column.id) && evidence.get(column.id)!.length === 0 - )); - let coverage: SensitivityScanCoverage = { kind: "unavailable", observedRows: 0 }; - if (scannableColumns.length > 0 && now() < deadline) { - try { - coverage = await this.values.scanTable({ - ...target, - columns: scannableColumns, - fullScanBudgetMs: this.options.fullScanBudgetMs ?? 5_000, - deadline, - }, (batch) => { - for (const item of batch) { - if (!evidence.has(item.columnId) || item.value === null) continue; - observed.set(item.columnId, (observed.get(item.columnId) ?? 0) + 1); - const matches = evidence.get(item.columnId)!; - if (matches.length === 0 && (item.characterLength ?? item.value.length) > 500) { - matches.push({ kind: "length", ruleId: "text.over_500_characters" }); - } else if (matches.length === 0) { - const match = contentEvidence(item.value); - if (match) matches.push(match); - else { - const candidates = nerCandidates.get(item.columnId)!; - if (candidates.length < maxNerValuesPerColumn && !candidates.includes(item.value)) { - candidates.push(item.value); - } - } - } - } - }, signal); - } catch (error) { - if (!(error instanceof CatalogConnectorError)) throw error; + const states = new Map(); + for (const target of targets) { + for (const column of target.columns) { + const metadataMatch = metadataEvidence(column); + const binary = UNSUPPORTED_BINARY_TYPE.test(column.dataType); + states.set(column.id, { + column, + evidence: metadataMatch + ? [metadataMatch] + : binary + ? [{ kind: "type", ruleId: "type.binary_uninspectable" }] + : [], + observedValues: 0, + nerCandidates: [], + coverage: metadataMatch || binary ? "metadata" : "no_values", + sampledTarget: 0, + }); } } - if (this.detector && (this.detector.isReady?.() ?? true) && !signal.aborted - && now() < deadline && (nerBudget?.remainingMs ?? 1) > 0) { - const candidates: LocalNerCandidate[] = []; - const maxCandidates = boundedCount(this.options.maxNerCandidatesPerTable, 2, 1_024); - candidateSelection: for (let valueIndex = 0; valueIndex < maxNerValuesPerColumn; valueIndex += 1) { - for (const column of target.columns) { - if (evidence.get(column.id)!.length > 0) continue; - const text = nerCandidates.get(column.id)![valueIndex]; - if (text === undefined) continue; - candidates.push({ columnId: column.id, text }); - if (candidates.length >= maxCandidates) break candidateSelection; + const completeTables = new Set(); + for (const [phaseIndex, phase] of SENSITIVITY_SAMPLE_PHASES.entries()) { + for (let offset = 0; offset < targets.length; offset += MAX_CONCURRENT_TABLE_SCANS) { + signal.throwIfAborted(); + const peerController = new AbortController(); + const scanSignal = AbortSignal.any([signal, peerController.signal]); + try { + await Promise.all(targets.slice(offset, offset + MAX_CONCURRENT_TABLE_SCANS).map(async (target) => { + if (completeTables.has(target.table.id)) return; + const columns = target.columns.filter((column) => { + const state = states.get(column.id)!; + return state.evidence.length === 0 + && (!phase.deepTextOnly || DEEP_TEXT_TYPE.test(column.dataType)); + }); + if (columns.length === 0) return; + const coverage = await this.values.scanTable({ + ...target, + columns, + valuesPerColumn: phase.additionalValuesPerColumn, + sampleOffset: phase.targetValuesPerColumn - phase.additionalValuesPerColumn, + sampleSeed: phase.sampleSeed, + queryTimeoutMs: this.options.queryTimeoutMs ?? 5_000, + ...(phaseIndex === 0 ? { fullScanThreshold: 1_000 } : {}), + }, (batch) => { + for (const item of batch) { + if (item.value === null) continue; + const state = states.get(item.columnId); + if (!state || state.evidence.length > 0) continue; + state.observedValues += 1; + if ((item.characterLength ?? item.value.length) > 500) { + state.evidence.push({ kind: "length", ruleId: "text.over_500_characters" }); + continue; + } + const match = contentEvidence(item.value); + if (match) { + state.evidence.push(match); + continue; + } + if (state.nerCandidates.length < maxNerValuesPerColumn + && !state.nerCandidates.includes(item.value)) { + state.nerCandidates.push(item.value); + } + } + }, scanSignal); + for (const column of columns) { + const state = states.get(column.id)!; + state.sampledTarget = Math.max(state.sampledTarget, phase.targetValuesPerColumn); + state.coverage = coverage.kind === "complete" + ? "complete" + : state.observedValues === 0 ? "no_values" : "sampled"; + } + if (coverage.kind === "complete") completeTables.add(target.table.id); + })); + } catch (error) { + peerController.abort(error); + throw error; } } - if (candidates.length > 0) { - const threshold = this.options.nerConfidenceThreshold ?? 0.8; - const nerStartedAt = now(); - const allowedNerMs = nerBudget - ? Math.max(0, nerBudget.remainingMs) - : Math.max(0, deadline - nerStartedAt); - const nerDeadline = Math.min(deadline, nerStartedAt + allowedNerMs); + } + + const nerBudget = sharedNerBudget ?? { remainingMs: 10_000 }; + if (this.detector && (this.detector.isReady?.() ?? true) && !signal.aborted + && nerBudget.remainingMs > 0) { + const maxCandidates = boundedCount(this.options.maxNerCandidatesPerTable, 2, 1_024); + const threshold = this.options.nerConfidenceThreshold ?? 0.8; + for (const target of targets) { + signal.throwIfAborted(); + if (nerBudget.remainingMs <= 0) break; + const candidates: LocalNerCandidate[] = []; + candidateSelection: for (let valueIndex = 0; valueIndex < maxNerValuesPerColumn; valueIndex += 1) { + for (const column of target.columns) { + const state = states.get(column.id)!; + if (state.evidence.length > 0) continue; + const text = state.nerCandidates[valueIndex]; + if (text === undefined) continue; + candidates.push({ columnId: column.id, text }); + if (candidates.length >= maxCandidates) break candidateSelection; + } + } + if (candidates.length === 0) continue; + const startedAt = now(); + const deadline = startedAt + nerBudget.remainingMs; try { for (let offset = 0; offset < candidates.length; offset += MAX_NER_CANDIDATES_PER_REQUEST) { - if (signal.aborted || now() >= nerDeadline) break; + if (signal.aborted || now() >= deadline) break; try { const detected = await this.detector.detect( candidates.slice(offset, offset + MAX_NER_CANDIDATES_PER_REQUEST), signal, - nerDeadline, + deadline, ); for (const item of detected) { - const matches = evidence.get(item.columnId); - if (!matches || matches.length > 0 || !Number.isFinite(item.confidence) + const state = states.get(item.columnId); + if (!state || state.evidence.length > 0 || !Number.isFinite(item.confidence) || item.confidence < threshold || item.confidence > 1) continue; const label = normalizedName(item.label).slice(0, 80); if (!label) continue; - matches.push({ + state.evidence.push({ kind: "ner", ruleId: "ner.entity", label, @@ -400,42 +456,43 @@ export class SensitivityClassifier { } } } finally { - if (nerBudget) { - const elapsedMs = Math.max(1, now() - nerStartedAt); - nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - elapsedMs); - } + nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - Math.max(1, now() - startedAt)); } } } - return target.columns.map((column) => { - const matches = evidence.get(column.id)!; - const count = observed.get(column.id) ?? 0; - const assessment: SensitivityAssessment = matches.length > 0 - ? "sensitive" - : unsupported.has(column.id) || count === 0 || coverage.kind !== "complete" - ? "unknown" - : "non_sensitive"; + return targets.flatMap((target) => target.columns.map((column) => { + const state = states.get(column.id)!; + const sensitive = state.evidence.length > 0; + const coverage = state.observedValues === 0 && !sensitive ? "no_values" : state.coverage; + const coverageEvidence: SensitivityEvidence[] = sensitive + ? state.evidence + : [{ + kind: "coverage", + ruleId: coverage === "complete" + ? "coverage.complete" + : coverage === "no_values" + ? "coverage.no_values" + : `coverage.sampled_${state.sampledTarget}`, + }]; return { columnId: column.id, - assessment, - proposedSensitive: assessment === "unknown" ? column.sensitive : assessment === "sensitive", - evidence: matches.length > 0 - ? matches - : assessment === "unknown" - ? [{ - kind: "coverage", - ruleId: unsupported.has(column.id) - ? "coverage.unsupported_type" - : coverage.kind === "unavailable" - ? "coverage.unavailable" - : count === 0 - ? "coverage.no_values" - : "coverage.incomplete", - }] - : [], - observedValues: count, + assessment: sensitive ? "sensitive" : "non_sensitive", + proposedSensitive: sensitive, + evidence: coverageEvidence, + observedValues: state.observedValues, + coverage, }; - }); + })); + } + + /** Convenience for focused callers and rule-level tests. Production orchestration uses assess(). */ + async assessTable( + target: SensitivityTableTarget, + signal: AbortSignal, + _retiredRunDeadline?: number, + nerBudget?: SensitivityNerBudget, + ): Promise { + return await this.assess([target], signal, nerBudget); } } diff --git a/backend/src/catalog/sensitivity-shadow.ts b/backend/src/catalog/sensitivity-shadow.ts index e7e3b18c..aeb34101 100644 --- a/backend/src/catalog/sensitivity-shadow.ts +++ b/backend/src/catalog/sensitivity-shadow.ts @@ -5,7 +5,10 @@ import { WorkspaceSecretStore } from "../workspaces/secret-store.js"; import { PythonLocalNerDetector } from "./local-ner-detector.js"; import { ConcreteCatalogPostgresAccess } from "./postgres-access.js"; import { createCatalogRepository } from "./repository.js"; -import { SensitivityAnalysisService } from "./sensitivity-analysis-service.js"; +import { + SENSITIVITY_POLICY_VERSION, + SensitivityAnalysisService, +} from "./sensitivity-analysis-service.js"; import { SensitivityClassifier } from "./sensitivity-classifier.js"; import { ConcreteSensitivityValueSource } from "./sensitivity-value-source.js"; @@ -60,23 +63,26 @@ async function main(): Promise { const suggestions = await new SensitivityAnalysisService( repository, new SensitivityClassifier(source, detector), - ).analyze(database.id, "all", [], AbortSignal.timeout(65_000)); - const assessments = { sensitive: 0, nonSensitive: 0, unknown: 0 }; + ).analyze(database.id, "all", [], new AbortController().signal); + const assessments = { sensitive: 0, nonSensitive: 0 }; + const coverage = { metadata: 0, complete: 0, sampled: 0, noValues: 0 }; const rules = new Map(); for (const suggestion of suggestions) { if (suggestion.assessment === "sensitive") assessments.sensitive += 1; - else if (suggestion.assessment === "non_sensitive") assessments.nonSensitive += 1; - else assessments.unknown += 1; + else assessments.nonSensitive += 1; + if (suggestion.coverage === "no_values") coverage.noValues += 1; + else coverage[suggestion.coverage] += 1; for (const evidence of suggestion.evidence) { rules.set(evidence.ruleId, (rules.get(evidence.ruleId) ?? 0) + 1); } } process.stdout.write(`${JSON.stringify({ ok: true, - policyVersion: "sensitivity-v1", + policyVersion: SENSITIVITY_POLICY_VERSION, nerEnabled: detector !== undefined, total: suggestions.length, assessments, + coverage, rules: Object.fromEntries([...rules].sort(([left], [right]) => left.localeCompare(right))), elapsedMs: Date.now() - startedAt, })}\n`); diff --git a/backend/src/catalog/sensitivity-value-source.ts b/backend/src/catalog/sensitivity-value-source.ts index d06d665b..24b73fac 100644 --- a/backend/src/catalog/sensitivity-value-source.ts +++ b/backend/src/catalog/sensitivity-value-source.ts @@ -8,82 +8,117 @@ import type { SensitivityValueObservation, SensitivityValueSource, } from "./sensitivity-classifier.js"; -import { CatalogConnectorError } from "./types.js"; +import { CatalogConnectorError, type CatalogColumn } from "./types.js"; const MAX_VALUE_CHARACTERS = 501; -const DEFAULT_BATCH_ROWS = 200; -const DEFAULT_SAMPLE_ROWS = 200; +const MAX_COLUMNS_PER_QUERY = 25; +const SAMPLE_OVERSCAN_FACTOR = 10; function quoteIdentifier(identifier: string): string { return `"${identifier.replaceAll('"', '""')}"`; } -function projections(request: SensitivityScanRequest): string { - return request.columns.flatMap((column, index) => { +function chunks(items: readonly T[], size: number): T[][] { + const result: T[][] = []; + for (let offset = 0; offset < items.length; offset += size) { + result.push(items.slice(offset, offset + size)); + } + return result; +} + +function tableReference(request: SensitivityScanRequest): string { + return `${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`; +} + +function samplePercentage(valuesPerColumn: number): number { + if (valuesPerColumn <= 300) return 30; + if (valuesPerColumn <= 700) return 70; + return 100; +} + +function flatValueQuery( + request: SensitivityScanRequest, + columns: readonly CatalogColumn[], + options: { complete: boolean; randomized: boolean }, +): string { + const projections = columns.map((column) => quoteIdentifier(column.name)).join(", "); + const perColumnLimit = options.complete + ? request.fullScanThreshold ?? request.valuesPerColumn + : request.valuesPerColumn; + const rowLimit = Math.max(perColumnLimit, perColumnLimit * SAMPLE_OVERSCAN_FACTOR); + const sample = options.complete + ? `SELECT ${projections} FROM ${tableReference(request)}` + : [ + `SELECT ${projections} FROM ${tableReference(request)}`, + ...(options.randomized + ? [`TABLESAMPLE SYSTEM (${samplePercentage(request.valuesPerColumn)}) REPEATABLE (${request.sampleSeed})`] + : []), + `LIMIT ${rowLimit} OFFSET ${request.sampleOffset}`, + ].join(" "); + const values = columns.map((column, index) => { const identifier = quoteIdentifier(column.name); return [ - `LEFT((${identifier})::text, ${MAX_VALUE_CHARACTERS}) AS "__value_${index}"`, - `CASE WHEN ${identifier} IS NULL THEN NULL ELSE char_length((${identifier})::text) END AS "__length_${index}"`, - ]; + `(${index}, LEFT((sampled.${identifier})::text, ${MAX_VALUE_CHARACTERS}),`, + `CASE WHEN sampled.${identifier} IS NULL THEN NULL`, + `ELSE char_length((sampled.${identifier})::text) END)`, + ].join(" "); }).join(", "); + return [ + `WITH sampled AS MATERIALIZED (${sample}),`, + "ranked AS (", + "SELECT value.__column_index, value.__value, value.__length,", + "row_number() OVER (PARTITION BY value.__column_index) AS __rank", + "FROM sampled", + `CROSS JOIN LATERAL (VALUES ${values}) AS value(__column_index, __value, __length)`, + "WHERE value.__value IS NOT NULL", + ")", + "SELECT __column_index, __value, __length FROM ranked", + `WHERE __rank <= ${perColumnLimit}`, + ].join(" "); } function observations( - request: SensitivityScanRequest, + columns: readonly CatalogColumn[], rows: readonly Record[], ): SensitivityValueObservation[] { - return rows.flatMap((row) => request.columns.map((column, index) => { - const sourceValue = row[`__value_${index}`]; - const sourceLength = row[`__length_${index}`]; - const value = sourceValue === null || sourceValue === undefined ? null : String(sourceValue); - const parsedLength = sourceLength === null || sourceLength === undefined + return rows.flatMap((row) => { + const index = Number(row.__column_index); + const column = Number.isSafeInteger(index) && index >= 0 ? columns[index] : undefined; + if (!column || row.__value === null || row.__value === undefined) return []; + const value = String(row.__value); + const parsedLength = row.__length === null || row.__length === undefined ? null - : Number(sourceLength); - return { + : Number(row.__length); + return [{ columnId: column.id, value, characterLength: parsedLength !== null && Number.isSafeInteger(parsedLength) && parsedLength >= 0 ? parsedLength - : value?.length ?? null, - }; - })); + : value.length, + }]; + }); } function cancelled(error: unknown): boolean { return Boolean(error && typeof error === "object" && "code" in error && error.code === "57014"); } -interface SensitivityValueSourceOptions { - now?: () => number; - batchRows?: number; - sampleRows?: number; -} - /** - * PostgreSQL value adapter. It owns bounded read mechanics and emits normalized values, never a - * sensitivity decision. + * Database-specific sampling adapter. Policy stays in SensitivityClassifier; this module only + * produces bounded, normalized non-null observations without persisting or logging values. */ export class ConcreteSensitivityValueSource implements SensitivityValueSource { - private readonly now: () => number; - private readonly batchRows: number; - private readonly sampleRows: number; - constructor( private readonly access: CatalogPostgresAccess, private readonly secretStore?: Pick, - options: SensitivityValueSourceOptions = {}, - ) { - this.now = options.now ?? Date.now; - this.batchRows = options.batchRows ?? DEFAULT_BATCH_ROWS; - this.sampleRows = options.sampleRows ?? DEFAULT_SAMPLE_ROWS; - } + ) {} async scanTable( request: SensitivityScanRequest, consume: (batch: readonly SensitivityValueObservation[]) => void | Promise, signal: AbortSignal, ): Promise { - if (request.columns.length === 0) return { kind: "unavailable", observedRows: 0 }; + if (request.columns.length === 0) return { kind: "complete", observedValues: 0 }; if (request.database.binding.transport === "rest_api") { return await this.scanRest(request, consume, signal); } @@ -97,69 +132,63 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource { ): Promise { const client = await this.access.connect(request.database, signal); let transactionOpen = false; - const startedAt = this.now(); - const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs); - let observedRows = 0; - let cursorOpen = false; + let savepointSequence = 0; + let observedValues = 0; try { - if (signal.aborted || this.now() >= request.deadline) { - return { kind: "sampled", observedRows: 0 }; - } + signal.throwIfAborted(); await client.query("BEGIN TRANSACTION READ ONLY", []); transactionOpen = true; await client.query("SELECT set_config('statement_timeout', $1, true)", [ - `${Math.max(1, Math.floor(fullDeadline - startedAt))}ms`, + `${Math.max(1, Math.floor(request.queryTimeoutMs))}ms`, ]); - await client.query("SAVEPOINT sensitivity_full_scan", []); - const cursor = [ - "DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR", - `SELECT ${projections(request)}`, - `FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`, - ].join(" "); - await client.query(cursor, []); - cursorOpen = true; - while (!signal.aborted && this.now() < fullDeadline) { - let rows: Array>; + const boundedQuery = async (sql: string): Promise> | undefined> => { + signal.throwIfAborted(); + savepointSequence += 1; + const savepoint = `sensitivity_scan_${savepointSequence}`; + await client.query(`SAVEPOINT ${savepoint}`, []); try { - await client.query("SELECT set_config('statement_timeout', $1, true)", [ - `${Math.max(1, Math.floor(fullDeadline - this.now()))}ms`, - ]); - rows = (await client.query( - `FETCH FORWARD ${this.batchRows} FROM sensitivity_full_scan_cursor`, - [], - )).rows; + return (await client.query(sql, [])).rows; } catch (error) { if (!cancelled(error)) throw error; - await client.query("ROLLBACK TO SAVEPOINT sensitivity_full_scan", []); - cursorOpen = false; - break; - } - if (rows.length > 0) { - observedRows += rows.length; - await consume(observations(request, rows)); - } - if (rows.length < this.batchRows) { - return { kind: "complete", observedRows }; + await client.query(`ROLLBACK TO SAVEPOINT ${savepoint}`, []); + return undefined; + } finally { + await client.query(`RELEASE SAVEPOINT ${savepoint}`, []).catch(() => undefined); } + }; + + let complete = false; + if (request.fullScanThreshold !== undefined) { + const probe = await boundedQuery( + `SELECT 1 AS __present FROM ${tableReference(request)} LIMIT ${request.fullScanThreshold + 1}`, + ); + complete = probe !== undefined && probe.length <= request.fullScanThreshold; } - if (signal.aborted || this.now() >= request.deadline) { - return { kind: "sampled", observedRows }; + for (const columnChunk of chunks(request.columns, MAX_COLUMNS_PER_QUERY)) { + signal.throwIfAborted(); + let rows = await boundedQuery(flatValueQuery(request, columnChunk, { + complete, + randomized: !complete, + })); + if (rows === undefined && complete) { + complete = false; + rows = await boundedQuery(flatValueQuery(request, columnChunk, { + complete: false, + randomized: true, + })); + } + if (!complete && (rows === undefined || rows.length === 0)) { + rows = await boundedQuery(flatValueQuery(request, columnChunk, { + complete: false, + randomized: false, + })); + } + if (rows === undefined) throw new CatalogConnectorError("Sensitivity sample query timed out"); + const batch = observations(columnChunk, rows); + observedValues += batch.length; + if (batch.length > 0) await consume(batch); } - if (cursorOpen) await client.query("CLOSE sensitivity_full_scan_cursor", []); - await client.query("RELEASE SAVEPOINT sensitivity_full_scan", []); - await client.query("SELECT set_config('statement_timeout', $1, true)", [ - `${Math.max(1, Math.floor(request.deadline - this.now()))}ms`, - ]); - const sampleSql = [ - `SELECT ${projections(request)}`, - `FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`, - "TABLESAMPLE SYSTEM (1) REPEATABLE (37)", - "LIMIT $1", - ].join(" "); - const sampledRows = (await client.query(sampleSql, [this.sampleRows])).rows; - observedRows += sampledRows.length; - if (sampledRows.length > 0) await consume(observations(request, sampledRows)); - return { kind: "sampled", observedRows }; + return { kind: complete ? "complete" : "sampled", observedValues }; } catch (error) { if (error instanceof CatalogConnectorError) throw error; throw new CatalogConnectorError("Sensitivity source scan failed"); @@ -180,9 +209,7 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource { request.database.workspaceId, auth === "none" ? [] : [CATALOG_SECRET_IDS.apiKey], ); - const startedAt = this.now(); - const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs); - let observedRows = 0; + let observedValues = 0; try { const headers: Record = { "content-type": "application/json" }; if (auth !== "none") { @@ -194,61 +221,66 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource { } const baseUrl = request.database.binding.baseUrl?.replace(/\/+$/u, ""); if (!baseUrl) throw new CatalogConnectorError("Database binding is incomplete"); - const runQuery = async (sql: string, deadline: number): Promise>> => { - const response = await fetch(`${baseUrl}/rpc/run_query`, { - method: "POST", - headers, - body: JSON.stringify({ query_text: sql }), - signal: AbortSignal.any([ - signal, - AbortSignal.timeout(Math.max(1, Math.floor(deadline - this.now()))), - ]), - }); - if (!response.ok) throw new CatalogConnectorError("REST sensitivity source scan failed"); - const body: unknown = await response.json(); - if (!Array.isArray(body) - || body.some((row) => !row || typeof row !== "object" || Array.isArray(row))) { - throw new CatalogConnectorError("REST sensitivity source response is invalid"); + const runQuery = async (sql: string): Promise> | undefined> => { + const timeout = AbortSignal.timeout(Math.max(1, Math.floor(request.queryTimeoutMs))); + try { + const response = await fetch(`${baseUrl}/rpc/run_query`, { + method: "POST", + headers, + body: JSON.stringify({ query_text: sql }), + signal: AbortSignal.any([signal, timeout]), + }); + if (!response.ok) throw new CatalogConnectorError("REST sensitivity source scan failed"); + const body: unknown = await response.json(); + if (!Array.isArray(body) + || body.some((row) => !row || typeof row !== "object" || Array.isArray(row))) { + throw new CatalogConnectorError("REST sensitivity source response is invalid"); + } + return body as Array>; + } catch (error) { + if (signal.aborted) throw error; + if (timeout.aborted) return undefined; + throw error; } - return body as Array>; }; - let offset = 0; - const baseSelect = [ - `SELECT ${projections(request)}`, - `FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`, - ].join(" "); - while (!signal.aborted) { - let rows: Array>; - try { - rows = await runQuery( - `${baseSelect} LIMIT ${this.batchRows} OFFSET ${offset}`, - fullDeadline, - ); - } catch (error) { - if (signal.aborted || this.now() < fullDeadline) throw error; - break; - } - observedRows += rows.length; - if (rows.length > 0) await consume(observations(request, rows)); - if (rows.length < this.batchRows) { - return { kind: offset === 0 ? "complete" : "sampled", observedRows }; - } - offset += rows.length; - if (this.now() >= fullDeadline) break; + let complete = false; + if (request.fullScanThreshold !== undefined) { + const probe = await runQuery( + `SELECT 1 AS __present FROM ${tableReference(request)} LIMIT ${request.fullScanThreshold + 1}`, + ); + complete = probe !== undefined && probe.length <= request.fullScanThreshold; } - if (signal.aborted || this.now() >= request.deadline) { - return { kind: "sampled", observedRows }; + let requestCount = request.fullScanThreshold === undefined ? 0 : 1; + for (const columnChunk of chunks(request.columns, MAX_COLUMNS_PER_QUERY)) { + signal.throwIfAborted(); + let rows = await runQuery(flatValueQuery(request, columnChunk, { + complete, + randomized: !complete, + })); + requestCount += 1; + if (rows === undefined && complete) { + complete = false; + rows = await runQuery(flatValueQuery(request, columnChunk, { + complete: false, + randomized: true, + })); + requestCount += 1; + } + if (!complete && (rows === undefined || rows.length === 0)) { + rows = await runQuery(flatValueQuery(request, columnChunk, { + complete: false, + randomized: false, + })); + requestCount += 1; + } + if (rows === undefined) throw new CatalogConnectorError("REST sensitivity sample query timed out"); + const batch = observations(columnChunk, rows); + observedValues += batch.length; + if (batch.length > 0) await consume(batch); } - const sampleSql = [ - baseSelect, - "TABLESAMPLE SYSTEM (1) REPEATABLE (37)", - `LIMIT ${this.sampleRows}`, - ].join(" "); - const sampledRows = await runQuery(sampleSql, request.deadline); - observedRows += sampledRows.length; - if (sampledRows.length > 0) await consume(observations(request, sampledRows)); - return { kind: "sampled", observedRows }; + // Multiple HTTP requests cannot share a source snapshot, so only one-request reads are complete. + return { kind: complete && requestCount === 1 ? "complete" : "sampled", observedValues }; } catch (error) { if (error instanceof CatalogConnectorError) throw error; throw new CatalogConnectorError("REST sensitivity source scan failed"); diff --git a/backend/src/routes/catalog-description-generation.ts b/backend/src/routes/catalog-description-generation.ts index 82af82bf..8103ab44 100644 --- a/backend/src/routes/catalog-description-generation.ts +++ b/backend/src/routes/catalog-description-generation.ts @@ -261,9 +261,9 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) { }); } if (error instanceof SensitivityAnalysisInterruptedError) { - return reply.code(504).send({ - code: "sensitivity_analysis_timeout", - message: "Sensitivity analysis reached its time limit. No assessments were applied.", + return reply.code(499).send({ + code: "sensitivity_analysis_interrupted", + message: "Sensitivity analysis was interrupted before completion. No assessments were applied.", }); } if (error instanceof CatalogConnectorError) { @@ -284,32 +284,6 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) { }); } -function untilAborted(operation: Promise, signal: AbortSignal): Promise { - if (signal.aborted) { - void operation.catch(() => undefined); - return Promise.reject(new SensitivityAnalysisInterruptedError()); - } - return new Promise((resolve, reject) => { - const abort = () => reject(new SensitivityAnalysisInterruptedError()); - signal.addEventListener("abort", abort, { once: true }); - if (signal.aborted) { - void operation.catch(() => undefined); - abort(); - return; - } - operation.then( - (value) => { - signal.removeEventListener("abort", abort); - resolve(value); - }, - (error: unknown) => { - signal.removeEventListener("abort", abort); - reject(error); - }, - ); - }); -} - function safeSuggestionHistoryError(reply: FastifyReply, error: unknown) { if (error instanceof CatalogUnavailableError) { return reply.code(503).send({ @@ -342,13 +316,22 @@ export function catalogDescriptionGenerationRoutes( try { const databaseId = idSchema.parse((request.params as { databaseId?: unknown }).databaseId); const input = suggestionSchema.parse(request.body); - const signal = AbortSignal.timeout(60_000); - const result = await untilAborted(deps.sensitivityAnalysisRunner.run( - databaseId, - input.scope, - "targetIds" in input ? input.targetIds : [], - signal, - ), signal); + const controller = new AbortController(); + const abort = () => controller.abort(); + request.raw.once("aborted", abort); + reply.raw.once("close", abort); + let result; + try { + result = await deps.sensitivityAnalysisRunner.run( + databaseId, + input.scope, + "targetIds" in input ? input.targetIds : [], + controller.signal, + ); + } finally { + request.raw.off("aborted", abort); + reply.raw.off("close", abort); + } return { suggestions: result.suggestions, run: publicSensitivityAnalysisRun(result.run), diff --git a/backend/test/catalog-description-generation-routes.test.ts b/backend/test/catalog-description-generation-routes.test.ts index 9bed0500..ff8b395e 100644 --- a/backend/test/catalog-description-generation-routes.test.ts +++ b/backend/test/catalog-description-generation-routes.test.ts @@ -77,7 +77,7 @@ async function setup( value: "ordinary", characterLength: 8, }))); - return { kind: "complete", observedRows: 1 }; + return { kind: "complete", observedValues: 1 }; }), }, ) { @@ -167,7 +167,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM scope: "all", engine: "local", modelId: null, - policyVersion: "sensitivity-v1", + policyVersion: "sensitivity-v2", status: "completed", total: 1, suggestedSensitive: 1, @@ -185,6 +185,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM sensitive: true, assessment: "sensitive", evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }], + coverage: "metadata", }], }); expect(await repository.getColumn(database.id, column.tableId, column.id)) @@ -239,10 +240,8 @@ test("assesses sensitive flags locally without persisting them or calling an LLM } }); -test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => { - const controller = new AbortController(); - controller.abort(); - const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal); +test("does not impose a global HTTP deadline on sensitivity analysis", async () => { + const timeout = vi.spyOn(AbortSignal, "timeout"); const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") }); try { @@ -252,12 +251,9 @@ test("stops sensitivity analysis at the HTTP deadline without creating a review" payload: { scope: "all" }, }); - expect(response.statusCode).toBe(504); - expect(response.json()).toEqual({ - code: "sensitivity_analysis_timeout", - message: "Sensitivity analysis reached its time limit. No assessments were applied.", - }); - expect(await repository.listSensitivityAnalysisRuns()).toEqual([]); + expect(response.statusCode).toBe(200); + expect(timeout).not.toHaveBeenCalled(); + expect(await repository.listSensitivityAnalysisRuns()).toHaveLength(1); } finally { timeout.mockRestore(); await app.close(); diff --git a/backend/test/catalog-sensitivity-analysis.test.ts b/backend/test/catalog-sensitivity-analysis.test.ts index 033dc079..f441f486 100644 --- a/backend/test/catalog-sensitivity-analysis.test.ts +++ b/backend/test/catalog-sensitivity-analysis.test.ts @@ -6,7 +6,9 @@ import { import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js"; import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js"; import type { + CatalogColumn, CatalogRepository, + CatalogTable, SensitivityAnalysisRun, WorkspaceDatabase, } from "../src/catalog/types.js"; @@ -24,13 +26,54 @@ const database = { binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" }, } satisfies WorkspaceDatabase; +function catalogTable(id: string, name: string): CatalogTable { + return { + id, + databaseId: database.id, + name, + sourceComment: null, + description: null, + generatedDescription: null, + lastSyncedDatabaseVersion: 1, + lastSyncedAt: "2026-09-02T08:00:00Z", + version: 1, + createdAt: "2026-09-02T08:00:00Z", + updatedAt: "2026-09-02T08:00:00Z", + }; +} + +function catalogColumn(id: string, tableId: string, name: string): CatalogColumn { + return { + id, + tableId, + name, + ordinalPosition: 1, + dataType: "text", + isNullable: true, + defaultExpression: null, + primaryKeyPosition: null, + isPrimaryKey: false, + isForeignKey: false, + foreignKeyCount: 0, + sourceComment: null, + description: null, + generatedDescription: null, + sensitive: false, + lastSyncedDatabaseVersion: 1, + lastSyncedAt: "2026-09-02T08:00:00Z", + version: 1, + createdAt: "2026-09-02T08:00:00Z", + updatedAt: "2026-09-02T08:00:00Z", + }; +} + const running: SensitivityAnalysisRun = { id: "22222222-2222-4222-8222-222222222222", databaseId: database.id, scope: "all", engine: "local", modelId: null, - policyVersion: "sensitivity-v1", + policyVersion: "sensitivity-v2", status: "running", total: 0, suggestedSensitive: 0, @@ -56,7 +99,7 @@ test("stops catalog selection when the request expires during a catalog read", a }), listTables, } as unknown as CatalogRepository; - const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier; + const classifier = { assess: vi.fn() } as unknown as SensitivityClassifier; const analysis = new SensitivityAnalysisService(repository, classifier); await expect(analysis.analyze( @@ -66,7 +109,71 @@ test("stops catalog selection when the request expires during a catalog read", a controller.signal, )).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError); expect(listTables).not.toHaveBeenCalled(); - expect(classifier.assessTable).not.toHaveBeenCalled(); + expect(classifier.assess).not.toHaveBeenCalled(); +}); + +test("classifies all selected tables in one breadth-first run and reports coverage", async () => { + const firstTable = catalogTable("33333333-3333-4333-8333-333333333333", "patients"); + const secondTable = catalogTable("44444444-4444-4444-8444-444444444444", "encounters"); + const firstColumn = catalogColumn( + "55555555-5555-4555-8555-555555555555", + firstTable.id, + "status", + ); + const secondColumn = catalogColumn( + "66666666-6666-4666-8666-666666666666", + secondTable.id, + "note", + ); + const repository = { + get: vi.fn(async () => database), + listTables: vi.fn(async () => [firstTable, secondTable]), + listColumns: vi.fn(async (_databaseId: string, tableId: string) => ( + tableId === firstTable.id ? [firstColumn] : [secondColumn] + )), + } as unknown as CatalogRepository; + const assess = vi.fn(async () => [ + { + columnId: firstColumn.id, + assessment: "non_sensitive" as const, + proposedSensitive: false, + evidence: [{ kind: "coverage" as const, ruleId: "coverage.sampled_1000" }], + observedValues: 1_000, + coverage: "sampled" as const, + }, + { + columnId: secondColumn.id, + assessment: "sensitive" as const, + proposedSensitive: true, + evidence: [{ kind: "content" as const, ruleId: "pii.email" }], + observedValues: 12, + coverage: "sampled" as const, + }, + ]); + const classifier = { assess } as unknown as SensitivityClassifier; + const onPrepared = vi.fn(); + const onProgress = vi.fn(); + + const suggestions = await new SensitivityAnalysisService(repository, classifier).analyze( + database.id, + "all", + [], + new AbortController().signal, + onPrepared, + onProgress, + ); + + expect(assess).toHaveBeenCalledOnce(); + expect(assess.mock.calls[0]![0]).toEqual([ + { database, table: firstTable, columns: [firstColumn] }, + { database, table: secondTable, columns: [secondColumn] }, + ]); + expect(onPrepared).toHaveBeenCalledWith(2); + expect(onProgress.mock.calls.map(([processed]) => processed)).toEqual([1, 2]); + expect(suggestions).toEqual([ + expect.objectContaining({ columnId: firstColumn.id, sensitive: false, coverage: "sampled" }), + expect.objectContaining({ columnId: secondColumn.id, sensitive: true, coverage: "sampled" }), + ]); }); test("marks a created run interrupted if the request deadline expires during persistence", async () => { @@ -94,6 +201,6 @@ test("marks a created run interrupted if the request deadline expires during per status: "interrupted", total: 0, unknown: 0, - errorSummary: "Local sensitivity analysis reached its time limit.", + errorSummary: "Local sensitivity analysis was interrupted before completion.", })); }); diff --git a/backend/test/catalog-sensitivity-classifier.test.ts b/backend/test/catalog-sensitivity-classifier.test.ts index 15dcf7be..3692ab75 100644 --- a/backend/test/catalog-sensitivity-classifier.test.ts +++ b/backend/test/catalog-sensitivity-classifier.test.ts @@ -76,7 +76,7 @@ test("one email hidden in a generically named column makes the whole column sens { columnId: target.id, value: "nessun contatto", characterLength: 16 }, { columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 }, ]], - coverage: { kind: "complete", observedRows: 2 }, + coverage: { kind: "complete", observedValues: 2 }, }); const classifier = new SensitivityClassifier(values); @@ -97,7 +97,7 @@ test("one text value longer than 500 characters makes the whole column sensitive const target = column({ name: "comment" }); const values = source({ batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]], - coverage: { kind: "sampled", observedRows: 1 }, + coverage: { kind: "sampled", observedValues: 1 }, }); const [assessment] = await new SensitivityClassifier(values).assessTable( @@ -112,7 +112,151 @@ test("one text value longer than 500 characters makes the whole column sensitive }); }); -test("complete coverage permits non-sensitive while empty columns remain unknown", async () => { +test("scans every table at 300 before advancing to 1,000 and 3,000 values", async () => { + const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" }; + const first = column({ name: "status" }); + const second = column({ + id: "88888888-8888-4888-8888-888888888888", + tableId: otherTable.id, + name: "comment", + }); + const calls: string[] = []; + const values: SensitivityValueSource = { + scanTable: vi.fn(async (request, consume) => { + calls.push(`${request.table.name}:${request.valuesPerColumn}:${request.sampleOffset}`); + await consume(request.columns.map((item) => ({ + columnId: item.id, + value: "ordinary", + characterLength: 8, + }))); + return { kind: "sampled", observedValues: request.columns.length }; + }), + }; + + await new SensitivityClassifier(values).assess([ + { database, table, columns: [first] }, + { database, table: otherTable, columns: [second] }, + ], new AbortController().signal); + + expect(calls).toEqual([ + "observations:300:0", + "events:300:0", + "observations:700:300", + "events:700:300", + "observations:2000:1000", + "events:2000:1000", + ]); +}); + +test("runs at most two table scans concurrently", async () => { + const targets = Array.from({ length: 3 }, (_, index) => { + const targetTable = { + ...table, + id: `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`, + name: `table_${index + 1}`, + }; + return { + database, + table: targetTable, + columns: [column({ + id: `10000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`, + tableId: targetTable.id, + name: `attribute_${index + 1}`, + })], + }; + }); + let active = 0; + let maximum = 0; + const values: SensitivityValueSource = { + scanTable: vi.fn(async () => { + active += 1; + maximum = Math.max(maximum, active); + await Promise.resolve(); + active -= 1; + return { kind: "complete", observedValues: 0 }; + }), + }; + + await new SensitivityClassifier(values).assess(targets, new AbortController().signal); + + expect(maximum).toBe(2); + expect(values.scanTable).toHaveBeenCalledTimes(3); +}); + +test("aborts a peer table scan when another concurrent source scan fails", async () => { + const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" }; + const first = column({ name: "status" }); + const second = column({ + id: "88888888-8888-4888-8888-888888888888", + tableId: otherTable.id, + name: "comment", + }); + let peerSignal: AbortSignal | undefined; + const failure = new CatalogConnectorError("source unavailable"); + const values: SensitivityValueSource = { + scanTable: vi.fn(async (request, _consume, scanSignal) => { + if (request.table.id === table.id) { + await Promise.resolve(); + throw failure; + } + peerSignal = scanSignal; + return await new Promise((_resolve, reject) => { + scanSignal.addEventListener("abort", () => reject(scanSignal.reason), { once: true }); + }); + }), + }; + + await expect(new SensitivityClassifier(values).assess([ + { database, table, columns: [first] }, + { database, table: otherTable, columns: [second] }, + ], new AbortController().signal)).rejects.toBe(failure); + + expect(peerSignal?.aborted).toBe(true); +}); + +test("stops sampling a column as soon as one value is sensitive", async () => { + const target = column(); + const values: SensitivityValueSource = { + scanTable: vi.fn(async (request, consume) => { + await consume([{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 }]); + return { kind: "sampled", observedValues: 1 }; + }), + }; + + const [assessment] = await new SensitivityClassifier(values).assessTable( + { database, table, columns: [target] }, + new AbortController().signal, + ); + + expect(values.scanTable).toHaveBeenCalledOnce(); + expect(assessment).toMatchObject({ assessment: "sensitive", proposedSensitive: true }); +}); + +test("stops non-text columns after the 1,000-value stage", async () => { + const target = column({ dataType: "integer", name: "sequence_number" }); + const values: SensitivityValueSource = { + scanTable: vi.fn(async (request, consume) => { + await consume([{ columnId: target.id, value: "42", characterLength: 2 }]); + return { kind: "sampled", observedValues: 1 }; + }), + }; + + const [assessment] = await new SensitivityClassifier(values).assessTable( + { database, table, columns: [target] }, + new AbortController().signal, + ); + + expect(vi.mocked(values.scanTable).mock.calls.map(([request]) => request.valuesPerColumn)) + .toEqual([300, 700]); + expect(assessment).toMatchObject({ + assessment: "non_sensitive", + proposedSensitive: false, + coverage: "sampled", + evidence: [{ kind: "coverage", ruleId: "coverage.sampled_1000" }], + }); +}); + +test("complete coverage classifies benign and empty columns as non-sensitive", async () => { const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" }); const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" }); const humanProtected = column({ @@ -126,7 +270,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown { columnId: empty.id, value: null, characterLength: null }, { columnId: humanProtected.id, value: "administrative", characterLength: 14 }, ]], - coverage: { kind: "complete", observedRows: 1 }, + coverage: { kind: "complete", observedValues: 1 }, }); const assessments = await new SensitivityClassifier(values).assessTable( @@ -138,7 +282,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }), expect.objectContaining({ columnId: empty.id, - assessment: "unknown", + assessment: "non_sensitive", proposedSensitive: false, evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }], }), @@ -150,11 +294,11 @@ test("complete coverage permits non-sensitive while empty columns remain unknown ]); }); -test("sampled coverage without a match is unknown and preserves the current human flag", async () => { +test("sampled coverage without a match proposes non-sensitive independently of the current flag", async () => { const target = column({ sensitive: true }); const values = source({ batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]], - coverage: { kind: "sampled", observedRows: 1 }, + coverage: { kind: "sampled", observedValues: 1 }, }); const [assessment] = await new SensitivityClassifier(values).assessTable( @@ -163,13 +307,13 @@ test("sampled coverage without a match is unknown and preserves the current huma ); expect(assessment).toMatchObject({ - assessment: "unknown", - proposedSensitive: true, - evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }], + assessment: "non_sensitive", + proposedSensitive: false, + evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }], }); }); -test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => { +test("an unavailable source fails the analysis instead of producing unknown decisions", async () => { const unresolved = column(); const metadataMatch = column({ id: "44444444-4444-4444-8444-444444444444", @@ -181,30 +325,17 @@ test("an unavailable source produces sanitized unknown evidence without losing m }), }; - const assessments = await new SensitivityClassifier(values).assessTable( + await expect(new SensitivityClassifier(values).assessTable( { database, table, columns: [unresolved, metadataMatch] }, new AbortController().signal, - ); - - expect(assessments).toEqual([ - expect.objectContaining({ - columnId: unresolved.id, - assessment: "unknown", - evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }], - }), - expect.objectContaining({ - columnId: metadataMatch.id, - assessment: "sensitive", - evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }], - }), - ]); + )).rejects.toBeInstanceOf(CatalogConnectorError); }); test("strong Italian PII metadata is sensitive even when the source column is empty", async () => { const target = column({ name: "codice_fiscale" }); const values = source({ batches: [], - coverage: { kind: "complete", observedRows: 0 }, + coverage: { kind: "complete", observedValues: 0 }, }); const [assessment] = await new SensitivityClassifier(values).assessTable( @@ -245,7 +376,7 @@ test.each([ const target = column(); const values = source({ batches: [[{ columnId: target.id, value, characterLength: value.length }]], - coverage: { kind: "complete", observedRows: 1 }, + coverage: { kind: "complete", observedValues: 1 }, }); const [assessment] = await new SensitivityClassifier(values).assessTable( @@ -267,13 +398,16 @@ test("does not make a malformed email decisive", async () => { value: "contatto a@b..com non valido", characterLength: 28, }]], - coverage: { kind: "complete", observedRows: 1 }, + coverage: { kind: "complete", observedValues: 1 }, })).assessTable( { database, table, columns: [target] }, new AbortController().signal, ); - expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] }); + expect(assessment).toMatchObject({ + assessment: "non_sensitive", + evidence: [{ kind: "coverage", ruleId: "coverage.complete" }], + }); }); test("finds a valid email after a malformed candidate in the same value", async () => { @@ -284,7 +418,7 @@ test("finds a valid email after a malformed candidate in the same value", async value: "contatto a@b..com; indirizzo valido mario.rossi@example.it", characterLength: 58, }]], - coverage: { kind: "complete", observedRows: 1 }, + coverage: { kind: "complete", observedValues: 1 }, })).assessTable( { database, table, columns: [target] }, new AbortController().signal, @@ -300,7 +434,7 @@ test("optional local NER evidence can make otherwise ambiguous Italian text sens const target = column(); const values = source({ batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]], - coverage: { kind: "sampled", observedRows: 1 }, + coverage: { kind: "sampled", observedValues: 1 }, }); const detector: LocalNerDetector = { detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]), @@ -331,14 +465,17 @@ test("does not wait for an optional NER worker that is still warming", async () const [assessment] = await new SensitivityClassifier(source({ batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]], - coverage: { kind: "sampled", observedRows: 1 }, + coverage: { kind: "sampled", observedValues: 1 }, }), detector).assessTable( { database, table, columns: [target] }, new AbortController().signal, ); expect(detector.detect).not.toHaveBeenCalled(); - expect(assessment).toMatchObject({ assessment: "unknown" }); + expect(assessment).toMatchObject({ + assessment: "non_sensitive", + evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }], + }); }); test("bounds each optional NER request when an installation raises the per-table work limit", async () => { @@ -359,7 +496,7 @@ test("bounds each optional NER request when an installation raises the per-table await new SensitivityClassifier(source({ batches: [observations], - coverage: { kind: "complete", observedRows: 8 }, + coverage: { kind: "complete", observedValues: 8 }, }), detector, { maxNerCandidatesPerTable: 136 }).assessTable( { database, table, columns }, new AbortController().signal, @@ -386,7 +523,7 @@ test("limits default NER work to two candidates spread across a wide table", asy value: `ordinary-${columnIndex}-${valueIndex}`, characterLength: 13, })))], - coverage: { kind: "complete", observedRows: 2 }, + coverage: { kind: "complete", observedValues: 2 }, }), detector).assessTable( { database, table, columns }, new AbortController().signal, @@ -402,7 +539,7 @@ test("shares a bounded NER time allowance across tables in one analysis run", as const target = column(); const values = source({ batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]], - coverage: { kind: "sampled", observedRows: 1 }, + coverage: { kind: "sampled", observedValues: 1 }, }); const detector: LocalNerDetector = { detect: vi.fn(async () => { @@ -430,11 +567,11 @@ test("shares a bounded NER time allowance across tables in one analysis run", as expect(nerBudget.remainingMs).toBe(0); }); -test("uninterpretable binary content remains unknown after complete coverage", async () => { +test("uninterpretable binary content is protected conservatively without scanning", async () => { const target = column({ dataType: "bytea" }); const values = source({ batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]], - coverage: { kind: "complete", observedRows: 1 }, + coverage: { kind: "complete", observedValues: 1 }, }); const [assessment] = await new SensitivityClassifier(values).assessTable( @@ -443,8 +580,9 @@ test("uninterpretable binary content remains unknown after complete coverage", a ); expect(assessment).toMatchObject({ - assessment: "unknown", - proposedSensitive: false, - evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }], + assessment: "sensitive", + proposedSensitive: true, + evidence: [{ kind: "type", ruleId: "type.binary_uninspectable" }], }); + expect(values.scanTable).not.toHaveBeenCalled(); }); diff --git a/backend/test/catalog-sensitivity-value-source.test.ts b/backend/test/catalog-sensitivity-value-source.test.ts index 269942a5..660d9ec6 100644 --- a/backend/test/catalog-sensitivity-value-source.test.ts +++ b/backend/test/catalog-sensitivity-value-source.test.ts @@ -1,12 +1,17 @@ -import { expect, test, vi } from "vitest"; import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; +import { expect, test, vi } from "vitest"; import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js"; -import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js"; -import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js"; -import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js"; import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js"; +import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js"; +import { + CatalogConnectorError, + type CatalogColumn, + type CatalogTable, + type WorkspaceDatabase, +} from "../src/catalog/types.js"; +import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js"; const database = { id: "11111111-1111-4111-8111-111111111111", @@ -60,128 +65,145 @@ function column(id: string, name: string): CatalogColumn { }; } -test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => { +function request(columns: readonly CatalogColumn[], overrides: Record = {}) { + return { + database, + table, + columns, + valuesPerColumn: 300, + sampleOffset: 0, + sampleSeed: 37, + queryTimeoutMs: 5_000, + fullScanThreshold: 1_000, + ...overrides, + }; +} + +test("uses bounded read-only PostgreSQL sampling for tables above 1,000 rows", async () => { const note = column("33333333-3333-4333-8333-333333333333", "note"); const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value'); - const fullRows = Array.from({ length: 200 }, () => ({ - __value_0: "ordinary", - __length_0: "8", - __value_1: null, - __length_1: null, - })); const query = vi.fn(async (sql: string) => { - if (sql.includes("TABLESAMPLE")) { - return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] }; + if (sql.startsWith("SELECT 1 AS __present")) { + return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) }; + } + if (sql.startsWith("WITH sampled")) { + return { rows: [ + { __column_index: 0, __value: "ordinary", __length: "8" }, + { __column_index: 1, __value: "mario.rossi@example.it", __length: 23 }, + ] }; } - if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows }; return { rows: [] }; }); const end = vi.fn(async () => undefined); const access: CatalogPostgresAccess = { connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient), }; - let clockCalls = 0; - const values = new ConcreteSensitivityValueSource(access, undefined, { - now: () => clockCalls++ < 3 ? 1_000 : 6_100, - }); - const consumed: unknown[] = []; + const consume = vi.fn(); - const coverage = await values.scanTable({ - database, - table, - columns: [note, contact], - fullScanBudgetMs: 5_000, - deadline: 61_000, - }, (batch) => consumed.push(...batch), new AbortController().signal); + await expect(new ConcreteSensitivityValueSource(access).scanTable( + request([note, contact]), + consume, + new AbortController().signal, + )).resolves.toEqual({ kind: "sampled", observedValues: 2 }); - expect(coverage).toEqual({ kind: "sampled", observedRows: 201 }); - expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 }); - expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null }); - expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 }); expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]); - expect(query.mock.calls.some(([sql]) => ( - String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT") - ))).toBe(true); - expect(query.mock.calls.some(([sql]) => String(sql) === ( - "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor" - ))).toBe(true); - expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false); - expect(query.mock.calls.some(([sql]) => ( - String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM') - ))).toBe(true); + expect(query).toHaveBeenCalledWith("SELECT set_config('statement_timeout', $1, true)", ["5000ms"]); + const sampleSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled")); + expect(sampleSql).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)'); + expect(sampleSql).toContain("REPEATABLE (37)"); + expect(sampleSql).toContain("LIMIT 3000 OFFSET 0"); + expect(sampleSql).toContain("CROSS JOIN LATERAL"); + expect(sampleSql).toContain("WHERE __rank <= 300"); + expect(consume).toHaveBeenCalledWith([ + { columnId: note.id, value: "ordinary", characterLength: 8 }, + { columnId: contact.id, value: "mario.rossi@example.it", characterLength: 23 }, + ]); expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]); expect(end).toHaveBeenCalledOnce(); }); -test("reports complete coverage when the final full-scan page is short", async () => { +test("fully scans a table when the 1,001-row probe proves it is small", async () => { const note = column("33333333-3333-4333-8333-333333333333", "note"); - const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD") - ? { rows: [{ __value_0: "ordinary", __length_0: 8 }] } - : { rows: [] }); - const end = vi.fn(async () => undefined); + const query = vi.fn(async (sql: string) => { + if (sql.startsWith("SELECT 1 AS __present")) return { rows: [{ __present: 1 }] }; + if (sql.startsWith("WITH sampled")) { + return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] }; + } + return { rows: [] }; + }); const access: CatalogPostgresAccess = { - connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient), + connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient), }; - const values = new ConcreteSensitivityValueSource(access); const consume = vi.fn(); - const coverage = await values.scanTable({ - database, - table, - columns: [note], - fullScanBudgetMs: 5_000, - deadline: Date.now() + 60_000, - }, consume, new AbortController().signal); + await expect(new ConcreteSensitivityValueSource(access).scanTable( + request([note]), + consume, + new AbortController().signal, + )).resolves.toEqual({ kind: "complete", observedValues: 1 }); - expect(coverage).toEqual({ kind: "complete", observedRows: 1 }); - expect(query.mock.calls.filter(([sql]) => ( - String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor" - ))).toHaveLength(1); + const valueSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled")); + expect(valueSql).not.toContain("TABLESAMPLE"); + expect(valueSql).toContain("WHERE __rank <= 1000"); expect(consume).toHaveBeenCalledWith([ { columnId: note.id, value: "ordinary", characterLength: 8 }, ]); }); -test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => { +test("falls back to sampling when the small-table probe reaches its query timeout", async () => { const note = column("33333333-3333-4333-8333-333333333333", "note"); - let fullScanAttempts = 0; const query = vi.fn(async (sql: string) => { - if (sql.includes("TABLESAMPLE")) { - return { rows: [{ __value_0: "sample", __length_0: 6 }] }; - } - if (sql.startsWith("FETCH FORWARD")) { - fullScanAttempts += 1; + if (sql.startsWith("SELECT 1 AS __present")) { throw Object.assign(new Error("statement timeout"), { code: "57014" }); } + if (sql.startsWith("WITH sampled")) { + return { rows: [{ __column_index: 0, __value: "sample", __length: 6 }] }; + } return { rows: [] }; }); - const end = vi.fn(async () => undefined); const access: CatalogPostgresAccess = { - connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient), + connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient), }; - const values = new ConcreteSensitivityValueSource(access); const consume = vi.fn(); - const coverage = await values.scanTable({ - database, - table, - columns: [note], - fullScanBudgetMs: 5_000, - deadline: Date.now() + 60_000, - }, consume, new AbortController().signal); - - expect(fullScanAttempts).toBe(1); - expect(coverage).toEqual({ kind: "sampled", observedRows: 1 }); - expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([ - "SAVEPOINT sensitivity_full_scan", - "ROLLBACK TO SAVEPOINT sensitivity_full_scan", - ])); - expect(consume).toHaveBeenCalledWith([ - { columnId: note.id, value: "sample", characterLength: 6 }, - ]); + await expect(new ConcreteSensitivityValueSource(access).scanTable( + request([note]), + consume, + new AbortController().signal, + )).resolves.toEqual({ kind: "sampled", observedValues: 1 }); + expect(query.mock.calls.map(([sql]) => String(sql))).toContain( + "ROLLBACK TO SAVEPOINT sensitivity_scan_1", + ); }); -test("scans a REST run_query binding without using PostgreSQL-wire access", async () => { +test("limits each source query to at most 25 columns", async () => { + const columns = Array.from({ length: 26 }, (_, index) => column( + `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`, + `attribute_${index + 1}`, + )); + const query = vi.fn(async (sql: string) => { + if (sql.startsWith("SELECT 1 AS __present")) { + return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) }; + } + if (sql.startsWith("WITH sampled")) { + return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] }; + } + return { rows: [] }; + }); + const access: CatalogPostgresAccess = { + connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient), + }; + + await new ConcreteSensitivityValueSource(access).scanTable( + request(columns), + vi.fn(), + new AbortController().signal, + ); + + expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2); +}); + +test("scans a REST run_query binding without PostgreSQL-wire access", async () => { const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-")); const credentialFile = join(root, "api-key"); writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 }); @@ -193,7 +215,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn })), } as unknown as WorkspaceSecretStore; const fetchMock = vi.fn(async () => new Response(JSON.stringify([ - { __value_0: "mario.rossi@example.it", __length_0: 23 }, + { __column_index: 0, __value: "mario.rossi@example.it", __length: 23 }, ]), { status: 200, headers: { "content-type": "application/json" } })); vi.stubGlobal("fetch", fetchMock); const access: CatalogPostgresAccess = { @@ -213,15 +235,12 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn const consume = vi.fn(); try { - await expect(values.scanTable({ + await expect(values.scanTable(request([note], { database: restDatabase, - table, - columns: [note], - fullScanBudgetMs: 5_000, - deadline: Date.now() + 60_000, - }, consume, new AbortController().signal)).resolves.toEqual({ - kind: "complete", - observedRows: 1, + fullScanThreshold: undefined, + }), consume, new AbortController().signal)).resolves.toEqual({ + kind: "sampled", + observedValues: 1, }); expect(access.connect).not.toHaveBeenCalled(); expect(fetchMock).toHaveBeenCalledWith( @@ -232,7 +251,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn }), ); const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body)); - expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0'); + expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)'); expect(consume).toHaveBeenCalledWith([ { columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 }, ]); @@ -243,74 +262,44 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn } }); -test("keeps multi-request REST scans conservative without a source transaction", async () => { - const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-")); - const credentialFile = join(root, "api-key"); - writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 }); - const secretStore = { - materialize: vi.fn(() => ({ - files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]), - release: vi.fn(), - })), - } as unknown as WorkspaceSecretStore; - const fetchMock = vi.fn() - .mockResolvedValueOnce(new Response(JSON.stringify([ - { __value_0: "ordinary", __length_0: 8 }, - ]), { status: 200 })) - .mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 })); - vi.stubGlobal("fetch", fetchMock); - const values = new ConcreteSensitivityValueSource({ - connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }), - }, secretStore, { batchRows: 1 }); - const restDatabase: WorkspaceDatabase = { - ...database, - binding: { - transport: "rest_api", - baseUrl: "https://dwh.example.test/root", - restPath: "/health", - restAuth: "x-api-key", - }, - }; - - try { - await expect(values.scanTable({ - database: restDatabase, - table, - columns: [column("33333333-3333-4333-8333-333333333333", "note")], - fullScanBudgetMs: 5_000, - deadline: Date.now() + 60_000, - }, vi.fn(), new AbortController().signal)).resolves.toEqual({ - kind: "sampled", - observedRows: 1, - }); - expect(fetchMock).toHaveBeenCalledTimes(2); - } finally { - vi.unstubAllGlobals(); - rmSync(root, { recursive: true, force: true }); - } -}); - -test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => { - const query = vi.fn(async () => ({ rows: [] })); - const end = vi.fn(async () => undefined); - const access: CatalogPostgresAccess = { - connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient), - }; - const now = vi.fn() - .mockReturnValueOnce(1_000) - .mockReturnValue(61_000); - const values = new ConcreteSensitivityValueSource(access, undefined, { now }); - - await expect(values.scanTable({ - database, - table, - columns: [column("33333333-3333-4333-8333-333333333333", "note")], - fullScanBudgetMs: 5_000, - deadline: 60_000, - }, vi.fn(), new AbortController().signal)).resolves.toEqual({ - kind: "sampled", - observedRows: 0, +test("falls back to a sequential bounded sample when randomized sampling times out", async () => { + const note = column("33333333-3333-4333-8333-333333333333", "note"); + const query = vi.fn(async (sql: string) => { + if (sql.startsWith("WITH sampled") && sql.includes("TABLESAMPLE")) { + throw Object.assign(new Error("raw source detail"), { code: "57014" }); + } + if (sql.startsWith("WITH sampled")) { + return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] }; + } + return { rows: [] }; }); - expect(query).not.toHaveBeenCalled(); - expect(end).toHaveBeenCalledOnce(); + const access: CatalogPostgresAccess = { + connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient), + }; + + await expect(new ConcreteSensitivityValueSource(access).scanTable( + request([note], { fullScanThreshold: undefined }), + vi.fn(), + new AbortController().signal, + )).resolves.toEqual({ kind: "sampled", observedValues: 1 }); + expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2); +}); + +test("fails explicitly when both randomized and sequential sample queries time out", async () => { + const note = column("33333333-3333-4333-8333-333333333333", "note"); + const query = vi.fn(async (sql: string) => { + if (sql.startsWith("WITH sampled")) { + throw Object.assign(new Error("raw source detail"), { code: "57014" }); + } + return { rows: [] }; + }); + const access: CatalogPostgresAccess = { + connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient), + }; + + await expect(new ConcreteSensitivityValueSource(access).scanTable( + request([note], { fullScanThreshold: undefined }), + vi.fn(), + new AbortController().signal, + )).rejects.toEqual(new CatalogConnectorError("Sensitivity sample query timed out")); }); diff --git a/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md b/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md index 4b971547..f6b60e84 100644 --- a/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md +++ b/docs/adr/0014-assess-sensitive-columns-locally-from-source-content.md @@ -1,5 +1,5 @@ --- -status: accepted +status: superseded by ADR-0015 --- # Assess sensitive columns locally from source content diff --git a/docs/adr/0015-use-progressive-sampling-for-sensitive-columns.md b/docs/adr/0015-use-progressive-sampling-for-sensitive-columns.md new file mode 100644 index 00000000..68f8d806 --- /dev/null +++ b/docs/adr/0015-use-progressive-sampling-for-sensitive-columns.md @@ -0,0 +1,33 @@ +--- +status: accepted +--- + +# Use progressive sampling for sensitive columns + +The `sensitivity-v1` wall-clock policy produced too many `unknown` assessments: a global +sixty-second deadline coupled the outcome of one column to table order, source latency, and optional +NER cost. Those outcomes were not useful for description-generation gating, because they did not +provide a usable draft Sensitive Data Flag. + +`sensitivity-v2` bounds database effort by inspected values rather than by one global clock. Tables +proven to contain at most 1,000 rows are fully scanned. Larger tables are processed breadth-first in +three passes: 300 values per unresolved column, 700 additional values to reach 1,000, then 2,000 +additional values to reach 3,000 for unresolved text, JSON, and XML columns. One positive rule or +NER finding is enough to stop later work for that column. At most two tables are scanned +concurrently. Source queries contain at most 25 columns and each has a five-second statement timeout. +An empty or timed-out randomized sample gets one sequential bounded retry; two timeouts fail the run. + +A completed v2 analysis returns only `sensitive` or `non_sensitive`. Sampled no-match, empty, and +all-null columns are proposed as `non_sensitive`, with coverage reported independently so the human +reviewer can judge the strength of the proposal. Binary or otherwise uninspectable column types are +proposed as `sensitive`. A source failure fails the analysis and returns no review; it is not +converted into `unknown`. The administrator can still set either final value. + +The HTTP operation has no global analysis deadline. It is canceled when the client disconnects or +the backend restarts. Historic and interrupted run records retain the database field named +`unknown` for compatibility, where it counts unprocessed columns rather than a v2 assessment. + +This decision supersedes ADR-0014 only for scan effort, coverage semantics, and the assessment +domain. ADR-0014 remains authoritative for the single local TypeScript decision point, the absence +of generative LLMs, transient human-reviewed drafts, the 500-character rule, and optional CPU-only +NER evidence. diff --git a/docs/architecture/components.md b/docs/architecture/components.md index b028d0b3..653b5d11 100644 --- a/docs/architecture/components.md +++ b/docs/architecture/components.md @@ -107,16 +107,17 @@ explicitly unlocked; it never resumes automatically. Sensitivity analysis is a synchronous administrative request and does not use the installation model catalog. Database-specific adapters stream bounded normalized values from read-only source connections; the TypeScript `SensitivityClassifier` is the single decision point for -`sensitive | non_sensitive | unknown`. Deterministic rules run first. A complete scan is attempted -for at most five seconds per table, then the adapter samples within the sixty-second request budget. -An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved short text, but it -cannot make or persist the decision itself. +`sensitive | non_sensitive`. Deterministic rules run first. Tables up to 1,000 rows are fully +scanned; larger tables use breadth-first targets of 300, 1,000, and 3,000 values, with the last pass +limited to text-like columns. Source queries have five-second limits, but the request has no global +analysis deadline. An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved +short text, but it cannot make or persist the decision itself. Each attempt has its own durable run and ordered sanitized events, separate from Description -Generation because its lifecycle and counters differ. The run records the local policy version, -coverage aggregates, and sanitized rule identifiers. Proposed flags, source values, NER spans, and -worker diagnostics remain transient. Only an explicit administrator save changes the human-owned -Sensitive Data Flag. +Generation because its lifecycle and counters differ. The run records the local policy version and +aggregate decision counts. Coverage, rule identifiers, proposed flags, source values, NER spans, +and worker diagnostics remain transient. Only an explicit administrator save changes the +human-owned Sensitive Data Flag. ## Main backend classes diff --git a/docs/operations/sensitivity-analysis.md b/docs/operations/sensitivity-analysis.md index 422b3573..3c6a95f1 100644 --- a/docs/operations/sensitivity-analysis.md +++ b/docs/operations/sensitivity-analysis.md @@ -7,7 +7,7 @@ may set either value, including overriding a `sensitive` proposal. ## Default policy -`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v1` +`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v2` policy combines: - normalized column-name rules for direct identifiers, credentials, and health data; @@ -18,22 +18,35 @@ policy combines: - a conservative length rule: any observed textual value longer than 500 characters makes the entire column sensitive. -One decisive value is enough to classify the column as `sensitive`. A complete scan with no match -may classify it as `non_sensitive`. Empty, all-null, binary/uninspectable, interrupted, and sampled -no-match columns are `unknown`; an `unknown` draft preserves the current human flag. +One decisive value is enough to classify the column as `sensitive` and removes it from subsequent +passes. Binary or otherwise uninspectable column types are also proposed as `sensitive`, because +their contents cannot be cleared by the textual rules. A completed analysis has only two draft +outcomes: `sensitive` and `non_sensitive`. Empty or all-null columns are `non_sensitive` with +`no_values` coverage; a sampled column with no match is `non_sensitive` with explicit sampled +coverage. The administrator remains free to reverse either proposal before saving it. Source reads are database-specific, but decisions are database-independent. PostgreSQL direct and REST `run_query` adapters project at most 501 characters per value, use only `SELECT`, and never -persist source values. A full scan gets five seconds per table. If it cannot finish, the adapter uses -a bounded repeatable sample within the sixty-second request deadline. PostgreSQL-wire reads run in a -read-only transaction and always end with rollback. A REST scan can claim complete coverage only -when it finishes in one request; multi-request pagination has no shared source transaction and is -therefore conservatively reported as sampled. +persist source values. Tables proven to contain at most 1,000 rows are fully scanned. Larger tables +are processed breadth-first so every table gets the cheapest pass before any table gets a deeper +one: -The HTTP operation stops waiting at sixty seconds. The same expiring signal is checked before and -after catalog selection, source access, progress writes, and every table. If it expires after a run -has been created, that run is finalized as `interrupted` and all not-decisively-processed columns -are counted as `unknown`; no review payload is returned from the timed-out request. +1. inspect up to 300 non-null values per unresolved column; +2. inspect up to 700 additional values, reaching a 1,000-value target; +3. for unresolved text, JSON, and XML columns only, inspect up to 2,000 additional values, reaching + a 3,000-value target. + +At most two tables are scanned concurrently, and the database adapter groups at most 25 columns in +one source query. Each probe or value query has a five-second statement timeout; PostgreSQL-wire +reads run in a read-only transaction and always end with rollback. Sampling is bounded and +repeatable for a policy version. If a randomized sample is empty or reaches its query timeout, the +adapter tries one sequential bounded sample; if that also times out, the source error fails the run +and returns no review instead of manufacturing `unknown` decisions. + +There is no global sixty-second analysis deadline. Work is bounded by sample counts, per-query +timeouts, and early column exits. The operation is interrupted only when its request connection is +aborted or the backend restarts. Historical or interrupted run counters named `unknown` represent +columns that were not processed; `unknown` is not a `sensitivity-v2` column assessment. History stores only the policy version, aggregate outcomes, timestamps, and fixed operational events. Sanitized rule IDs are returned in the transient review and shadow report, not persisted. @@ -116,13 +129,16 @@ Enabling NER by default requires all of these gates: 1. the pinned artifact and `MODEL_SHA256SUMS` are archived with the installation inventory; 2. the Python dependency/license inventory contains only redistribution-compatible licenses; -3. the CPU benchmark stays within the configured deadlines and does not use a GPU; +3. the CPU benchmark stays within the configured NER allowance and does not use a GPU; 4. the labeled Italian evaluation meets thresholds approved by the product owner. -If a gate fails, leave NER disabled. The deterministic policy remains available and unresolved -columns remain `unknown` rather than being sent to an internal or external LLM. +If a gate fails, leave NER disabled. The deterministic policy remains available and produces the +binary draft from its scan coverage; no content is sent to an internal or external LLM. The first aggregate PSD shadow comparison is recorded in [`2026-09-02-psd-sensitivity-shadow.md`](../reports/2026-09-02-psd-sensitivity-shadow.md). On the -local CPU runner, NER found additional entities but reduced total coverage inside the 60-second -deadline, so the accepted setting remains disabled by default. +local CPU runner, NER found additional entities but reduced total coverage under the superseded +global deadline, so the accepted setting remains disabled by default pending a new v2 benchmark. +The deterministic progressive PSD run is recorded in +[`2026-09-03-psd-progressive-sensitivity-shadow.md`](../reports/2026-09-03-psd-progressive-sensitivity-shadow.md): +it assessed all 2,275 columns with zero `unknown` decisions and left NER disabled. diff --git a/docs/reports/2026-09-03-psd-progressive-sensitivity-shadow.md b/docs/reports/2026-09-03-psd-progressive-sensitivity-shadow.md new file mode 100644 index 00000000..34890527 --- /dev/null +++ b/docs/reports/2026-09-03-psd-progressive-sensitivity-shadow.md @@ -0,0 +1,49 @@ +# PSD progressive sensitivity shadow evaluation + +Date: 2026-09-03 + +This report records an aggregate, non-mutating evaluation of `sensitivity-v2` against the PSD +workspace. The configured connector accessed the source data warehouse with its read-only role. +The shadow command did not create an analysis run, update local catalog metadata, or save Sensitive +Data Flags. No database, table, column, source value, or matched span was emitted. + +The final post-fix run used the local Docker CPU environment, without NER. It inspected all 2,275 +catalog columns through the progressive 300, 1,000, and text-only 3,000-value policy. + +| Sensitive | Non-sensitive | Unknown decisions | Analysis time | +| ---: | ---: | ---: | ---: | +| 343 | 1,932 | 0 | 50,082 ms | + +Coverage was reported independently from the decision: + +| Metadata decision | Complete scan | Sampled | No observed values | +| ---: | ---: | ---: | ---: | +| 39 | 0 | 2,128 | 108 | + +The sampled no-match population comprised 1,337 columns ending after the 1,000-value target and +487 text-like columns ending after the 3,000-value target. Positive findings were: + +| Rule | Columns | +| --- | ---: | +| `pii.phone_number` | 202 | +| `text.over_500_characters` | 52 | +| `metadata.health` | 34 | +| `health.clinical_term` | 25 | +| `pii.italian_vat` | 8 | +| `pii.email` | 7 | +| `metadata.direct_identifier` | 5 | +| `pii.uuid` | 5 | +| `financial.payment_card` | 3 | +| `pii.italian_fiscal_code` | 2 | + +Three successful v2 diagnostic runs produced the same decisions and aggregate rule counts. Their +times ranged from 49,253 to 130,818 ms, showing that source load still affects latency even though +it no longer changes the outcome through a global deadline. An earlier run exposed an intermittent +randomized-query timeout. The adapter now retries that case once with a sequential bounded query; +a regression test covers the fallback, while two consecutive timeouts still fail the whole analysis +instead of creating `unknown` decisions. + +This is a coverage and operational benchmark, not a precision/recall acceptance test. In +particular, the 202 phone-number findings and every other rule family still require human review or +a separately approved labeled corpus before their false-positive rate can be measured. NER remains +disabled by default. diff --git a/frontend/src/api/catalog-databases.ts b/frontend/src/api/catalog-databases.ts index 97e773fa..21da4be8 100644 --- a/frontend/src/api/catalog-databases.ts +++ b/frontend/src/api/catalog-databases.ts @@ -134,14 +134,15 @@ export interface SensitivityReviewItem { version: number; currentSensitive: boolean; sensitive: boolean; - assessment: "sensitive" | "non_sensitive" | "unknown"; + assessment: "sensitive" | "non_sensitive"; evidence: Array<{ - kind: "metadata" | "content" | "length" | "ner" | "coverage"; + kind: "metadata" | "content" | "length" | "ner" | "coverage" | "type"; ruleId: string; label?: string; confidence?: number; }>; observedValues: number; + coverage: "metadata" | "complete" | "sampled" | "no_values"; } export interface SensitivityAnalysisResult { diff --git a/frontend/src/api/client.test.ts b/frontend/src/api/client.test.ts index 7c5e1e3c..2113b3ce 100644 --- a/frontend/src/api/client.test.ts +++ b/frontend/src/api/client.test.ts @@ -149,6 +149,7 @@ test.each([ ["catalog_table_not_found", "One or more selected catalog tables were not found."], ["sensitivity_source_unavailable", "The database content could not be read for sensitivity analysis. No assessments were applied."], ["sensitivity_analysis_timeout", "Sensitivity analysis reached its time limit. No assessments were applied."], + ["sensitivity_analysis_interrupted", "Sensitivity analysis was interrupted before completion. No assessments were applied."], ["sensitive_data_suggestion_history_request_invalid", "Sensitivity analysis history parameters are invalid."], ["sensitive_data_suggestion_history_failed", "Sensitivity analysis history could not be loaded."], ["sensitive_data_suggestion_run_not_found", "The sensitivity analysis run was not found."], diff --git a/frontend/src/api/client.ts b/frontend/src/api/client.ts index 47bac65b..6bd3dc2b 100644 --- a/frontend/src/api/client.ts +++ b/frontend/src/api/client.ts @@ -28,6 +28,7 @@ const safeErrorCodes = new Set([ "sensitive_data_suggestion_no_columns", "sensitivity_source_unavailable", "sensitivity_analysis_timeout", + "sensitivity_analysis_interrupted", "sensitive_data_suggestion_failed", "sensitive_data_suggestion_history_request_invalid", "sensitive_data_suggestion_history_failed", @@ -94,6 +95,7 @@ const localCodeMessages: Record = { sensitive_data_suggestion_no_columns: "The selected scope contains no catalog columns to assess.", sensitivity_source_unavailable: "The database content could not be read for sensitivity analysis. No assessments were applied.", sensitivity_analysis_timeout: "Sensitivity analysis reached its time limit. No assessments were applied.", + sensitivity_analysis_interrupted: "Sensitivity analysis was interrupted before completion. No assessments were applied.", sensitive_data_suggestion_failed: "Sensitivity analysis failed before review. No changes were applied.", sensitive_data_suggestion_history_request_invalid: "Sensitivity analysis history parameters are invalid.", sensitive_data_suggestion_history_failed: "Sensitivity analysis history could not be loaded.", diff --git a/frontend/src/shell/DatabaseManagementPage.test.tsx b/frontend/src/shell/DatabaseManagementPage.test.tsx index 8e0269ad..ef07fb37 100644 --- a/frontend/src/shell/DatabaseManagementPage.test.tsx +++ b/frontend/src/shell/DatabaseManagementPage.test.tsx @@ -167,7 +167,7 @@ function makeSensitivityAnalysisRun( databaseId: "11111111-1111-4111-8111-111111111111", engine: "local", modelId: null, - policyVersion: "sensitivity-v1", + policyVersion: "sensitivity-v2", scope: "selected_columns", status: "completed", total: 2, @@ -2081,6 +2081,7 @@ test("requests database-level sensitivity analysis for the only selected databas assessment: "sensitive", evidence: [{ kind: "content", ruleId: "pii.email" }], observedValues: 1, + coverage: "sampled", }], }); }), @@ -2162,6 +2163,7 @@ test("requests sensitivity analysis only for selected tables", async () => { assessment: "sensitive", evidence: [{ kind: "metadata", ruleId: "metadata.health" }], observedValues: 0, + coverage: "metadata", }], }); }), @@ -2236,6 +2238,7 @@ test("allows a human downgrade and saves only explicit sensitivity changes", asy assessment: "sensitive", evidence: [{ kind: "content", ruleId: "pii.email" }], observedValues: 1, + coverage: "sampled", }, { columnId: nameColumn.id, @@ -2246,8 +2249,9 @@ test("allows a human downgrade and saves only explicit sensitivity changes", asy currentSensitive: true, sensitive: false, assessment: "non_sensitive", - evidence: [], + evidence: [{ kind: "coverage", ruleId: "coverage.complete" }], observedValues: 2, + coverage: "complete", }, ], }); diff --git a/frontend/src/shell/database-management/SensitiveDataReviewDrawer.test.tsx b/frontend/src/shell/database-management/SensitiveDataReviewDrawer.test.tsx index d09e6448..0f32a9db 100644 --- a/frontend/src/shell/database-management/SensitiveDataReviewDrawer.test.tsx +++ b/frontend/src/shell/database-management/SensitiveDataReviewDrawer.test.tsx @@ -21,6 +21,7 @@ const suggestions: SensitivityReviewItem[] = [ assessment: "sensitive", evidence: [{ kind: "content", ruleId: "pii.email" }], observedValues: 1, + coverage: "sampled", }, { columnId: "22222222-2222-4222-8222-222222222222", @@ -33,6 +34,7 @@ const suggestions: SensitivityReviewItem[] = [ assessment: "sensitive", evidence: [{ kind: "metadata", ruleId: "metadata.health" }], observedValues: 0, + coverage: "metadata", }, ]; diff --git a/frontend/src/shell/database-management/SensitiveDataReviewDrawer.tsx b/frontend/src/shell/database-management/SensitiveDataReviewDrawer.tsx index ca437018..77508e99 100644 --- a/frontend/src/shell/database-management/SensitiveDataReviewDrawer.tsx +++ b/frontend/src/shell/database-management/SensitiveDataReviewDrawer.tsx @@ -183,7 +183,7 @@ export function SensitiveDataReviewDrawer({ ? suggestion.evidence.map((item) => item.label ? `${item.ruleId} (${item.label}${item.confidence === undefined ? "" : ` ${Math.round(item.confidence * 100)}%`})` : item.ruleId).join(", ") - : "no sensitive match"}. Observed values: {suggestion.observedValues}. + : "no sensitive match"}. Coverage: {suggestion.coverage.replace("_", " ")}. Observed values: {suggestion.observedValues}.

diff --git a/frontend/src/shell/database-management/SensitivityAnalysisHistoryDrawer.tsx b/frontend/src/shell/database-management/SensitivityAnalysisHistoryDrawer.tsx index 31437235..b5e89b58 100644 --- a/frontend/src/shell/database-management/SensitivityAnalysisHistoryDrawer.tsx +++ b/frontend/src/shell/database-management/SensitivityAnalysisHistoryDrawer.tsx @@ -145,7 +145,7 @@ export function SensitivityAnalysisHistoryDrawer({ ["total", "Total columns"], ["suggestedSensitive", "Sensitive"], ["suggestedNonSensitive", "Not sensitive"], - ["unknown", "Unknown"], + ["unknown", "Unprocessed"], ] as const; return (