feat: sample sensitive columns progressively
This commit is contained in:
+13
-12
@@ -98,16 +98,17 @@ column. The KPI strip reads installation-wide or selected-database aggregates fr
|
||||
description history, and sensitive-field review/history use the production APIs in right-side
|
||||
drawers rather than prototype fixtures; closing a history drawer does not stop its background run.
|
||||
|
||||
Sensitive-field review is now driven by the versioned local `sensitivity-v1` policy, not by a
|
||||
Sensitive-field review is now driven by the versioned local `sensitivity-v2` policy, not by a
|
||||
catalog model. The backend reads selected source tables through read-only, database-specific
|
||||
adapters and makes every `sensitive | non_sensitive | unknown` decision in the TypeScript
|
||||
`SensitivityClassifier`. A single validated match protects the column; a full scan is limited to
|
||||
five seconds per table before sampling and the whole request to sixty seconds. Draft assessments
|
||||
remain transient until an administrator explicitly saves them. Optional GLiNER2 evidence is
|
||||
CPU-only, offline, opt-in, and never replaces the deterministic decision point; see
|
||||
`docs/operations/sensitivity-analysis.md`. The aggregate PSD shadow comparison kept NER disabled by
|
||||
default because its extra findings did not offset the coverage lost to inference within the global
|
||||
deadline; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`.
|
||||
adapters and makes every `sensitive | non_sensitive` draft decision in the TypeScript
|
||||
`SensitivityClassifier`. A single validated match protects the column. Tables up to 1,000 rows are
|
||||
fully scanned; larger tables use breadth-first 300, 1,000, and text-only 3,000-value targets, with a
|
||||
five-second limit per source query and no global request deadline. Source failures fail the run
|
||||
instead of yielding `unknown`; coverage remains visible separately from the proposal. Draft
|
||||
assessments remain transient until an administrator explicitly saves them. Optional GLiNER2
|
||||
evidence is CPU-only, offline, opt-in, and never replaces the deterministic decision point; see
|
||||
`docs/operations/sensitivity-analysis.md`. The earlier v1 PSD shadow comparison kept NER disabled by
|
||||
default; see `docs/reports/2026-09-02-psd-sensitivity-shadow.md`. A v2 PSD benchmark is still due.
|
||||
|
||||
Physical membership, source
|
||||
comments, column types/default/nullability/PK positions, and constraint-level ordered FK pairs are
|
||||
@@ -169,9 +170,9 @@ AI Description Generation uses the catalog's human-owned Sensitive Data Flag. Th
|
||||
`false`, including for newly synchronized columns. An administrator may request a local sensitivity
|
||||
analysis for one selected database, selected tables, or selected columns. One deterministic
|
||||
TypeScript classifier combines metadata, bounded source-content rules, and optional CPU-only NER;
|
||||
no generative model decides the result. Its `sensitive`, `non_sensitive`, or `unknown` assessments
|
||||
remain an unsaved draft until the human reviews and saves any chosen flag changes, including a
|
||||
downgrade to non-sensitive.
|
||||
no generative model decides the result. Its `sensitive` or `non_sensitive` assessments remain an
|
||||
unsaved draft until the human reviews and saves any chosen flag changes, including a downgrade to
|
||||
non-sensitive. Coverage is reported separately; interrupted history may count unprocessed columns.
|
||||
Each started analysis records a separate Sensitivity Analysis Run with aggregate counters and safe
|
||||
ordered events. This operational history never stores per-column assessments, source values,
|
||||
matched spans, prompts, or free-form diagnostics; reloading still discards an unsaved review draft.
|
||||
|
||||
@@ -14,7 +14,7 @@ import type {
|
||||
} from "./types.js";
|
||||
|
||||
const interruptedMessage = "Local sensitivity analysis was interrupted by backend restart.";
|
||||
const deadlineMessage = "Local sensitivity analysis reached its time limit.";
|
||||
const interruptedDuringRunMessage = "Local sensitivity analysis was interrupted before completion.";
|
||||
const failedMessage = "Local sensitivity analysis failed.";
|
||||
|
||||
function ensureActive(signal: AbortSignal): void {
|
||||
@@ -98,14 +98,12 @@ export class SensitivityAnalysisRunner {
|
||||
const suggestedNonSensitive = batch.filter(
|
||||
(suggestion) => suggestion.assessment === "non_sensitive",
|
||||
).length;
|
||||
const unknown = batch.filter((suggestion) => suggestion.assessment === "unknown").length;
|
||||
const current = await this.repository.getSensitivityAnalysisRun(started.id);
|
||||
ensureActive(signal);
|
||||
if (!current) throw new Error("Sensitivity Analysis Run disappeared");
|
||||
const progress = await this.repository.updateSensitivityAnalysisRun(started.id, {
|
||||
suggestedSensitive: current.suggestedSensitive + suggestedSensitive,
|
||||
suggestedNonSensitive: current.suggestedNonSensitive + suggestedNonSensitive,
|
||||
unknown: current.unknown + unknown,
|
||||
});
|
||||
if (!progress) throw new Error("Sensitivity Analysis Run disappeared");
|
||||
processedSensitive += suggestedSensitive;
|
||||
@@ -126,7 +124,6 @@ export class SensitivityAnalysisRunner {
|
||||
const suggestedNonSensitive = suggestions.filter(
|
||||
(suggestion) => suggestion.assessment === "non_sensitive",
|
||||
).length;
|
||||
const unknown = suggestions.filter((suggestion) => suggestion.assessment === "unknown").length;
|
||||
await this.repository.appendSensitivityAnalysisEvent(
|
||||
started.id,
|
||||
"info",
|
||||
@@ -140,7 +137,7 @@ export class SensitivityAnalysisRunner {
|
||||
total: suggestions.length,
|
||||
suggestedSensitive,
|
||||
suggestedNonSensitive,
|
||||
unknown,
|
||||
unknown: 0,
|
||||
finishedAt: new Date().toISOString(),
|
||||
errorSummary: null,
|
||||
});
|
||||
@@ -149,7 +146,7 @@ export class SensitivityAnalysisRunner {
|
||||
return { suggestions, run: completed };
|
||||
} catch (error) {
|
||||
const interrupted = signal.aborted || error instanceof SensitivityAnalysisInterruptedError;
|
||||
const message = interrupted ? deadlineMessage : failedMessage;
|
||||
const message = interrupted ? interruptedDuringRunMessage : failedMessage;
|
||||
await this.repository.updateSensitivityAnalysisRun(started.id, {
|
||||
status: interrupted ? "interrupted" : "failed",
|
||||
...(interrupted ? {
|
||||
|
||||
@@ -12,7 +12,7 @@ import type {
|
||||
} from "./types.js";
|
||||
|
||||
export type { SensitivityAnalysisScope } from "./types.js";
|
||||
export const SENSITIVITY_POLICY_VERSION = "sensitivity-v1";
|
||||
export const SENSITIVITY_POLICY_VERSION = "sensitivity-v2";
|
||||
|
||||
interface SelectedColumn {
|
||||
table: CatalogTable;
|
||||
@@ -30,6 +30,7 @@ export interface SensitivityReviewItem {
|
||||
assessment: SensitivityColumnAssessment["assessment"];
|
||||
evidence: readonly SensitivityEvidence[];
|
||||
observedValues: number;
|
||||
coverage: SensitivityColumnAssessment["coverage"];
|
||||
}
|
||||
|
||||
export class SensitivityAnalysisTargetNotFoundError extends Error {
|
||||
@@ -48,7 +49,7 @@ export class SensitivityAnalysisDuplicateTargetIdsError extends Error {
|
||||
|
||||
export class SensitivityAnalysisInterruptedError extends Error {
|
||||
constructor() {
|
||||
super("sensitivity analysis deadline exceeded");
|
||||
super("sensitivity analysis interrupted");
|
||||
this.name = "SensitivityAnalysisInterruptedError";
|
||||
}
|
||||
}
|
||||
@@ -69,7 +70,7 @@ export class SensitivityAnalysisService {
|
||||
constructor(
|
||||
private readonly repository: CatalogRepository,
|
||||
private readonly classifier: SensitivityClassifier,
|
||||
private readonly options: { runBudgetMs?: number; nerBudgetMs?: number; now?: () => number } = {},
|
||||
private readonly options: { nerBudgetMs?: number } = {},
|
||||
) {}
|
||||
|
||||
private async selectColumns(
|
||||
@@ -116,8 +117,6 @@ export class SensitivityAnalysisService {
|
||||
onPrepared?: (total: number) => void | Promise<void>,
|
||||
onProgress?: (processed: number, suggestions: readonly SensitivityReviewItem[]) => void | Promise<void>,
|
||||
): Promise<readonly SensitivityReviewItem[]> {
|
||||
const now = this.options.now ?? Date.now;
|
||||
const deadline = now() + (this.options.runBudgetMs ?? 60_000);
|
||||
const configuredNerBudget = this.options.nerBudgetMs ?? 10_000;
|
||||
const nerBudget: SensitivityNerBudget = {
|
||||
remainingMs: Number.isFinite(configuredNerBudget) && configuredNerBudget >= 0
|
||||
@@ -137,20 +136,23 @@ export class SensitivityAnalysisService {
|
||||
items.push(item);
|
||||
byTable.set(item.table.id, items);
|
||||
}
|
||||
const suggestions: SensitivityReviewItem[] = [];
|
||||
for (const items of byTable.values()) {
|
||||
ensureActive(signal);
|
||||
const tableTargets = [...byTable.values()].map((items) => {
|
||||
const first = items[0]!;
|
||||
const assessments = await this.classifier.assessTable({
|
||||
return {
|
||||
database,
|
||||
table: first.table,
|
||||
columns: items.map(({ column }) => column),
|
||||
}, signal, deadline, nerBudget);
|
||||
};
|
||||
});
|
||||
const assessments = await this.classifier.assess(tableTargets, signal, nerBudget);
|
||||
ensureActive(signal);
|
||||
const assessmentById = new Map(assessments.map((assessment) => [
|
||||
assessment.columnId,
|
||||
assessment,
|
||||
]));
|
||||
const suggestions: SensitivityReviewItem[] = [];
|
||||
for (const items of byTable.values()) {
|
||||
ensureActive(signal);
|
||||
const batch = items.map(({ table, column }) => {
|
||||
const assessment = assessmentById.get(column.id)!;
|
||||
return {
|
||||
@@ -164,6 +166,7 @@ export class SensitivityAnalysisService {
|
||||
assessment: assessment.assessment,
|
||||
evidence: assessment.evidence,
|
||||
observedValues: assessment.observedValues,
|
||||
coverage: assessment.coverage,
|
||||
};
|
||||
});
|
||||
suggestions.push(...batch);
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
import { CatalogConnectorError, type CatalogColumn, type CatalogTable, type WorkspaceDatabase } from "./types.js";
|
||||
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "./types.js";
|
||||
import { findPhoneNumbersInText } from "libphonenumber-js/max";
|
||||
import validator from "validator";
|
||||
|
||||
export type SensitivityAssessment = "sensitive" | "non_sensitive" | "unknown";
|
||||
export type SensitivityAssessment = "sensitive" | "non_sensitive";
|
||||
|
||||
export interface SensitivityEvidence {
|
||||
kind: "metadata" | "content" | "length" | "ner" | "coverage";
|
||||
kind: "metadata" | "content" | "length" | "ner" | "coverage" | "type";
|
||||
ruleId: string;
|
||||
label?: string;
|
||||
confidence?: number;
|
||||
@@ -18,8 +18,8 @@ export interface SensitivityValueObservation {
|
||||
}
|
||||
|
||||
export interface SensitivityScanCoverage {
|
||||
kind: "complete" | "sampled" | "unavailable";
|
||||
observedRows: number;
|
||||
kind: "complete" | "sampled";
|
||||
observedValues: number;
|
||||
}
|
||||
|
||||
export interface SensitivityTableScan {
|
||||
@@ -31,8 +31,11 @@ export interface SensitivityScanRequest {
|
||||
database: WorkspaceDatabase;
|
||||
table: CatalogTable;
|
||||
columns: readonly CatalogColumn[];
|
||||
fullScanBudgetMs: number;
|
||||
deadline: number;
|
||||
valuesPerColumn: number;
|
||||
sampleOffset: number;
|
||||
sampleSeed: number;
|
||||
queryTimeoutMs: number;
|
||||
fullScanThreshold?: number;
|
||||
}
|
||||
|
||||
export interface SensitivityValueSource {
|
||||
@@ -76,6 +79,7 @@ export interface SensitivityColumnAssessment {
|
||||
proposedSensitive: boolean;
|
||||
evidence: readonly SensitivityEvidence[];
|
||||
observedValues: number;
|
||||
coverage: "metadata" | "complete" | "sampled" | "no_values";
|
||||
}
|
||||
|
||||
export interface SensitivityTableTarget {
|
||||
@@ -94,7 +98,15 @@ const CREDENTIAL_NAME = /(?:^|_)(?:api_key|credential|password|passwd|private_ke
|
||||
const HEALTH_NAME = /(?:^|_)(?:anamnesi|clinical|diagnos(?:i|is)|health|medical|patient|patologia|therapy|terapia)(?:_|$)/u;
|
||||
const CLINICAL_TERM = /(?:^|[^\p{L}])(?:allergi[ae]|anamnesi|carcinoma|chemioterapia|diabete|diagnos[ei]|epatite|farmac[io]|gravidanza|hiv|metastasi|neoplasia|patologia|radioterapia|referto|terapia|tumore)(?:$|[^\p{L}])/iu;
|
||||
const UNSUPPORTED_BINARY_TYPE = /(?:^|\s)(?:binary|blob|bytea|image|varbinary)(?:\s|$|\()/iu;
|
||||
const DEEP_TEXT_TYPE = /(?:^|\s)(?:char|character|citext|clob|json|jsonb|nchar|nvarchar|string|text|varchar|xml)(?:\s|$|\()/iu;
|
||||
const MAX_NER_CANDIDATES_PER_REQUEST = 128;
|
||||
const MAX_CONCURRENT_TABLE_SCANS = 2;
|
||||
|
||||
export const SENSITIVITY_SAMPLE_PHASES = [
|
||||
{ targetValuesPerColumn: 300, additionalValuesPerColumn: 300, sampleSeed: 37, deepTextOnly: false },
|
||||
{ targetValuesPerColumn: 1_000, additionalValuesPerColumn: 700, sampleSeed: 73, deepTextOnly: false },
|
||||
{ targetValuesPerColumn: 3_000, additionalValuesPerColumn: 2_000, sampleSeed: 109, deepTextOnly: true },
|
||||
] as const;
|
||||
|
||||
function normalizedName(value: string): string {
|
||||
return value.normalize("NFKD")
|
||||
@@ -284,14 +296,22 @@ function contentEvidence(value: string): SensitivityEvidence | undefined {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
interface ColumnState {
|
||||
column: CatalogColumn;
|
||||
evidence: SensitivityEvidence[];
|
||||
observedValues: number;
|
||||
nerCandidates: string[];
|
||||
coverage: "metadata" | "complete" | "sampled" | "no_values";
|
||||
sampledTarget: number;
|
||||
}
|
||||
|
||||
/** Sole decision module for local column-level sensitivity assessments. */
|
||||
export class SensitivityClassifier {
|
||||
constructor(
|
||||
private readonly values: SensitivityValueSource,
|
||||
private readonly detector?: LocalNerDetector,
|
||||
private readonly options: {
|
||||
fullScanBudgetMs?: number;
|
||||
runBudgetMs?: number;
|
||||
queryTimeoutMs?: number;
|
||||
nerConfidenceThreshold?: number;
|
||||
maxNerValuesPerColumn?: number;
|
||||
maxNerCandidatesPerTable?: number;
|
||||
@@ -299,95 +319,131 @@ export class SensitivityClassifier {
|
||||
} = {},
|
||||
) {}
|
||||
|
||||
async assessTable(
|
||||
target: SensitivityTableTarget,
|
||||
async assess(
|
||||
targets: readonly SensitivityTableTarget[],
|
||||
signal: AbortSignal,
|
||||
runDeadline?: number,
|
||||
nerBudget?: SensitivityNerBudget,
|
||||
sharedNerBudget?: SensitivityNerBudget,
|
||||
): Promise<readonly SensitivityColumnAssessment[]> {
|
||||
const now = this.options.now ?? Date.now;
|
||||
const deadline = runDeadline ?? now() + (this.options.runBudgetMs ?? 60_000);
|
||||
const evidence = new Map(target.columns.map((column) => {
|
||||
const match = metadataEvidence(column);
|
||||
return [column.id, match ? [match] : [] as SensitivityEvidence[]];
|
||||
}));
|
||||
const observed = new Map(target.columns.map((column) => [column.id, 0]));
|
||||
const nerCandidates = new Map(target.columns.map((column) => [column.id, [] as string[]]));
|
||||
const maxNerValuesPerColumn = boundedCount(this.options.maxNerValuesPerColumn, 8, 8);
|
||||
const unsupported = new Set(target.columns
|
||||
.filter((column) => UNSUPPORTED_BINARY_TYPE.test(column.dataType))
|
||||
.map((column) => column.id));
|
||||
const scannableColumns = target.columns.filter((column) => (
|
||||
!unsupported.has(column.id) && evidence.get(column.id)!.length === 0
|
||||
));
|
||||
let coverage: SensitivityScanCoverage = { kind: "unavailable", observedRows: 0 };
|
||||
if (scannableColumns.length > 0 && now() < deadline) {
|
||||
try {
|
||||
coverage = await this.values.scanTable({
|
||||
...target,
|
||||
columns: scannableColumns,
|
||||
fullScanBudgetMs: this.options.fullScanBudgetMs ?? 5_000,
|
||||
deadline,
|
||||
}, (batch) => {
|
||||
for (const item of batch) {
|
||||
if (!evidence.has(item.columnId) || item.value === null) continue;
|
||||
observed.set(item.columnId, (observed.get(item.columnId) ?? 0) + 1);
|
||||
const matches = evidence.get(item.columnId)!;
|
||||
if (matches.length === 0 && (item.characterLength ?? item.value.length) > 500) {
|
||||
matches.push({ kind: "length", ruleId: "text.over_500_characters" });
|
||||
} else if (matches.length === 0) {
|
||||
const match = contentEvidence(item.value);
|
||||
if (match) matches.push(match);
|
||||
else {
|
||||
const candidates = nerCandidates.get(item.columnId)!;
|
||||
if (candidates.length < maxNerValuesPerColumn && !candidates.includes(item.value)) {
|
||||
candidates.push(item.value);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}, signal);
|
||||
} catch (error) {
|
||||
if (!(error instanceof CatalogConnectorError)) throw error;
|
||||
const states = new Map<string, ColumnState>();
|
||||
for (const target of targets) {
|
||||
for (const column of target.columns) {
|
||||
const metadataMatch = metadataEvidence(column);
|
||||
const binary = UNSUPPORTED_BINARY_TYPE.test(column.dataType);
|
||||
states.set(column.id, {
|
||||
column,
|
||||
evidence: metadataMatch
|
||||
? [metadataMatch]
|
||||
: binary
|
||||
? [{ kind: "type", ruleId: "type.binary_uninspectable" }]
|
||||
: [],
|
||||
observedValues: 0,
|
||||
nerCandidates: [],
|
||||
coverage: metadataMatch || binary ? "metadata" : "no_values",
|
||||
sampledTarget: 0,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const completeTables = new Set<string>();
|
||||
for (const [phaseIndex, phase] of SENSITIVITY_SAMPLE_PHASES.entries()) {
|
||||
for (let offset = 0; offset < targets.length; offset += MAX_CONCURRENT_TABLE_SCANS) {
|
||||
signal.throwIfAborted();
|
||||
const peerController = new AbortController();
|
||||
const scanSignal = AbortSignal.any([signal, peerController.signal]);
|
||||
try {
|
||||
await Promise.all(targets.slice(offset, offset + MAX_CONCURRENT_TABLE_SCANS).map(async (target) => {
|
||||
if (completeTables.has(target.table.id)) return;
|
||||
const columns = target.columns.filter((column) => {
|
||||
const state = states.get(column.id)!;
|
||||
return state.evidence.length === 0
|
||||
&& (!phase.deepTextOnly || DEEP_TEXT_TYPE.test(column.dataType));
|
||||
});
|
||||
if (columns.length === 0) return;
|
||||
const coverage = await this.values.scanTable({
|
||||
...target,
|
||||
columns,
|
||||
valuesPerColumn: phase.additionalValuesPerColumn,
|
||||
sampleOffset: phase.targetValuesPerColumn - phase.additionalValuesPerColumn,
|
||||
sampleSeed: phase.sampleSeed,
|
||||
queryTimeoutMs: this.options.queryTimeoutMs ?? 5_000,
|
||||
...(phaseIndex === 0 ? { fullScanThreshold: 1_000 } : {}),
|
||||
}, (batch) => {
|
||||
for (const item of batch) {
|
||||
if (item.value === null) continue;
|
||||
const state = states.get(item.columnId);
|
||||
if (!state || state.evidence.length > 0) continue;
|
||||
state.observedValues += 1;
|
||||
if ((item.characterLength ?? item.value.length) > 500) {
|
||||
state.evidence.push({ kind: "length", ruleId: "text.over_500_characters" });
|
||||
continue;
|
||||
}
|
||||
const match = contentEvidence(item.value);
|
||||
if (match) {
|
||||
state.evidence.push(match);
|
||||
continue;
|
||||
}
|
||||
if (state.nerCandidates.length < maxNerValuesPerColumn
|
||||
&& !state.nerCandidates.includes(item.value)) {
|
||||
state.nerCandidates.push(item.value);
|
||||
}
|
||||
}
|
||||
}, scanSignal);
|
||||
for (const column of columns) {
|
||||
const state = states.get(column.id)!;
|
||||
state.sampledTarget = Math.max(state.sampledTarget, phase.targetValuesPerColumn);
|
||||
state.coverage = coverage.kind === "complete"
|
||||
? "complete"
|
||||
: state.observedValues === 0 ? "no_values" : "sampled";
|
||||
}
|
||||
if (coverage.kind === "complete") completeTables.add(target.table.id);
|
||||
}));
|
||||
} catch (error) {
|
||||
peerController.abort(error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const nerBudget = sharedNerBudget ?? { remainingMs: 10_000 };
|
||||
if (this.detector && (this.detector.isReady?.() ?? true) && !signal.aborted
|
||||
&& now() < deadline && (nerBudget?.remainingMs ?? 1) > 0) {
|
||||
const candidates: LocalNerCandidate[] = [];
|
||||
&& nerBudget.remainingMs > 0) {
|
||||
const maxCandidates = boundedCount(this.options.maxNerCandidatesPerTable, 2, 1_024);
|
||||
const threshold = this.options.nerConfidenceThreshold ?? 0.8;
|
||||
for (const target of targets) {
|
||||
signal.throwIfAborted();
|
||||
if (nerBudget.remainingMs <= 0) break;
|
||||
const candidates: LocalNerCandidate[] = [];
|
||||
candidateSelection: for (let valueIndex = 0; valueIndex < maxNerValuesPerColumn; valueIndex += 1) {
|
||||
for (const column of target.columns) {
|
||||
if (evidence.get(column.id)!.length > 0) continue;
|
||||
const text = nerCandidates.get(column.id)![valueIndex];
|
||||
const state = states.get(column.id)!;
|
||||
if (state.evidence.length > 0) continue;
|
||||
const text = state.nerCandidates[valueIndex];
|
||||
if (text === undefined) continue;
|
||||
candidates.push({ columnId: column.id, text });
|
||||
if (candidates.length >= maxCandidates) break candidateSelection;
|
||||
}
|
||||
}
|
||||
if (candidates.length > 0) {
|
||||
const threshold = this.options.nerConfidenceThreshold ?? 0.8;
|
||||
const nerStartedAt = now();
|
||||
const allowedNerMs = nerBudget
|
||||
? Math.max(0, nerBudget.remainingMs)
|
||||
: Math.max(0, deadline - nerStartedAt);
|
||||
const nerDeadline = Math.min(deadline, nerStartedAt + allowedNerMs);
|
||||
if (candidates.length === 0) continue;
|
||||
const startedAt = now();
|
||||
const deadline = startedAt + nerBudget.remainingMs;
|
||||
try {
|
||||
for (let offset = 0; offset < candidates.length; offset += MAX_NER_CANDIDATES_PER_REQUEST) {
|
||||
if (signal.aborted || now() >= nerDeadline) break;
|
||||
if (signal.aborted || now() >= deadline) break;
|
||||
try {
|
||||
const detected = await this.detector.detect(
|
||||
candidates.slice(offset, offset + MAX_NER_CANDIDATES_PER_REQUEST),
|
||||
signal,
|
||||
nerDeadline,
|
||||
deadline,
|
||||
);
|
||||
for (const item of detected) {
|
||||
const matches = evidence.get(item.columnId);
|
||||
if (!matches || matches.length > 0 || !Number.isFinite(item.confidence)
|
||||
const state = states.get(item.columnId);
|
||||
if (!state || state.evidence.length > 0 || !Number.isFinite(item.confidence)
|
||||
|| item.confidence < threshold || item.confidence > 1) continue;
|
||||
const label = normalizedName(item.label).slice(0, 80);
|
||||
if (!label) continue;
|
||||
matches.push({
|
||||
state.evidence.push({
|
||||
kind: "ner",
|
||||
ruleId: "ner.entity",
|
||||
label,
|
||||
@@ -400,42 +456,43 @@ export class SensitivityClassifier {
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
if (nerBudget) {
|
||||
const elapsedMs = Math.max(1, now() - nerStartedAt);
|
||||
nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - elapsedMs);
|
||||
}
|
||||
nerBudget.remainingMs = Math.max(0, nerBudget.remainingMs - Math.max(1, now() - startedAt));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return target.columns.map((column) => {
|
||||
const matches = evidence.get(column.id)!;
|
||||
const count = observed.get(column.id) ?? 0;
|
||||
const assessment: SensitivityAssessment = matches.length > 0
|
||||
? "sensitive"
|
||||
: unsupported.has(column.id) || count === 0 || coverage.kind !== "complete"
|
||||
? "unknown"
|
||||
: "non_sensitive";
|
||||
return targets.flatMap((target) => target.columns.map((column) => {
|
||||
const state = states.get(column.id)!;
|
||||
const sensitive = state.evidence.length > 0;
|
||||
const coverage = state.observedValues === 0 && !sensitive ? "no_values" : state.coverage;
|
||||
const coverageEvidence: SensitivityEvidence[] = sensitive
|
||||
? state.evidence
|
||||
: [{
|
||||
kind: "coverage",
|
||||
ruleId: coverage === "complete"
|
||||
? "coverage.complete"
|
||||
: coverage === "no_values"
|
||||
? "coverage.no_values"
|
||||
: `coverage.sampled_${state.sampledTarget}`,
|
||||
}];
|
||||
return {
|
||||
columnId: column.id,
|
||||
assessment,
|
||||
proposedSensitive: assessment === "unknown" ? column.sensitive : assessment === "sensitive",
|
||||
evidence: matches.length > 0
|
||||
? matches
|
||||
: assessment === "unknown"
|
||||
? [{
|
||||
kind: "coverage",
|
||||
ruleId: unsupported.has(column.id)
|
||||
? "coverage.unsupported_type"
|
||||
: coverage.kind === "unavailable"
|
||||
? "coverage.unavailable"
|
||||
: count === 0
|
||||
? "coverage.no_values"
|
||||
: "coverage.incomplete",
|
||||
}]
|
||||
: [],
|
||||
observedValues: count,
|
||||
assessment: sensitive ? "sensitive" : "non_sensitive",
|
||||
proposedSensitive: sensitive,
|
||||
evidence: coverageEvidence,
|
||||
observedValues: state.observedValues,
|
||||
coverage,
|
||||
};
|
||||
});
|
||||
}));
|
||||
}
|
||||
|
||||
/** Convenience for focused callers and rule-level tests. Production orchestration uses assess(). */
|
||||
async assessTable(
|
||||
target: SensitivityTableTarget,
|
||||
signal: AbortSignal,
|
||||
_retiredRunDeadline?: number,
|
||||
nerBudget?: SensitivityNerBudget,
|
||||
): Promise<readonly SensitivityColumnAssessment[]> {
|
||||
return await this.assess([target], signal, nerBudget);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,7 +5,10 @@ import { WorkspaceSecretStore } from "../workspaces/secret-store.js";
|
||||
import { PythonLocalNerDetector } from "./local-ner-detector.js";
|
||||
import { ConcreteCatalogPostgresAccess } from "./postgres-access.js";
|
||||
import { createCatalogRepository } from "./repository.js";
|
||||
import { SensitivityAnalysisService } from "./sensitivity-analysis-service.js";
|
||||
import {
|
||||
SENSITIVITY_POLICY_VERSION,
|
||||
SensitivityAnalysisService,
|
||||
} from "./sensitivity-analysis-service.js";
|
||||
import { SensitivityClassifier } from "./sensitivity-classifier.js";
|
||||
import { ConcreteSensitivityValueSource } from "./sensitivity-value-source.js";
|
||||
|
||||
@@ -60,23 +63,26 @@ async function main(): Promise<void> {
|
||||
const suggestions = await new SensitivityAnalysisService(
|
||||
repository,
|
||||
new SensitivityClassifier(source, detector),
|
||||
).analyze(database.id, "all", [], AbortSignal.timeout(65_000));
|
||||
const assessments = { sensitive: 0, nonSensitive: 0, unknown: 0 };
|
||||
).analyze(database.id, "all", [], new AbortController().signal);
|
||||
const assessments = { sensitive: 0, nonSensitive: 0 };
|
||||
const coverage = { metadata: 0, complete: 0, sampled: 0, noValues: 0 };
|
||||
const rules = new Map<string, number>();
|
||||
for (const suggestion of suggestions) {
|
||||
if (suggestion.assessment === "sensitive") assessments.sensitive += 1;
|
||||
else if (suggestion.assessment === "non_sensitive") assessments.nonSensitive += 1;
|
||||
else assessments.unknown += 1;
|
||||
else assessments.nonSensitive += 1;
|
||||
if (suggestion.coverage === "no_values") coverage.noValues += 1;
|
||||
else coverage[suggestion.coverage] += 1;
|
||||
for (const evidence of suggestion.evidence) {
|
||||
rules.set(evidence.ruleId, (rules.get(evidence.ruleId) ?? 0) + 1);
|
||||
}
|
||||
}
|
||||
process.stdout.write(`${JSON.stringify({
|
||||
ok: true,
|
||||
policyVersion: "sensitivity-v1",
|
||||
policyVersion: SENSITIVITY_POLICY_VERSION,
|
||||
nerEnabled: detector !== undefined,
|
||||
total: suggestions.length,
|
||||
assessments,
|
||||
coverage,
|
||||
rules: Object.fromEntries([...rules].sort(([left], [right]) => left.localeCompare(right))),
|
||||
elapsedMs: Date.now() - startedAt,
|
||||
})}\n`);
|
||||
|
||||
@@ -8,82 +8,117 @@ import type {
|
||||
SensitivityValueObservation,
|
||||
SensitivityValueSource,
|
||||
} from "./sensitivity-classifier.js";
|
||||
import { CatalogConnectorError } from "./types.js";
|
||||
import { CatalogConnectorError, type CatalogColumn } from "./types.js";
|
||||
|
||||
const MAX_VALUE_CHARACTERS = 501;
|
||||
const DEFAULT_BATCH_ROWS = 200;
|
||||
const DEFAULT_SAMPLE_ROWS = 200;
|
||||
const MAX_COLUMNS_PER_QUERY = 25;
|
||||
const SAMPLE_OVERSCAN_FACTOR = 10;
|
||||
|
||||
function quoteIdentifier(identifier: string): string {
|
||||
return `"${identifier.replaceAll('"', '""')}"`;
|
||||
}
|
||||
|
||||
function projections(request: SensitivityScanRequest): string {
|
||||
return request.columns.flatMap((column, index) => {
|
||||
function chunks<T>(items: readonly T[], size: number): T[][] {
|
||||
const result: T[][] = [];
|
||||
for (let offset = 0; offset < items.length; offset += size) {
|
||||
result.push(items.slice(offset, offset + size));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
function tableReference(request: SensitivityScanRequest): string {
|
||||
return `${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`;
|
||||
}
|
||||
|
||||
function samplePercentage(valuesPerColumn: number): number {
|
||||
if (valuesPerColumn <= 300) return 30;
|
||||
if (valuesPerColumn <= 700) return 70;
|
||||
return 100;
|
||||
}
|
||||
|
||||
function flatValueQuery(
|
||||
request: SensitivityScanRequest,
|
||||
columns: readonly CatalogColumn[],
|
||||
options: { complete: boolean; randomized: boolean },
|
||||
): string {
|
||||
const projections = columns.map((column) => quoteIdentifier(column.name)).join(", ");
|
||||
const perColumnLimit = options.complete
|
||||
? request.fullScanThreshold ?? request.valuesPerColumn
|
||||
: request.valuesPerColumn;
|
||||
const rowLimit = Math.max(perColumnLimit, perColumnLimit * SAMPLE_OVERSCAN_FACTOR);
|
||||
const sample = options.complete
|
||||
? `SELECT ${projections} FROM ${tableReference(request)}`
|
||||
: [
|
||||
`SELECT ${projections} FROM ${tableReference(request)}`,
|
||||
...(options.randomized
|
||||
? [`TABLESAMPLE SYSTEM (${samplePercentage(request.valuesPerColumn)}) REPEATABLE (${request.sampleSeed})`]
|
||||
: []),
|
||||
`LIMIT ${rowLimit} OFFSET ${request.sampleOffset}`,
|
||||
].join(" ");
|
||||
const values = columns.map((column, index) => {
|
||||
const identifier = quoteIdentifier(column.name);
|
||||
return [
|
||||
`LEFT((${identifier})::text, ${MAX_VALUE_CHARACTERS}) AS "__value_${index}"`,
|
||||
`CASE WHEN ${identifier} IS NULL THEN NULL ELSE char_length((${identifier})::text) END AS "__length_${index}"`,
|
||||
];
|
||||
`(${index}, LEFT((sampled.${identifier})::text, ${MAX_VALUE_CHARACTERS}),`,
|
||||
`CASE WHEN sampled.${identifier} IS NULL THEN NULL`,
|
||||
`ELSE char_length((sampled.${identifier})::text) END)`,
|
||||
].join(" ");
|
||||
}).join(", ");
|
||||
return [
|
||||
`WITH sampled AS MATERIALIZED (${sample}),`,
|
||||
"ranked AS (",
|
||||
"SELECT value.__column_index, value.__value, value.__length,",
|
||||
"row_number() OVER (PARTITION BY value.__column_index) AS __rank",
|
||||
"FROM sampled",
|
||||
`CROSS JOIN LATERAL (VALUES ${values}) AS value(__column_index, __value, __length)`,
|
||||
"WHERE value.__value IS NOT NULL",
|
||||
")",
|
||||
"SELECT __column_index, __value, __length FROM ranked",
|
||||
`WHERE __rank <= ${perColumnLimit}`,
|
||||
].join(" ");
|
||||
}
|
||||
|
||||
function observations(
|
||||
request: SensitivityScanRequest,
|
||||
columns: readonly CatalogColumn[],
|
||||
rows: readonly Record<string, unknown>[],
|
||||
): SensitivityValueObservation[] {
|
||||
return rows.flatMap((row) => request.columns.map((column, index) => {
|
||||
const sourceValue = row[`__value_${index}`];
|
||||
const sourceLength = row[`__length_${index}`];
|
||||
const value = sourceValue === null || sourceValue === undefined ? null : String(sourceValue);
|
||||
const parsedLength = sourceLength === null || sourceLength === undefined
|
||||
return rows.flatMap((row) => {
|
||||
const index = Number(row.__column_index);
|
||||
const column = Number.isSafeInteger(index) && index >= 0 ? columns[index] : undefined;
|
||||
if (!column || row.__value === null || row.__value === undefined) return [];
|
||||
const value = String(row.__value);
|
||||
const parsedLength = row.__length === null || row.__length === undefined
|
||||
? null
|
||||
: Number(sourceLength);
|
||||
return {
|
||||
: Number(row.__length);
|
||||
return [{
|
||||
columnId: column.id,
|
||||
value,
|
||||
characterLength: parsedLength !== null && Number.isSafeInteger(parsedLength) && parsedLength >= 0
|
||||
? parsedLength
|
||||
: value?.length ?? null,
|
||||
};
|
||||
}));
|
||||
: value.length,
|
||||
}];
|
||||
});
|
||||
}
|
||||
|
||||
function cancelled(error: unknown): boolean {
|
||||
return Boolean(error && typeof error === "object" && "code" in error && error.code === "57014");
|
||||
}
|
||||
|
||||
interface SensitivityValueSourceOptions {
|
||||
now?: () => number;
|
||||
batchRows?: number;
|
||||
sampleRows?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* PostgreSQL value adapter. It owns bounded read mechanics and emits normalized values, never a
|
||||
* sensitivity decision.
|
||||
* Database-specific sampling adapter. Policy stays in SensitivityClassifier; this module only
|
||||
* produces bounded, normalized non-null observations without persisting or logging values.
|
||||
*/
|
||||
export class ConcreteSensitivityValueSource implements SensitivityValueSource {
|
||||
private readonly now: () => number;
|
||||
private readonly batchRows: number;
|
||||
private readonly sampleRows: number;
|
||||
|
||||
constructor(
|
||||
private readonly access: CatalogPostgresAccess,
|
||||
private readonly secretStore?: Pick<WorkspaceSecretStore, "materialize">,
|
||||
options: SensitivityValueSourceOptions = {},
|
||||
) {
|
||||
this.now = options.now ?? Date.now;
|
||||
this.batchRows = options.batchRows ?? DEFAULT_BATCH_ROWS;
|
||||
this.sampleRows = options.sampleRows ?? DEFAULT_SAMPLE_ROWS;
|
||||
}
|
||||
) {}
|
||||
|
||||
async scanTable(
|
||||
request: SensitivityScanRequest,
|
||||
consume: (batch: readonly SensitivityValueObservation[]) => void | Promise<void>,
|
||||
signal: AbortSignal,
|
||||
): Promise<SensitivityScanCoverage> {
|
||||
if (request.columns.length === 0) return { kind: "unavailable", observedRows: 0 };
|
||||
if (request.columns.length === 0) return { kind: "complete", observedValues: 0 };
|
||||
if (request.database.binding.transport === "rest_api") {
|
||||
return await this.scanRest(request, consume, signal);
|
||||
}
|
||||
@@ -97,69 +132,63 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
|
||||
): Promise<SensitivityScanCoverage> {
|
||||
const client = await this.access.connect(request.database, signal);
|
||||
let transactionOpen = false;
|
||||
const startedAt = this.now();
|
||||
const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
|
||||
let observedRows = 0;
|
||||
let cursorOpen = false;
|
||||
let savepointSequence = 0;
|
||||
let observedValues = 0;
|
||||
try {
|
||||
if (signal.aborted || this.now() >= request.deadline) {
|
||||
return { kind: "sampled", observedRows: 0 };
|
||||
}
|
||||
signal.throwIfAborted();
|
||||
await client.query("BEGIN TRANSACTION READ ONLY", []);
|
||||
transactionOpen = true;
|
||||
await client.query("SELECT set_config('statement_timeout', $1, true)", [
|
||||
`${Math.max(1, Math.floor(fullDeadline - startedAt))}ms`,
|
||||
`${Math.max(1, Math.floor(request.queryTimeoutMs))}ms`,
|
||||
]);
|
||||
await client.query("SAVEPOINT sensitivity_full_scan", []);
|
||||
const cursor = [
|
||||
"DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR",
|
||||
`SELECT ${projections(request)}`,
|
||||
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
|
||||
].join(" ");
|
||||
await client.query(cursor, []);
|
||||
cursorOpen = true;
|
||||
while (!signal.aborted && this.now() < fullDeadline) {
|
||||
let rows: Array<Record<string, unknown>>;
|
||||
const boundedQuery = async (sql: string): Promise<Array<Record<string, unknown>> | undefined> => {
|
||||
signal.throwIfAborted();
|
||||
savepointSequence += 1;
|
||||
const savepoint = `sensitivity_scan_${savepointSequence}`;
|
||||
await client.query(`SAVEPOINT ${savepoint}`, []);
|
||||
try {
|
||||
await client.query("SELECT set_config('statement_timeout', $1, true)", [
|
||||
`${Math.max(1, Math.floor(fullDeadline - this.now()))}ms`,
|
||||
]);
|
||||
rows = (await client.query(
|
||||
`FETCH FORWARD ${this.batchRows} FROM sensitivity_full_scan_cursor`,
|
||||
[],
|
||||
)).rows;
|
||||
return (await client.query(sql, [])).rows;
|
||||
} catch (error) {
|
||||
if (!cancelled(error)) throw error;
|
||||
await client.query("ROLLBACK TO SAVEPOINT sensitivity_full_scan", []);
|
||||
cursorOpen = false;
|
||||
break;
|
||||
await client.query(`ROLLBACK TO SAVEPOINT ${savepoint}`, []);
|
||||
return undefined;
|
||||
} finally {
|
||||
await client.query(`RELEASE SAVEPOINT ${savepoint}`, []).catch(() => undefined);
|
||||
}
|
||||
if (rows.length > 0) {
|
||||
observedRows += rows.length;
|
||||
await consume(observations(request, rows));
|
||||
};
|
||||
|
||||
let complete = false;
|
||||
if (request.fullScanThreshold !== undefined) {
|
||||
const probe = await boundedQuery(
|
||||
`SELECT 1 AS __present FROM ${tableReference(request)} LIMIT ${request.fullScanThreshold + 1}`,
|
||||
);
|
||||
complete = probe !== undefined && probe.length <= request.fullScanThreshold;
|
||||
}
|
||||
if (rows.length < this.batchRows) {
|
||||
return { kind: "complete", observedRows };
|
||||
for (const columnChunk of chunks(request.columns, MAX_COLUMNS_PER_QUERY)) {
|
||||
signal.throwIfAborted();
|
||||
let rows = await boundedQuery(flatValueQuery(request, columnChunk, {
|
||||
complete,
|
||||
randomized: !complete,
|
||||
}));
|
||||
if (rows === undefined && complete) {
|
||||
complete = false;
|
||||
rows = await boundedQuery(flatValueQuery(request, columnChunk, {
|
||||
complete: false,
|
||||
randomized: true,
|
||||
}));
|
||||
}
|
||||
if (!complete && (rows === undefined || rows.length === 0)) {
|
||||
rows = await boundedQuery(flatValueQuery(request, columnChunk, {
|
||||
complete: false,
|
||||
randomized: false,
|
||||
}));
|
||||
}
|
||||
if (signal.aborted || this.now() >= request.deadline) {
|
||||
return { kind: "sampled", observedRows };
|
||||
if (rows === undefined) throw new CatalogConnectorError("Sensitivity sample query timed out");
|
||||
const batch = observations(columnChunk, rows);
|
||||
observedValues += batch.length;
|
||||
if (batch.length > 0) await consume(batch);
|
||||
}
|
||||
if (cursorOpen) await client.query("CLOSE sensitivity_full_scan_cursor", []);
|
||||
await client.query("RELEASE SAVEPOINT sensitivity_full_scan", []);
|
||||
await client.query("SELECT set_config('statement_timeout', $1, true)", [
|
||||
`${Math.max(1, Math.floor(request.deadline - this.now()))}ms`,
|
||||
]);
|
||||
const sampleSql = [
|
||||
`SELECT ${projections(request)}`,
|
||||
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
|
||||
"TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
|
||||
"LIMIT $1",
|
||||
].join(" ");
|
||||
const sampledRows = (await client.query(sampleSql, [this.sampleRows])).rows;
|
||||
observedRows += sampledRows.length;
|
||||
if (sampledRows.length > 0) await consume(observations(request, sampledRows));
|
||||
return { kind: "sampled", observedRows };
|
||||
return { kind: complete ? "complete" : "sampled", observedValues };
|
||||
} catch (error) {
|
||||
if (error instanceof CatalogConnectorError) throw error;
|
||||
throw new CatalogConnectorError("Sensitivity source scan failed");
|
||||
@@ -180,9 +209,7 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
|
||||
request.database.workspaceId,
|
||||
auth === "none" ? [] : [CATALOG_SECRET_IDS.apiKey],
|
||||
);
|
||||
const startedAt = this.now();
|
||||
const fullDeadline = Math.min(request.deadline, startedAt + request.fullScanBudgetMs);
|
||||
let observedRows = 0;
|
||||
let observedValues = 0;
|
||||
try {
|
||||
const headers: Record<string, string> = { "content-type": "application/json" };
|
||||
if (auth !== "none") {
|
||||
@@ -194,15 +221,14 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
|
||||
}
|
||||
const baseUrl = request.database.binding.baseUrl?.replace(/\/+$/u, "");
|
||||
if (!baseUrl) throw new CatalogConnectorError("Database binding is incomplete");
|
||||
const runQuery = async (sql: string, deadline: number): Promise<Array<Record<string, unknown>>> => {
|
||||
const runQuery = async (sql: string): Promise<Array<Record<string, unknown>> | undefined> => {
|
||||
const timeout = AbortSignal.timeout(Math.max(1, Math.floor(request.queryTimeoutMs)));
|
||||
try {
|
||||
const response = await fetch(`${baseUrl}/rpc/run_query`, {
|
||||
method: "POST",
|
||||
headers,
|
||||
body: JSON.stringify({ query_text: sql }),
|
||||
signal: AbortSignal.any([
|
||||
signal,
|
||||
AbortSignal.timeout(Math.max(1, Math.floor(deadline - this.now()))),
|
||||
]),
|
||||
signal: AbortSignal.any([signal, timeout]),
|
||||
});
|
||||
if (!response.ok) throw new CatalogConnectorError("REST sensitivity source scan failed");
|
||||
const body: unknown = await response.json();
|
||||
@@ -211,44 +237,50 @@ export class ConcreteSensitivityValueSource implements SensitivityValueSource {
|
||||
throw new CatalogConnectorError("REST sensitivity source response is invalid");
|
||||
}
|
||||
return body as Array<Record<string, unknown>>;
|
||||
} catch (error) {
|
||||
if (signal.aborted) throw error;
|
||||
if (timeout.aborted) return undefined;
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
let offset = 0;
|
||||
const baseSelect = [
|
||||
`SELECT ${projections(request)}`,
|
||||
`FROM ${quoteIdentifier(request.database.schema)}.${quoteIdentifier(request.table.name)}`,
|
||||
].join(" ");
|
||||
while (!signal.aborted) {
|
||||
let rows: Array<Record<string, unknown>>;
|
||||
try {
|
||||
rows = await runQuery(
|
||||
`${baseSelect} LIMIT ${this.batchRows} OFFSET ${offset}`,
|
||||
fullDeadline,
|
||||
let complete = false;
|
||||
if (request.fullScanThreshold !== undefined) {
|
||||
const probe = await runQuery(
|
||||
`SELECT 1 AS __present FROM ${tableReference(request)} LIMIT ${request.fullScanThreshold + 1}`,
|
||||
);
|
||||
} catch (error) {
|
||||
if (signal.aborted || this.now() < fullDeadline) throw error;
|
||||
break;
|
||||
complete = probe !== undefined && probe.length <= request.fullScanThreshold;
|
||||
}
|
||||
observedRows += rows.length;
|
||||
if (rows.length > 0) await consume(observations(request, rows));
|
||||
if (rows.length < this.batchRows) {
|
||||
return { kind: offset === 0 ? "complete" : "sampled", observedRows };
|
||||
let requestCount = request.fullScanThreshold === undefined ? 0 : 1;
|
||||
for (const columnChunk of chunks(request.columns, MAX_COLUMNS_PER_QUERY)) {
|
||||
signal.throwIfAborted();
|
||||
let rows = await runQuery(flatValueQuery(request, columnChunk, {
|
||||
complete,
|
||||
randomized: !complete,
|
||||
}));
|
||||
requestCount += 1;
|
||||
if (rows === undefined && complete) {
|
||||
complete = false;
|
||||
rows = await runQuery(flatValueQuery(request, columnChunk, {
|
||||
complete: false,
|
||||
randomized: true,
|
||||
}));
|
||||
requestCount += 1;
|
||||
}
|
||||
offset += rows.length;
|
||||
if (this.now() >= fullDeadline) break;
|
||||
if (!complete && (rows === undefined || rows.length === 0)) {
|
||||
rows = await runQuery(flatValueQuery(request, columnChunk, {
|
||||
complete: false,
|
||||
randomized: false,
|
||||
}));
|
||||
requestCount += 1;
|
||||
}
|
||||
if (signal.aborted || this.now() >= request.deadline) {
|
||||
return { kind: "sampled", observedRows };
|
||||
if (rows === undefined) throw new CatalogConnectorError("REST sensitivity sample query timed out");
|
||||
const batch = observations(columnChunk, rows);
|
||||
observedValues += batch.length;
|
||||
if (batch.length > 0) await consume(batch);
|
||||
}
|
||||
const sampleSql = [
|
||||
baseSelect,
|
||||
"TABLESAMPLE SYSTEM (1) REPEATABLE (37)",
|
||||
`LIMIT ${this.sampleRows}`,
|
||||
].join(" ");
|
||||
const sampledRows = await runQuery(sampleSql, request.deadline);
|
||||
observedRows += sampledRows.length;
|
||||
if (sampledRows.length > 0) await consume(observations(request, sampledRows));
|
||||
return { kind: "sampled", observedRows };
|
||||
// Multiple HTTP requests cannot share a source snapshot, so only one-request reads are complete.
|
||||
return { kind: complete && requestCount === 1 ? "complete" : "sampled", observedValues };
|
||||
} catch (error) {
|
||||
if (error instanceof CatalogConnectorError) throw error;
|
||||
throw new CatalogConnectorError("REST sensitivity source scan failed");
|
||||
|
||||
@@ -261,9 +261,9 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
|
||||
});
|
||||
}
|
||||
if (error instanceof SensitivityAnalysisInterruptedError) {
|
||||
return reply.code(504).send({
|
||||
code: "sensitivity_analysis_timeout",
|
||||
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
|
||||
return reply.code(499).send({
|
||||
code: "sensitivity_analysis_interrupted",
|
||||
message: "Sensitivity analysis was interrupted before completion. No assessments were applied.",
|
||||
});
|
||||
}
|
||||
if (error instanceof CatalogConnectorError) {
|
||||
@@ -284,32 +284,6 @@ function safeSuggestionError(reply: FastifyReply, error: unknown) {
|
||||
});
|
||||
}
|
||||
|
||||
function untilAborted<T>(operation: Promise<T>, signal: AbortSignal): Promise<T> {
|
||||
if (signal.aborted) {
|
||||
void operation.catch(() => undefined);
|
||||
return Promise.reject(new SensitivityAnalysisInterruptedError());
|
||||
}
|
||||
return new Promise<T>((resolve, reject) => {
|
||||
const abort = () => reject(new SensitivityAnalysisInterruptedError());
|
||||
signal.addEventListener("abort", abort, { once: true });
|
||||
if (signal.aborted) {
|
||||
void operation.catch(() => undefined);
|
||||
abort();
|
||||
return;
|
||||
}
|
||||
operation.then(
|
||||
(value) => {
|
||||
signal.removeEventListener("abort", abort);
|
||||
resolve(value);
|
||||
},
|
||||
(error: unknown) => {
|
||||
signal.removeEventListener("abort", abort);
|
||||
reject(error);
|
||||
},
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
function safeSuggestionHistoryError(reply: FastifyReply, error: unknown) {
|
||||
if (error instanceof CatalogUnavailableError) {
|
||||
return reply.code(503).send({
|
||||
@@ -342,13 +316,22 @@ export function catalogDescriptionGenerationRoutes(
|
||||
try {
|
||||
const databaseId = idSchema.parse((request.params as { databaseId?: unknown }).databaseId);
|
||||
const input = suggestionSchema.parse(request.body);
|
||||
const signal = AbortSignal.timeout(60_000);
|
||||
const result = await untilAborted(deps.sensitivityAnalysisRunner.run(
|
||||
const controller = new AbortController();
|
||||
const abort = () => controller.abort();
|
||||
request.raw.once("aborted", abort);
|
||||
reply.raw.once("close", abort);
|
||||
let result;
|
||||
try {
|
||||
result = await deps.sensitivityAnalysisRunner.run(
|
||||
databaseId,
|
||||
input.scope,
|
||||
"targetIds" in input ? input.targetIds : [],
|
||||
signal,
|
||||
), signal);
|
||||
controller.signal,
|
||||
);
|
||||
} finally {
|
||||
request.raw.off("aborted", abort);
|
||||
reply.raw.off("close", abort);
|
||||
}
|
||||
return {
|
||||
suggestions: result.suggestions,
|
||||
run: publicSensitivityAnalysisRun(result.run),
|
||||
|
||||
@@ -77,7 +77,7 @@ async function setup(
|
||||
value: "ordinary",
|
||||
characterLength: 8,
|
||||
})));
|
||||
return { kind: "complete", observedRows: 1 };
|
||||
return { kind: "complete", observedValues: 1 };
|
||||
}),
|
||||
},
|
||||
) {
|
||||
@@ -167,7 +167,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
|
||||
scope: "all",
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
policyVersion: "sensitivity-v2",
|
||||
status: "completed",
|
||||
total: 1,
|
||||
suggestedSensitive: 1,
|
||||
@@ -185,6 +185,7 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
|
||||
sensitive: true,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
|
||||
coverage: "metadata",
|
||||
}],
|
||||
});
|
||||
expect(await repository.getColumn(database.id, column.tableId, column.id))
|
||||
@@ -239,10 +240,8 @@ test("assesses sensitive flags locally without persisting them or calling an LLM
|
||||
}
|
||||
});
|
||||
|
||||
test("stops sensitivity analysis at the HTTP deadline without creating a review", async () => {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
const timeout = vi.spyOn(AbortSignal, "timeout").mockReturnValue(controller.signal);
|
||||
test("does not impose a global HTTP deadline on sensitivity analysis", async () => {
|
||||
const timeout = vi.spyOn(AbortSignal, "timeout");
|
||||
const { app, repository, database } = await setup({ complete: vi.fn(async () => "unused") });
|
||||
|
||||
try {
|
||||
@@ -252,12 +251,9 @@ test("stops sensitivity analysis at the HTTP deadline without creating a review"
|
||||
payload: { scope: "all" },
|
||||
});
|
||||
|
||||
expect(response.statusCode).toBe(504);
|
||||
expect(response.json()).toEqual({
|
||||
code: "sensitivity_analysis_timeout",
|
||||
message: "Sensitivity analysis reached its time limit. No assessments were applied.",
|
||||
});
|
||||
expect(await repository.listSensitivityAnalysisRuns()).toEqual([]);
|
||||
expect(response.statusCode).toBe(200);
|
||||
expect(timeout).not.toHaveBeenCalled();
|
||||
expect(await repository.listSensitivityAnalysisRuns()).toHaveLength(1);
|
||||
} finally {
|
||||
timeout.mockRestore();
|
||||
await app.close();
|
||||
|
||||
@@ -6,7 +6,9 @@ import {
|
||||
import { SensitivityAnalysisRunner } from "../src/catalog/sensitivity-analysis-runner.js";
|
||||
import type { SensitivityClassifier } from "../src/catalog/sensitivity-classifier.js";
|
||||
import type {
|
||||
CatalogColumn,
|
||||
CatalogRepository,
|
||||
CatalogTable,
|
||||
SensitivityAnalysisRun,
|
||||
WorkspaceDatabase,
|
||||
} from "../src/catalog/types.js";
|
||||
@@ -24,13 +26,54 @@ const database = {
|
||||
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
|
||||
} satisfies WorkspaceDatabase;
|
||||
|
||||
function catalogTable(id: string, name: string): CatalogTable {
|
||||
return {
|
||||
id,
|
||||
databaseId: database.id,
|
||||
name,
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
};
|
||||
}
|
||||
|
||||
function catalogColumn(id: string, tableId: string, name: string): CatalogColumn {
|
||||
return {
|
||||
id,
|
||||
tableId,
|
||||
name,
|
||||
ordinalPosition: 1,
|
||||
dataType: "text",
|
||||
isNullable: true,
|
||||
defaultExpression: null,
|
||||
primaryKeyPosition: null,
|
||||
isPrimaryKey: false,
|
||||
isForeignKey: false,
|
||||
foreignKeyCount: 0,
|
||||
sourceComment: null,
|
||||
description: null,
|
||||
generatedDescription: null,
|
||||
sensitive: false,
|
||||
lastSyncedDatabaseVersion: 1,
|
||||
lastSyncedAt: "2026-09-02T08:00:00Z",
|
||||
version: 1,
|
||||
createdAt: "2026-09-02T08:00:00Z",
|
||||
updatedAt: "2026-09-02T08:00:00Z",
|
||||
};
|
||||
}
|
||||
|
||||
const running: SensitivityAnalysisRun = {
|
||||
id: "22222222-2222-4222-8222-222222222222",
|
||||
databaseId: database.id,
|
||||
scope: "all",
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
policyVersion: "sensitivity-v2",
|
||||
status: "running",
|
||||
total: 0,
|
||||
suggestedSensitive: 0,
|
||||
@@ -56,7 +99,7 @@ test("stops catalog selection when the request expires during a catalog read", a
|
||||
}),
|
||||
listTables,
|
||||
} as unknown as CatalogRepository;
|
||||
const classifier = { assessTable: vi.fn() } as unknown as SensitivityClassifier;
|
||||
const classifier = { assess: vi.fn() } as unknown as SensitivityClassifier;
|
||||
const analysis = new SensitivityAnalysisService(repository, classifier);
|
||||
|
||||
await expect(analysis.analyze(
|
||||
@@ -66,7 +109,71 @@ test("stops catalog selection when the request expires during a catalog read", a
|
||||
controller.signal,
|
||||
)).rejects.toBeInstanceOf(SensitivityAnalysisInterruptedError);
|
||||
expect(listTables).not.toHaveBeenCalled();
|
||||
expect(classifier.assessTable).not.toHaveBeenCalled();
|
||||
expect(classifier.assess).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test("classifies all selected tables in one breadth-first run and reports coverage", async () => {
|
||||
const firstTable = catalogTable("33333333-3333-4333-8333-333333333333", "patients");
|
||||
const secondTable = catalogTable("44444444-4444-4444-8444-444444444444", "encounters");
|
||||
const firstColumn = catalogColumn(
|
||||
"55555555-5555-4555-8555-555555555555",
|
||||
firstTable.id,
|
||||
"status",
|
||||
);
|
||||
const secondColumn = catalogColumn(
|
||||
"66666666-6666-4666-8666-666666666666",
|
||||
secondTable.id,
|
||||
"note",
|
||||
);
|
||||
const repository = {
|
||||
get: vi.fn(async () => database),
|
||||
listTables: vi.fn(async () => [firstTable, secondTable]),
|
||||
listColumns: vi.fn(async (_databaseId: string, tableId: string) => (
|
||||
tableId === firstTable.id ? [firstColumn] : [secondColumn]
|
||||
)),
|
||||
} as unknown as CatalogRepository;
|
||||
const assess = vi.fn(async () => [
|
||||
{
|
||||
columnId: firstColumn.id,
|
||||
assessment: "non_sensitive" as const,
|
||||
proposedSensitive: false,
|
||||
evidence: [{ kind: "coverage" as const, ruleId: "coverage.sampled_1000" }],
|
||||
observedValues: 1_000,
|
||||
coverage: "sampled" as const,
|
||||
},
|
||||
{
|
||||
columnId: secondColumn.id,
|
||||
assessment: "sensitive" as const,
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "content" as const, ruleId: "pii.email" }],
|
||||
observedValues: 12,
|
||||
coverage: "sampled" as const,
|
||||
},
|
||||
]);
|
||||
const classifier = { assess } as unknown as SensitivityClassifier;
|
||||
const onPrepared = vi.fn();
|
||||
const onProgress = vi.fn();
|
||||
|
||||
const suggestions = await new SensitivityAnalysisService(repository, classifier).analyze(
|
||||
database.id,
|
||||
"all",
|
||||
[],
|
||||
new AbortController().signal,
|
||||
onPrepared,
|
||||
onProgress,
|
||||
);
|
||||
|
||||
expect(assess).toHaveBeenCalledOnce();
|
||||
expect(assess.mock.calls[0]![0]).toEqual([
|
||||
{ database, table: firstTable, columns: [firstColumn] },
|
||||
{ database, table: secondTable, columns: [secondColumn] },
|
||||
]);
|
||||
expect(onPrepared).toHaveBeenCalledWith(2);
|
||||
expect(onProgress.mock.calls.map(([processed]) => processed)).toEqual([1, 2]);
|
||||
expect(suggestions).toEqual([
|
||||
expect.objectContaining({ columnId: firstColumn.id, sensitive: false, coverage: "sampled" }),
|
||||
expect.objectContaining({ columnId: secondColumn.id, sensitive: true, coverage: "sampled" }),
|
||||
]);
|
||||
});
|
||||
|
||||
test("marks a created run interrupted if the request deadline expires during persistence", async () => {
|
||||
@@ -94,6 +201,6 @@ test("marks a created run interrupted if the request deadline expires during per
|
||||
status: "interrupted",
|
||||
total: 0,
|
||||
unknown: 0,
|
||||
errorSummary: "Local sensitivity analysis reached its time limit.",
|
||||
errorSummary: "Local sensitivity analysis was interrupted before completion.",
|
||||
}));
|
||||
});
|
||||
|
||||
@@ -76,7 +76,7 @@ test("one email hidden in a generically named column makes the whole column sens
|
||||
{ columnId: target.id, value: "nessun contatto", characterLength: 16 },
|
||||
{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 },
|
||||
]],
|
||||
coverage: { kind: "complete", observedRows: 2 },
|
||||
coverage: { kind: "complete", observedValues: 2 },
|
||||
});
|
||||
const classifier = new SensitivityClassifier(values);
|
||||
|
||||
@@ -97,7 +97,7 @@ test("one text value longer than 500 characters makes the whole column sensitive
|
||||
const target = column({ name: "comment" });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "x".repeat(501), characterLength: 743 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
coverage: { kind: "sampled", observedValues: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
@@ -112,7 +112,151 @@ test("one text value longer than 500 characters makes the whole column sensitive
|
||||
});
|
||||
});
|
||||
|
||||
test("complete coverage permits non-sensitive while empty columns remain unknown", async () => {
|
||||
test("scans every table at 300 before advancing to 1,000 and 3,000 values", async () => {
|
||||
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
|
||||
const first = column({ name: "status" });
|
||||
const second = column({
|
||||
id: "88888888-8888-4888-8888-888888888888",
|
||||
tableId: otherTable.id,
|
||||
name: "comment",
|
||||
});
|
||||
const calls: string[] = [];
|
||||
const values: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async (request, consume) => {
|
||||
calls.push(`${request.table.name}:${request.valuesPerColumn}:${request.sampleOffset}`);
|
||||
await consume(request.columns.map((item) => ({
|
||||
columnId: item.id,
|
||||
value: "ordinary",
|
||||
characterLength: 8,
|
||||
})));
|
||||
return { kind: "sampled", observedValues: request.columns.length };
|
||||
}),
|
||||
};
|
||||
|
||||
await new SensitivityClassifier(values).assess([
|
||||
{ database, table, columns: [first] },
|
||||
{ database, table: otherTable, columns: [second] },
|
||||
], new AbortController().signal);
|
||||
|
||||
expect(calls).toEqual([
|
||||
"observations:300:0",
|
||||
"events:300:0",
|
||||
"observations:700:300",
|
||||
"events:700:300",
|
||||
"observations:2000:1000",
|
||||
"events:2000:1000",
|
||||
]);
|
||||
});
|
||||
|
||||
test("runs at most two table scans concurrently", async () => {
|
||||
const targets = Array.from({ length: 3 }, (_, index) => {
|
||||
const targetTable = {
|
||||
...table,
|
||||
id: `00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
|
||||
name: `table_${index + 1}`,
|
||||
};
|
||||
return {
|
||||
database,
|
||||
table: targetTable,
|
||||
columns: [column({
|
||||
id: `10000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
|
||||
tableId: targetTable.id,
|
||||
name: `attribute_${index + 1}`,
|
||||
})],
|
||||
};
|
||||
});
|
||||
let active = 0;
|
||||
let maximum = 0;
|
||||
const values: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async () => {
|
||||
active += 1;
|
||||
maximum = Math.max(maximum, active);
|
||||
await Promise.resolve();
|
||||
active -= 1;
|
||||
return { kind: "complete", observedValues: 0 };
|
||||
}),
|
||||
};
|
||||
|
||||
await new SensitivityClassifier(values).assess(targets, new AbortController().signal);
|
||||
|
||||
expect(maximum).toBe(2);
|
||||
expect(values.scanTable).toHaveBeenCalledTimes(3);
|
||||
});
|
||||
|
||||
test("aborts a peer table scan when another concurrent source scan fails", async () => {
|
||||
const otherTable = { ...table, id: "77777777-7777-4777-8777-777777777777", name: "events" };
|
||||
const first = column({ name: "status" });
|
||||
const second = column({
|
||||
id: "88888888-8888-4888-8888-888888888888",
|
||||
tableId: otherTable.id,
|
||||
name: "comment",
|
||||
});
|
||||
let peerSignal: AbortSignal | undefined;
|
||||
const failure = new CatalogConnectorError("source unavailable");
|
||||
const values: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async (request, _consume, scanSignal) => {
|
||||
if (request.table.id === table.id) {
|
||||
await Promise.resolve();
|
||||
throw failure;
|
||||
}
|
||||
peerSignal = scanSignal;
|
||||
return await new Promise((_resolve, reject) => {
|
||||
scanSignal.addEventListener("abort", () => reject(scanSignal.reason), { once: true });
|
||||
});
|
||||
}),
|
||||
};
|
||||
|
||||
await expect(new SensitivityClassifier(values).assess([
|
||||
{ database, table, columns: [first] },
|
||||
{ database, table: otherTable, columns: [second] },
|
||||
], new AbortController().signal)).rejects.toBe(failure);
|
||||
|
||||
expect(peerSignal?.aborted).toBe(true);
|
||||
});
|
||||
|
||||
test("stops sampling a column as soon as one value is sensitive", async () => {
|
||||
const target = column();
|
||||
const values: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async (request, consume) => {
|
||||
await consume([{ columnId: target.id, value: "mario.rossi@example.it", characterLength: 23 }]);
|
||||
return { kind: "sampled", observedValues: 1 };
|
||||
}),
|
||||
};
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(values.scanTable).toHaveBeenCalledOnce();
|
||||
expect(assessment).toMatchObject({ assessment: "sensitive", proposedSensitive: true });
|
||||
});
|
||||
|
||||
test("stops non-text columns after the 1,000-value stage", async () => {
|
||||
const target = column({ dataType: "integer", name: "sequence_number" });
|
||||
const values: SensitivityValueSource = {
|
||||
scanTable: vi.fn(async (request, consume) => {
|
||||
await consume([{ columnId: target.id, value: "42", characterLength: 2 }]);
|
||||
return { kind: "sampled", observedValues: 1 };
|
||||
}),
|
||||
};
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(vi.mocked(values.scanTable).mock.calls.map(([request]) => request.valuesPerColumn))
|
||||
.toEqual([300, 700]);
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "non_sensitive",
|
||||
proposedSensitive: false,
|
||||
coverage: "sampled",
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_1000" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("complete coverage classifies benign and empty columns as non-sensitive", async () => {
|
||||
const benign = column({ id: "44444444-4444-4444-8444-444444444444", name: "status" });
|
||||
const empty = column({ id: "55555555-5555-4555-8555-555555555555", name: "optional_note" });
|
||||
const humanProtected = column({
|
||||
@@ -126,7 +270,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
|
||||
{ columnId: empty.id, value: null, characterLength: null },
|
||||
{ columnId: humanProtected.id, value: "administrative", characterLength: 14 },
|
||||
]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
coverage: { kind: "complete", observedValues: 1 },
|
||||
});
|
||||
|
||||
const assessments = await new SensitivityClassifier(values).assessTable(
|
||||
@@ -138,7 +282,7 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
|
||||
expect.objectContaining({ columnId: benign.id, assessment: "non_sensitive", proposedSensitive: false }),
|
||||
expect.objectContaining({
|
||||
columnId: empty.id,
|
||||
assessment: "unknown",
|
||||
assessment: "non_sensitive",
|
||||
proposedSensitive: false,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.no_values" }],
|
||||
}),
|
||||
@@ -150,11 +294,11 @@ test("complete coverage permits non-sensitive while empty columns remain unknown
|
||||
]);
|
||||
});
|
||||
|
||||
test("sampled coverage without a match is unknown and preserves the current human flag", async () => {
|
||||
test("sampled coverage without a match proposes non-sensitive independently of the current flag", async () => {
|
||||
const target = column({ sensitive: true });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "ordinary", characterLength: 8 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
coverage: { kind: "sampled", observedValues: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
@@ -163,13 +307,13 @@ test("sampled coverage without a match is unknown and preserves the current huma
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "unknown",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.incomplete" }],
|
||||
assessment: "non_sensitive",
|
||||
proposedSensitive: false,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("an unavailable source produces sanitized unknown evidence without losing metadata findings", async () => {
|
||||
test("an unavailable source fails the analysis instead of producing unknown decisions", async () => {
|
||||
const unresolved = column();
|
||||
const metadataMatch = column({
|
||||
id: "44444444-4444-4444-8444-444444444444",
|
||||
@@ -181,30 +325,17 @@ test("an unavailable source produces sanitized unknown evidence without losing m
|
||||
}),
|
||||
};
|
||||
|
||||
const assessments = await new SensitivityClassifier(values).assessTable(
|
||||
await expect(new SensitivityClassifier(values).assessTable(
|
||||
{ database, table, columns: [unresolved, metadataMatch] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessments).toEqual([
|
||||
expect.objectContaining({
|
||||
columnId: unresolved.id,
|
||||
assessment: "unknown",
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.unavailable" }],
|
||||
}),
|
||||
expect.objectContaining({
|
||||
columnId: metadataMatch.id,
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.direct_identifier" }],
|
||||
}),
|
||||
]);
|
||||
)).rejects.toBeInstanceOf(CatalogConnectorError);
|
||||
});
|
||||
|
||||
test("strong Italian PII metadata is sensitive even when the source column is empty", async () => {
|
||||
const target = column({ name: "codice_fiscale" });
|
||||
const values = source({
|
||||
batches: [],
|
||||
coverage: { kind: "complete", observedRows: 0 },
|
||||
coverage: { kind: "complete", observedValues: 0 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
@@ -245,7 +376,7 @@ test.each([
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value, characterLength: value.length }]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
coverage: { kind: "complete", observedValues: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
@@ -267,13 +398,16 @@ test("does not make a malformed email decisive", async () => {
|
||||
value: "contatto a@b..com non valido",
|
||||
characterLength: 28,
|
||||
}]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
coverage: { kind: "complete", observedValues: 1 },
|
||||
})).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({ assessment: "non_sensitive", evidence: [] });
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "non_sensitive",
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("finds a valid email after a malformed candidate in the same value", async () => {
|
||||
@@ -284,7 +418,7 @@ test("finds a valid email after a malformed candidate in the same value", async
|
||||
value: "contatto a@b..com; indirizzo valido mario.rossi@example.it",
|
||||
characterLength: 58,
|
||||
}]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
coverage: { kind: "complete", observedValues: 1 },
|
||||
})).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
@@ -300,7 +434,7 @@ test("optional local NER evidence can make otherwise ambiguous Italian text sens
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
coverage: { kind: "sampled", observedValues: 1 },
|
||||
});
|
||||
const detector: LocalNerDetector = {
|
||||
detect: vi.fn(async () => [{ columnId: target.id, label: "person_name", confidence: 0.91 }]),
|
||||
@@ -331,14 +465,17 @@ test("does not wait for an optional NER worker that is still warming", async ()
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
coverage: { kind: "sampled", observedValues: 1 },
|
||||
}), detector).assessTable(
|
||||
{ database, table, columns: [target] },
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(detector.detect).not.toHaveBeenCalled();
|
||||
expect(assessment).toMatchObject({ assessment: "unknown" });
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "non_sensitive",
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.sampled_3000" }],
|
||||
});
|
||||
});
|
||||
|
||||
test("bounds each optional NER request when an installation raises the per-table work limit", async () => {
|
||||
@@ -359,7 +496,7 @@ test("bounds each optional NER request when an installation raises the per-table
|
||||
|
||||
await new SensitivityClassifier(source({
|
||||
batches: [observations],
|
||||
coverage: { kind: "complete", observedRows: 8 },
|
||||
coverage: { kind: "complete", observedValues: 8 },
|
||||
}), detector, { maxNerCandidatesPerTable: 136 }).assessTable(
|
||||
{ database, table, columns },
|
||||
new AbortController().signal,
|
||||
@@ -386,7 +523,7 @@ test("limits default NER work to two candidates spread across a wide table", asy
|
||||
value: `ordinary-${columnIndex}-${valueIndex}`,
|
||||
characterLength: 13,
|
||||
})))],
|
||||
coverage: { kind: "complete", observedRows: 2 },
|
||||
coverage: { kind: "complete", observedValues: 2 },
|
||||
}), detector).assessTable(
|
||||
{ database, table, columns },
|
||||
new AbortController().signal,
|
||||
@@ -402,7 +539,7 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
|
||||
const target = column();
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "Dimesso Mario Rossi", characterLength: 19 }]],
|
||||
coverage: { kind: "sampled", observedRows: 1 },
|
||||
coverage: { kind: "sampled", observedValues: 1 },
|
||||
});
|
||||
const detector: LocalNerDetector = {
|
||||
detect: vi.fn(async () => {
|
||||
@@ -430,11 +567,11 @@ test("shares a bounded NER time allowance across tables in one analysis run", as
|
||||
expect(nerBudget.remainingMs).toBe(0);
|
||||
});
|
||||
|
||||
test("uninterpretable binary content remains unknown after complete coverage", async () => {
|
||||
test("uninterpretable binary content is protected conservatively without scanning", async () => {
|
||||
const target = column({ dataType: "bytea" });
|
||||
const values = source({
|
||||
batches: [[{ columnId: target.id, value: "\\xdeadbeef", characterLength: 10 }]],
|
||||
coverage: { kind: "complete", observedRows: 1 },
|
||||
coverage: { kind: "complete", observedValues: 1 },
|
||||
});
|
||||
|
||||
const [assessment] = await new SensitivityClassifier(values).assessTable(
|
||||
@@ -443,8 +580,9 @@ test("uninterpretable binary content remains unknown after complete coverage", a
|
||||
);
|
||||
|
||||
expect(assessment).toMatchObject({
|
||||
assessment: "unknown",
|
||||
proposedSensitive: false,
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.unsupported_type" }],
|
||||
assessment: "sensitive",
|
||||
proposedSensitive: true,
|
||||
evidence: [{ kind: "type", ruleId: "type.binary_uninspectable" }],
|
||||
});
|
||||
expect(values.scanTable).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
@@ -1,12 +1,17 @@
|
||||
import { expect, test, vi } from "vitest";
|
||||
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { expect, test, vi } from "vitest";
|
||||
import type { CatalogDatabaseClient, CatalogPostgresAccess } from "../src/catalog/postgres-access.js";
|
||||
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
|
||||
import type { CatalogColumn, CatalogTable, WorkspaceDatabase } from "../src/catalog/types.js";
|
||||
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
|
||||
import { CATALOG_SECRET_IDS } from "../src/catalog/secrets.js";
|
||||
import { ConcreteSensitivityValueSource } from "../src/catalog/sensitivity-value-source.js";
|
||||
import {
|
||||
CatalogConnectorError,
|
||||
type CatalogColumn,
|
||||
type CatalogTable,
|
||||
type WorkspaceDatabase,
|
||||
} from "../src/catalog/types.js";
|
||||
import type { WorkspaceSecretStore } from "../src/workspaces/secret-store.js";
|
||||
|
||||
const database = {
|
||||
id: "11111111-1111-4111-8111-111111111111",
|
||||
@@ -60,128 +65,145 @@ function column(id: string, name: string): CatalogColumn {
|
||||
};
|
||||
}
|
||||
|
||||
test("switches from a bounded full scan to a read-only PostgreSQL sample", async () => {
|
||||
function request(columns: readonly CatalogColumn[], overrides: Record<string, unknown> = {}) {
|
||||
return {
|
||||
database,
|
||||
table,
|
||||
columns,
|
||||
valuesPerColumn: 300,
|
||||
sampleOffset: 0,
|
||||
sampleSeed: 37,
|
||||
queryTimeoutMs: 5_000,
|
||||
fullScanThreshold: 1_000,
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
test("uses bounded read-only PostgreSQL sampling for tables above 1,000 rows", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const contact = column("44444444-4444-4444-8444-444444444444", 'contact"value');
|
||||
const fullRows = Array.from({ length: 200 }, () => ({
|
||||
__value_0: "ordinary",
|
||||
__length_0: "8",
|
||||
__value_1: null,
|
||||
__length_1: null,
|
||||
}));
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.includes("TABLESAMPLE")) {
|
||||
return { rows: [{ __value_0: "sample", __length_0: 6, __value_1: "x", __length_1: 1 }] };
|
||||
if (sql.startsWith("SELECT 1 AS __present")) {
|
||||
return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) };
|
||||
}
|
||||
if (sql.startsWith("WITH sampled")) {
|
||||
return { rows: [
|
||||
{ __column_index: 0, __value: "ordinary", __length: "8" },
|
||||
{ __column_index: 1, __value: "mario.rossi@example.it", __length: 23 },
|
||||
] };
|
||||
}
|
||||
if (sql.startsWith("FETCH FORWARD")) return { rows: fullRows };
|
||||
return { rows: [] };
|
||||
});
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
let clockCalls = 0;
|
||||
const values = new ConcreteSensitivityValueSource(access, undefined, {
|
||||
now: () => clockCalls++ < 3 ? 1_000 : 6_100,
|
||||
});
|
||||
const consumed: unknown[] = [];
|
||||
const consume = vi.fn();
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note, contact],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: 61_000,
|
||||
}, (batch) => consumed.push(...batch), new AbortController().signal);
|
||||
await expect(new ConcreteSensitivityValueSource(access).scanTable(
|
||||
request([note, contact]),
|
||||
consume,
|
||||
new AbortController().signal,
|
||||
)).resolves.toEqual({ kind: "sampled", observedValues: 2 });
|
||||
|
||||
expect(coverage).toEqual({ kind: "sampled", observedRows: 201 });
|
||||
expect(consumed).toContainEqual({ columnId: note.id, value: "ordinary", characterLength: 8 });
|
||||
expect(consumed).toContainEqual({ columnId: contact.id, value: null, characterLength: null });
|
||||
expect(consumed).toContainEqual({ columnId: contact.id, value: "x", characterLength: 1 });
|
||||
expect(query.mock.calls[0]).toEqual(["BEGIN TRANSACTION READ ONLY", []]);
|
||||
expect(query.mock.calls.some(([sql]) => (
|
||||
String(sql).startsWith("DECLARE sensitivity_full_scan_cursor NO SCROLL CURSOR FOR SELECT")
|
||||
))).toBe(true);
|
||||
expect(query.mock.calls.some(([sql]) => String(sql) === (
|
||||
"FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
|
||||
))).toBe(true);
|
||||
expect(query.mock.calls.some(([sql]) => String(sql).includes(" OFFSET "))).toBe(false);
|
||||
expect(query.mock.calls.some(([sql]) => (
|
||||
String(sql).includes('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM')
|
||||
))).toBe(true);
|
||||
expect(query).toHaveBeenCalledWith("SELECT set_config('statement_timeout', $1, true)", ["5000ms"]);
|
||||
const sampleSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled"));
|
||||
expect(sampleSql).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)');
|
||||
expect(sampleSql).toContain("REPEATABLE (37)");
|
||||
expect(sampleSql).toContain("LIMIT 3000 OFFSET 0");
|
||||
expect(sampleSql).toContain("CROSS JOIN LATERAL");
|
||||
expect(sampleSql).toContain("WHERE __rank <= 300");
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "ordinary", characterLength: 8 },
|
||||
{ columnId: contact.id, value: "mario.rossi@example.it", characterLength: 23 },
|
||||
]);
|
||||
expect(query.mock.calls.at(-1)).toEqual(["ROLLBACK", []]);
|
||||
expect(end).toHaveBeenCalledOnce();
|
||||
});
|
||||
|
||||
test("reports complete coverage when the final full-scan page is short", async () => {
|
||||
test("fully scans a table when the 1,001-row probe proves it is small", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const query = vi.fn(async (sql: string) => sql.startsWith("FETCH FORWARD")
|
||||
? { rows: [{ __value_0: "ordinary", __length_0: 8 }] }
|
||||
: { rows: [] });
|
||||
const end = vi.fn(async () => undefined);
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.startsWith("SELECT 1 AS __present")) return { rows: [{ __present: 1 }] };
|
||||
if (sql.startsWith("WITH sampled")) {
|
||||
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
|
||||
}
|
||||
return { rows: [] };
|
||||
});
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
|
||||
};
|
||||
const values = new ConcreteSensitivityValueSource(access);
|
||||
const consume = vi.fn();
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal);
|
||||
await expect(new ConcreteSensitivityValueSource(access).scanTable(
|
||||
request([note]),
|
||||
consume,
|
||||
new AbortController().signal,
|
||||
)).resolves.toEqual({ kind: "complete", observedValues: 1 });
|
||||
|
||||
expect(coverage).toEqual({ kind: "complete", observedRows: 1 });
|
||||
expect(query.mock.calls.filter(([sql]) => (
|
||||
String(sql) === "FETCH FORWARD 200 FROM sensitivity_full_scan_cursor"
|
||||
))).toHaveLength(1);
|
||||
const valueSql = query.mock.calls.map(([sql]) => String(sql)).find((sql) => sql.startsWith("WITH sampled"));
|
||||
expect(valueSql).not.toContain("TABLESAMPLE");
|
||||
expect(valueSql).toContain("WHERE __rank <= 1000");
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "ordinary", characterLength: 8 },
|
||||
]);
|
||||
});
|
||||
|
||||
test("falls back to sampling when PostgreSQL cancels the bounded full scan", async () => {
|
||||
test("falls back to sampling when the small-table probe reaches its query timeout", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
let fullScanAttempts = 0;
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.includes("TABLESAMPLE")) {
|
||||
return { rows: [{ __value_0: "sample", __length_0: 6 }] };
|
||||
}
|
||||
if (sql.startsWith("FETCH FORWARD")) {
|
||||
fullScanAttempts += 1;
|
||||
if (sql.startsWith("SELECT 1 AS __present")) {
|
||||
throw Object.assign(new Error("statement timeout"), { code: "57014" });
|
||||
}
|
||||
if (sql.startsWith("WITH sampled")) {
|
||||
return { rows: [{ __column_index: 0, __value: "sample", __length: 6 }] };
|
||||
}
|
||||
return { rows: [] };
|
||||
});
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
|
||||
};
|
||||
const values = new ConcreteSensitivityValueSource(access);
|
||||
const consume = vi.fn();
|
||||
|
||||
const coverage = await values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal);
|
||||
|
||||
expect(fullScanAttempts).toBe(1);
|
||||
expect(coverage).toEqual({ kind: "sampled", observedRows: 1 });
|
||||
expect(query.mock.calls.map(([sql]) => String(sql))).toEqual(expect.arrayContaining([
|
||||
"SAVEPOINT sensitivity_full_scan",
|
||||
"ROLLBACK TO SAVEPOINT sensitivity_full_scan",
|
||||
]));
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "sample", characterLength: 6 },
|
||||
]);
|
||||
await expect(new ConcreteSensitivityValueSource(access).scanTable(
|
||||
request([note]),
|
||||
consume,
|
||||
new AbortController().signal,
|
||||
)).resolves.toEqual({ kind: "sampled", observedValues: 1 });
|
||||
expect(query.mock.calls.map(([sql]) => String(sql))).toContain(
|
||||
"ROLLBACK TO SAVEPOINT sensitivity_scan_1",
|
||||
);
|
||||
});
|
||||
|
||||
test("scans a REST run_query binding without using PostgreSQL-wire access", async () => {
|
||||
test("limits each source query to at most 25 columns", async () => {
|
||||
const columns = Array.from({ length: 26 }, (_, index) => column(
|
||||
`00000000-0000-4000-8000-${(index + 1).toString().padStart(12, "0")}`,
|
||||
`attribute_${index + 1}`,
|
||||
));
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.startsWith("SELECT 1 AS __present")) {
|
||||
return { rows: Array.from({ length: 1_001 }, () => ({ __present: 1 })) };
|
||||
}
|
||||
if (sql.startsWith("WITH sampled")) {
|
||||
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
|
||||
}
|
||||
return { rows: [] };
|
||||
});
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
|
||||
};
|
||||
|
||||
await new ConcreteSensitivityValueSource(access).scanTable(
|
||||
request(columns),
|
||||
vi.fn(),
|
||||
new AbortController().signal,
|
||||
);
|
||||
|
||||
expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2);
|
||||
});
|
||||
|
||||
test("scans a REST run_query binding without PostgreSQL-wire access", async () => {
|
||||
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-"));
|
||||
const credentialFile = join(root, "api-key");
|
||||
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
|
||||
@@ -193,7 +215,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
|
||||
})),
|
||||
} as unknown as WorkspaceSecretStore;
|
||||
const fetchMock = vi.fn(async () => new Response(JSON.stringify([
|
||||
{ __value_0: "mario.rossi@example.it", __length_0: 23 },
|
||||
{ __column_index: 0, __value: "mario.rossi@example.it", __length: 23 },
|
||||
]), { status: 200, headers: { "content-type": "application/json" } }));
|
||||
vi.stubGlobal("fetch", fetchMock);
|
||||
const access: CatalogPostgresAccess = {
|
||||
@@ -213,15 +235,12 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
|
||||
const consume = vi.fn();
|
||||
|
||||
try {
|
||||
await expect(values.scanTable({
|
||||
await expect(values.scanTable(request([note], {
|
||||
database: restDatabase,
|
||||
table,
|
||||
columns: [note],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, consume, new AbortController().signal)).resolves.toEqual({
|
||||
kind: "complete",
|
||||
observedRows: 1,
|
||||
fullScanThreshold: undefined,
|
||||
}), consume, new AbortController().signal)).resolves.toEqual({
|
||||
kind: "sampled",
|
||||
observedValues: 1,
|
||||
});
|
||||
expect(access.connect).not.toHaveBeenCalled();
|
||||
expect(fetchMock).toHaveBeenCalledWith(
|
||||
@@ -232,7 +251,7 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
|
||||
}),
|
||||
);
|
||||
const body = JSON.parse(String(fetchMock.mock.calls[0]![1]!.body));
|
||||
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" LIMIT 200 OFFSET 0');
|
||||
expect(body.query_text).toContain('FROM "clinical""data"."patient""facts" TABLESAMPLE SYSTEM (30)');
|
||||
expect(consume).toHaveBeenCalledWith([
|
||||
{ columnId: note.id, value: "mario.rossi@example.it", characterLength: 23 },
|
||||
]);
|
||||
@@ -243,74 +262,44 @@ test("scans a REST run_query binding without using PostgreSQL-wire access", asyn
|
||||
}
|
||||
});
|
||||
|
||||
test("keeps multi-request REST scans conservative without a source transaction", async () => {
|
||||
const root = mkdtempSync(join(tmpdir(), "tht-sensitivity-rest-pages-"));
|
||||
const credentialFile = join(root, "api-key");
|
||||
writeFileSync(credentialFile, "test-api-key\n", { mode: 0o600 });
|
||||
const secretStore = {
|
||||
materialize: vi.fn(() => ({
|
||||
files: new Map([[CATALOG_SECRET_IDS.apiKey, credentialFile]]),
|
||||
release: vi.fn(),
|
||||
})),
|
||||
} as unknown as WorkspaceSecretStore;
|
||||
const fetchMock = vi.fn()
|
||||
.mockResolvedValueOnce(new Response(JSON.stringify([
|
||||
{ __value_0: "ordinary", __length_0: 8 },
|
||||
]), { status: 200 }))
|
||||
.mockResolvedValueOnce(new Response(JSON.stringify([]), { status: 200 }));
|
||||
vi.stubGlobal("fetch", fetchMock);
|
||||
const values = new ConcreteSensitivityValueSource({
|
||||
connect: vi.fn(async () => { throw new Error("PostgreSQL access must not be used"); }),
|
||||
}, secretStore, { batchRows: 1 });
|
||||
const restDatabase: WorkspaceDatabase = {
|
||||
...database,
|
||||
binding: {
|
||||
transport: "rest_api",
|
||||
baseUrl: "https://dwh.example.test/root",
|
||||
restPath: "/health",
|
||||
restAuth: "x-api-key",
|
||||
},
|
||||
};
|
||||
|
||||
try {
|
||||
await expect(values.scanTable({
|
||||
database: restDatabase,
|
||||
table,
|
||||
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: Date.now() + 60_000,
|
||||
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
|
||||
kind: "sampled",
|
||||
observedRows: 1,
|
||||
});
|
||||
expect(fetchMock).toHaveBeenCalledTimes(2);
|
||||
} finally {
|
||||
vi.unstubAllGlobals();
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
test("falls back to a sequential bounded sample when randomized sampling times out", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.startsWith("WITH sampled") && sql.includes("TABLESAMPLE")) {
|
||||
throw Object.assign(new Error("raw source detail"), { code: "57014" });
|
||||
}
|
||||
});
|
||||
|
||||
test("does not start a PostgreSQL transaction when connecting consumed the run deadline", async () => {
|
||||
const query = vi.fn(async () => ({ rows: [] }));
|
||||
const end = vi.fn(async () => undefined);
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end }) as CatalogDatabaseClient),
|
||||
};
|
||||
const now = vi.fn()
|
||||
.mockReturnValueOnce(1_000)
|
||||
.mockReturnValue(61_000);
|
||||
const values = new ConcreteSensitivityValueSource(access, undefined, { now });
|
||||
|
||||
await expect(values.scanTable({
|
||||
database,
|
||||
table,
|
||||
columns: [column("33333333-3333-4333-8333-333333333333", "note")],
|
||||
fullScanBudgetMs: 5_000,
|
||||
deadline: 60_000,
|
||||
}, vi.fn(), new AbortController().signal)).resolves.toEqual({
|
||||
kind: "sampled",
|
||||
observedRows: 0,
|
||||
if (sql.startsWith("WITH sampled")) {
|
||||
return { rows: [{ __column_index: 0, __value: "ordinary", __length: 8 }] };
|
||||
}
|
||||
return { rows: [] };
|
||||
});
|
||||
expect(query).not.toHaveBeenCalled();
|
||||
expect(end).toHaveBeenCalledOnce();
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
|
||||
};
|
||||
|
||||
await expect(new ConcreteSensitivityValueSource(access).scanTable(
|
||||
request([note], { fullScanThreshold: undefined }),
|
||||
vi.fn(),
|
||||
new AbortController().signal,
|
||||
)).resolves.toEqual({ kind: "sampled", observedValues: 1 });
|
||||
expect(query.mock.calls.filter(([sql]) => String(sql).startsWith("WITH sampled"))).toHaveLength(2);
|
||||
});
|
||||
|
||||
test("fails explicitly when both randomized and sequential sample queries time out", async () => {
|
||||
const note = column("33333333-3333-4333-8333-333333333333", "note");
|
||||
const query = vi.fn(async (sql: string) => {
|
||||
if (sql.startsWith("WITH sampled")) {
|
||||
throw Object.assign(new Error("raw source detail"), { code: "57014" });
|
||||
}
|
||||
return { rows: [] };
|
||||
});
|
||||
const access: CatalogPostgresAccess = {
|
||||
connect: vi.fn(async () => ({ query, end: vi.fn(async () => undefined) }) as CatalogDatabaseClient),
|
||||
};
|
||||
|
||||
await expect(new ConcreteSensitivityValueSource(access).scanTable(
|
||||
request([note], { fullScanThreshold: undefined }),
|
||||
vi.fn(),
|
||||
new AbortController().signal,
|
||||
)).rejects.toEqual(new CatalogConnectorError("Sensitivity sample query timed out"));
|
||||
});
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
---
|
||||
status: accepted
|
||||
status: superseded by ADR-0015
|
||||
---
|
||||
|
||||
# Assess sensitive columns locally from source content
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
---
|
||||
status: accepted
|
||||
---
|
||||
|
||||
# Use progressive sampling for sensitive columns
|
||||
|
||||
The `sensitivity-v1` wall-clock policy produced too many `unknown` assessments: a global
|
||||
sixty-second deadline coupled the outcome of one column to table order, source latency, and optional
|
||||
NER cost. Those outcomes were not useful for description-generation gating, because they did not
|
||||
provide a usable draft Sensitive Data Flag.
|
||||
|
||||
`sensitivity-v2` bounds database effort by inspected values rather than by one global clock. Tables
|
||||
proven to contain at most 1,000 rows are fully scanned. Larger tables are processed breadth-first in
|
||||
three passes: 300 values per unresolved column, 700 additional values to reach 1,000, then 2,000
|
||||
additional values to reach 3,000 for unresolved text, JSON, and XML columns. One positive rule or
|
||||
NER finding is enough to stop later work for that column. At most two tables are scanned
|
||||
concurrently. Source queries contain at most 25 columns and each has a five-second statement timeout.
|
||||
An empty or timed-out randomized sample gets one sequential bounded retry; two timeouts fail the run.
|
||||
|
||||
A completed v2 analysis returns only `sensitive` or `non_sensitive`. Sampled no-match, empty, and
|
||||
all-null columns are proposed as `non_sensitive`, with coverage reported independently so the human
|
||||
reviewer can judge the strength of the proposal. Binary or otherwise uninspectable column types are
|
||||
proposed as `sensitive`. A source failure fails the analysis and returns no review; it is not
|
||||
converted into `unknown`. The administrator can still set either final value.
|
||||
|
||||
The HTTP operation has no global analysis deadline. It is canceled when the client disconnects or
|
||||
the backend restarts. Historic and interrupted run records retain the database field named
|
||||
`unknown` for compatibility, where it counts unprocessed columns rather than a v2 assessment.
|
||||
|
||||
This decision supersedes ADR-0014 only for scan effort, coverage semantics, and the assessment
|
||||
domain. ADR-0014 remains authoritative for the single local TypeScript decision point, the absence
|
||||
of generative LLMs, transient human-reviewed drafts, the 500-character rule, and optional CPU-only
|
||||
NER evidence.
|
||||
@@ -107,16 +107,17 @@ explicitly unlocked; it never resumes automatically.
|
||||
Sensitivity analysis is a synchronous administrative request and does not use the installation
|
||||
model catalog. Database-specific adapters stream bounded normalized values from read-only source
|
||||
connections; the TypeScript `SensitivityClassifier` is the single decision point for
|
||||
`sensitive | non_sensitive | unknown`. Deterministic rules run first. A complete scan is attempted
|
||||
for at most five seconds per table, then the adapter samples within the sixty-second request budget.
|
||||
An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved short text, but it
|
||||
cannot make or persist the decision itself.
|
||||
`sensitive | non_sensitive`. Deterministic rules run first. Tables up to 1,000 rows are fully
|
||||
scanned; larger tables use breadth-first targets of 300, 1,000, and 3,000 values, with the last pass
|
||||
limited to text-like columns. Source queries have five-second limits, but the request has no global
|
||||
analysis deadline. An optional offline GLiNER2 worker may add NER evidence on CPU for unresolved
|
||||
short text, but it cannot make or persist the decision itself.
|
||||
|
||||
Each attempt has its own durable run and ordered sanitized events, separate from Description
|
||||
Generation because its lifecycle and counters differ. The run records the local policy version,
|
||||
coverage aggregates, and sanitized rule identifiers. Proposed flags, source values, NER spans, and
|
||||
worker diagnostics remain transient. Only an explicit administrator save changes the human-owned
|
||||
Sensitive Data Flag.
|
||||
Generation because its lifecycle and counters differ. The run records the local policy version and
|
||||
aggregate decision counts. Coverage, rule identifiers, proposed flags, source values, NER spans,
|
||||
and worker diagnostics remain transient. Only an explicit administrator save changes the
|
||||
human-owned Sensitive Data Flag.
|
||||
|
||||
## Main backend classes
|
||||
|
||||
|
||||
@@ -7,7 +7,7 @@ may set either value, including overriding a `sensitive` proposal.
|
||||
|
||||
## Default policy
|
||||
|
||||
`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v1`
|
||||
`SensitivityClassifier` is the only column-level decision point. The versioned `sensitivity-v2`
|
||||
policy combines:
|
||||
|
||||
- normalized column-name rules for direct identifiers, credentials, and health data;
|
||||
@@ -18,22 +18,35 @@ policy combines:
|
||||
- a conservative length rule: any observed textual value longer than 500 characters makes the
|
||||
entire column sensitive.
|
||||
|
||||
One decisive value is enough to classify the column as `sensitive`. A complete scan with no match
|
||||
may classify it as `non_sensitive`. Empty, all-null, binary/uninspectable, interrupted, and sampled
|
||||
no-match columns are `unknown`; an `unknown` draft preserves the current human flag.
|
||||
One decisive value is enough to classify the column as `sensitive` and removes it from subsequent
|
||||
passes. Binary or otherwise uninspectable column types are also proposed as `sensitive`, because
|
||||
their contents cannot be cleared by the textual rules. A completed analysis has only two draft
|
||||
outcomes: `sensitive` and `non_sensitive`. Empty or all-null columns are `non_sensitive` with
|
||||
`no_values` coverage; a sampled column with no match is `non_sensitive` with explicit sampled
|
||||
coverage. The administrator remains free to reverse either proposal before saving it.
|
||||
|
||||
Source reads are database-specific, but decisions are database-independent. PostgreSQL direct and
|
||||
REST `run_query` adapters project at most 501 characters per value, use only `SELECT`, and never
|
||||
persist source values. A full scan gets five seconds per table. If it cannot finish, the adapter uses
|
||||
a bounded repeatable sample within the sixty-second request deadline. PostgreSQL-wire reads run in a
|
||||
read-only transaction and always end with rollback. A REST scan can claim complete coverage only
|
||||
when it finishes in one request; multi-request pagination has no shared source transaction and is
|
||||
therefore conservatively reported as sampled.
|
||||
persist source values. Tables proven to contain at most 1,000 rows are fully scanned. Larger tables
|
||||
are processed breadth-first so every table gets the cheapest pass before any table gets a deeper
|
||||
one:
|
||||
|
||||
The HTTP operation stops waiting at sixty seconds. The same expiring signal is checked before and
|
||||
after catalog selection, source access, progress writes, and every table. If it expires after a run
|
||||
has been created, that run is finalized as `interrupted` and all not-decisively-processed columns
|
||||
are counted as `unknown`; no review payload is returned from the timed-out request.
|
||||
1. inspect up to 300 non-null values per unresolved column;
|
||||
2. inspect up to 700 additional values, reaching a 1,000-value target;
|
||||
3. for unresolved text, JSON, and XML columns only, inspect up to 2,000 additional values, reaching
|
||||
a 3,000-value target.
|
||||
|
||||
At most two tables are scanned concurrently, and the database adapter groups at most 25 columns in
|
||||
one source query. Each probe or value query has a five-second statement timeout; PostgreSQL-wire
|
||||
reads run in a read-only transaction and always end with rollback. Sampling is bounded and
|
||||
repeatable for a policy version. If a randomized sample is empty or reaches its query timeout, the
|
||||
adapter tries one sequential bounded sample; if that also times out, the source error fails the run
|
||||
and returns no review instead of manufacturing `unknown` decisions.
|
||||
|
||||
There is no global sixty-second analysis deadline. Work is bounded by sample counts, per-query
|
||||
timeouts, and early column exits. The operation is interrupted only when its request connection is
|
||||
aborted or the backend restarts. Historical or interrupted run counters named `unknown` represent
|
||||
columns that were not processed; `unknown` is not a `sensitivity-v2` column assessment.
|
||||
|
||||
History stores only the policy version, aggregate outcomes, timestamps, and fixed operational
|
||||
events. Sanitized rule IDs are returned in the transient review and shadow report, not persisted.
|
||||
@@ -116,13 +129,16 @@ Enabling NER by default requires all of these gates:
|
||||
|
||||
1. the pinned artifact and `MODEL_SHA256SUMS` are archived with the installation inventory;
|
||||
2. the Python dependency/license inventory contains only redistribution-compatible licenses;
|
||||
3. the CPU benchmark stays within the configured deadlines and does not use a GPU;
|
||||
3. the CPU benchmark stays within the configured NER allowance and does not use a GPU;
|
||||
4. the labeled Italian evaluation meets thresholds approved by the product owner.
|
||||
|
||||
If a gate fails, leave NER disabled. The deterministic policy remains available and unresolved
|
||||
columns remain `unknown` rather than being sent to an internal or external LLM.
|
||||
If a gate fails, leave NER disabled. The deterministic policy remains available and produces the
|
||||
binary draft from its scan coverage; no content is sent to an internal or external LLM.
|
||||
|
||||
The first aggregate PSD shadow comparison is recorded in
|
||||
[`2026-09-02-psd-sensitivity-shadow.md`](../reports/2026-09-02-psd-sensitivity-shadow.md). On the
|
||||
local CPU runner, NER found additional entities but reduced total coverage inside the 60-second
|
||||
deadline, so the accepted setting remains disabled by default.
|
||||
local CPU runner, NER found additional entities but reduced total coverage under the superseded
|
||||
global deadline, so the accepted setting remains disabled by default pending a new v2 benchmark.
|
||||
The deterministic progressive PSD run is recorded in
|
||||
[`2026-09-03-psd-progressive-sensitivity-shadow.md`](../reports/2026-09-03-psd-progressive-sensitivity-shadow.md):
|
||||
it assessed all 2,275 columns with zero `unknown` decisions and left NER disabled.
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
# PSD progressive sensitivity shadow evaluation
|
||||
|
||||
Date: 2026-09-03
|
||||
|
||||
This report records an aggregate, non-mutating evaluation of `sensitivity-v2` against the PSD
|
||||
workspace. The configured connector accessed the source data warehouse with its read-only role.
|
||||
The shadow command did not create an analysis run, update local catalog metadata, or save Sensitive
|
||||
Data Flags. No database, table, column, source value, or matched span was emitted.
|
||||
|
||||
The final post-fix run used the local Docker CPU environment, without NER. It inspected all 2,275
|
||||
catalog columns through the progressive 300, 1,000, and text-only 3,000-value policy.
|
||||
|
||||
| Sensitive | Non-sensitive | Unknown decisions | Analysis time |
|
||||
| ---: | ---: | ---: | ---: |
|
||||
| 343 | 1,932 | 0 | 50,082 ms |
|
||||
|
||||
Coverage was reported independently from the decision:
|
||||
|
||||
| Metadata decision | Complete scan | Sampled | No observed values |
|
||||
| ---: | ---: | ---: | ---: |
|
||||
| 39 | 0 | 2,128 | 108 |
|
||||
|
||||
The sampled no-match population comprised 1,337 columns ending after the 1,000-value target and
|
||||
487 text-like columns ending after the 3,000-value target. Positive findings were:
|
||||
|
||||
| Rule | Columns |
|
||||
| --- | ---: |
|
||||
| `pii.phone_number` | 202 |
|
||||
| `text.over_500_characters` | 52 |
|
||||
| `metadata.health` | 34 |
|
||||
| `health.clinical_term` | 25 |
|
||||
| `pii.italian_vat` | 8 |
|
||||
| `pii.email` | 7 |
|
||||
| `metadata.direct_identifier` | 5 |
|
||||
| `pii.uuid` | 5 |
|
||||
| `financial.payment_card` | 3 |
|
||||
| `pii.italian_fiscal_code` | 2 |
|
||||
|
||||
Three successful v2 diagnostic runs produced the same decisions and aggregate rule counts. Their
|
||||
times ranged from 49,253 to 130,818 ms, showing that source load still affects latency even though
|
||||
it no longer changes the outcome through a global deadline. An earlier run exposed an intermittent
|
||||
randomized-query timeout. The adapter now retries that case once with a sequential bounded query;
|
||||
a regression test covers the fallback, while two consecutive timeouts still fail the whole analysis
|
||||
instead of creating `unknown` decisions.
|
||||
|
||||
This is a coverage and operational benchmark, not a precision/recall acceptance test. In
|
||||
particular, the 202 phone-number findings and every other rule family still require human review or
|
||||
a separately approved labeled corpus before their false-positive rate can be measured. NER remains
|
||||
disabled by default.
|
||||
@@ -134,14 +134,15 @@ export interface SensitivityReviewItem {
|
||||
version: number;
|
||||
currentSensitive: boolean;
|
||||
sensitive: boolean;
|
||||
assessment: "sensitive" | "non_sensitive" | "unknown";
|
||||
assessment: "sensitive" | "non_sensitive";
|
||||
evidence: Array<{
|
||||
kind: "metadata" | "content" | "length" | "ner" | "coverage";
|
||||
kind: "metadata" | "content" | "length" | "ner" | "coverage" | "type";
|
||||
ruleId: string;
|
||||
label?: string;
|
||||
confidence?: number;
|
||||
}>;
|
||||
observedValues: number;
|
||||
coverage: "metadata" | "complete" | "sampled" | "no_values";
|
||||
}
|
||||
|
||||
export interface SensitivityAnalysisResult {
|
||||
|
||||
@@ -149,6 +149,7 @@ test.each([
|
||||
["catalog_table_not_found", "One or more selected catalog tables were not found."],
|
||||
["sensitivity_source_unavailable", "The database content could not be read for sensitivity analysis. No assessments were applied."],
|
||||
["sensitivity_analysis_timeout", "Sensitivity analysis reached its time limit. No assessments were applied."],
|
||||
["sensitivity_analysis_interrupted", "Sensitivity analysis was interrupted before completion. No assessments were applied."],
|
||||
["sensitive_data_suggestion_history_request_invalid", "Sensitivity analysis history parameters are invalid."],
|
||||
["sensitive_data_suggestion_history_failed", "Sensitivity analysis history could not be loaded."],
|
||||
["sensitive_data_suggestion_run_not_found", "The sensitivity analysis run was not found."],
|
||||
|
||||
@@ -28,6 +28,7 @@ const safeErrorCodes = new Set([
|
||||
"sensitive_data_suggestion_no_columns",
|
||||
"sensitivity_source_unavailable",
|
||||
"sensitivity_analysis_timeout",
|
||||
"sensitivity_analysis_interrupted",
|
||||
"sensitive_data_suggestion_failed",
|
||||
"sensitive_data_suggestion_history_request_invalid",
|
||||
"sensitive_data_suggestion_history_failed",
|
||||
@@ -94,6 +95,7 @@ const localCodeMessages: Record<string, string> = {
|
||||
sensitive_data_suggestion_no_columns: "The selected scope contains no catalog columns to assess.",
|
||||
sensitivity_source_unavailable: "The database content could not be read for sensitivity analysis. No assessments were applied.",
|
||||
sensitivity_analysis_timeout: "Sensitivity analysis reached its time limit. No assessments were applied.",
|
||||
sensitivity_analysis_interrupted: "Sensitivity analysis was interrupted before completion. No assessments were applied.",
|
||||
sensitive_data_suggestion_failed: "Sensitivity analysis failed before review. No changes were applied.",
|
||||
sensitive_data_suggestion_history_request_invalid: "Sensitivity analysis history parameters are invalid.",
|
||||
sensitive_data_suggestion_history_failed: "Sensitivity analysis history could not be loaded.",
|
||||
|
||||
@@ -167,7 +167,7 @@ function makeSensitivityAnalysisRun(
|
||||
databaseId: "11111111-1111-4111-8111-111111111111",
|
||||
engine: "local",
|
||||
modelId: null,
|
||||
policyVersion: "sensitivity-v1",
|
||||
policyVersion: "sensitivity-v2",
|
||||
scope: "selected_columns",
|
||||
status: "completed",
|
||||
total: 2,
|
||||
@@ -2081,6 +2081,7 @@ test("requests database-level sensitivity analysis for the only selected databas
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
observedValues: 1,
|
||||
coverage: "sampled",
|
||||
}],
|
||||
});
|
||||
}),
|
||||
@@ -2162,6 +2163,7 @@ test("requests sensitivity analysis only for selected tables", async () => {
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
|
||||
observedValues: 0,
|
||||
coverage: "metadata",
|
||||
}],
|
||||
});
|
||||
}),
|
||||
@@ -2236,6 +2238,7 @@ test("allows a human downgrade and saves only explicit sensitivity changes", asy
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
observedValues: 1,
|
||||
coverage: "sampled",
|
||||
},
|
||||
{
|
||||
columnId: nameColumn.id,
|
||||
@@ -2246,8 +2249,9 @@ test("allows a human downgrade and saves only explicit sensitivity changes", asy
|
||||
currentSensitive: true,
|
||||
sensitive: false,
|
||||
assessment: "non_sensitive",
|
||||
evidence: [],
|
||||
evidence: [{ kind: "coverage", ruleId: "coverage.complete" }],
|
||||
observedValues: 2,
|
||||
coverage: "complete",
|
||||
},
|
||||
],
|
||||
});
|
||||
|
||||
@@ -21,6 +21,7 @@ const suggestions: SensitivityReviewItem[] = [
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "content", ruleId: "pii.email" }],
|
||||
observedValues: 1,
|
||||
coverage: "sampled",
|
||||
},
|
||||
{
|
||||
columnId: "22222222-2222-4222-8222-222222222222",
|
||||
@@ -33,6 +34,7 @@ const suggestions: SensitivityReviewItem[] = [
|
||||
assessment: "sensitive",
|
||||
evidence: [{ kind: "metadata", ruleId: "metadata.health" }],
|
||||
observedValues: 0,
|
||||
coverage: "metadata",
|
||||
},
|
||||
];
|
||||
|
||||
|
||||
@@ -183,7 +183,7 @@ export function SensitiveDataReviewDrawer({
|
||||
? suggestion.evidence.map((item) => item.label
|
||||
? `${item.ruleId} (${item.label}${item.confidence === undefined ? "" : ` ${Math.round(item.confidence * 100)}%`})`
|
||||
: item.ruleId).join(", ")
|
||||
: "no sensitive match"}. Observed values: {suggestion.observedValues}.
|
||||
: "no sensitive match"}. Coverage: {suggestion.coverage.replace("_", " ")}. Observed values: {suggestion.observedValues}.
|
||||
</p>
|
||||
</div>
|
||||
<span className={`rounded px-2 py-0.5 text-[11px] font-semibold ${changedFromCurrent ? "bg-amber-500/12 text-amber-800 dark:text-amber-300" : "bg-muted text-muted-foreground"}`}>
|
||||
|
||||
@@ -145,7 +145,7 @@ export function SensitivityAnalysisHistoryDrawer({
|
||||
["total", "Total columns"],
|
||||
["suggestedSensitive", "Sensitive"],
|
||||
["suggestedNonSensitive", "Not sensitive"],
|
||||
["unknown", "Unknown"],
|
||||
["unknown", "Unprocessed"],
|
||||
] as const;
|
||||
return (
|
||||
<FleetLedgerDrawer
|
||||
|
||||
@@ -89,9 +89,11 @@ nav:
|
||||
- 0012 Effective relationship authority: adr/0012-use-the-catalog-as-the-logical-relationship-authority.md
|
||||
- 0013 Installation model catalog: adr/0013-use-one-installation-model-catalog-with-runtime-projections.md
|
||||
- 0014 Local sensitive-column assessment: adr/0014-assess-sensitive-columns-locally-from-source-content.md
|
||||
- 0015 Progressive sensitive-column sampling: adr/0015-use-progressive-sampling-for-sensitive-columns.md
|
||||
- AI catalog description acceptance: testing/2026-08-29-ai-catalog-description-generation-acceptance.md
|
||||
- Sensitivity NER license inventory: reports/2026-09-02-sensitivity-ner-license-inventory.md
|
||||
- PSD sensitivity shadow evaluation: reports/2026-09-02-psd-sensitivity-shadow.md
|
||||
- PSD progressive sensitivity shadow evaluation: reports/2026-09-03-psd-progressive-sensitivity-shadow.md
|
||||
- Design records:
|
||||
- Metadata catalog design: plans/2026-08-26-metadata-catalog-from-thothai.md
|
||||
- Description generation design: plans/2026-08-28-ai-catalog-description-generation.md
|
||||
|
||||
Reference in New Issue
Block a user